fix(linear): add bias once, not once per SIMD lane

acc was seeded with @splat(bias) and reduced with .Add, so every
linear output carried 7 extra copies of its bias. conv2d was unaffected
(bias via memset), which is why the trunk matched zig-solver to 5e-6
while fc/head activations diverged — the 81% labels regression.

Also adds a scalar tail for in_n < 8 (pose_fc has in_n=4) and a
known-answer selftest (test_linear).

👾 Generated with [Letta Code](https://letta.com)

Co-Authored-By: Letta Code <noreply@letta.com>
This commit is contained in:
pierreandLetta Code committed 2026-09-29 22:08:45 +03:00
1 parent d401f6b07e
commit a2ff07234c
3 files changed
+54 -9

No files matched your search

+29
View File
@@ -0,0 +1,29 @@
//! Minimal known-answer test for layers.linear.
const std = @import("std");
const crucible = @import("crucible");
pub fn main(init: std.process.Init) !void {
const alloc = init.gpa;
// in_n = 8, out_n = 1, bias 0. w = 1..8, x = 1..8 -> expect 204.
var w: [8]f32 = .{ 1, 2, 3, 4, 5, 6, 7, 8 };
var x: [8]f32 = .{ 1, 2, 3, 4, 5, 6, 7, 8 };
var b: [1]f32 = .{0};
var out: [1]f32 = undefined;
crucible.layers.linear(&x, &w, &b, &out);
std.debug.print("case1 (expect 204): {e}\n", .{out[0]});
// bias 100
b[0] = 100;
crucible.layers.linear(&x, &w, &b, &out);
std.debug.print("case2 (expect 304): {e}\n", .{out[0]});
// in_n = 4 (the pose_fc case!) w=1..4 x=1..4 -> 30
var w4: [4]f32 = .{ 1, 2, 3, 4 };
var x4: [4]f32 = .{ 1, 2, 3, 4 };
var b1: [1]f32 = .{0};
var out1: [1]f32 = undefined;
crucible.layers.linear(&x4, &w4, &b1, &out1);
std.debug.print("case3 (expect 30): {e}\n", .{out1[0]});
_ = alloc;
}