diff --git a/src/letta.zig b/src/letta.zig index 9b7a8b8..8a9a5bc 100644 --- a/src/letta.zig +++ b/src/letta.zig @@ -120,8 +120,12 @@ pub fn streamInferConv( var line_len: usize = 0; var buf: [16 * 1024]u8 = undefined; + var stream_ok = false; // did we see a proper result/completion line? while (true) { - const n = stdout_file.readStreaming(io, &.{&buf}) catch break; + const n = stdout_file.readStreaming(io, &.{&buf}) catch |e| { + std.debug.print("inferon/letta: stdout read error {s} — reply may be TRUNCATED\n", .{@errorName(e)}); + break; + }; if (n == 0) break; raw_tail.appendSlice(allocator, buf[0..n]) catch {}; if (raw_tail.items.len > 4096) { @@ -133,6 +137,7 @@ pub fn streamInferConv( if (ch == '\n') { if (line_len > 0) { handleLine(allocator, line_buf[0..line_len], ctx, onEvent, &reply); + if (std.mem.indexOf(u8, line_buf[0..line_len], "\"type\":\"result\"") != null) stream_ok = true; line_len = 0; } } else if (line_len < line_buf.len) { @@ -141,8 +146,21 @@ pub fn streamInferConv( } } } + // CRITICAL: the last line may arrive WITHOUT a trailing newline (EOF + // right after the final JSON). It is still a complete line — dropping + // it silently truncated replies (the text of the last assistant message + // or the result line was lost). + if (line_len > 0) { + handleLine(allocator, line_buf[0..line_len], ctx, onEvent, &reply); + if (std.mem.indexOf(u8, line_buf[0..line_len], "\"type\":\"result\"") != null) stream_ok = true; + } _ = child.wait(io) catch {}; + if (!stream_ok) { + std.debug.print("inferon/letta: stream ended WITHOUT result line — reply likely TRUNCATED ({d} bytes collected). tail:\n{s}\n", .{ reply.items.len, raw_tail.items }); + } + std.debug.print("inferon/letta: stream complete, reply {d} bytes, result_line={}\n", .{ reply.items.len, stream_ok }); + if (reply.items.len == 0 and raw_tail.items.len > 0) { std.debug.print("inferon/letta: EMPTY reply, raw stream tail:\n{s}\n", .{raw_tail.items}); }