// llama.cpp streaming — tests the REAL completeStream against SSE that is split // across network chunks at awkward offsets (mid-"data:" prefix, mid-JSON). This is // the exact failure mode that produced empty responses; the parser must reassemble. const { test } = require('node:test'); const assert = require('node:assert'); const { completeStream } = require('../packages/inference-service/src/providers/llamacpp'); // Build a full SSE body, then hand it back in arbitrarily-split chunks via a fake fetch. function mockLlamaServer(chunks) { global.fetch = async () => ({ ok: true, body: (async function* () { for (const c of chunks) yield Buffer.from(c); })(), }); } async function collect() { let text = '', final = null; for await (const ev of completeStream('prompt', {})) { if (ev.done) final = ev; else text += ev.response; } return { text, final }; } const sse = (obj) => `data: ${JSON.stringify(obj)}\n\n`; test('reassembles content when SSE lines are split across chunks', async () => { const body = sse({ choices: [{ delta: { content: 'Good' } }] }) + sse({ choices: [{ delta: { content: ' morning' }, finish_reason: null }] }) + sse({ choices: [{ delta: {}, finish_reason: 'stop' }], model: 'gemma-test' }) + sse({ choices: [], usage: { completion_tokens: 5, prompt_tokens: 11 } }) + 'data: [DONE]\n\n'; // Split at nasty offsets: mid-prefix and mid-JSON const chunks = [body.slice(0, 9), body.slice(9, 40), body.slice(40, 41), body.slice(41, 110), body.slice(110)]; mockLlamaServer(chunks); const { text, final } = await collect(); assert.strictEqual(text, 'Good morning'); assert.strictEqual(final.model, 'gemma-test'); assert.strictEqual(final.tokenCount, 16, 'completion + prompt tokens'); assert.strictEqual(final.done, true); }); test('a single malformed line is skipped without killing the stream', async () => { const body = sse({ choices: [{ delta: { content: 'A' } }] }) + 'data: {not valid json\n\n' + sse({ choices: [{ delta: { content: 'B' }, finish_reason: 'stop' }], model: 'm' }) + 'data: [DONE]\n\n'; mockLlamaServer([body]); const { text } = await collect(); assert.strictEqual(text, 'AB', 'content before and after the bad line both survive'); }); test('empty stream yields a done event with zero tokens, not a throw', async () => { mockLlamaServer(['data: [DONE]\n\n']); const { text, final } = await collect(); assert.strictEqual(text, ''); assert.strictEqual(final.tokenCount, 0); });