Files
nexusAI/test/llamacpp-stream.test.js
T
2026-08-17 01:48:28 -07:00

64 lines
2.5 KiB
JavaScript

// llama.cpp streaming — tests the REAL completeStream against SSE that is split
// across network chunks at awkward offsets (mid-"data:" prefix, mid-JSON). This is
// the exact failure mode that produced empty responses; the parser must reassemble.
const { test } = require('node:test');
const assert = require('node:assert');
const { completeStream } = require('../packages/inference-service/src/providers/llamacpp');
// Build a full SSE body, then hand it back in arbitrarily-split chunks via a fake fetch.
function mockLlamaServer(chunks) {
global.fetch = async () => ({
ok: true,
body: (async function* () {
for (const c of chunks) yield Buffer.from(c);
})(),
});
}
async function collect() {
let text = '', final = null;
for await (const ev of completeStream('prompt', {})) {
if (ev.done) final = ev; else text += ev.response;
}
return { text, final };
}
const sse = (obj) => `data: ${JSON.stringify(obj)}\n\n`;
test('reassembles content when SSE lines are split across chunks', async () => {
const body =
sse({ choices: [{ delta: { content: 'Good' } }] }) +
sse({ choices: [{ delta: { content: ' morning' }, finish_reason: null }] }) +
sse({ choices: [{ delta: {}, finish_reason: 'stop' }], model: 'gemma-test' }) +
sse({ choices: [], usage: { completion_tokens: 5, prompt_tokens: 11 } }) +
'data: [DONE]\n\n';
// Split at nasty offsets: mid-prefix and mid-JSON
const chunks = [body.slice(0, 9), body.slice(9, 40), body.slice(40, 41), body.slice(41, 110), body.slice(110)];
mockLlamaServer(chunks);
const { text, final } = await collect();
assert.strictEqual(text, 'Good morning');
assert.strictEqual(final.model, 'gemma-test');
assert.strictEqual(final.tokenCount, 16, 'completion + prompt tokens');
assert.strictEqual(final.done, true);
});
test('a single malformed line is skipped without killing the stream', async () => {
const body =
sse({ choices: [{ delta: { content: 'A' } }] }) +
'data: {not valid json\n\n' +
sse({ choices: [{ delta: { content: 'B' }, finish_reason: 'stop' }], model: 'm' }) +
'data: [DONE]\n\n';
mockLlamaServer([body]);
const { text } = await collect();
assert.strictEqual(text, 'AB', 'content before and after the bad line both survive');
});
test('empty stream yields a done event with zero tokens, not a throw', async () => {
mockLlamaServer(['data: [DONE]\n\n']);
const { text, final } = await collect();
assert.strictEqual(text, '');
assert.strictEqual(final.tokenCount, 0);
});