64 lines
2.5 KiB
JavaScript
64 lines
2.5 KiB
JavaScript
// llama.cpp streaming — tests the REAL completeStream against SSE that is split
|
|
// across network chunks at awkward offsets (mid-"data:" prefix, mid-JSON). This is
|
|
// the exact failure mode that produced empty responses; the parser must reassemble.
|
|
const { test } = require('node:test');
|
|
const assert = require('node:assert');
|
|
const { completeStream } = require('../packages/inference-service/src/providers/llamacpp');
|
|
|
|
// Build a full SSE body, then hand it back in arbitrarily-split chunks via a fake fetch.
|
|
function mockLlamaServer(chunks) {
|
|
global.fetch = async () => ({
|
|
ok: true,
|
|
body: (async function* () {
|
|
for (const c of chunks) yield Buffer.from(c);
|
|
})(),
|
|
});
|
|
}
|
|
|
|
async function collect() {
|
|
let text = '', final = null;
|
|
for await (const ev of completeStream('prompt', {})) {
|
|
if (ev.done) final = ev; else text += ev.response;
|
|
}
|
|
return { text, final };
|
|
}
|
|
|
|
const sse = (obj) => `data: ${JSON.stringify(obj)}\n\n`;
|
|
|
|
test('reassembles content when SSE lines are split across chunks', async () => {
|
|
const body =
|
|
sse({ choices: [{ delta: { content: 'Good' } }] }) +
|
|
sse({ choices: [{ delta: { content: ' morning' }, finish_reason: null }] }) +
|
|
sse({ choices: [{ delta: {}, finish_reason: 'stop' }], model: 'gemma-test' }) +
|
|
sse({ choices: [], usage: { completion_tokens: 5, prompt_tokens: 11 } }) +
|
|
'data: [DONE]\n\n';
|
|
|
|
// Split at nasty offsets: mid-prefix and mid-JSON
|
|
const chunks = [body.slice(0, 9), body.slice(9, 40), body.slice(40, 41), body.slice(41, 110), body.slice(110)];
|
|
mockLlamaServer(chunks);
|
|
|
|
const { text, final } = await collect();
|
|
assert.strictEqual(text, 'Good morning');
|
|
assert.strictEqual(final.model, 'gemma-test');
|
|
assert.strictEqual(final.tokenCount, 16, 'completion + prompt tokens');
|
|
assert.strictEqual(final.done, true);
|
|
});
|
|
|
|
test('a single malformed line is skipped without killing the stream', async () => {
|
|
const body =
|
|
sse({ choices: [{ delta: { content: 'A' } }] }) +
|
|
'data: {not valid json\n\n' +
|
|
sse({ choices: [{ delta: { content: 'B' }, finish_reason: 'stop' }], model: 'm' }) +
|
|
'data: [DONE]\n\n';
|
|
mockLlamaServer([body]);
|
|
|
|
const { text } = await collect();
|
|
assert.strictEqual(text, 'AB', 'content before and after the bad line both survive');
|
|
});
|
|
|
|
test('empty stream yields a done event with zero tokens, not a throw', async () => {
|
|
mockLlamaServer(['data: [DONE]\n\n']);
|
|
const { text, final } = await collect();
|
|
assert.strictEqual(text, '');
|
|
assert.strictEqual(final.tokenCount, 0);
|
|
}); |