From b87b2fc930917c3c1df3e95fce29b9769818e8fe Mon Sep 17 00:00:00 2001 From: Storme-bit Date: Sun, 16 Aug 2026 06:28:16 -0700 Subject: [PATCH] empty string fix --- .../src/providers/llamacpp.js | 33 ++++++++++--------- 1 file changed, 17 insertions(+), 16 deletions(-) diff --git a/packages/inference-service/src/providers/llamacpp.js b/packages/inference-service/src/providers/llamacpp.js index 6477343..4ddd64e 100644 --- a/packages/inference-service/src/providers/llamacpp.js +++ b/packages/inference-service/src/providers/llamacpp.js @@ -66,49 +66,50 @@ async function* completeStream(prompt, options = {}) { if (!res.ok) throw new Error(`llama.cpp error: ${res.status} ${res.statusText}`); - //SSE lines can be split across network chunks, so we need to buffer them until we have a complete line + // SSE lines can be split across network chunks, so we buffer the incomplete + // tail and only process whole lines. let buffer = ''; - function* processLine(line){ + // Handles ONE line. Yields at most one content chunk. Never touches `lines` + // or the network buffer — the reader loop below owns that. + function* processLine(line) { if (!line.startsWith('data: ') || line === 'data: [DONE]') return; let json; - try { json = JSON.parse(line.slice(6)); } catch (err) { - logger.error('[llamacpp] Skipping unparseable SSE line:', line.slice(0,120)); + logger.warn('[llamacpp] Skipping unparseable SSE line:', line.slice(0, 120)); return; } + const delta = json.choices?.[0]?.delta?.content; - - const delta = json.choices?.[0]?.delta?.content ?? ''; - - if (json.choices?.[0]?.finish_reason === 'stop' ) { + if (json.choices?.[0]?.finish_reason === 'stop') { finalModel = json.model ?? finalModel; - } // usage arrives in a separate final chunk with empty choices array - if (json.usage){ + if (json.usage) { finalTokenCount = (json.usage.completion_tokens ?? 0) + (json.usage.prompt_tokens ?? 0); } - for (const line of lines) { - yield* processLine(line.trim()); - } + if (delta) yield { response: delta, done: false }; } for await (const chunk of res.body) { buffer += Buffer.from(chunk).toString('utf-8'); const lines = buffer.split('\n'); - buffer = lines.pop() ?? ""; // keep the last line in the buffer, it may be incomplete + buffer = lines.pop() ?? ''; // last element may be incomplete — keep for next chunk + + for (const line of lines) { + yield* processLine(line.trim()); + } } - //Flush anything left in buffer after stream closes - if (buffer.trim()){ + // Flush anything left in the buffer after the stream closes + if (buffer.trim()) { yield* processLine(buffer.trim()); }