fixes for summary prompt, chat route logger, QDrant orphan cleanup, llamacpp stream buffering

This commit is contained in:
Storme-bit
2026-08-16 06:05:33 -07:00
parent e4908193bd
commit 0248fcb58c
6 changed files with 76 additions and 22 deletions
@@ -66,29 +66,52 @@ async function* completeStream(prompt, options = {}) {
if (!res.ok)
throw new Error(`llama.cpp error: ${res.status} ${res.statusText}`);
for await (const chunk of res.body) {
const lines = Buffer.from(chunk)
.toString("utf8")
.split("\n")
.filter((l) => l.startsWith("data: ") && l !== "data: [DONE]");
//SSE lines can be split across network chunks, so we need to buffer them until we have a complete line
let buffer = '';
function* processLine(line){
if (!line.startsWith('data: ') || line === 'data: [DONE]') return;
let json;
try {
json = JSON.parse(line.slice(6));
} catch (err) {
logger.error('[llamacpp] Skipping unparseable SSE line:', line.slice(0,120));
return;
}
const delta = json.choices?.[0]?.delta?.content ?? '';
if (json.choices?.[0]?.finish_reason === 'stop' ) {
finalModel = json.model ?? finalModel;
}
// usage arrives in a separate final chunk with empty choices array
if (json.usage){
finalTokenCount = (json.usage.completion_tokens ?? 0) + (json.usage.prompt_tokens ?? 0);
}
for (const line of lines) {
const json = JSON.parse(line.slice(6));
const delta = json.choices?.[0]?.delta?.content;
if (json.choices?.[0]?.finish_reason === 'stop') {
finalModel = json.model ?? finalModel;
}
// usage arrives in a separate final chunk with empty choices array
if (json.usage) {
finalTokenCount = (json.usage.completion_tokens ?? 0) + (json.usage.prompt_tokens ?? 0);
}
if (delta) yield { response: delta, done: false };
yield* processLine(line.trim());
}
}
for await (const chunk of res.body) {
buffer += Buffer.from(chunk).toString('utf-8');
const lines = buffer.split('\n');
buffer = lines.pop() ?? ""; // keep the last line in the buffer, it may be incomplete
}
//Flush anything left in buffer after stream closes
if (buffer.trim()){
yield* processLine(buffer.trim());
}
logger.info('[llamacpp] finalTokenCount:', finalTokenCount);
yield { response: '', done: true, model: finalModel, tokenCount: finalTokenCount };