utility inference layer 3
This commit is contained in:
@@ -1,7 +1,5 @@
|
||||
const { getEnv, SERVICES, SUMMARIES, logger } = require('@nexusai/shared');
|
||||
const { getEnv, SERVICES, SUMMARIES, logger, utilityInference } = require('@nexusai/shared');
|
||||
|
||||
const EXTRACTION_URL = getEnv('EXTRACTION_URL', 'http://localhost:11434');
|
||||
const EXTRACTION_MODEL = getEnv('EXTRACTION_MODEL', 'qwen2.5:3b');
|
||||
const MEMORY_URL = getEnv('MEMORY_SERVICE_URL', SERVICES.MEMORY_URL);
|
||||
|
||||
const THRESHOLD_TOKENS = parseInt(getEnv('SUMMARY_THRESHOLD_TOKENS', SUMMARIES.THRESHOLD_TOKENS));
|
||||
@@ -35,41 +33,20 @@ Do not include greetings, sign-offs, or filler. Output only the summary text.
|
||||
Conversation:
|
||||
${context}`;
|
||||
|
||||
return [
|
||||
'<|im_start|>user', // ChatML for qwen2.5
|
||||
instruction,
|
||||
'<|im_end|>',
|
||||
'<|im_start|>assistant',
|
||||
].join('\n');
|
||||
// No ChatML wrapper — the model's own prompt template is applied server-side
|
||||
// by the inference service's /utility/complete route (Ollama /api/chat).
|
||||
return instruction;
|
||||
}
|
||||
|
||||
async function generateSummary(episodes, existingSummary = null) {
|
||||
const prompt = buildSummaryPrompt(episodes, existingSummary);
|
||||
const user = buildSummaryPrompt(episodes, existingSummary);
|
||||
|
||||
const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model: EXTRACTION_MODEL,
|
||||
prompt,
|
||||
stream: false,
|
||||
options: {
|
||||
temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
||||
num_predict: 500, // generous but bounded — keeps summaries from running long
|
||||
},
|
||||
}),
|
||||
const content = await utilityInference({
|
||||
user,
|
||||
temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
||||
maxTokens: 500, // generous but bounded — keeps summaries from running long
|
||||
});
|
||||
|
||||
if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
||||
const data = await res.json();
|
||||
|
||||
|
||||
const raw = data.response?.trim() ?? '';
|
||||
// Strip any leaked ChatML tokens Qwen echoes back
|
||||
const content = raw
|
||||
.replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
||||
.replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
||||
.trim();
|
||||
return content;
|
||||
}
|
||||
|
||||
@@ -148,4 +125,4 @@ async function triggerSummary(session) {
|
||||
);
|
||||
}
|
||||
|
||||
module.exports = { triggerSummary, maybeSummarize };
|
||||
module.exports = { triggerSummary, maybeSummarize };
|
||||
|
||||
Reference in New Issue
Block a user