Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a223b753e5 | ||
|
|
67e2b446e4 |
@@ -1,4 +1,4 @@
|
|||||||
const { SERVICES, getEnv, SUMMARIES } = require('@nexusai/shared');
|
const { SERVICES, getEnv, SUMMARIES, utilityInference } = require('@nexusai/shared');
|
||||||
const {
|
const {
|
||||||
getSessionSummariesForProject,
|
getSessionSummariesForProject,
|
||||||
getProjectOverviewSummary,
|
getProjectOverviewSummary,
|
||||||
@@ -9,9 +9,6 @@ const {
|
|||||||
const { getEpisodesByProject } = require('../episodic');
|
const { getEpisodesByProject } = require('../episodic');
|
||||||
const { getProject } = require('../db/projects');
|
const { getProject } = require('../db/projects');
|
||||||
|
|
||||||
const EXTRACTION_URL = getEnv('EXTRACTION_URL', 'http://localhost:11434');
|
|
||||||
const EXTRACTION_MODEL = getEnv('EXTRACTION_MODEL', 'qwen2.5:3b');
|
|
||||||
|
|
||||||
const MAX_SUMMARY_CHARS = SUMMARIES.MAX_SUMMARY_CHARS; // generous ceiling before we truncate input
|
const MAX_SUMMARY_CHARS = SUMMARIES.MAX_SUMMARY_CHARS; // generous ceiling before we truncate input
|
||||||
|
|
||||||
function buildProjectSummaryPrompt(projectName, sessionSummaries) {
|
function buildProjectSummaryPrompt(projectName, sessionSummaries) {
|
||||||
@@ -24,8 +21,9 @@ function buildProjectSummaryPrompt(projectName, sessionSummaries) {
|
|||||||
summaryBlock = summaryBlock.slice(-MAX_SUMMARY_CHARS);
|
summaryBlock = summaryBlock.slice(-MAX_SUMMARY_CHARS);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// No ChatML wrapper — the model's own prompt template is applied server-side
|
||||||
|
// by the inference service's /utility/complete route (Ollama /api/chat).
|
||||||
return [
|
return [
|
||||||
'<|im_start|>user',
|
|
||||||
`The following are session summaries from a project called "${projectName}".`,
|
`The following are session summaries from a project called "${projectName}".`,
|
||||||
'Write a project overview covering: goals, progress, key decisions, and current state.',
|
'Write a project overview covering: goals, progress, key decisions, and current state.',
|
||||||
'Scale the length to the material — use multiple paragraphs for complex projects, a few sentences for simple ones.',
|
'Scale the length to the material — use multiple paragraphs for complex projects, a few sentences for simple ones.',
|
||||||
@@ -33,8 +31,6 @@ function buildProjectSummaryPrompt(projectName, sessionSummaries) {
|
|||||||
'Write in third person. Output only the overview text, no headings or labels.',
|
'Write in third person. Output only the overview text, no headings or labels.',
|
||||||
'',
|
'',
|
||||||
summaryBlock,
|
summaryBlock,
|
||||||
'<|im_end|>',
|
|
||||||
'<|im_start|>assistant',
|
|
||||||
].join('\n');
|
].join('\n');
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -49,8 +45,9 @@ function buildProjectSummaryFromEpisodesPrompt(projectName, episodes) {
|
|||||||
episodeBlock = episodeBlock.slice(-MAX_SUMMARY_CHARS);
|
episodeBlock = episodeBlock.slice(-MAX_SUMMARY_CHARS);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// No ChatML wrapper — the model's own prompt template is applied server-side
|
||||||
|
// by the inference service's /utility/complete route (Ollama /api/chat).
|
||||||
return [
|
return [
|
||||||
'<|im_start|>user',
|
|
||||||
`The following are conversations from a project called "${projectName}".`,
|
`The following are conversations from a project called "${projectName}".`,
|
||||||
'Write a project overview covering: goals, progress, key decisions, and current state.',
|
'Write a project overview covering: goals, progress, key decisions, and current state.',
|
||||||
'Scale the length to the material — use multiple paragraphs for complex projects, a few sentences for simple ones.',
|
'Scale the length to the material — use multiple paragraphs for complex projects, a few sentences for simple ones.',
|
||||||
@@ -58,58 +55,17 @@ function buildProjectSummaryFromEpisodesPrompt(projectName, episodes) {
|
|||||||
'Write in third person. Output only the overview text, no headings or labels.',
|
'Write in third person. Output only the overview text, no headings or labels.',
|
||||||
'',
|
'',
|
||||||
episodeBlock,
|
episodeBlock,
|
||||||
'<|im_end|>',
|
|
||||||
'<|im_start|>assistant',
|
|
||||||
].join('\n');
|
].join('\n');
|
||||||
}
|
}
|
||||||
|
|
||||||
async function generateProjectSummaryFromEpisodes(projectName, episodes) {
|
async function generateProjectSummaryFromEpisodes(projectName, episodes) {
|
||||||
const prompt = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
|
const user = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
|
||||||
|
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
||||||
const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
|
||||||
method: 'POST',
|
|
||||||
headers: { 'Content-Type': 'application/json' },
|
|
||||||
body: JSON.stringify({
|
|
||||||
model: EXTRACTION_MODEL,
|
|
||||||
prompt,
|
|
||||||
stream: false,
|
|
||||||
options: { temperature: 0.2, num_predict: 1200 },
|
|
||||||
}),
|
|
||||||
});
|
|
||||||
|
|
||||||
if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
|
||||||
const data = await res.json();
|
|
||||||
|
|
||||||
const raw = data.response?.trim() ?? '';
|
|
||||||
return raw
|
|
||||||
.replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
|
||||||
.replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
|
||||||
.trim();
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async function generateProjectSummary(projectName, sessionSummaries) {
|
async function generateProjectSummary(projectName, sessionSummaries) {
|
||||||
const prompt = buildProjectSummaryPrompt(projectName, sessionSummaries);
|
const user = buildProjectSummaryPrompt(projectName, sessionSummaries);
|
||||||
|
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
||||||
const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
|
||||||
method: 'POST',
|
|
||||||
headers: { 'Content-Type': 'application/json' },
|
|
||||||
body: JSON.stringify({
|
|
||||||
model: EXTRACTION_MODEL,
|
|
||||||
prompt,
|
|
||||||
stream: false,
|
|
||||||
// No format: 'json' — we want free-text narrative, same as session summarization
|
|
||||||
options: { temperature: 0.2, num_predict: 1200 },
|
|
||||||
}),
|
|
||||||
});
|
|
||||||
|
|
||||||
if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
|
||||||
const data = await res.json();
|
|
||||||
|
|
||||||
const raw = data.response?.trim() ?? '';
|
|
||||||
return raw
|
|
||||||
.replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
|
||||||
.replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
|
||||||
.trim();
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Main entry point — called by the route handler
|
// Main entry point — called by the route handler
|
||||||
|
|||||||
@@ -1,7 +1,5 @@
|
|||||||
const { getEnv, SERVICES, SUMMARIES, logger } = require('@nexusai/shared');
|
const { getEnv, SERVICES, SUMMARIES, logger, utilityInference } = require('@nexusai/shared');
|
||||||
|
|
||||||
const EXTRACTION_URL = getEnv('EXTRACTION_URL', 'http://localhost:11434');
|
|
||||||
const EXTRACTION_MODEL = getEnv('EXTRACTION_MODEL', 'qwen2.5:3b');
|
|
||||||
const MEMORY_URL = getEnv('MEMORY_SERVICE_URL', SERVICES.MEMORY_URL);
|
const MEMORY_URL = getEnv('MEMORY_SERVICE_URL', SERVICES.MEMORY_URL);
|
||||||
|
|
||||||
const THRESHOLD_TOKENS = parseInt(getEnv('SUMMARY_THRESHOLD_TOKENS', SUMMARIES.THRESHOLD_TOKENS));
|
const THRESHOLD_TOKENS = parseInt(getEnv('SUMMARY_THRESHOLD_TOKENS', SUMMARIES.THRESHOLD_TOKENS));
|
||||||
@@ -35,41 +33,20 @@ Do not include greetings, sign-offs, or filler. Output only the summary text.
|
|||||||
Conversation:
|
Conversation:
|
||||||
${context}`;
|
${context}`;
|
||||||
|
|
||||||
return [
|
// No ChatML wrapper — the model's own prompt template is applied server-side
|
||||||
'<|im_start|>user', // ChatML for qwen2.5
|
// by the inference service's /utility/complete route (Ollama /api/chat).
|
||||||
instruction,
|
return instruction;
|
||||||
'<|im_end|>',
|
|
||||||
'<|im_start|>assistant',
|
|
||||||
].join('\n');
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async function generateSummary(episodes, existingSummary = null) {
|
async function generateSummary(episodes, existingSummary = null) {
|
||||||
const prompt = buildSummaryPrompt(episodes, existingSummary);
|
const user = buildSummaryPrompt(episodes, existingSummary);
|
||||||
|
|
||||||
const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
const content = await utilityInference({
|
||||||
method: 'POST',
|
user,
|
||||||
headers: { 'Content-Type': 'application/json' },
|
|
||||||
body: JSON.stringify({
|
|
||||||
model: EXTRACTION_MODEL,
|
|
||||||
prompt,
|
|
||||||
stream: false,
|
|
||||||
options: {
|
|
||||||
temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
||||||
num_predict: 500, // generous but bounded — keeps summaries from running long
|
maxTokens: 500, // generous but bounded — keeps summaries from running long
|
||||||
},
|
|
||||||
}),
|
|
||||||
});
|
});
|
||||||
|
|
||||||
if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
|
||||||
const data = await res.json();
|
|
||||||
|
|
||||||
|
|
||||||
const raw = data.response?.trim() ?? '';
|
|
||||||
// Strip any leaked ChatML tokens Qwen echoes back
|
|
||||||
const content = raw
|
|
||||||
.replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
|
||||||
.replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
|
||||||
.trim();
|
|
||||||
return content;
|
return content;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user