utility inference layer 3
This commit is contained in:
@@ -0,0 +1,189 @@
|
||||
diff -ruN nexusai-baseline/packages/memory-service/src/summarization/project.js nexusai/packages/memory-service/src/summarization/project.js
|
||||
--- nexusai-baseline/packages/memory-service/src/summarization/project.js 2026-08-17 17:05:22.398445363 +0000
|
||||
+++ nexusai/packages/memory-service/src/summarization/project.js 2026-08-17 17:05:50.408024878 +0000
|
||||
@@ -1,4 +1,4 @@
|
||||
-const { SERVICES, getEnv, SUMMARIES } = require('@nexusai/shared');
|
||||
+const { SERVICES, getEnv, SUMMARIES, utilityInference } = require('@nexusai/shared');
|
||||
const {
|
||||
getSessionSummariesForProject,
|
||||
getProjectOverviewSummary,
|
||||
@@ -9,9 +9,6 @@
|
||||
const { getEpisodesByProject } = require('../episodic');
|
||||
const { getProject } = require('../db/projects');
|
||||
|
||||
-const EXTRACTION_URL = getEnv('EXTRACTION_URL', 'http://localhost:11434');
|
||||
-const EXTRACTION_MODEL = getEnv('EXTRACTION_MODEL', 'qwen2.5:3b');
|
||||
-
|
||||
const MAX_SUMMARY_CHARS = SUMMARIES.MAX_SUMMARY_CHARS; // generous ceiling before we truncate input
|
||||
|
||||
function buildProjectSummaryPrompt(projectName, sessionSummaries) {
|
||||
@@ -24,8 +21,9 @@
|
||||
summaryBlock = summaryBlock.slice(-MAX_SUMMARY_CHARS);
|
||||
}
|
||||
|
||||
+ // No ChatML wrapper — the model's own prompt template is applied server-side
|
||||
+ // by the inference service's /utility/complete route (Ollama /api/chat).
|
||||
return [
|
||||
- '<|im_start|>user',
|
||||
`The following are session summaries from a project called "${projectName}".`,
|
||||
'Write a project overview covering: goals, progress, key decisions, and current state.',
|
||||
'Scale the length to the material — use multiple paragraphs for complex projects, a few sentences for simple ones.',
|
||||
@@ -33,8 +31,6 @@
|
||||
'Write in third person. Output only the overview text, no headings or labels.',
|
||||
'',
|
||||
summaryBlock,
|
||||
- '<|im_end|>',
|
||||
- '<|im_start|>assistant',
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
@@ -49,8 +45,9 @@
|
||||
episodeBlock = episodeBlock.slice(-MAX_SUMMARY_CHARS);
|
||||
}
|
||||
|
||||
+ // No ChatML wrapper — the model's own prompt template is applied server-side
|
||||
+ // by the inference service's /utility/complete route (Ollama /api/chat).
|
||||
return [
|
||||
- '<|im_start|>user',
|
||||
`The following are conversations from a project called "${projectName}".`,
|
||||
'Write a project overview covering: goals, progress, key decisions, and current state.',
|
||||
'Scale the length to the material — use multiple paragraphs for complex projects, a few sentences for simple ones.',
|
||||
@@ -58,58 +55,17 @@
|
||||
'Write in third person. Output only the overview text, no headings or labels.',
|
||||
'',
|
||||
episodeBlock,
|
||||
- '<|im_end|>',
|
||||
- '<|im_start|>assistant',
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
async function generateProjectSummaryFromEpisodes(projectName, episodes) {
|
||||
- const prompt = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
|
||||
-
|
||||
- const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
||||
- method: 'POST',
|
||||
- headers: { 'Content-Type': 'application/json' },
|
||||
- body: JSON.stringify({
|
||||
- model: EXTRACTION_MODEL,
|
||||
- prompt,
|
||||
- stream: false,
|
||||
- options: { temperature: 0.2, num_predict: 1200 },
|
||||
- }),
|
||||
- });
|
||||
-
|
||||
- if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
||||
- const data = await res.json();
|
||||
-
|
||||
- const raw = data.response?.trim() ?? '';
|
||||
- return raw
|
||||
- .replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
||||
- .replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
||||
- .trim();
|
||||
+ const user = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
|
||||
+ return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
||||
}
|
||||
|
||||
async function generateProjectSummary(projectName, sessionSummaries) {
|
||||
- const prompt = buildProjectSummaryPrompt(projectName, sessionSummaries);
|
||||
-
|
||||
- const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
||||
- method: 'POST',
|
||||
- headers: { 'Content-Type': 'application/json' },
|
||||
- body: JSON.stringify({
|
||||
- model: EXTRACTION_MODEL,
|
||||
- prompt,
|
||||
- stream: false,
|
||||
- // No format: 'json' — we want free-text narrative, same as session summarization
|
||||
- options: { temperature: 0.2, num_predict: 1200 },
|
||||
- }),
|
||||
- });
|
||||
-
|
||||
- if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
||||
- const data = await res.json();
|
||||
-
|
||||
- const raw = data.response?.trim() ?? '';
|
||||
- return raw
|
||||
- .replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
||||
- .replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
||||
- .trim();
|
||||
+ const user = buildProjectSummaryPrompt(projectName, sessionSummaries);
|
||||
+ return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
||||
}
|
||||
|
||||
// Main entry point — called by the route handler
|
||||
@@ -142,4 +98,4 @@
|
||||
}
|
||||
}
|
||||
|
||||
-module.exports = { generateAndStoreProjectSummary };
|
||||
\ No newline at end of file
|
||||
+module.exports = { generateAndStoreProjectSummary };
|
||||
diff -ruN nexusai-baseline/packages/orchestration-service/src/services/summarization.js nexusai/packages/orchestration-service/src/services/summarization.js
|
||||
--- nexusai-baseline/packages/orchestration-service/src/services/summarization.js 2026-08-17 17:05:22.397285190 +0000
|
||||
+++ nexusai/packages/orchestration-service/src/services/summarization.js 2026-08-17 17:05:38.180024151 +0000
|
||||
@@ -1,7 +1,5 @@
|
||||
-const { getEnv, SERVICES, SUMMARIES, logger } = require('@nexusai/shared');
|
||||
+const { getEnv, SERVICES, SUMMARIES, logger, utilityInference } = require('@nexusai/shared');
|
||||
|
||||
-const EXTRACTION_URL = getEnv('EXTRACTION_URL', 'http://localhost:11434');
|
||||
-const EXTRACTION_MODEL = getEnv('EXTRACTION_MODEL', 'qwen2.5:3b');
|
||||
const MEMORY_URL = getEnv('MEMORY_SERVICE_URL', SERVICES.MEMORY_URL);
|
||||
|
||||
const THRESHOLD_TOKENS = parseInt(getEnv('SUMMARY_THRESHOLD_TOKENS', SUMMARIES.THRESHOLD_TOKENS));
|
||||
@@ -35,41 +33,20 @@
|
||||
Conversation:
|
||||
${context}`;
|
||||
|
||||
- return [
|
||||
- '<|im_start|>user', // ChatML for qwen2.5
|
||||
- instruction,
|
||||
- '<|im_end|>',
|
||||
- '<|im_start|>assistant',
|
||||
- ].join('\n');
|
||||
+ // No ChatML wrapper — the model's own prompt template is applied server-side
|
||||
+ // by the inference service's /utility/complete route (Ollama /api/chat).
|
||||
+ return instruction;
|
||||
}
|
||||
|
||||
async function generateSummary(episodes, existingSummary = null) {
|
||||
- const prompt = buildSummaryPrompt(episodes, existingSummary);
|
||||
+ const user = buildSummaryPrompt(episodes, existingSummary);
|
||||
|
||||
- const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
||||
- method: 'POST',
|
||||
- headers: { 'Content-Type': 'application/json' },
|
||||
- body: JSON.stringify({
|
||||
- model: EXTRACTION_MODEL,
|
||||
- prompt,
|
||||
- stream: false,
|
||||
- options: {
|
||||
- temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
||||
- num_predict: 500, // generous but bounded — keeps summaries from running long
|
||||
- },
|
||||
- }),
|
||||
+ const content = await utilityInference({
|
||||
+ user,
|
||||
+ temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
||||
+ maxTokens: 500, // generous but bounded — keeps summaries from running long
|
||||
});
|
||||
|
||||
- if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
||||
- const data = await res.json();
|
||||
-
|
||||
-
|
||||
- const raw = data.response?.trim() ?? '';
|
||||
- // Strip any leaked ChatML tokens Qwen echoes back
|
||||
- const content = raw
|
||||
- .replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
||||
- .replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
||||
- .trim();
|
||||
return content;
|
||||
}
|
||||
|
||||
@@ -148,4 +125,4 @@
|
||||
);
|
||||
}
|
||||
|
||||
-module.exports = { triggerSummary, maybeSummarize };
|
||||
\ No newline at end of file
|
||||
+module.exports = { triggerSummary, maybeSummarize };
|
||||
@@ -1,4 +1,4 @@
|
||||
const { SERVICES, getEnv, SUMMARIES } = require('@nexusai/shared');
|
||||
const { SERVICES, getEnv, SUMMARIES, utilityInference } = require('@nexusai/shared');
|
||||
const {
|
||||
getSessionSummariesForProject,
|
||||
getProjectOverviewSummary,
|
||||
@@ -9,9 +9,6 @@ const {
|
||||
const { getEpisodesByProject } = require('../episodic');
|
||||
const { getProject } = require('../db/projects');
|
||||
|
||||
const EXTRACTION_URL = getEnv('EXTRACTION_URL', 'http://localhost:11434');
|
||||
const EXTRACTION_MODEL = getEnv('EXTRACTION_MODEL', 'qwen2.5:3b');
|
||||
|
||||
const MAX_SUMMARY_CHARS = SUMMARIES.MAX_SUMMARY_CHARS; // generous ceiling before we truncate input
|
||||
|
||||
function buildProjectSummaryPrompt(projectName, sessionSummaries) {
|
||||
@@ -24,8 +21,9 @@ function buildProjectSummaryPrompt(projectName, sessionSummaries) {
|
||||
summaryBlock = summaryBlock.slice(-MAX_SUMMARY_CHARS);
|
||||
}
|
||||
|
||||
// No ChatML wrapper — the model's own prompt template is applied server-side
|
||||
// by the inference service's /utility/complete route (Ollama /api/chat).
|
||||
return [
|
||||
'<|im_start|>user',
|
||||
`The following are session summaries from a project called "${projectName}".`,
|
||||
'Write a project overview covering: goals, progress, key decisions, and current state.',
|
||||
'Scale the length to the material — use multiple paragraphs for complex projects, a few sentences for simple ones.',
|
||||
@@ -33,8 +31,6 @@ function buildProjectSummaryPrompt(projectName, sessionSummaries) {
|
||||
'Write in third person. Output only the overview text, no headings or labels.',
|
||||
'',
|
||||
summaryBlock,
|
||||
'<|im_end|>',
|
||||
'<|im_start|>assistant',
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
@@ -49,8 +45,9 @@ function buildProjectSummaryFromEpisodesPrompt(projectName, episodes) {
|
||||
episodeBlock = episodeBlock.slice(-MAX_SUMMARY_CHARS);
|
||||
}
|
||||
|
||||
// No ChatML wrapper — the model's own prompt template is applied server-side
|
||||
// by the inference service's /utility/complete route (Ollama /api/chat).
|
||||
return [
|
||||
'<|im_start|>user',
|
||||
`The following are conversations from a project called "${projectName}".`,
|
||||
'Write a project overview covering: goals, progress, key decisions, and current state.',
|
||||
'Scale the length to the material — use multiple paragraphs for complex projects, a few sentences for simple ones.',
|
||||
@@ -58,58 +55,17 @@ function buildProjectSummaryFromEpisodesPrompt(projectName, episodes) {
|
||||
'Write in third person. Output only the overview text, no headings or labels.',
|
||||
'',
|
||||
episodeBlock,
|
||||
'<|im_end|>',
|
||||
'<|im_start|>assistant',
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
async function generateProjectSummaryFromEpisodes(projectName, episodes) {
|
||||
const prompt = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
|
||||
|
||||
const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model: EXTRACTION_MODEL,
|
||||
prompt,
|
||||
stream: false,
|
||||
options: { temperature: 0.2, num_predict: 1200 },
|
||||
}),
|
||||
});
|
||||
|
||||
if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
||||
const data = await res.json();
|
||||
|
||||
const raw = data.response?.trim() ?? '';
|
||||
return raw
|
||||
.replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
||||
.replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
||||
.trim();
|
||||
const user = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
|
||||
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
||||
}
|
||||
|
||||
async function generateProjectSummary(projectName, sessionSummaries) {
|
||||
const prompt = buildProjectSummaryPrompt(projectName, sessionSummaries);
|
||||
|
||||
const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model: EXTRACTION_MODEL,
|
||||
prompt,
|
||||
stream: false,
|
||||
// No format: 'json' — we want free-text narrative, same as session summarization
|
||||
options: { temperature: 0.2, num_predict: 1200 },
|
||||
}),
|
||||
});
|
||||
|
||||
if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
||||
const data = await res.json();
|
||||
|
||||
const raw = data.response?.trim() ?? '';
|
||||
return raw
|
||||
.replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
||||
.replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
||||
.trim();
|
||||
const user = buildProjectSummaryPrompt(projectName, sessionSummaries);
|
||||
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
||||
}
|
||||
|
||||
// Main entry point — called by the route handler
|
||||
@@ -142,4 +98,4 @@ async function generateAndStoreProjectSummary(projectId) {
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = { generateAndStoreProjectSummary };
|
||||
module.exports = { generateAndStoreProjectSummary };
|
||||
|
||||
@@ -1,7 +1,5 @@
|
||||
const { getEnv, SERVICES, SUMMARIES, logger } = require('@nexusai/shared');
|
||||
const { getEnv, SERVICES, SUMMARIES, logger, utilityInference } = require('@nexusai/shared');
|
||||
|
||||
const EXTRACTION_URL = getEnv('EXTRACTION_URL', 'http://localhost:11434');
|
||||
const EXTRACTION_MODEL = getEnv('EXTRACTION_MODEL', 'qwen2.5:3b');
|
||||
const MEMORY_URL = getEnv('MEMORY_SERVICE_URL', SERVICES.MEMORY_URL);
|
||||
|
||||
const THRESHOLD_TOKENS = parseInt(getEnv('SUMMARY_THRESHOLD_TOKENS', SUMMARIES.THRESHOLD_TOKENS));
|
||||
@@ -35,41 +33,20 @@ Do not include greetings, sign-offs, or filler. Output only the summary text.
|
||||
Conversation:
|
||||
${context}`;
|
||||
|
||||
return [
|
||||
'<|im_start|>user', // ChatML for qwen2.5
|
||||
instruction,
|
||||
'<|im_end|>',
|
||||
'<|im_start|>assistant',
|
||||
].join('\n');
|
||||
// No ChatML wrapper — the model's own prompt template is applied server-side
|
||||
// by the inference service's /utility/complete route (Ollama /api/chat).
|
||||
return instruction;
|
||||
}
|
||||
|
||||
async function generateSummary(episodes, existingSummary = null) {
|
||||
const prompt = buildSummaryPrompt(episodes, existingSummary);
|
||||
const user = buildSummaryPrompt(episodes, existingSummary);
|
||||
|
||||
const res = await fetch(`${EXTRACTION_URL}/api/generate`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model: EXTRACTION_MODEL,
|
||||
prompt,
|
||||
stream: false,
|
||||
options: {
|
||||
temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
||||
num_predict: 500, // generous but bounded — keeps summaries from running long
|
||||
},
|
||||
}),
|
||||
const content = await utilityInference({
|
||||
user,
|
||||
temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
||||
maxTokens: 500, // generous but bounded — keeps summaries from running long
|
||||
});
|
||||
|
||||
if (!res.ok) throw new Error(`Ollama responded ${res.status}`);
|
||||
const data = await res.json();
|
||||
|
||||
|
||||
const raw = data.response?.trim() ?? '';
|
||||
// Strip any leaked ChatML tokens Qwen echoes back
|
||||
const content = raw
|
||||
.replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
||||
.replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
||||
.trim();
|
||||
return content;
|
||||
}
|
||||
|
||||
@@ -148,4 +125,4 @@ async function triggerSummary(session) {
|
||||
);
|
||||
}
|
||||
|
||||
module.exports = { triggerSummary, maybeSummarize };
|
||||
module.exports = { triggerSummary, maybeSummarize };
|
||||
|
||||
Reference in New Issue
Block a user