From 551c7f03ec2f6c41b099b71cf9fc86ee0ad3e1c1 Mon Sep 17 00:00:00 2001 From: Storme-bit Date: Mon, 17 Aug 2026 23:48:15 -0700 Subject: [PATCH] utility inference cleanup --- docs/services/entity-extraction.md | 4 +- docs/services/memory-service.md | 3 +- docs/services/orchestration-service.md | 2 - docs/services/summarization.md | 73 ++++++++----------- .../src/summarization/project.js | 12 ++- .../src/services/summarization.js | 4 +- packages/shared/src/config/constants.js | 7 ++ 7 files changed, 53 insertions(+), 52 deletions(-) diff --git a/docs/services/entity-extraction.md b/docs/services/entity-extraction.md index e16085d..20cf322 100644 --- a/docs/services/entity-extraction.md +++ b/docs/services/entity-extraction.md @@ -2,7 +2,7 @@ **Location:** `packages/memory-service/src/entities/extraction.js` **Triggered by:** Episode creation (`POST /episodes`) -**Model:** `qwen2.5:3b` via Ollama (configurable via `EXTRACTION_MODEL` env var) +**Model:** the utility model served by the inference service (`/utility/complete`), configurable via `UTILITY_MODEL` on the inference service ## Purpose @@ -28,7 +28,7 @@ swallowed. | Setting | Value | Notes | |---|---|---| -| Model | `qwen2.5:3b` | Ollama, configurable via `EXTRACTION_MODEL` | +| Model | utility model | Served by inference-service `/utility/complete`, set via `UTILITY_MODEL` | | Temperature | 0.1 | Low for consistent, deterministic output | | `num_predict` | 1500 | Higher ceiling to accommodate entity + relationship JSON | | `format` | `'json'` | Ollama constrained decoding — enforces valid JSON output | diff --git a/docs/services/memory-service.md b/docs/services/memory-service.md index b2331be..ee24b34 100644 --- a/docs/services/memory-service.md +++ b/docs/services/memory-service.md @@ -28,8 +28,7 @@ relationship extraction and embeds results into Qdrant. | SQLITE_PATH | Yes | — | Path to SQLite database file | | QDRANT_URL | No | http://localhost:6333 | Qdrant instance URL | | EMBEDDING_SERVICE_URL | No | http://localhost:3003 | Embedding service URL | -| EXTRACTION_URL | No | http://localhost:11434 | Ollama URL for entity extraction | -| EXTRACTION_MODEL | No | qwen2.5:3b | Ollama model used for entity extraction | +| INFERENCE_SERVICE_URL | No | http://localhost:3001 | Inference service URL — entity extraction routes through its `/utility/complete` endpoint | ## Internal Structure diff --git a/docs/services/orchestration-service.md b/docs/services/orchestration-service.md index 6239e80..62b7265 100644 --- a/docs/services/orchestration-service.md +++ b/docs/services/orchestration-service.md @@ -30,8 +30,6 @@ or inference services — all traffic flows through orchestration. | LLAMA_SERVER_URL | No | http://localhost:8080 | Direct llama-server URL for /models/props | | QDRANT_URL | No | http://localhost:6333 | Qdrant URL for semantic search | | CORS_ORIGIN | No | http://localhost:5173 | Allowed origin for CORS requests | -| EXTRACTION_URL | No | http://localhost:11434 | Ollama URL for summarisation | -| EXTRACTION_MODEL | No | qwen2.5:3b | Ollama model used for summarisation | ## Internal Structure diff --git a/docs/services/summarization.md b/docs/services/summarization.md index bc51ace..1a12a48 100644 --- a/docs/services/summarization.md +++ b/docs/services/summarization.md @@ -6,7 +6,7 @@ the full context window with raw episodes. **Location:** `packages/orchestration-service/src/services/summarization.js` **Triggered by:** `chat/index.js` after every episode write (fire-and-forget) -**Model:** `qwen2.5:3b` via Ollama on Mini PC 1 (192.168.0.81) +**Model:** the utility model served by the inference service (`/utility/complete`), set via `UTILITY_MODEL` (backed by Ollama on Mini PC 1, 192.168.0.81) --- @@ -56,47 +56,50 @@ not all episodes in the session. --- -## Ollama Request +## Utility Inference Request + +Summaries are generated through the shared `utilityInference()` helper, which +POSTs to the inference service's `/utility/complete` endpoint. `buildSummaryPrompt` +returns a plain instruction string (no template tags) passed as the `user` message: ```js -{ - model: EXTRACTION_MODEL, // qwen2.5:3b (set via EXTRACTION_MODEL env var) - prompt: buildSummaryPrompt(episodesToSummarize, existingSummary), - stream: false, - // No format: 'json' — free-text output required for summaries - options: { - temperature: 0.2, - num_predict: 500, - }, -} +const content = await utilityInference({ + user: buildSummaryPrompt(episodesToSummarize, existingSummary), + temperature: SUMMARIES.TEMPERATURE, // 0.2 + maxTokens: SUMMARIES.SESSION_GEN_MAX_TOKENS, // 500 +}); ``` -`temperature: 0.2` is slightly higher than extraction (0.1) — summaries -benefit from some fluency. `num_predict: 500` gives room for 5 thorough -sentences without risk of runoff. +`TEMPERATURE` (0.2) is slightly higher than extraction (0.1) — summaries benefit +from some fluency. `SESSION_GEN_MAX_TOKENS` (500) gives room for ~5 thorough +sentences without runoff. Both live in `@nexusai/shared` `SUMMARIES` constants. + +There is no `json: true` here — summaries are free-text, unlike entity extraction. --- ## Prompt Format -ChatML format — native to qwen2.5: +The prompt is plain text describing the task; the model's own prompt template +(ChatML for qwen, etc.) is applied **server-side** by the inference service via +Ollama's `/api/chat`. No `<|im_start|>` tags belong in this codebase, and the +utility model can be swapped (via `UTILITY_MODEL` on the inference service) with +no prompt changes here. + +Fresh summary instruction: ``` -<|im_start|>user Summarize the conversation below in 3-5 sentences. Write in third person. Do not quote directly — paraphrase only. Do not include greetings, sign-offs, or filler. Output only the summary text. Conversation: {context} -<|im_end|> -<|im_start|>assistant ``` -For cumulative updates, the instruction and context change: +Cumulative update instruction: ``` -<|im_start|>user Update the summary below to incorporate the new exchanges. Write 3-5 sentences in third person. Do not quote directly — paraphrase only. Do not include greetings, sign-offs, or filler. Output only the updated summary text. @@ -106,35 +109,22 @@ Previous summary: New exchanges: {context} -<|im_end|> -<|im_start|>assistant ``` ### Input truncation Episode context is truncated to `MAX_CHARS = 3000` characters, keeping the -most recent exchanges (sliced from the end). This keeps Qwen focused and +most recent exchanges (sliced from the end). This keeps the model focused and prevents the prompt from exceeding its effective context window. --- -## ChatML Token Stripping +## Output Handling -Qwen occasionally echoes ChatML tokens back into its response. The raw output -is cleaned before saving: - -```js -const raw = data.response?.trim() ?? ''; -const content = raw - .replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '') - .replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '') - .trim(); -return content; -``` - -Without this, leaked tokens get stored in the summary and then injected -back into the next summarisation prompt — causing the model to append a new -summary after the old one rather than replacing it. +Because `/api/chat` applies and removes the prompt template server-side, the +returned text is already clean — the previous ChatML token-stripping step (and +the class of bug where leaked tokens got stored and re-injected into the next +summarisation prompt) no longer applies. --- @@ -207,8 +197,7 @@ Set in `packages/orchestration-service/src/.env`: | Variable | Default | Description | |---|---|---| -| `EXTRACTION_URL` | `http://localhost:11434` | Ollama instance URL | -| `EXTRACTION_MODEL` | `qwen2.5:3b` | Model for summarisation | +| `INFERENCE_SERVICE_URL` | `http://localhost:3001` | Inference service — summaries route through its `/utility/complete` endpoint (model set via `UTILITY_MODEL` there) | | `MEMORY_SERVICE_URL` | `http://localhost:3002` | Memory service URL | | `SUMMARY_THRESHOLD_TOKENS` | `200` | Token threshold before summarisation triggers | | `SUMMARY_MAX_TOKENS` | `800` | Max summary length before a new row is created | diff --git a/packages/memory-service/src/summarization/project.js b/packages/memory-service/src/summarization/project.js index be5648b..6a4c0e2 100644 --- a/packages/memory-service/src/summarization/project.js +++ b/packages/memory-service/src/summarization/project.js @@ -60,12 +60,20 @@ function buildProjectSummaryFromEpisodesPrompt(projectName, episodes) { async function generateProjectSummaryFromEpisodes(projectName, episodes) { const user = buildProjectSummaryFromEpisodesPrompt(projectName, episodes); - return utilityInference({ user, temperature: 0.2, maxTokens: 1200 }); + return utilityInference({ + user, + temperature: SUMMARIES.TEMPERATURE, + maxTokens: SUMMARIES.PROJECT_GEN_MAX_TOKENS, + }); } async function generateProjectSummary(projectName, sessionSummaries) { const user = buildProjectSummaryPrompt(projectName, sessionSummaries); - return utilityInference({ user, temperature: 0.2, maxTokens: 1200 }); + return utilityInference({ + user, + temperature: SUMMARIES.TEMPERATURE, + maxTokens: SUMMARIES.PROJECT_GEN_MAX_TOKENS, + }); } // Main entry point — called by the route handler diff --git a/packages/orchestration-service/src/services/summarization.js b/packages/orchestration-service/src/services/summarization.js index ef67a44..4b1ee9a 100644 --- a/packages/orchestration-service/src/services/summarization.js +++ b/packages/orchestration-service/src/services/summarization.js @@ -43,8 +43,8 @@ async function generateSummary(episodes, existingSummary = null) { const content = await utilityInference({ user, - temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency - maxTokens: 500, // generous but bounded — keeps summaries from running long + temperature: SUMMARIES.TEMPERATURE, + maxTokens: SUMMARIES.SESSION_GEN_MAX_TOKENS, }); return content; diff --git a/packages/shared/src/config/constants.js b/packages/shared/src/config/constants.js index f63f7b0..4436dcb 100644 --- a/packages/shared/src/config/constants.js +++ b/packages/shared/src/config/constants.js @@ -78,6 +78,13 @@ const SUMMARIES = { MIN_EPISODES_SINCE: 5, // don't resummarize until N new episodes since last summary MAX_SUMMARY_CHARS: 8000, // max chars to include from recent episodes when generating summary (to control prompt size) MAX_PROJECT_EPISODE_LIMIT: 200, // max number of episodes to consider from the entire project when generating summary (to control prompt size) + + // Generation params for the utility model (passed to utilityInference). + // Distinct from MAX_SUMMARY_TOKENS above, which gates STORED summary size; + // these two cap GENERATION length (num_predict) per summary type. + TEMPERATURE: 0.2, // slightly higher than entities (0.1) — summaries benefit from some fluency + SESSION_GEN_MAX_TOKENS: 500, // num_predict for a session summary (3-5 sentences) + PROJECT_GEN_MAX_TOKENS: 1200, // num_predict for a project overview (multi-paragraph) } const ENTITIES = {