utility inference cleanup
This commit is contained in:
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
**Location:** `packages/memory-service/src/entities/extraction.js`
|
**Location:** `packages/memory-service/src/entities/extraction.js`
|
||||||
**Triggered by:** Episode creation (`POST /episodes`)
|
**Triggered by:** Episode creation (`POST /episodes`)
|
||||||
**Model:** `qwen2.5:3b` via Ollama (configurable via `EXTRACTION_MODEL` env var)
|
**Model:** the utility model served by the inference service (`/utility/complete`), configurable via `UTILITY_MODEL` on the inference service
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
|
|
||||||
@@ -28,7 +28,7 @@ swallowed.
|
|||||||
|
|
||||||
| Setting | Value | Notes |
|
| Setting | Value | Notes |
|
||||||
|---|---|---|
|
|---|---|---|
|
||||||
| Model | `qwen2.5:3b` | Ollama, configurable via `EXTRACTION_MODEL` |
|
| Model | utility model | Served by inference-service `/utility/complete`, set via `UTILITY_MODEL` |
|
||||||
| Temperature | 0.1 | Low for consistent, deterministic output |
|
| Temperature | 0.1 | Low for consistent, deterministic output |
|
||||||
| `num_predict` | 1500 | Higher ceiling to accommodate entity + relationship JSON |
|
| `num_predict` | 1500 | Higher ceiling to accommodate entity + relationship JSON |
|
||||||
| `format` | `'json'` | Ollama constrained decoding — enforces valid JSON output |
|
| `format` | `'json'` | Ollama constrained decoding — enforces valid JSON output |
|
||||||
|
|||||||
@@ -28,8 +28,7 @@ relationship extraction and embeds results into Qdrant.
|
|||||||
| SQLITE_PATH | Yes | — | Path to SQLite database file |
|
| SQLITE_PATH | Yes | — | Path to SQLite database file |
|
||||||
| QDRANT_URL | No | http://localhost:6333 | Qdrant instance URL |
|
| QDRANT_URL | No | http://localhost:6333 | Qdrant instance URL |
|
||||||
| EMBEDDING_SERVICE_URL | No | http://localhost:3003 | Embedding service URL |
|
| EMBEDDING_SERVICE_URL | No | http://localhost:3003 | Embedding service URL |
|
||||||
| EXTRACTION_URL | No | http://localhost:11434 | Ollama URL for entity extraction |
|
| INFERENCE_SERVICE_URL | No | http://localhost:3001 | Inference service URL — entity extraction routes through its `/utility/complete` endpoint |
|
||||||
| EXTRACTION_MODEL | No | qwen2.5:3b | Ollama model used for entity extraction |
|
|
||||||
|
|
||||||
## Internal Structure
|
## Internal Structure
|
||||||
|
|
||||||
|
|||||||
@@ -30,8 +30,6 @@ or inference services — all traffic flows through orchestration.
|
|||||||
| LLAMA_SERVER_URL | No | http://localhost:8080 | Direct llama-server URL for /models/props |
|
| LLAMA_SERVER_URL | No | http://localhost:8080 | Direct llama-server URL for /models/props |
|
||||||
| QDRANT_URL | No | http://localhost:6333 | Qdrant URL for semantic search |
|
| QDRANT_URL | No | http://localhost:6333 | Qdrant URL for semantic search |
|
||||||
| CORS_ORIGIN | No | http://localhost:5173 | Allowed origin for CORS requests |
|
| CORS_ORIGIN | No | http://localhost:5173 | Allowed origin for CORS requests |
|
||||||
| EXTRACTION_URL | No | http://localhost:11434 | Ollama URL for summarisation |
|
|
||||||
| EXTRACTION_MODEL | No | qwen2.5:3b | Ollama model used for summarisation |
|
|
||||||
|
|
||||||
## Internal Structure
|
## Internal Structure
|
||||||
|
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ the full context window with raw episodes.
|
|||||||
|
|
||||||
**Location:** `packages/orchestration-service/src/services/summarization.js`
|
**Location:** `packages/orchestration-service/src/services/summarization.js`
|
||||||
**Triggered by:** `chat/index.js` after every episode write (fire-and-forget)
|
**Triggered by:** `chat/index.js` after every episode write (fire-and-forget)
|
||||||
**Model:** `qwen2.5:3b` via Ollama on Mini PC 1 (192.168.0.81)
|
**Model:** the utility model served by the inference service (`/utility/complete`), set via `UTILITY_MODEL` (backed by Ollama on Mini PC 1, 192.168.0.81)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -56,47 +56,50 @@ not all episodes in the session.
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Ollama Request
|
## Utility Inference Request
|
||||||
|
|
||||||
|
Summaries are generated through the shared `utilityInference()` helper, which
|
||||||
|
POSTs to the inference service's `/utility/complete` endpoint. `buildSummaryPrompt`
|
||||||
|
returns a plain instruction string (no template tags) passed as the `user` message:
|
||||||
|
|
||||||
```js
|
```js
|
||||||
{
|
const content = await utilityInference({
|
||||||
model: EXTRACTION_MODEL, // qwen2.5:3b (set via EXTRACTION_MODEL env var)
|
user: buildSummaryPrompt(episodesToSummarize, existingSummary),
|
||||||
prompt: buildSummaryPrompt(episodesToSummarize, existingSummary),
|
temperature: SUMMARIES.TEMPERATURE, // 0.2
|
||||||
stream: false,
|
maxTokens: SUMMARIES.SESSION_GEN_MAX_TOKENS, // 500
|
||||||
// No format: 'json' — free-text output required for summaries
|
});
|
||||||
options: {
|
|
||||||
temperature: 0.2,
|
|
||||||
num_predict: 500,
|
|
||||||
},
|
|
||||||
}
|
|
||||||
```
|
```
|
||||||
|
|
||||||
`temperature: 0.2` is slightly higher than extraction (0.1) — summaries
|
`TEMPERATURE` (0.2) is slightly higher than extraction (0.1) — summaries benefit
|
||||||
benefit from some fluency. `num_predict: 500` gives room for 5 thorough
|
from some fluency. `SESSION_GEN_MAX_TOKENS` (500) gives room for ~5 thorough
|
||||||
sentences without risk of runoff.
|
sentences without runoff. Both live in `@nexusai/shared` `SUMMARIES` constants.
|
||||||
|
|
||||||
|
There is no `json: true` here — summaries are free-text, unlike entity extraction.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Prompt Format
|
## Prompt Format
|
||||||
|
|
||||||
ChatML format — native to qwen2.5:
|
The prompt is plain text describing the task; the model's own prompt template
|
||||||
|
(ChatML for qwen, etc.) is applied **server-side** by the inference service via
|
||||||
|
Ollama's `/api/chat`. No `<|im_start|>` tags belong in this codebase, and the
|
||||||
|
utility model can be swapped (via `UTILITY_MODEL` on the inference service) with
|
||||||
|
no prompt changes here.
|
||||||
|
|
||||||
|
Fresh summary instruction:
|
||||||
|
|
||||||
```
|
```
|
||||||
<|im_start|>user
|
|
||||||
Summarize the conversation below in 3-5 sentences.
|
Summarize the conversation below in 3-5 sentences.
|
||||||
Write in third person. Do not quote directly — paraphrase only.
|
Write in third person. Do not quote directly — paraphrase only.
|
||||||
Do not include greetings, sign-offs, or filler. Output only the summary text.
|
Do not include greetings, sign-offs, or filler. Output only the summary text.
|
||||||
|
|
||||||
Conversation:
|
Conversation:
|
||||||
{context}
|
{context}
|
||||||
<|im_end|>
|
|
||||||
<|im_start|>assistant
|
|
||||||
```
|
```
|
||||||
|
|
||||||
For cumulative updates, the instruction and context change:
|
Cumulative update instruction:
|
||||||
|
|
||||||
```
|
```
|
||||||
<|im_start|>user
|
|
||||||
Update the summary below to incorporate the new exchanges.
|
Update the summary below to incorporate the new exchanges.
|
||||||
Write 3-5 sentences in third person. Do not quote directly — paraphrase only.
|
Write 3-5 sentences in third person. Do not quote directly — paraphrase only.
|
||||||
Do not include greetings, sign-offs, or filler. Output only the updated summary text.
|
Do not include greetings, sign-offs, or filler. Output only the updated summary text.
|
||||||
@@ -106,35 +109,22 @@ Previous summary:
|
|||||||
|
|
||||||
New exchanges:
|
New exchanges:
|
||||||
{context}
|
{context}
|
||||||
<|im_end|>
|
|
||||||
<|im_start|>assistant
|
|
||||||
```
|
```
|
||||||
|
|
||||||
### Input truncation
|
### Input truncation
|
||||||
|
|
||||||
Episode context is truncated to `MAX_CHARS = 3000` characters, keeping the
|
Episode context is truncated to `MAX_CHARS = 3000` characters, keeping the
|
||||||
most recent exchanges (sliced from the end). This keeps Qwen focused and
|
most recent exchanges (sliced from the end). This keeps the model focused and
|
||||||
prevents the prompt from exceeding its effective context window.
|
prevents the prompt from exceeding its effective context window.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## ChatML Token Stripping
|
## Output Handling
|
||||||
|
|
||||||
Qwen occasionally echoes ChatML tokens back into its response. The raw output
|
Because `/api/chat` applies and removes the prompt template server-side, the
|
||||||
is cleaned before saving:
|
returned text is already clean — the previous ChatML token-stripping step (and
|
||||||
|
the class of bug where leaked tokens got stored and re-injected into the next
|
||||||
```js
|
summarisation prompt) no longer applies.
|
||||||
const raw = data.response?.trim() ?? '';
|
|
||||||
const content = raw
|
|
||||||
.replace(/<\|im_start\|>.*?<\|im_end\|>/gs, '')
|
|
||||||
.replace(/<\|im_start\|>|<\|im_end\|>|<\|im_sep\|>/g, '')
|
|
||||||
.trim();
|
|
||||||
return content;
|
|
||||||
```
|
|
||||||
|
|
||||||
Without this, leaked tokens get stored in the summary and then injected
|
|
||||||
back into the next summarisation prompt — causing the model to append a new
|
|
||||||
summary after the old one rather than replacing it.
|
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -207,8 +197,7 @@ Set in `packages/orchestration-service/src/.env`:
|
|||||||
|
|
||||||
| Variable | Default | Description |
|
| Variable | Default | Description |
|
||||||
|---|---|---|
|
|---|---|---|
|
||||||
| `EXTRACTION_URL` | `http://localhost:11434` | Ollama instance URL |
|
| `INFERENCE_SERVICE_URL` | `http://localhost:3001` | Inference service — summaries route through its `/utility/complete` endpoint (model set via `UTILITY_MODEL` there) |
|
||||||
| `EXTRACTION_MODEL` | `qwen2.5:3b` | Model for summarisation |
|
|
||||||
| `MEMORY_SERVICE_URL` | `http://localhost:3002` | Memory service URL |
|
| `MEMORY_SERVICE_URL` | `http://localhost:3002` | Memory service URL |
|
||||||
| `SUMMARY_THRESHOLD_TOKENS` | `200` | Token threshold before summarisation triggers |
|
| `SUMMARY_THRESHOLD_TOKENS` | `200` | Token threshold before summarisation triggers |
|
||||||
| `SUMMARY_MAX_TOKENS` | `800` | Max summary length before a new row is created |
|
| `SUMMARY_MAX_TOKENS` | `800` | Max summary length before a new row is created |
|
||||||
|
|||||||
@@ -60,12 +60,20 @@ function buildProjectSummaryFromEpisodesPrompt(projectName, episodes) {
|
|||||||
|
|
||||||
async function generateProjectSummaryFromEpisodes(projectName, episodes) {
|
async function generateProjectSummaryFromEpisodes(projectName, episodes) {
|
||||||
const user = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
|
const user = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
|
||||||
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
return utilityInference({
|
||||||
|
user,
|
||||||
|
temperature: SUMMARIES.TEMPERATURE,
|
||||||
|
maxTokens: SUMMARIES.PROJECT_GEN_MAX_TOKENS,
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
async function generateProjectSummary(projectName, sessionSummaries) {
|
async function generateProjectSummary(projectName, sessionSummaries) {
|
||||||
const user = buildProjectSummaryPrompt(projectName, sessionSummaries);
|
const user = buildProjectSummaryPrompt(projectName, sessionSummaries);
|
||||||
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
return utilityInference({
|
||||||
|
user,
|
||||||
|
temperature: SUMMARIES.TEMPERATURE,
|
||||||
|
maxTokens: SUMMARIES.PROJECT_GEN_MAX_TOKENS,
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
// Main entry point — called by the route handler
|
// Main entry point — called by the route handler
|
||||||
|
|||||||
@@ -43,8 +43,8 @@ async function generateSummary(episodes, existingSummary = null) {
|
|||||||
|
|
||||||
const content = await utilityInference({
|
const content = await utilityInference({
|
||||||
user,
|
user,
|
||||||
temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
temperature: SUMMARIES.TEMPERATURE,
|
||||||
maxTokens: 500, // generous but bounded — keeps summaries from running long
|
maxTokens: SUMMARIES.SESSION_GEN_MAX_TOKENS,
|
||||||
});
|
});
|
||||||
|
|
||||||
return content;
|
return content;
|
||||||
|
|||||||
@@ -78,6 +78,13 @@ const SUMMARIES = {
|
|||||||
MIN_EPISODES_SINCE: 5, // don't resummarize until N new episodes since last summary
|
MIN_EPISODES_SINCE: 5, // don't resummarize until N new episodes since last summary
|
||||||
MAX_SUMMARY_CHARS: 8000, // max chars to include from recent episodes when generating summary (to control prompt size)
|
MAX_SUMMARY_CHARS: 8000, // max chars to include from recent episodes when generating summary (to control prompt size)
|
||||||
MAX_PROJECT_EPISODE_LIMIT: 200, // max number of episodes to consider from the entire project when generating summary (to control prompt size)
|
MAX_PROJECT_EPISODE_LIMIT: 200, // max number of episodes to consider from the entire project when generating summary (to control prompt size)
|
||||||
|
|
||||||
|
// Generation params for the utility model (passed to utilityInference).
|
||||||
|
// Distinct from MAX_SUMMARY_TOKENS above, which gates STORED summary size;
|
||||||
|
// these two cap GENERATION length (num_predict) per summary type.
|
||||||
|
TEMPERATURE: 0.2, // slightly higher than entities (0.1) — summaries benefit from some fluency
|
||||||
|
SESSION_GEN_MAX_TOKENS: 500, // num_predict for a session summary (3-5 sentences)
|
||||||
|
PROJECT_GEN_MAX_TOKENS: 1200, // num_predict for a project overview (multi-paragraph)
|
||||||
}
|
}
|
||||||
|
|
||||||
const ENTITIES = {
|
const ENTITIES = {
|
||||||
|
|||||||
Reference in New Issue
Block a user