From 8f8177615c7e749d7625756acd2bdefff81eea26 Mon Sep 17 00:00:00 2001 From: Storme-bit Date: Sun, 16 Aug 2026 23:25:40 -0700 Subject: [PATCH] documentation update --- docs/services/entity-extraction.md | 26 ++++++++++++-- docs/services/inference-service.md | 32 +++++++++++++++++ docs/services/memory-service.md | 36 ++++++++++++++++++- .../memory-service/src/entities/extraction.js | 11 +++++- 4 files changed, 100 insertions(+), 5 deletions(-) diff --git a/docs/services/entity-extraction.md b/docs/services/entity-extraction.md index 6e43a0e..a9e5b7b 100644 --- a/docs/services/entity-extraction.md +++ b/docs/services/entity-extraction.md @@ -99,7 +99,7 @@ returns without writing anything. For each entity in `parsed.entities`: -1. Validate `name`, `type` (must be in `ENTITY_TYPES`), and not in `IGNORED_NAMES` +1. Validate `name`, `type` (must be in `ENTITY_TYPES`), and not a greeting via `isIgnoredName(name)` 2. Call `upsertEntity(name, type, notes)`: - **Insert**: creates new row with `mention_count = 1`, `source = 'extraction'` - **Conflict** on `(name, type)`: increments `mention_count`, updates `last_seen_at`, preserves existing `notes` if new extraction returns null @@ -109,7 +109,27 @@ For each entity in `parsed.entities`: **Valid entity types:** `person`, `place`, `project`, `technology`, `concept`, `organization` -**Stoplist (ignored names):** `good morning`, `good night`, `hello`, `goodbye`, `thanks`, `thank you` +**Greeting filter:** `isIgnoredName(name)` normalizes the candidate before +matching — lowercased, punctuation stripped (`/[^\w\s]/g`), trimmed — then +tests membership against the `IGNORED_NAMES` set. Normalizing means variants +like `"Good morning!"` or `" hello "` are caught, not just exact lowercase +matches. Current set: `good morning`, `good night`, `good evening`, +`good afternoon`, `hello`, `hi`, `hey`, `goodbye`, `bye`, `thanks`, +`thank you`, `morning`. + +The extraction prompt also instructs the model not to emit greetings or +conversational filler as entities, so the filter is a backstop rather than the +sole defense — necessary because the model (qwen2.5:3b) will occasionally label +a greeting as a `concept` or `topic`, both valid types that the `ENTITY_TYPES` +check alone won't reject. + +> **Note:** the filter runs only at extraction time; it does not retroactively +> remove greetings stored before the filter (or an entry) existed. Pre-filter +> stragglers must be deleted manually via `DELETE /entities/:id`, which now +> also removes the Qdrant vector (see `memory-service.md` → Delete Behaviour). +> `\w` is ASCII-only, so non-ASCII greetings (e.g. `buenos días`) normalize +> imperfectly — switch to `/[^\p{L}\p{N}\s]/gu` if the set gains non-ASCII +> entries. ## Relationship Processing @@ -137,4 +157,4 @@ All steps after the initial model call are wrapped in a single outer try/catch. If Ollama is unreachable, returns a non-200 status, or the JSON cannot be parsed, the function logs at `warn` level and returns. There is no retry logic. Individual entity embedding failures are caught per-entity and logged at `warn` -level without affecting other entities in the same batch. +level without affecting other entities in the same batch. \ No newline at end of file diff --git a/docs/services/inference-service.md b/docs/services/inference-service.md index f558725..f90a39f 100644 --- a/docs/services/inference-service.md +++ b/docs/services/inference-service.md @@ -147,4 +147,36 @@ data: [DONE] `model` and `tokenCount` are captured from the llama.cpp `finish_reason: stop` chunk and emitted on the done event. +### SSE parsing (buffered) + +`llama-server` emits SSE events as `data:` lines, but a single network chunk +from `res.body` is **not** guaranteed to contain whole lines — an event can be +split across two chunks (mid-`data:` prefix or even mid-JSON). The provider +therefore accumulates a string `buffer`, splits on `\n`, and retains the final +(possibly incomplete) element for the next chunk rather than parsing each raw +chunk in isolation: + +```js +let buffer = ''; +for await (const chunk of res.body) { + buffer += Buffer.from(chunk).toString('utf-8'); + const lines = buffer.split('\n'); + buffer = lines.pop() ?? ''; // keep incomplete tail for next chunk + for (const line of lines) yield* processLine(line.trim()); +} +if (buffer.trim()) yield* processLine(buffer.trim()); // flush remainder +``` + +`processLine()` handles exactly one line: it skips non-`data:` lines and +`[DONE]`, guards `JSON.parse` in a try/catch (a malformed line logs a warning +and is skipped rather than throwing and killing the stream), extracts +`delta.content`, and captures `model` / token usage from the terminal chunks. +Content is emitted only when a delta is present (`if (delta) yield { response, +done: false }`). + +> This mirrors the same buffering the orchestration service uses when reading +> the inference stream. Without it, long responses intermittently die +> mid-stream on a chunk boundary, and a run where every content delta lands on +> a split line surfaces as an empty response with a `done` event. + For all HTTP endpoints, see `api-routes.md`. \ No newline at end of file diff --git a/docs/services/memory-service.md b/docs/services/memory-service.md index a5c7974..c103054 100644 --- a/docs/services/memory-service.md +++ b/docs/services/memory-service.md @@ -184,6 +184,40 @@ service is responsible only for CRUD — generation logic lives in orchestration > For full details on trigger conditions, prompt format, cumulative updates, > and ChatML token stripping, see `summarization.md`. +## Delete Behaviour (SQLite + Qdrant consistency) + +SQLite cascades handle relational cleanup, but Qdrant is a separate store and +must be cleaned explicitly. Each delete path that removes embedded rows also +removes the corresponding vectors: + +| Delete | SQLite effect | Qdrant cleanup | +|---|---|---| +| `DELETE /episodes/:id` | Row removed | `semantic.deleteEpisode(id)` — vector by point ID | +| `DELETE /sessions/by-external/:id` | Session + episodes cascade-deleted | `semantic.deleteEpisodesBySession(id)` — **payload-filter** delete on `sessionId` | +| `DELETE /entities/:id` | Row removed, relationships cascade | `semantic.deleteEntity(id)` — vector by point ID | + +All three Qdrant deletes are **fire-and-forget** with error logging, matching +the fire-and-forget write path — a Qdrant failure logs but does not fail the +delete. + +The session path uses a **payload-filter** delete (matching on the `sessionId` +field in the vector payload) rather than enumerating episode point IDs. This +matters because the SQLite cascade has already removed the episode rows by the +time cleanup runs, so there are no IDs left to enumerate — the filter deletes +by payload regardless. It also cleans up any pre-existing orphans for that +session as a side effect. + +> **Not cleaned on session delete:** entity vectors. Entities are shared across +> sessions and projects (`UNIQUE(name, type)` is global), so deleting one +> session must not remove entities that other sessions still reference. Entity +> vector lifecycle is tied to explicit entity deletion and the (planned) memory +> consolidation / orphan-cleanup pass. + +> **Historical orphans:** vectors orphaned by session deletes *before* this +> cleanup existed are not removed retroactively. A one-time sweep (scroll the +> `episodes` collection, delete points whose `sessionId` no longer exists in +> SQLite) clears them. + ## Project Delete Behaviour Deleting a project runs as a transaction — it first nulls out `project_id` @@ -197,4 +231,4 @@ const doDelete = db.transaction(() => { }); ``` -For all HTTP endpoints, see `api-routes.md`. +For all HTTP endpoints, see `api-routes.md`. \ No newline at end of file diff --git a/packages/memory-service/src/entities/extraction.js b/packages/memory-service/src/entities/extraction.js index cd6a1f7..29088bf 100644 --- a/packages/memory-service/src/entities/extraction.js +++ b/packages/memory-service/src/entities/extraction.js @@ -7,7 +7,15 @@ const EXTRACTION_MODEL = getEnv('EXTRACTION_MODEL', 'qwen2.5:3b'); // ChatML for const EMBEDDING_SERVICE_URL = getEnv('EMBEDDING_SERVICE_URL', SERVICES.EMBEDDING_URL); const ENTITY_TYPES = ENTITIES.TYPES; -const IGNORED_NAMES = ['good morning', 'good night', 'hello', 'goodbye', 'thanks', 'thank you']; +const IGNORED_NAMES = new Set([ + 'good morning', 'good night', 'good evening', 'good afternoon', + 'hello', 'hi', 'hey', 'goodbye', 'bye', 'thanks', 'thank you', 'morning', +]); + +function isIgnoredName(name) { + const normalized = name.toLowerCase().replace(/[^\w\s]/g, '').trim(); + return IGNORED_NAMES.has(normalized); +} // NOTE: This prompt uses ChatML format (<|im_start|> / <|im_end|> tags), which is // specific to qwen-family models. If EXTRACTION_MODEL is changed to a Llama-family @@ -31,6 +39,7 @@ function buildExtractionPrompt(userMessage, aiResponse, knownEntities = []) { `Entity types: ${ENTITY_TYPES.join(', ')}`, 'Use "character" for any fictional, game, or media characters (e.g. characters from anime, games, books, TV shows, movies)', 'Use "person" only for real people', + 'Do not extract greetings, pleasantries, or conversational filler (e.g. "good morning", "thanks") as entities.', 'For each entity provide:', ' "name": short proper noun only (max 4 words)', ' "type": one of the valid types',