From c31fb786f4a806623b37825ef4988c67fb96a92b Mon Sep 17 00:00:00 2001 From: Storme-bit Date: Sun, 23 Aug 2026 23:26:51 -0700 Subject: [PATCH] documentation update --- docs-update.patch | 270 ++++++++++++++++++++++++++++++++++++++++++++++ scratch.js | 31 ------ 2 files changed, 270 insertions(+), 31 deletions(-) create mode 100644 docs-update.patch delete mode 100644 scratch.js diff --git a/docs-update.patch b/docs-update.patch new file mode 100644 index 0000000..bb0062b --- /dev/null +++ b/docs-update.patch @@ -0,0 +1,270 @@ +diff -ruN a/docs/reference/API-routes.md b/docs/reference/API-routes.md +--- a/docs/reference/API-routes.md 2026-08-24 06:09:15.372210893 +0000 ++++ b/docs/reference/API-routes.md 2026-08-24 06:09:15.375774861 +0000 +@@ -205,6 +205,9 @@ + | `scoreThreshold` | float | 0–1 | Minimum similarity score for Qdrant results | + | `semanticWeight` | float | 0–5 | RRF weight for Qdrant semantic results | + | `keywordWeight` | float | 0–5 | RRF weight for FTS5 keyword results (`0` = disabled) | ++| `contextBudget` | integer | — | Token budget for context assembly (char/4 estimation) | ++| `entityWeight` | float | — | Scoring bonus for entity-linked episodes in the context pool | ++| `minRecentEpisodes` | integer | — | Guaranteed floor of recent episodes always included in context | + | `modelsFolderPath` | string | — | Path to folder containing .gguf files | + | `temperature` | float | 0–2 | Inference randomness | + | `repeatPenalty` | float | 1–2 | Repeat token penalty | +@@ -253,6 +256,7 @@ + | GET | /sessions/by-external/:externalId | Get session by external ID | + | PATCH | /sessions/by-external/:externalId | Update session fields | + | DELETE | /sessions/by-external/:externalId | Delete session (cascades to episodes) | ++| GET | /sessions/:id/entity-ids | Entity IDs linked to this session's episodes (non-project memory isolation scoping) | + + > Route ordering: `by-external/:externalId` must be defined before `/:id` + > to prevent `by-external` being captured as an ID param. +@@ -279,6 +283,8 @@ + | GET | /sessions/:id/episodes?limit=&offset= | Paginated episodes for a session | + | GET | /sessions/:id/episode-stats | Aggregate: count, total tokens, max id (summarization threshold check) | + | GET | /sessions/:id/episodes/since/:afterId | Episodes newer than :afterId, chronological (un-summarized tail) | ++| POST | /episodes/touch | Batch access-tracking bump — increments `access_count`, sets `last_accessed_at` | ++| GET | /sessions/:id/consolidation-candidates | Dry-run consolidation scoring (aging score, floors applied) | + | DELETE | /episodes/:id | Delete episode (SQLite + Qdrant cleanup) | + + > Route ordering: `/episodes/search` must be defined before `/episodes/:id`. +@@ -293,6 +299,21 @@ + } + ``` + ++**POST /episodes/touch — body:** ++```json ++{ "ids": [54, 55, 56] } ++``` ++Returns `{ "touched": 3 }`. Called fire-and-forget by orchestration after ++budget selection. Nonexistent IDs are silently skipped; `touched` echoes the ++request count, not rows matched. ++ ++**GET /sessions/:id/consolidation-candidates** — returns ++`{ eligible, reason?, candidates: [...] }` where each candidate is ++`{ id, access_count, created_at, last_accessed_at, preview, aging_score }`, ++sorted by `aging_score` ascending (most eligible first). Sessions under ++`CONSOLIDATION.MIN_SESSION_EPISODES` return `eligible: false` with a `reason`. ++Observe-only — nothing is modified. ++ + ### Projects + + | Method | Path | Description | +diff -ruN a/docs/reference/testing.md b/docs/reference/testing.md +--- a/docs/reference/testing.md 2026-08-24 06:09:15.373637589 +0000 ++++ b/docs/reference/testing.md 2026-08-24 06:09:15.376597678 +0000 +@@ -18,6 +18,7 @@ + | `test/summarization.test.js` | Summarization decision logic | `maybeSummarize` (orchestration) | + | `test/entity-extraction.test.js` | Greeting + regurgitation guards | `mentionedIn`, `isIgnoredName` (memory-service) | + | `test/schema.test.js` | Fresh-DB schema completeness | `schema.js` string (memory-service) | ++| `test/trivial-turn.test.js` | Greeting/trivial-turn detection | `isTrivialTurn` (shared) | + | `test/migrations.test.js` | Migration version-stepping | `migrate` (memory-service) | + + Tests import the **real** functions rather than reimplementing logic — the +@@ -44,3 +45,10 @@ + logic (ranking, tokenizing, version-stepping, decision branches) over wiring. + Several of these tests were written *after* a bug slipped through — each new + class of mistake is worth a case so it can't recur silently. ++ ++When a mocked service call changes shape (e.g. the `utilityInference` refactor ++moving summarization from Ollama's `/api/generate` to the inference service's ++`/utility/complete`), the mock URL router must move with it — an "unexpected ++fetch" throw in these tests usually means the code under test evolved, not ++broke. Runner-contract tests (migrations) inject stub no-op migration arrays ++rather than letting the real SQL-bearing migrations hit the minimal fake db. +diff -ruN a/docs/roadmap.md b/docs/roadmap.md +--- a/docs/roadmap.md 2026-08-24 06:09:15.371175009 +0000 ++++ b/docs/roadmap.md 2026-08-24 06:09:15.374781393 +0000 +@@ -74,8 +74,9 @@ + + ### 3. Memory Consolidation Lifecycle + Prevents long-term memory degradation and enables compression. +-- [ ] Episode aging — score/weight episodes by recency and access frequency +-- [ ] Consolidation pass — merge related low-weight episodes into summary nodes ++- [x] Episode aging — `access_count` + `last_accessed_at` columns (v2 migration), batch touch on retrieval selection (`POST /episodes/touch`), aging score `access_count / (1 + days since last access)` ++- [x] Dry-run candidates endpoint — `GET /sessions/:id/consolidation-candidates` with age + session-size floors (observe-only phase before destructive pass) ++- [ ] Consolidation pass — merge related low-weight episodes into summary nodes (incl. Qdrant vector + entity link cleanup) + - [ ] Orphan cleanup — remove entities no longer referenced by active episodes + + ### 4. User Preference Model +@@ -90,11 +91,11 @@ + - [ ] Confidence bands — FAST PATH (memory lookup only) vs FULL (LLM + context) + - [ ] Fast-path handlers — direct memory queries, session lookups, factual recalls + +-### 6. Smarter Context Assembly *(inspired by acid2lake)* ++### 6. Smarter Context Assembly *(inspired by acid2lake)* ✅ + Budget-aware context selection instead of dumping all relevant memory into the prompt. +-- [ ] Token budget manager in orchestration +-- [ ] Priority scoring — recency × relevance × entity weight +-- [ ] Configurable context budget via env var ++- [x] Token budget manager in orchestration (`selectWithinBudget`, char/4 estimation on stored text) ++- [x] Priority scoring — RRF fusion + recency + entity boost (`buildScoredPool`) ++- [x] Configurable via settings (`contextBudget`, `entityWeight`, `minRecentEpisodes`) — live, no restart + + ### 7. Procedural Memory Store *(inspired by acid2lake)* + Learns "how NexusAI has successfully handled this type of request before." +@@ -227,4 +228,4 @@ + + --- + +-*Last updated: April 2026* +\ No newline at end of file ++*Last updated: August 2026* +\ No newline at end of file +diff -ruN a/docs/services/memory-service.md b/docs/services/memory-service.md +--- a/docs/services/memory-service.md 2026-08-24 06:09:15.372796767 +0000 ++++ b/docs/services/memory-service.md 2026-08-24 06:09:15.376191053 +0000 +@@ -78,6 +78,11 @@ + ```js + const migrations = [ + (_db) => {}, // v0 → v1: consolidated baseline (historical ALTERs folded into schema.js) ++ (db) => { // v1 → v2: access tracking for consolidation lifecycle ++ db.exec(`ALTER TABLE episodes ADD COLUMN last_accessed_at INTEGER`); ++ db.exec(`ALTER TABLE episodes ADD COLUMN access_count INTEGER NOT NULL DEFAULT 0`); ++ db.exec(`UPDATE episodes SET last_accessed_at = created_at`); // backfill ++ }, + ]; + const LATEST_VERSION = migrations.length; // derived, never hand-maintained + ``` +@@ -121,6 +126,12 @@ + - `foreign_keys = ON` — enforces referential integrity and cascade deletes + - PRAGMAs set via `db.pragma()`, not `db.exec()` + ++> **Copying a live WAL database:** `cp` on the `.db` file alone silently loses ++> everything in the un-checkpointed `-wal` file (recent writes, even the ++> migration version stamp). Always use ++> `sqlite3 nexusai.db "VACUUM INTO './copy.db'"` (or `.backup`) — safe while ++> the service is running, produces a complete single-file snapshot. ++ + ### Dynamic Updates + + Both `updateSession` and `updateProject` build their `SET` clause dynamically +@@ -204,6 +215,28 @@ + > For full details on trigger conditions, prompt format, cumulative updates, + > and ChatML token stripping, see `summarization.md`. + ++## Access Tracking & Consolidation (dry-run) ++ ++Every episode selected into a chat context window (budget-selected, not the ++guaranteed-recency floor) gets an access bump via `POST /episodes/touch` — ++`access_count` incremented, `last_accessed_at` set to `Date.now()` (ms). ++Called fire-and-forget from orchestration; a failure loses one increment, ++nothing more. ++ ++`GET /sessions/:id/consolidation-candidates` scores episodes by ++`access_count / (1 + days since last access)` — never-accessed episodes fall ++back to `created_at` for the recency term and score exactly 0 (most eligible). ++Two floors apply: episodes younger than `CONSOLIDATION.MIN_AGE_DAYS` are ++excluded in SQL; sessions under `CONSOLIDATION.MIN_SESSION_EPISODES` return ++`eligible: false` before scoring runs. The endpoint is observe-only — the ++destructive pass (merge → summarize → Qdrant cleanup → orphan sweep) is not ++yet built. ++ ++> **Unit note:** `created_at` is unix **seconds** (`unixepoch()`); ++> `last_accessed_at` is unix **milliseconds** (`Date.now()`). The scoring ++> query normalizes with `created_at * 1000`. Keep this in mind for any new ++> queries touching both columns. ++ + ## Delete Behaviour (SQLite + Qdrant consistency) + + SQLite cascades handle relational cleanup, but Qdrant is a separate store and +diff -ruN a/docs/services/orchestration-service.md b/docs/services/orchestration-service.md +--- a/docs/services/orchestration-service.md 2026-08-24 06:09:15.373384709 +0000 ++++ b/docs/services/orchestration-service.md 2026-08-24 06:09:15.376427273 +0000 +@@ -73,6 +73,9 @@ + | `scoreThreshold` | 0.5 | Minimum similarity score for Qdrant semantic results | + | `semanticWeight` | 1.0 | RRF weight for Qdrant semantic results | + | `keywordWeight` | 0 | RRF weight for FTS5 keyword results (`0` = disabled) | ++| `contextBudget` | — | Token budget for context assembly (char/4 estimation on stored text) | ++| `entityWeight` | — | Scoring bonus for entity-linked episodes in the context pool | ++| `minRecentEpisodes` | — | Guaranteed floor of recent episodes always included in context | + | `modelsFolderPath` | `/mnt/nexus-models` | Path to folder containing .gguf files | + | `temperature` | 0.7 | Inference temperature | + | `repeatPenalty` | 1.1 | Repeat token penalty | +@@ -101,35 +104,43 @@ + + 4. **Recent episode retrieval** — fetch most recent episodes (`recentEpisodeLimit`). + +-5. **Fused episode retrieval** — runs semantic (Qdrant) and keyword (FTS5) ++5. **Trivial-turn gate** — greetings/pleasantries (`isTrivialTurn`) skip all ++ retrieval (semantic, keyword, entity); recent history alone is the context. ++ Breaks the greeting → marginal-retrieval → confabulation loop. ++ ++6. **Fused episode retrieval** — runs semantic (Qdrant) and keyword (FTS5) + search in parallel, then merges results via Reciprocal Rank Fusion (RRF). +- Both paths are filtered against `recentIds` before fusion. FTS is scoped +- to the current session or all project sessions. If `keywordWeight` is `0`, +- the FTS call is skipped entirely. Non-critical — failures fall back to +- whichever strategy succeeded. +- +-6. **Entity search** — query `entities` Qdrant collection filtered by +- `projectId`. Returns entity IDs alongside Qdrant payload data (the Qdrant +- point ID equals the SQLite entity ID). Non-critical. +- +-7. **Graph neighborhood expansion** — call `POST /graph/neighbors` on +- memory-service with the entity IDs from step 6. Returns a 1-hop subgraph +- `{ nodes, edges }` — entity objects plus the relationships connecting them. +- If no entities were found or the graph call fails, falls back to flat entity +- list (no edges). Non-critical. +- +-8. **Prompt assembly** — combine system prompt, graph context, fused episodes, +- recent episodes, and user message. ++ The query is embedded once and shared with entity search. Both paths are ++ filtered against `recentIds` before fusion. FTS is scoped to the current ++ session or all project sessions. If `keywordWeight` is `0`, the FTS call ++ is skipped entirely. Non-critical — failures fall back to whichever ++ strategy succeeded. ++ ++7. **Entity search + graph expansion** — query `entities` Qdrant collection ++ (project-scoped, or session-scoped via `/sessions/:id/entity-ids` for ++ non-project chats). Entity IDs are expanded into a 1-hop subgraph via ++ `POST /graph/neighbors`; on failure, falls back to flat entity list. ++ Non-critical. ++ ++8. **Scored pool + budget selection** — `buildScoredPool` combines RRF scores, ++ recency, and entity-linkage bonus; `selectWithinBudget` fills `contextBudget` ++ (char/4 token estimation on stored text) above a guaranteed floor of ++ `minRecentEpisodes` recent episodes. Selected episode IDs are then reported ++ to `POST /episodes/touch` fire-and-forget (access tracking for the ++ consolidation lifecycle). ++ ++9. **Prompt assembly** — combine system prompt, graph context, selected ++ episodes, guaranteed recent episodes, and user message. + +-9. **Inference** — send to inference service. `/chat` awaits full response; +- `/chat/stream` pipes SSE chunks to the client. ++10. **Inference** — send to inference service. `/chat` awaits full response; ++ `/chat/stream` pipes SSE chunks to the client. + +-10. **Episode write** — write exchange back to memory with `projectId`. ++11. **Episode write** — write exchange back to memory with `projectId`. + +-11. **Summarisation trigger** — `triggerSummary(session, allEpisodes)` called ++12. **Summarisation trigger** — `triggerSummary(session)` called + fire-and-forget. See `summarization.md` for full details. + +-12. **Auto-naming** — on first message with no session name, fires a secondary ++13. **Auto-naming** — on first message with no session name, fires a secondary + inference call (max 20 tokens, temperature 0.3) to generate a session name. + + ### Prompt Structure +diff -ruN a/docs/services/shared.md b/docs/services/shared.md +--- a/docs/services/shared.md 2026-08-24 06:09:15.371834893 +0000 ++++ b/docs/services/shared.md 2026-08-24 06:09:15.374906804 +0000 +@@ -203,6 +203,16 @@ + SUMMARY_MIN_EPISODES=5 + ``` + ++#### `CONSOLIDATION` ++ ++Controls the memory consolidation lifecycle (currently dry-run only). ++ ++| Key | Value | Description | ++|---|---|---| ++| `MIN_AGE_DAYS` | `7` | Episodes younger than this are never consolidation candidates | ++| `MIN_SESSION_EPISODES` | `20` | Sessions with fewer episodes are skipped entirely | ++| `CANDIDATE_LIMIT` | `50` | Max candidates returned per scoring query | ++ + #### `SQLITE` + + | Key | Value | Description | diff --git a/scratch.js b/scratch.js deleted file mode 100644 index 4def023..0000000 --- a/scratch.js +++ /dev/null @@ -1,31 +0,0 @@ -// scratch.js — run with: SQLITE_PATH=./packages/memory-service/data/test-copy.db node scratch.js -const { getDB } = require('./packages/memory-service/src/db'); -const episodic = require('./packages/memory-service/src/episodic'); - -const db = getDB(); - -// Backdate half the episodes to 10-40 days old, give some fake access history -const eps = db.prepare('SELECT id FROM episodes').all(); -const now = Math.floor(Date.now() / 1000); - -for (const { id } of eps) { - if (Math.random() < 0.5) continue; // leave half young, so the age floor stays testable - const daysOld = 10 + Math.floor(Math.random() * 30); - const createdAt = now - daysOld * 86400; - const touched = Math.random() < 0.6; - db.prepare(` - UPDATE episodes - SET created_at = ?, - access_count = ?, - last_accessed_at = ? - WHERE id = ? - `).run( - createdAt, - touched ? 1 + Math.floor(Math.random() * 5) : 0, - touched ? (createdAt + Math.floor(Math.random() * daysOld) * 86400) * 1000 : null, // accessed sometime after creation, in ms - id - ); -} - -// Now run the actual candidate function against the doctored data -console.table(episodic.getConsolidationCandidates(19)); // whatever session id has rows \ No newline at end of file