documentation update
This commit is contained in:
@@ -0,0 +1,270 @@
|
|||||||
|
diff -ruN a/docs/reference/API-routes.md b/docs/reference/API-routes.md
|
||||||
|
--- a/docs/reference/API-routes.md 2026-08-24 06:09:15.372210893 +0000
|
||||||
|
+++ b/docs/reference/API-routes.md 2026-08-24 06:09:15.375774861 +0000
|
||||||
|
@@ -205,6 +205,9 @@
|
||||||
|
| `scoreThreshold` | float | 0–1 | Minimum similarity score for Qdrant results |
|
||||||
|
| `semanticWeight` | float | 0–5 | RRF weight for Qdrant semantic results |
|
||||||
|
| `keywordWeight` | float | 0–5 | RRF weight for FTS5 keyword results (`0` = disabled) |
|
||||||
|
+| `contextBudget` | integer | — | Token budget for context assembly (char/4 estimation) |
|
||||||
|
+| `entityWeight` | float | — | Scoring bonus for entity-linked episodes in the context pool |
|
||||||
|
+| `minRecentEpisodes` | integer | — | Guaranteed floor of recent episodes always included in context |
|
||||||
|
| `modelsFolderPath` | string | — | Path to folder containing .gguf files |
|
||||||
|
| `temperature` | float | 0–2 | Inference randomness |
|
||||||
|
| `repeatPenalty` | float | 1–2 | Repeat token penalty |
|
||||||
|
@@ -253,6 +256,7 @@
|
||||||
|
| GET | /sessions/by-external/:externalId | Get session by external ID |
|
||||||
|
| PATCH | /sessions/by-external/:externalId | Update session fields |
|
||||||
|
| DELETE | /sessions/by-external/:externalId | Delete session (cascades to episodes) |
|
||||||
|
+| GET | /sessions/:id/entity-ids | Entity IDs linked to this session's episodes (non-project memory isolation scoping) |
|
||||||
|
|
||||||
|
> Route ordering: `by-external/:externalId` must be defined before `/:id`
|
||||||
|
> to prevent `by-external` being captured as an ID param.
|
||||||
|
@@ -279,6 +283,8 @@
|
||||||
|
| GET | /sessions/:id/episodes?limit=&offset= | Paginated episodes for a session |
|
||||||
|
| GET | /sessions/:id/episode-stats | Aggregate: count, total tokens, max id (summarization threshold check) |
|
||||||
|
| GET | /sessions/:id/episodes/since/:afterId | Episodes newer than :afterId, chronological (un-summarized tail) |
|
||||||
|
+| POST | /episodes/touch | Batch access-tracking bump — increments `access_count`, sets `last_accessed_at` |
|
||||||
|
+| GET | /sessions/:id/consolidation-candidates | Dry-run consolidation scoring (aging score, floors applied) |
|
||||||
|
| DELETE | /episodes/:id | Delete episode (SQLite + Qdrant cleanup) |
|
||||||
|
|
||||||
|
> Route ordering: `/episodes/search` must be defined before `/episodes/:id`.
|
||||||
|
@@ -293,6 +299,21 @@
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
+**POST /episodes/touch — body:**
|
||||||
|
+```json
|
||||||
|
+{ "ids": [54, 55, 56] }
|
||||||
|
+```
|
||||||
|
+Returns `{ "touched": 3 }`. Called fire-and-forget by orchestration after
|
||||||
|
+budget selection. Nonexistent IDs are silently skipped; `touched` echoes the
|
||||||
|
+request count, not rows matched.
|
||||||
|
+
|
||||||
|
+**GET /sessions/:id/consolidation-candidates** — returns
|
||||||
|
+`{ eligible, reason?, candidates: [...] }` where each candidate is
|
||||||
|
+`{ id, access_count, created_at, last_accessed_at, preview, aging_score }`,
|
||||||
|
+sorted by `aging_score` ascending (most eligible first). Sessions under
|
||||||
|
+`CONSOLIDATION.MIN_SESSION_EPISODES` return `eligible: false` with a `reason`.
|
||||||
|
+Observe-only — nothing is modified.
|
||||||
|
+
|
||||||
|
### Projects
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
diff -ruN a/docs/reference/testing.md b/docs/reference/testing.md
|
||||||
|
--- a/docs/reference/testing.md 2026-08-24 06:09:15.373637589 +0000
|
||||||
|
+++ b/docs/reference/testing.md 2026-08-24 06:09:15.376597678 +0000
|
||||||
|
@@ -18,6 +18,7 @@
|
||||||
|
| `test/summarization.test.js` | Summarization decision logic | `maybeSummarize` (orchestration) |
|
||||||
|
| `test/entity-extraction.test.js` | Greeting + regurgitation guards | `mentionedIn`, `isIgnoredName` (memory-service) |
|
||||||
|
| `test/schema.test.js` | Fresh-DB schema completeness | `schema.js` string (memory-service) |
|
||||||
|
+| `test/trivial-turn.test.js` | Greeting/trivial-turn detection | `isTrivialTurn` (shared) |
|
||||||
|
| `test/migrations.test.js` | Migration version-stepping | `migrate` (memory-service) |
|
||||||
|
|
||||||
|
Tests import the **real** functions rather than reimplementing logic — the
|
||||||
|
@@ -44,3 +45,10 @@
|
||||||
|
logic (ranking, tokenizing, version-stepping, decision branches) over wiring.
|
||||||
|
Several of these tests were written *after* a bug slipped through — each new
|
||||||
|
class of mistake is worth a case so it can't recur silently.
|
||||||
|
+
|
||||||
|
+When a mocked service call changes shape (e.g. the `utilityInference` refactor
|
||||||
|
+moving summarization from Ollama's `/api/generate` to the inference service's
|
||||||
|
+`/utility/complete`), the mock URL router must move with it — an "unexpected
|
||||||
|
+fetch" throw in these tests usually means the code under test evolved, not
|
||||||
|
+broke. Runner-contract tests (migrations) inject stub no-op migration arrays
|
||||||
|
+rather than letting the real SQL-bearing migrations hit the minimal fake db.
|
||||||
|
diff -ruN a/docs/roadmap.md b/docs/roadmap.md
|
||||||
|
--- a/docs/roadmap.md 2026-08-24 06:09:15.371175009 +0000
|
||||||
|
+++ b/docs/roadmap.md 2026-08-24 06:09:15.374781393 +0000
|
||||||
|
@@ -74,8 +74,9 @@
|
||||||
|
|
||||||
|
### 3. Memory Consolidation Lifecycle
|
||||||
|
Prevents long-term memory degradation and enables compression.
|
||||||
|
-- [ ] Episode aging — score/weight episodes by recency and access frequency
|
||||||
|
-- [ ] Consolidation pass — merge related low-weight episodes into summary nodes
|
||||||
|
+- [x] Episode aging — `access_count` + `last_accessed_at` columns (v2 migration), batch touch on retrieval selection (`POST /episodes/touch`), aging score `access_count / (1 + days since last access)`
|
||||||
|
+- [x] Dry-run candidates endpoint — `GET /sessions/:id/consolidation-candidates` with age + session-size floors (observe-only phase before destructive pass)
|
||||||
|
+- [ ] Consolidation pass — merge related low-weight episodes into summary nodes (incl. Qdrant vector + entity link cleanup)
|
||||||
|
- [ ] Orphan cleanup — remove entities no longer referenced by active episodes
|
||||||
|
|
||||||
|
### 4. User Preference Model
|
||||||
|
@@ -90,11 +91,11 @@
|
||||||
|
- [ ] Confidence bands — FAST PATH (memory lookup only) vs FULL (LLM + context)
|
||||||
|
- [ ] Fast-path handlers — direct memory queries, session lookups, factual recalls
|
||||||
|
|
||||||
|
-### 6. Smarter Context Assembly *(inspired by acid2lake)*
|
||||||
|
+### 6. Smarter Context Assembly *(inspired by acid2lake)* ✅
|
||||||
|
Budget-aware context selection instead of dumping all relevant memory into the prompt.
|
||||||
|
-- [ ] Token budget manager in orchestration
|
||||||
|
-- [ ] Priority scoring — recency × relevance × entity weight
|
||||||
|
-- [ ] Configurable context budget via env var
|
||||||
|
+- [x] Token budget manager in orchestration (`selectWithinBudget`, char/4 estimation on stored text)
|
||||||
|
+- [x] Priority scoring — RRF fusion + recency + entity boost (`buildScoredPool`)
|
||||||
|
+- [x] Configurable via settings (`contextBudget`, `entityWeight`, `minRecentEpisodes`) — live, no restart
|
||||||
|
|
||||||
|
### 7. Procedural Memory Store *(inspired by acid2lake)*
|
||||||
|
Learns "how NexusAI has successfully handled this type of request before."
|
||||||
|
@@ -227,4 +228,4 @@
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
-*Last updated: April 2026*
|
||||||
|
\ No newline at end of file
|
||||||
|
+*Last updated: August 2026*
|
||||||
|
\ No newline at end of file
|
||||||
|
diff -ruN a/docs/services/memory-service.md b/docs/services/memory-service.md
|
||||||
|
--- a/docs/services/memory-service.md 2026-08-24 06:09:15.372796767 +0000
|
||||||
|
+++ b/docs/services/memory-service.md 2026-08-24 06:09:15.376191053 +0000
|
||||||
|
@@ -78,6 +78,11 @@
|
||||||
|
```js
|
||||||
|
const migrations = [
|
||||||
|
(_db) => {}, // v0 → v1: consolidated baseline (historical ALTERs folded into schema.js)
|
||||||
|
+ (db) => { // v1 → v2: access tracking for consolidation lifecycle
|
||||||
|
+ db.exec(`ALTER TABLE episodes ADD COLUMN last_accessed_at INTEGER`);
|
||||||
|
+ db.exec(`ALTER TABLE episodes ADD COLUMN access_count INTEGER NOT NULL DEFAULT 0`);
|
||||||
|
+ db.exec(`UPDATE episodes SET last_accessed_at = created_at`); // backfill
|
||||||
|
+ },
|
||||||
|
];
|
||||||
|
const LATEST_VERSION = migrations.length; // derived, never hand-maintained
|
||||||
|
```
|
||||||
|
@@ -121,6 +126,12 @@
|
||||||
|
- `foreign_keys = ON` — enforces referential integrity and cascade deletes
|
||||||
|
- PRAGMAs set via `db.pragma()`, not `db.exec()`
|
||||||
|
|
||||||
|
+> **Copying a live WAL database:** `cp` on the `.db` file alone silently loses
|
||||||
|
+> everything in the un-checkpointed `-wal` file (recent writes, even the
|
||||||
|
+> migration version stamp). Always use
|
||||||
|
+> `sqlite3 nexusai.db "VACUUM INTO './copy.db'"` (or `.backup`) — safe while
|
||||||
|
+> the service is running, produces a complete single-file snapshot.
|
||||||
|
+
|
||||||
|
### Dynamic Updates
|
||||||
|
|
||||||
|
Both `updateSession` and `updateProject` build their `SET` clause dynamically
|
||||||
|
@@ -204,6 +215,28 @@
|
||||||
|
> For full details on trigger conditions, prompt format, cumulative updates,
|
||||||
|
> and ChatML token stripping, see `summarization.md`.
|
||||||
|
|
||||||
|
+## Access Tracking & Consolidation (dry-run)
|
||||||
|
+
|
||||||
|
+Every episode selected into a chat context window (budget-selected, not the
|
||||||
|
+guaranteed-recency floor) gets an access bump via `POST /episodes/touch` —
|
||||||
|
+`access_count` incremented, `last_accessed_at` set to `Date.now()` (ms).
|
||||||
|
+Called fire-and-forget from orchestration; a failure loses one increment,
|
||||||
|
+nothing more.
|
||||||
|
+
|
||||||
|
+`GET /sessions/:id/consolidation-candidates` scores episodes by
|
||||||
|
+`access_count / (1 + days since last access)` — never-accessed episodes fall
|
||||||
|
+back to `created_at` for the recency term and score exactly 0 (most eligible).
|
||||||
|
+Two floors apply: episodes younger than `CONSOLIDATION.MIN_AGE_DAYS` are
|
||||||
|
+excluded in SQL; sessions under `CONSOLIDATION.MIN_SESSION_EPISODES` return
|
||||||
|
+`eligible: false` before scoring runs. The endpoint is observe-only — the
|
||||||
|
+destructive pass (merge → summarize → Qdrant cleanup → orphan sweep) is not
|
||||||
|
+yet built.
|
||||||
|
+
|
||||||
|
+> **Unit note:** `created_at` is unix **seconds** (`unixepoch()`);
|
||||||
|
+> `last_accessed_at` is unix **milliseconds** (`Date.now()`). The scoring
|
||||||
|
+> query normalizes with `created_at * 1000`. Keep this in mind for any new
|
||||||
|
+> queries touching both columns.
|
||||||
|
+
|
||||||
|
## Delete Behaviour (SQLite + Qdrant consistency)
|
||||||
|
|
||||||
|
SQLite cascades handle relational cleanup, but Qdrant is a separate store and
|
||||||
|
diff -ruN a/docs/services/orchestration-service.md b/docs/services/orchestration-service.md
|
||||||
|
--- a/docs/services/orchestration-service.md 2026-08-24 06:09:15.373384709 +0000
|
||||||
|
+++ b/docs/services/orchestration-service.md 2026-08-24 06:09:15.376427273 +0000
|
||||||
|
@@ -73,6 +73,9 @@
|
||||||
|
| `scoreThreshold` | 0.5 | Minimum similarity score for Qdrant semantic results |
|
||||||
|
| `semanticWeight` | 1.0 | RRF weight for Qdrant semantic results |
|
||||||
|
| `keywordWeight` | 0 | RRF weight for FTS5 keyword results (`0` = disabled) |
|
||||||
|
+| `contextBudget` | — | Token budget for context assembly (char/4 estimation on stored text) |
|
||||||
|
+| `entityWeight` | — | Scoring bonus for entity-linked episodes in the context pool |
|
||||||
|
+| `minRecentEpisodes` | — | Guaranteed floor of recent episodes always included in context |
|
||||||
|
| `modelsFolderPath` | `/mnt/nexus-models` | Path to folder containing .gguf files |
|
||||||
|
| `temperature` | 0.7 | Inference temperature |
|
||||||
|
| `repeatPenalty` | 1.1 | Repeat token penalty |
|
||||||
|
@@ -101,35 +104,43 @@
|
||||||
|
|
||||||
|
4. **Recent episode retrieval** — fetch most recent episodes (`recentEpisodeLimit`).
|
||||||
|
|
||||||
|
-5. **Fused episode retrieval** — runs semantic (Qdrant) and keyword (FTS5)
|
||||||
|
+5. **Trivial-turn gate** — greetings/pleasantries (`isTrivialTurn`) skip all
|
||||||
|
+ retrieval (semantic, keyword, entity); recent history alone is the context.
|
||||||
|
+ Breaks the greeting → marginal-retrieval → confabulation loop.
|
||||||
|
+
|
||||||
|
+6. **Fused episode retrieval** — runs semantic (Qdrant) and keyword (FTS5)
|
||||||
|
search in parallel, then merges results via Reciprocal Rank Fusion (RRF).
|
||||||
|
- Both paths are filtered against `recentIds` before fusion. FTS is scoped
|
||||||
|
- to the current session or all project sessions. If `keywordWeight` is `0`,
|
||||||
|
- the FTS call is skipped entirely. Non-critical — failures fall back to
|
||||||
|
- whichever strategy succeeded.
|
||||||
|
-
|
||||||
|
-6. **Entity search** — query `entities` Qdrant collection filtered by
|
||||||
|
- `projectId`. Returns entity IDs alongside Qdrant payload data (the Qdrant
|
||||||
|
- point ID equals the SQLite entity ID). Non-critical.
|
||||||
|
-
|
||||||
|
-7. **Graph neighborhood expansion** — call `POST /graph/neighbors` on
|
||||||
|
- memory-service with the entity IDs from step 6. Returns a 1-hop subgraph
|
||||||
|
- `{ nodes, edges }` — entity objects plus the relationships connecting them.
|
||||||
|
- If no entities were found or the graph call fails, falls back to flat entity
|
||||||
|
- list (no edges). Non-critical.
|
||||||
|
-
|
||||||
|
-8. **Prompt assembly** — combine system prompt, graph context, fused episodes,
|
||||||
|
- recent episodes, and user message.
|
||||||
|
+ The query is embedded once and shared with entity search. Both paths are
|
||||||
|
+ filtered against `recentIds` before fusion. FTS is scoped to the current
|
||||||
|
+ session or all project sessions. If `keywordWeight` is `0`, the FTS call
|
||||||
|
+ is skipped entirely. Non-critical — failures fall back to whichever
|
||||||
|
+ strategy succeeded.
|
||||||
|
+
|
||||||
|
+7. **Entity search + graph expansion** — query `entities` Qdrant collection
|
||||||
|
+ (project-scoped, or session-scoped via `/sessions/:id/entity-ids` for
|
||||||
|
+ non-project chats). Entity IDs are expanded into a 1-hop subgraph via
|
||||||
|
+ `POST /graph/neighbors`; on failure, falls back to flat entity list.
|
||||||
|
+ Non-critical.
|
||||||
|
+
|
||||||
|
+8. **Scored pool + budget selection** — `buildScoredPool` combines RRF scores,
|
||||||
|
+ recency, and entity-linkage bonus; `selectWithinBudget` fills `contextBudget`
|
||||||
|
+ (char/4 token estimation on stored text) above a guaranteed floor of
|
||||||
|
+ `minRecentEpisodes` recent episodes. Selected episode IDs are then reported
|
||||||
|
+ to `POST /episodes/touch` fire-and-forget (access tracking for the
|
||||||
|
+ consolidation lifecycle).
|
||||||
|
+
|
||||||
|
+9. **Prompt assembly** — combine system prompt, graph context, selected
|
||||||
|
+ episodes, guaranteed recent episodes, and user message.
|
||||||
|
|
||||||
|
-9. **Inference** — send to inference service. `/chat` awaits full response;
|
||||||
|
- `/chat/stream` pipes SSE chunks to the client.
|
||||||
|
+10. **Inference** — send to inference service. `/chat` awaits full response;
|
||||||
|
+ `/chat/stream` pipes SSE chunks to the client.
|
||||||
|
|
||||||
|
-10. **Episode write** — write exchange back to memory with `projectId`.
|
||||||
|
+11. **Episode write** — write exchange back to memory with `projectId`.
|
||||||
|
|
||||||
|
-11. **Summarisation trigger** — `triggerSummary(session, allEpisodes)` called
|
||||||
|
+12. **Summarisation trigger** — `triggerSummary(session)` called
|
||||||
|
fire-and-forget. See `summarization.md` for full details.
|
||||||
|
|
||||||
|
-12. **Auto-naming** — on first message with no session name, fires a secondary
|
||||||
|
+13. **Auto-naming** — on first message with no session name, fires a secondary
|
||||||
|
inference call (max 20 tokens, temperature 0.3) to generate a session name.
|
||||||
|
|
||||||
|
### Prompt Structure
|
||||||
|
diff -ruN a/docs/services/shared.md b/docs/services/shared.md
|
||||||
|
--- a/docs/services/shared.md 2026-08-24 06:09:15.371834893 +0000
|
||||||
|
+++ b/docs/services/shared.md 2026-08-24 06:09:15.374906804 +0000
|
||||||
|
@@ -203,6 +203,16 @@
|
||||||
|
SUMMARY_MIN_EPISODES=5
|
||||||
|
```
|
||||||
|
|
||||||
|
+#### `CONSOLIDATION`
|
||||||
|
+
|
||||||
|
+Controls the memory consolidation lifecycle (currently dry-run only).
|
||||||
|
+
|
||||||
|
+| Key | Value | Description |
|
||||||
|
+|---|---|---|
|
||||||
|
+| `MIN_AGE_DAYS` | `7` | Episodes younger than this are never consolidation candidates |
|
||||||
|
+| `MIN_SESSION_EPISODES` | `20` | Sessions with fewer episodes are skipped entirely |
|
||||||
|
+| `CANDIDATE_LIMIT` | `50` | Max candidates returned per scoring query |
|
||||||
|
+
|
||||||
|
#### `SQLITE`
|
||||||
|
|
||||||
|
| Key | Value | Description |
|
||||||
-31
@@ -1,31 +0,0 @@
|
|||||||
// scratch.js — run with: SQLITE_PATH=./packages/memory-service/data/test-copy.db node scratch.js
|
|
||||||
const { getDB } = require('./packages/memory-service/src/db');
|
|
||||||
const episodic = require('./packages/memory-service/src/episodic');
|
|
||||||
|
|
||||||
const db = getDB();
|
|
||||||
|
|
||||||
// Backdate half the episodes to 10-40 days old, give some fake access history
|
|
||||||
const eps = db.prepare('SELECT id FROM episodes').all();
|
|
||||||
const now = Math.floor(Date.now() / 1000);
|
|
||||||
|
|
||||||
for (const { id } of eps) {
|
|
||||||
if (Math.random() < 0.5) continue; // leave half young, so the age floor stays testable
|
|
||||||
const daysOld = 10 + Math.floor(Math.random() * 30);
|
|
||||||
const createdAt = now - daysOld * 86400;
|
|
||||||
const touched = Math.random() < 0.6;
|
|
||||||
db.prepare(`
|
|
||||||
UPDATE episodes
|
|
||||||
SET created_at = ?,
|
|
||||||
access_count = ?,
|
|
||||||
last_accessed_at = ?
|
|
||||||
WHERE id = ?
|
|
||||||
`).run(
|
|
||||||
createdAt,
|
|
||||||
touched ? 1 + Math.floor(Math.random() * 5) : 0,
|
|
||||||
touched ? (createdAt + Math.floor(Math.random() * daysOld) * 86400) * 1000 : null, // accessed sometime after creation, in ms
|
|
||||||
id
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Now run the actual candidate function against the doctored data
|
|
||||||
console.table(episodic.getConsolidationCandidates(19)); // whatever session id has rows
|
|
||||||
Reference in New Issue
Block a user