diff --git a/package.json b/package.json index 5dbf18a..3856ca2 100644 --- a/package.json +++ b/package.json @@ -9,7 +9,8 @@ "embedding": "npm -w packages/embedding-service start", "inference": "npm -w packages/inference-service start", "orchestration": "npm -w packages/orchestration-service start", - "mini1": "concurrently -n memory,embedding -c blue,green \"npm run memory\" \"npm run embedding\"" + "mini1": "concurrently -n memory,embedding -c blue,green \"npm run memory\" \"npm run embedding\"", + "test": "node --test" }, "engines": { "node": ">=22.0.0" diff --git a/packages/orchestration-service/src/chat/index.js b/packages/orchestration-service/src/chat/index.js index 905c85c..f0d5b41 100644 --- a/packages/orchestration-service/src/chat/index.js +++ b/packages/orchestration-service/src/chat/index.js @@ -417,4 +417,4 @@ async function chatStream(externalId, userMessage, onChunk, options = {}) { } } -module.exports = { chat, chatStream }; +module.exports = { chat, chatStream, fuseEpisodeResults, buildScoredPool, selectWithinBudget, estimateTokens }; diff --git a/packages/orchestration-service/src/services/summarization.js b/packages/orchestration-service/src/services/summarization.js index 0a2f52d..e47bd5b 100644 --- a/packages/orchestration-service/src/services/summarization.js +++ b/packages/orchestration-service/src/services/summarization.js @@ -148,4 +148,4 @@ async function triggerSummary(session) { ); } -module.exports = { triggerSummary }; \ No newline at end of file +module.exports = { triggerSummary, maybeSummarize }; \ No newline at end of file diff --git a/test/fts-query.test.js b/test/fts-query.test.js new file mode 100644 index 0000000..efc8d03 --- /dev/null +++ b/test/fts-query.test.js @@ -0,0 +1,70 @@ +// FTS keyword search — tests the REAL buildFtsQuery against a real FTS5 table. +// Uses Node's built-in node:sqlite (no external deps) to build a throwaway index, +// so this runs offline with nothing installed beyond Node itself. +const { test, before } = require('node:test'); +const assert = require('node:assert'); +const { buildFtsQuery } = require('../packages/memory-service/src/episodic/index'); + +let DatabaseSync; +try { + ({ DatabaseSync } = require('node:sqlite')); +} catch { + DatabaseSync = null; +} + +const ROWS = [ + 'To configure the Qdrant collection you set the vector size and distance metric', + 'The embedding service wraps nomic-embed-text via Ollama', + 'Reciprocal rank fusion merges semantic and keyword results', + 'I had a great morning walk today', +]; + +let db; +before(() => { + if (!DatabaseSync) return; + db = new DatabaseSync(':memory:'); + db.exec('CREATE VIRTUAL TABLE fts USING fts5(body)'); + const ins = db.prepare('INSERT INTO fts(rowid, body) VALUES (?, ?)'); + ROWS.forEach((r, i) => ins.run(i + 1, r)); +}); + +const matchRows = (match) => + db.prepare('SELECT rowid FROM fts WHERE fts MATCH ? ORDER BY rank').all(match).map(r => r.rowid); + +// --- Pure buildFtsQuery behaviour (runs without sqlite) --- + +test('tokenizes and OR-joins significant terms, dropping stopwords', () => { + assert.strictEqual( + buildFtsQuery('How do I configure the Qdrant collection?'), + '"configure" OR "qdrant" OR "collection"', + ); +}); + +test('all-stopword / empty input returns null so caller can skip keyword search', () => { + assert.strictEqual(buildFtsQuery('how do I do it?'), null); + assert.strictEqual(buildFtsQuery(' '), null); + assert.strictEqual(buildFtsQuery('!!! ??? ...'), null); +}); + +test('FTS-significant tokens in the message are neutralized as literals', () => { + // "OR" and the stray quote/asterisk must not become operators + const q = buildFtsQuery('fusion OR results" hack*'); + assert.strictEqual(q, '"fusion" OR "results" OR "hack"'); +}); + +// --- Real FTS5 matching (skips cleanly if node:sqlite is unavailable) --- + +test('a conversational query matches on shared terms (the whole point of the fix)', { skip: !DatabaseSync }, () => { + const q = buildFtsQuery('How do I configure the Qdrant collection?'); + assert.deepStrictEqual(matchRows(q), [1], 'should hit the Qdrant-collection episode'); +}); + +test('the OLD whole-phrase approach would NOT have matched (regression guard)', { skip: !DatabaseSync }, () => { + const oldPhrase = `"${'How do I configure the Qdrant collection?'.replace(/"/g, '""')}"`; + assert.deepStrictEqual(matchRows(oldPhrase), [], 'verbatim phrase never appears → the bug we fixed'); +}); + +test('injection-y input runs without error and matches real terms', { skip: !DatabaseSync }, () => { + const q = buildFtsQuery('fusion OR results" hack*'); + assert.deepStrictEqual(matchRows(q), [3]); +}); \ No newline at end of file diff --git a/test/fusion.test.js b/test/fusion.test.js new file mode 100644 index 0000000..c7911a6 --- /dev/null +++ b/test/fusion.test.js @@ -0,0 +1,57 @@ +// Reciprocal Rank Fusion — tests the REAL fuseEpisodeResults from the orchestration +// chat pipeline (previously test-fusion.js reimplemented the math, which could drift). +const { test } = require('node:test'); +const assert = require('node:assert'); +const { fuseEpisodeResults } = require('../packages/orchestration-service/src/chat/index'); + +// fuseEpisodeResults returns {episode, score}[]; map to episodes for readability +const fuse = (sem, kw, opts) => fuseEpisodeResults(sem, kw, opts).map(r => r.episode); + +const semantic = [ + { id: 1, user_message: 'ep1 — semantic only, rank 1' }, + { id: 2, user_message: 'ep2 — in both lists, rank 2 semantic' }, + { id: 3, user_message: 'ep3 — in both lists, rank 3 semantic' }, +]; +const keyword = [ + { id: 3, user_message: 'ep3 — rank 1 FTS' }, + { id: 2, user_message: 'ep2 — rank 2 FTS' }, + { id: 4, user_message: 'ep4 — FTS only, rank 3' }, +]; + +test('episodes appearing in both lists rank highest', () => { + const r = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 1, limit: 5 }); + assert.ok([2, 3].includes(r[0].id), 'a dual-list episode should be rank 1'); + const idx = id => r.findIndex(e => e.id === id); + assert.ok(idx(1) > idx(2), 'ep1 (semantic-only) should rank below ep2 (both lists)'); +}); + +test('keywordWeight:0 → pure semantic passthrough in original order', () => { + const r = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 0, limit: 5 }); + assert.strictEqual(r.length, 3, 'only the 3 semantic episodes survive'); + assert.strictEqual(r[0].id, 1); + assert.strictEqual(r[1].id, 2); +}); + +test('limit is respected', () => { + const r = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 1, limit: 2 }); + assert.strictEqual(r.length, 2); +}); + +test('no overlap → both episodes appear, semantic tie-breaks first', () => { + const r = fuse( + [{ id: 10, user_message: 'sem' }], + [{ id: 20, user_message: 'fts' }], + { semanticWeight: 1, keywordWeight: 1, limit: 5 }, + ); + assert.strictEqual(r.length, 2); + assert.strictEqual(r[0].id, 10, 'equal rank+weight: semantic inserted first, so it tie-breaks ahead'); +}); + +test('raising keywordWeight lifts a keyword-only hit above a weak semantic tail', () => { + // ep4 is keyword-only at FTS rank 3; with a high keyword weight it should + // outrank ep1 (semantic rank 1) once keywordWeight dominates. + const lowKw = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 0.1, limit: 5 }); + const highKw = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 5, limit: 5 }); + const rank = (list, id) => list.findIndex(e => e.id === id); + assert.ok(rank(highKw, 4) < rank(lowKw, 4), 'ep4 should climb as keywordWeight increases'); +}); \ No newline at end of file diff --git a/test/llamacpp-stream.test.js b/test/llamacpp-stream.test.js new file mode 100644 index 0000000..c5e41b1 --- /dev/null +++ b/test/llamacpp-stream.test.js @@ -0,0 +1,64 @@ +// llama.cpp streaming — tests the REAL completeStream against SSE that is split +// across network chunks at awkward offsets (mid-"data:" prefix, mid-JSON). This is +// the exact failure mode that produced empty responses; the parser must reassemble. +const { test } = require('node:test'); +const assert = require('node:assert'); +const { completeStream } = require('../packages/inference-service/src/providers/llamacpp'); + +// Build a full SSE body, then hand it back in arbitrarily-split chunks via a fake fetch. +function mockLlamaServer(chunks) { + global.fetch = async () => ({ + ok: true, + body: (async function* () { + for (const c of chunks) yield Buffer.from(c); + })(), + }); +} + +async function collect() { + let text = '', final = null; + for await (const ev of completeStream('prompt', {})) { + if (ev.done) final = ev; else text += ev.response; + } + return { text, final }; +} + +const sse = (obj) => `data: ${JSON.stringify(obj)}\n\n`; + +test('reassembles content when SSE lines are split across chunks', async () => { + const body = + sse({ choices: [{ delta: { content: 'Good' } }] }) + + sse({ choices: [{ delta: { content: ' morning' }, finish_reason: null }] }) + + sse({ choices: [{ delta: {}, finish_reason: 'stop' }], model: 'gemma-test' }) + + sse({ choices: [], usage: { completion_tokens: 5, prompt_tokens: 11 } }) + + 'data: [DONE]\n\n'; + + // Split at nasty offsets: mid-prefix and mid-JSON + const chunks = [body.slice(0, 9), body.slice(9, 40), body.slice(40, 41), body.slice(41, 110), body.slice(110)]; + mockLlamaServer(chunks); + + const { text, final } = await collect(); + assert.strictEqual(text, 'Good morning'); + assert.strictEqual(final.model, 'gemma-test'); + assert.strictEqual(final.tokenCount, 16, 'completion + prompt tokens'); + assert.strictEqual(final.done, true); +}); + +test('a single malformed line is skipped without killing the stream', async () => { + const body = + sse({ choices: [{ delta: { content: 'A' } }] }) + + 'data: {not valid json\n\n' + + sse({ choices: [{ delta: { content: 'B' }, finish_reason: 'stop' }], model: 'm' }) + + 'data: [DONE]\n\n'; + mockLlamaServer([body]); + + const { text } = await collect(); + assert.strictEqual(text, 'AB', 'content before and after the bad line both survive'); +}); + +test('empty stream yields a done event with zero tokens, not a throw', async () => { + mockLlamaServer(['data: [DONE]\n\n']); + const { text, final } = await collect(); + assert.strictEqual(text, ''); + assert.strictEqual(final.tokenCount, 0); +}); \ No newline at end of file diff --git a/test/summarization.test.js b/test/summarization.test.js new file mode 100644 index 0000000..9d0930d --- /dev/null +++ b/test/summarization.test.js @@ -0,0 +1,80 @@ +// Session summarization — tests the REAL maybeSummarize decision logic against +// mocked memory + Ollama endpoints. Guards the refactor that moved data-fetching +// into the summarizer (episode-stats + episodes/since) instead of receiving the +// whole session. Constants: THRESHOLD_TOKENS=200, MIN_EPISODES_SINCE=5, MAX_SUMMARY_TOKENS=800. +const { test } = require('node:test'); +const assert = require('node:assert'); +const { maybeSummarize } = require('../packages/orchestration-service/src/services/summarization'); + +const json = (obj) => ({ ok: true, status: 200, json: async () => obj }); + +// Route mocked fetches by URL and record what got written. +function mockMemory({ stats, summaries, since }) { + const calls = { posted: null, patched: null, sinceAfterId: null }; + global.fetch = async (url, opts = {}) => { + const u = String(url); + if (u.endsWith('/episode-stats')) return json(stats); + if (u.endsWith('/summaries') && (!opts.method || opts.method === 'GET')) return json(summaries); + if (u.includes('/episodes/since/')) { + calls.sinceAfterId = Number(u.split('/since/').at(-1)); + return json(since.filter(ep => ep.id > calls.sinceAfterId)); + } + if (u.includes('/api/generate')) return json({ response: 'A concise third-person summary.' }); + if (u.endsWith('/summaries') && opts.method === 'POST') { calls.posted = JSON.parse(opts.body); return json({ id: 99 }); } + if (/\/summaries\/\d+$/.test(u) && opts.method === 'PATCH') { calls.patched = JSON.parse(opts.body); return json({ ok: true }); } + throw new Error('unexpected fetch: ' + u); + }; + return calls; +} + +const eps = (n, startId = 1, tok = 50) => + Array.from({ length: n }, (_, i) => ({ id: startId + i, user_message: `u${startId + i}`, ai_response: `a${startId + i}`, token_count: tok })); + +test('under the token threshold → nothing is written', async () => { + const c = mockMemory({ stats: { totalTokens: 150, count: 3, maxId: 3 }, summaries: [], since: eps(3) }); + await maybeSummarize({ id: 1 }); + assert.strictEqual(c.posted, null); + assert.strictEqual(c.patched, null); +}); + +test('over threshold, no prior summary → POST covering all episodes', async () => { + const c = mockMemory({ stats: { totalTokens: 300, count: 6, maxId: 6 }, summaries: [], since: eps(6) }); + await maybeSummarize({ id: 2 }); + assert.ok(c.posted && !c.patched, 'creates via POST'); + assert.strictEqual(c.sinceAfterId, 0, 'fetches since/0 = whole session when no summary exists'); + assert.strictEqual(c.posted.episodeRange, '1-6'); + assert.strictEqual(c.posted.tokenCount, 300, 'token count comes from the stats aggregate'); +}); + +test('over threshold, small prior summary, ≥5 new → PATCH the tail only', async () => { + const c = mockMemory({ + stats: { totalTokens: 550, count: 11, maxId: 11 }, + summaries: [{ id: 42, content: 'short summary', episode_range: '1-6' }], + since: eps(11), + }); + await maybeSummarize({ id: 3 }); + assert.ok(c.patched && !c.posted, 'updates via PATCH'); + assert.strictEqual(c.sinceAfterId, 6, 'only fetches the un-summarized tail'); + assert.strictEqual(c.patched.episodeRange, '7-11'); +}); + +test('the MIN_EPISODES_SINCE guard blocks re-summarizing too soon', async () => { + const c = mockMemory({ + stats: { totalTokens: 450, count: 9, maxId: 9 }, + summaries: [{ id: 43, content: 'short', episode_range: '1-6' }], + since: eps(9), // only 3 new (7,8,9) < 5 + }); + await maybeSummarize({ id: 4 }); + assert.strictEqual(c.posted, null); + assert.strictEqual(c.patched, null); +}); + +test('a large existing summary forces a fresh row instead of appending', async () => { + const c = mockMemory({ + stats: { totalTokens: 900, count: 15, maxId: 15 }, + summaries: [{ id: 44, content: 'x'.repeat(850), episode_range: '1-6' }], // > MAX_SUMMARY_TOKENS + since: eps(15), + }); + await maybeSummarize({ id: 5 }); + assert.ok(c.posted && !c.patched, 'large summary → POST fresh row'); +}); \ No newline at end of file