test harness

This commit is contained in:
Storme-bit
2026-08-17 01:48:28 -07:00
parent b7788780fa
commit 42ad66c798
7 changed files with 275 additions and 3 deletions
+70
View File
@@ -0,0 +1,70 @@
// FTS keyword search — tests the REAL buildFtsQuery against a real FTS5 table.
// Uses Node's built-in node:sqlite (no external deps) to build a throwaway index,
// so this runs offline with nothing installed beyond Node itself.
const { test, before } = require('node:test');
const assert = require('node:assert');
const { buildFtsQuery } = require('../packages/memory-service/src/episodic/index');
let DatabaseSync;
try {
({ DatabaseSync } = require('node:sqlite'));
} catch {
DatabaseSync = null;
}
const ROWS = [
'To configure the Qdrant collection you set the vector size and distance metric',
'The embedding service wraps nomic-embed-text via Ollama',
'Reciprocal rank fusion merges semantic and keyword results',
'I had a great morning walk today',
];
let db;
before(() => {
if (!DatabaseSync) return;
db = new DatabaseSync(':memory:');
db.exec('CREATE VIRTUAL TABLE fts USING fts5(body)');
const ins = db.prepare('INSERT INTO fts(rowid, body) VALUES (?, ?)');
ROWS.forEach((r, i) => ins.run(i + 1, r));
});
const matchRows = (match) =>
db.prepare('SELECT rowid FROM fts WHERE fts MATCH ? ORDER BY rank').all(match).map(r => r.rowid);
// --- Pure buildFtsQuery behaviour (runs without sqlite) ---
test('tokenizes and OR-joins significant terms, dropping stopwords', () => {
assert.strictEqual(
buildFtsQuery('How do I configure the Qdrant collection?'),
'"configure" OR "qdrant" OR "collection"',
);
});
test('all-stopword / empty input returns null so caller can skip keyword search', () => {
assert.strictEqual(buildFtsQuery('how do I do it?'), null);
assert.strictEqual(buildFtsQuery(' '), null);
assert.strictEqual(buildFtsQuery('!!! ??? ...'), null);
});
test('FTS-significant tokens in the message are neutralized as literals', () => {
// "OR" and the stray quote/asterisk must not become operators
const q = buildFtsQuery('fusion OR results" hack*');
assert.strictEqual(q, '"fusion" OR "results" OR "hack"');
});
// --- Real FTS5 matching (skips cleanly if node:sqlite is unavailable) ---
test('a conversational query matches on shared terms (the whole point of the fix)', { skip: !DatabaseSync }, () => {
const q = buildFtsQuery('How do I configure the Qdrant collection?');
assert.deepStrictEqual(matchRows(q), [1], 'should hit the Qdrant-collection episode');
});
test('the OLD whole-phrase approach would NOT have matched (regression guard)', { skip: !DatabaseSync }, () => {
const oldPhrase = `"${'How do I configure the Qdrant collection?'.replace(/"/g, '""')}"`;
assert.deepStrictEqual(matchRows(oldPhrase), [], 'verbatim phrase never appears → the bug we fixed');
});
test('injection-y input runs without error and matches real terms', { skip: !DatabaseSync }, () => {
const q = buildFtsQuery('fusion OR results" hack*');
assert.deepStrictEqual(matchRows(q), [3]);
});
+57
View File
@@ -0,0 +1,57 @@
// Reciprocal Rank Fusion — tests the REAL fuseEpisodeResults from the orchestration
// chat pipeline (previously test-fusion.js reimplemented the math, which could drift).
const { test } = require('node:test');
const assert = require('node:assert');
const { fuseEpisodeResults } = require('../packages/orchestration-service/src/chat/index');
// fuseEpisodeResults returns {episode, score}[]; map to episodes for readability
const fuse = (sem, kw, opts) => fuseEpisodeResults(sem, kw, opts).map(r => r.episode);
const semantic = [
{ id: 1, user_message: 'ep1 — semantic only, rank 1' },
{ id: 2, user_message: 'ep2 — in both lists, rank 2 semantic' },
{ id: 3, user_message: 'ep3 — in both lists, rank 3 semantic' },
];
const keyword = [
{ id: 3, user_message: 'ep3 — rank 1 FTS' },
{ id: 2, user_message: 'ep2 — rank 2 FTS' },
{ id: 4, user_message: 'ep4 — FTS only, rank 3' },
];
test('episodes appearing in both lists rank highest', () => {
const r = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 1, limit: 5 });
assert.ok([2, 3].includes(r[0].id), 'a dual-list episode should be rank 1');
const idx = id => r.findIndex(e => e.id === id);
assert.ok(idx(1) > idx(2), 'ep1 (semantic-only) should rank below ep2 (both lists)');
});
test('keywordWeight:0 → pure semantic passthrough in original order', () => {
const r = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 0, limit: 5 });
assert.strictEqual(r.length, 3, 'only the 3 semantic episodes survive');
assert.strictEqual(r[0].id, 1);
assert.strictEqual(r[1].id, 2);
});
test('limit is respected', () => {
const r = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 1, limit: 2 });
assert.strictEqual(r.length, 2);
});
test('no overlap → both episodes appear, semantic tie-breaks first', () => {
const r = fuse(
[{ id: 10, user_message: 'sem' }],
[{ id: 20, user_message: 'fts' }],
{ semanticWeight: 1, keywordWeight: 1, limit: 5 },
);
assert.strictEqual(r.length, 2);
assert.strictEqual(r[0].id, 10, 'equal rank+weight: semantic inserted first, so it tie-breaks ahead');
});
test('raising keywordWeight lifts a keyword-only hit above a weak semantic tail', () => {
// ep4 is keyword-only at FTS rank 3; with a high keyword weight it should
// outrank ep1 (semantic rank 1) once keywordWeight dominates.
const lowKw = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 0.1, limit: 5 });
const highKw = fuse(semantic, keyword, { semanticWeight: 1, keywordWeight: 5, limit: 5 });
const rank = (list, id) => list.findIndex(e => e.id === id);
assert.ok(rank(highKw, 4) < rank(lowKw, 4), 'ep4 should climb as keywordWeight increases');
});
+64
View File
@@ -0,0 +1,64 @@
// llama.cpp streaming — tests the REAL completeStream against SSE that is split
// across network chunks at awkward offsets (mid-"data:" prefix, mid-JSON). This is
// the exact failure mode that produced empty responses; the parser must reassemble.
const { test } = require('node:test');
const assert = require('node:assert');
const { completeStream } = require('../packages/inference-service/src/providers/llamacpp');
// Build a full SSE body, then hand it back in arbitrarily-split chunks via a fake fetch.
function mockLlamaServer(chunks) {
global.fetch = async () => ({
ok: true,
body: (async function* () {
for (const c of chunks) yield Buffer.from(c);
})(),
});
}
async function collect() {
let text = '', final = null;
for await (const ev of completeStream('prompt', {})) {
if (ev.done) final = ev; else text += ev.response;
}
return { text, final };
}
const sse = (obj) => `data: ${JSON.stringify(obj)}\n\n`;
test('reassembles content when SSE lines are split across chunks', async () => {
const body =
sse({ choices: [{ delta: { content: 'Good' } }] }) +
sse({ choices: [{ delta: { content: ' morning' }, finish_reason: null }] }) +
sse({ choices: [{ delta: {}, finish_reason: 'stop' }], model: 'gemma-test' }) +
sse({ choices: [], usage: { completion_tokens: 5, prompt_tokens: 11 } }) +
'data: [DONE]\n\n';
// Split at nasty offsets: mid-prefix and mid-JSON
const chunks = [body.slice(0, 9), body.slice(9, 40), body.slice(40, 41), body.slice(41, 110), body.slice(110)];
mockLlamaServer(chunks);
const { text, final } = await collect();
assert.strictEqual(text, 'Good morning');
assert.strictEqual(final.model, 'gemma-test');
assert.strictEqual(final.tokenCount, 16, 'completion + prompt tokens');
assert.strictEqual(final.done, true);
});
test('a single malformed line is skipped without killing the stream', async () => {
const body =
sse({ choices: [{ delta: { content: 'A' } }] }) +
'data: {not valid json\n\n' +
sse({ choices: [{ delta: { content: 'B' }, finish_reason: 'stop' }], model: 'm' }) +
'data: [DONE]\n\n';
mockLlamaServer([body]);
const { text } = await collect();
assert.strictEqual(text, 'AB', 'content before and after the bad line both survive');
});
test('empty stream yields a done event with zero tokens, not a throw', async () => {
mockLlamaServer(['data: [DONE]\n\n']);
const { text, final } = await collect();
assert.strictEqual(text, '');
assert.strictEqual(final.tokenCount, 0);
});
+80
View File
@@ -0,0 +1,80 @@
// Session summarization — tests the REAL maybeSummarize decision logic against
// mocked memory + Ollama endpoints. Guards the refactor that moved data-fetching
// into the summarizer (episode-stats + episodes/since) instead of receiving the
// whole session. Constants: THRESHOLD_TOKENS=200, MIN_EPISODES_SINCE=5, MAX_SUMMARY_TOKENS=800.
const { test } = require('node:test');
const assert = require('node:assert');
const { maybeSummarize } = require('../packages/orchestration-service/src/services/summarization');
const json = (obj) => ({ ok: true, status: 200, json: async () => obj });
// Route mocked fetches by URL and record what got written.
function mockMemory({ stats, summaries, since }) {
const calls = { posted: null, patched: null, sinceAfterId: null };
global.fetch = async (url, opts = {}) => {
const u = String(url);
if (u.endsWith('/episode-stats')) return json(stats);
if (u.endsWith('/summaries') && (!opts.method || opts.method === 'GET')) return json(summaries);
if (u.includes('/episodes/since/')) {
calls.sinceAfterId = Number(u.split('/since/').at(-1));
return json(since.filter(ep => ep.id > calls.sinceAfterId));
}
if (u.includes('/api/generate')) return json({ response: 'A concise third-person summary.' });
if (u.endsWith('/summaries') && opts.method === 'POST') { calls.posted = JSON.parse(opts.body); return json({ id: 99 }); }
if (/\/summaries\/\d+$/.test(u) && opts.method === 'PATCH') { calls.patched = JSON.parse(opts.body); return json({ ok: true }); }
throw new Error('unexpected fetch: ' + u);
};
return calls;
}
const eps = (n, startId = 1, tok = 50) =>
Array.from({ length: n }, (_, i) => ({ id: startId + i, user_message: `u${startId + i}`, ai_response: `a${startId + i}`, token_count: tok }));
test('under the token threshold → nothing is written', async () => {
const c = mockMemory({ stats: { totalTokens: 150, count: 3, maxId: 3 }, summaries: [], since: eps(3) });
await maybeSummarize({ id: 1 });
assert.strictEqual(c.posted, null);
assert.strictEqual(c.patched, null);
});
test('over threshold, no prior summary → POST covering all episodes', async () => {
const c = mockMemory({ stats: { totalTokens: 300, count: 6, maxId: 6 }, summaries: [], since: eps(6) });
await maybeSummarize({ id: 2 });
assert.ok(c.posted && !c.patched, 'creates via POST');
assert.strictEqual(c.sinceAfterId, 0, 'fetches since/0 = whole session when no summary exists');
assert.strictEqual(c.posted.episodeRange, '1-6');
assert.strictEqual(c.posted.tokenCount, 300, 'token count comes from the stats aggregate');
});
test('over threshold, small prior summary, ≥5 new → PATCH the tail only', async () => {
const c = mockMemory({
stats: { totalTokens: 550, count: 11, maxId: 11 },
summaries: [{ id: 42, content: 'short summary', episode_range: '1-6' }],
since: eps(11),
});
await maybeSummarize({ id: 3 });
assert.ok(c.patched && !c.posted, 'updates via PATCH');
assert.strictEqual(c.sinceAfterId, 6, 'only fetches the un-summarized tail');
assert.strictEqual(c.patched.episodeRange, '7-11');
});
test('the MIN_EPISODES_SINCE guard blocks re-summarizing too soon', async () => {
const c = mockMemory({
stats: { totalTokens: 450, count: 9, maxId: 9 },
summaries: [{ id: 43, content: 'short', episode_range: '1-6' }],
since: eps(9), // only 3 new (7,8,9) < 5
});
await maybeSummarize({ id: 4 });
assert.strictEqual(c.posted, null);
assert.strictEqual(c.patched, null);
});
test('a large existing summary forces a fresh row instead of appending', async () => {
const c = mockMemory({
stats: { totalTokens: 900, count: 15, maxId: 15 },
summaries: [{ id: 44, content: 'x'.repeat(850), episode_range: '1-6' }], // > MAX_SUMMARY_TOKENS
since: eps(15),
});
await maybeSummarize({ id: 5 });
assert.ok(c.posted && !c.patched, 'large summary → POST fresh row');
});