diff --git a/nexusai-fts-tokenizing.patch b/nexusai-fts-tokenizing.patch new file mode 100644 index 0000000..5ae66d4 --- /dev/null +++ b/nexusai-fts-tokenizing.patch @@ -0,0 +1,69 @@ +diff -ruN nexusai-baseline/packages/memory-service/src/episodic/index.js nexusai/packages/memory-service/src/episodic/index.js +--- nexusai-baseline/packages/memory-service/src/episodic/index.js 2026-08-17 08:00:53.663975458 +0000 ++++ nexusai/packages/memory-service/src/episodic/index.js 2026-08-17 08:02:22.352734961 +0000 +@@ -3,6 +3,36 @@ + const semantic = require('../semantic'); + const { extractAndStoreEntities } = require('../entities/extraction') + ++// Common English function words that add noise to keyword matching. Kept short ++// on purpose — FTS5's BM25 ranking already down-weights frequent terms; this is ++// mainly to stop an all-stopword query ("how do I do this") from matching everything. ++const FTS_STOPWORDS = new Set([ ++ 'the','a','an','and','or','but','if','then','of','to','in','on','at','for', ++ 'with','is','are','was','were','be','been','being','do','does','did','how', ++ 'what','why','when','where','who','which','this','that','these','those', ++ 'i','you','it','we','they','me','my','your','can','could','would','should', ++ 'will','about','as','by','from','so','just','get','got','have','has','had', ++]); ++ ++// Turn a free-text message into an FTS5 MATCH expression. The message is tokenized ++// into significant terms, each wrapped in quotes (so any FTS5-significant token in ++// the message — e.g. a literal "OR" or "*" — is treated as a search term, not an ++// operator), and the terms are OR-joined for recall. Returns null when nothing ++// usable remains, so the caller can skip keyword search rather than run a ++// degenerate query. ++// ++// This replaces the previous behaviour of quoting the ENTIRE message as one phrase, ++// which required the whole message to appear verbatim and made keyword recall ~nil. ++function buildFtsQuery(query) { ++ const tokens = query ++ .toLowerCase() ++ .split(/[^\p{L}\p{N}]+/u) // split on any non-letter/number, unicode-aware ++ .filter(t => t.length >= 2 && !FTS_STOPWORDS.has(t)); ++ ++ if (tokens.length === 0) return null; ++ return tokens.map(t => `"${t}"`).join(' OR '); ++} ++ + // --Sessions -------------------------------------------------- + + // Creates a new session with the given external ID and optional metadata +@@ -199,7 +229,9 @@ + // Searches episodes using FTS5 full-text search, ordered by relevance, with a limit + function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds = null) { + const db = getDB(); +- const safeQuery = `"${query.replace(/"/g, '""')}"`; ++ const match = buildFtsQuery(query); ++ if (!match) return []; // no usable keyword terms — let semantic retrieval carry this turn ++ + if (sessionIds && sessionIds.length > 0) { + const ph = sessionIds.map(() => '?').join(','); + return db.prepare(` +@@ -209,7 +241,7 @@ + AND e.session_id IN (${ph}) + ORDER BY rank + LIMIT ? +- `).all(safeQuery, ...sessionIds, limit).map(parseRow); ++ `).all(match, ...sessionIds, limit).map(parseRow); + } + return db.prepare(` + SELECT e.* FROM episodes e +@@ -217,7 +249,7 @@ + WHERE episodes_fts MATCH ? + ORDER BY rank + LIMIT ? +- `).all(safeQuery, limit).map(parseRow); ++ `).all(match, limit).map(parseRow); + } + + // Deletes an episode by its ID diff --git a/packages/memory-service/src/episodic/index.js b/packages/memory-service/src/episodic/index.js index a8c9caf..b00f856 100644 --- a/packages/memory-service/src/episodic/index.js +++ b/packages/memory-service/src/episodic/index.js @@ -3,6 +3,36 @@ const { EPISODIC, getEnv, SERVICES, parseRow, formatEpisodeText, SUMMARIES, logg const semantic = require('../semantic'); const { extractAndStoreEntities } = require('../entities/extraction') +// Common English function words that add noise to keyword matching. Kept short +// on purpose — FTS5's BM25 ranking already down-weights frequent terms; this is +// mainly to stop an all-stopword query ("how do I do this") from matching everything. +const FTS_STOPWORDS = new Set([ + 'the','a','an','and','or','but','if','then','of','to','in','on','at','for', + 'with','is','are','was','were','be','been','being','do','does','did','how', + 'what','why','when','where','who','which','this','that','these','those', + 'i','you','it','we','they','me','my','your','can','could','would','should', + 'will','about','as','by','from','so','just','get','got','have','has','had', +]); + +// Turn a free-text message into an FTS5 MATCH expression. The message is tokenized +// into significant terms, each wrapped in quotes (so any FTS5-significant token in +// the message — e.g. a literal "OR" or "*" — is treated as a search term, not an +// operator), and the terms are OR-joined for recall. Returns null when nothing +// usable remains, so the caller can skip keyword search rather than run a +// degenerate query. +// +// This replaces the previous behaviour of quoting the ENTIRE message as one phrase, +// which required the whole message to appear verbatim and made keyword recall ~nil. +function buildFtsQuery(query) { + const tokens = query + .toLowerCase() + .split(/[^\p{L}\p{N}]+/u) // split on any non-letter/number, unicode-aware + .filter(t => t.length >= 2 && !FTS_STOPWORDS.has(t)); + + if (tokens.length === 0) return null; + return tokens.map(t => `"${t}"`).join(' OR '); +} + // --Sessions -------------------------------------------------- // Creates a new session with the given external ID and optional metadata @@ -199,7 +229,9 @@ function getEpisodesSince(sessionId, afterId) { // Searches episodes using FTS5 full-text search, ordered by relevance, with a limit function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds = null) { const db = getDB(); - const safeQuery = `"${query.replace(/"/g, '""')}"`; + const match = buildFtsQuery(query); + if (!match) return []; // no usable keyword terms — let semantic retrieval carry this turn + if (sessionIds && sessionIds.length > 0) { const ph = sessionIds.map(() => '?').join(','); return db.prepare(` @@ -209,7 +241,7 @@ function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds AND e.session_id IN (${ph}) ORDER BY rank LIMIT ? - `).all(safeQuery, ...sessionIds, limit).map(parseRow); + `).all(match, ...sessionIds, limit).map(parseRow); } return db.prepare(` SELECT e.* FROM episodes e @@ -217,7 +249,7 @@ function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds WHERE episodes_fts MATCH ? ORDER BY rank LIMIT ? - `).all(safeQuery, limit).map(parseRow); + `).all(match, limit).map(parseRow); } // Deletes an episode by its ID