fts tokenization fix

This commit is contained in:
Storme-bit
2026-08-17 01:07:21 -07:00
parent 8d9a0b5403
commit b7788780fa
-69
View File
@@ -1,69 +0,0 @@
diff -ruN nexusai-baseline/packages/memory-service/src/episodic/index.js nexusai/packages/memory-service/src/episodic/index.js
--- nexusai-baseline/packages/memory-service/src/episodic/index.js 2026-08-17 08:00:53.663975458 +0000
+++ nexusai/packages/memory-service/src/episodic/index.js 2026-08-17 08:02:22.352734961 +0000
@@ -3,6 +3,36 @@
const semantic = require('../semantic');
const { extractAndStoreEntities } = require('../entities/extraction')
+// Common English function words that add noise to keyword matching. Kept short
+// on purpose — FTS5's BM25 ranking already down-weights frequent terms; this is
+// mainly to stop an all-stopword query ("how do I do this") from matching everything.
+const FTS_STOPWORDS = new Set([
+ 'the','a','an','and','or','but','if','then','of','to','in','on','at','for',
+ 'with','is','are','was','were','be','been','being','do','does','did','how',
+ 'what','why','when','where','who','which','this','that','these','those',
+ 'i','you','it','we','they','me','my','your','can','could','would','should',
+ 'will','about','as','by','from','so','just','get','got','have','has','had',
+]);
+
+// Turn a free-text message into an FTS5 MATCH expression. The message is tokenized
+// into significant terms, each wrapped in quotes (so any FTS5-significant token in
+// the message — e.g. a literal "OR" or "*" — is treated as a search term, not an
+// operator), and the terms are OR-joined for recall. Returns null when nothing
+// usable remains, so the caller can skip keyword search rather than run a
+// degenerate query.
+//
+// This replaces the previous behaviour of quoting the ENTIRE message as one phrase,
+// which required the whole message to appear verbatim and made keyword recall ~nil.
+function buildFtsQuery(query) {
+ const tokens = query
+ .toLowerCase()
+ .split(/[^\p{L}\p{N}]+/u) // split on any non-letter/number, unicode-aware
+ .filter(t => t.length >= 2 && !FTS_STOPWORDS.has(t));
+
+ if (tokens.length === 0) return null;
+ return tokens.map(t => `"${t}"`).join(' OR ');
+}
+
// --Sessions --------------------------------------------------
// Creates a new session with the given external ID and optional metadata
@@ -199,7 +229,9 @@
// Searches episodes using FTS5 full-text search, ordered by relevance, with a limit
function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds = null) {
const db = getDB();
- const safeQuery = `"${query.replace(/"/g, '""')}"`;
+ const match = buildFtsQuery(query);
+ if (!match) return []; // no usable keyword terms — let semantic retrieval carry this turn
+
if (sessionIds && sessionIds.length > 0) {
const ph = sessionIds.map(() => '?').join(',');
return db.prepare(`
@@ -209,7 +241,7 @@
AND e.session_id IN (${ph})
ORDER BY rank
LIMIT ?
- `).all(safeQuery, ...sessionIds, limit).map(parseRow);
+ `).all(match, ...sessionIds, limit).map(parseRow);
}
return db.prepare(`
SELECT e.* FROM episodes e
@@ -217,7 +249,7 @@
WHERE episodes_fts MATCH ?
ORDER BY rank
LIMIT ?
- `).all(safeQuery, limit).map(parseRow);
+ `).all(match, limit).map(parseRow);
}
// Deletes an episode by its ID