fts tokenization fix
This commit is contained in:
@@ -1,69 +0,0 @@
|
||||
diff -ruN nexusai-baseline/packages/memory-service/src/episodic/index.js nexusai/packages/memory-service/src/episodic/index.js
|
||||
--- nexusai-baseline/packages/memory-service/src/episodic/index.js 2026-08-17 08:00:53.663975458 +0000
|
||||
+++ nexusai/packages/memory-service/src/episodic/index.js 2026-08-17 08:02:22.352734961 +0000
|
||||
@@ -3,6 +3,36 @@
|
||||
const semantic = require('../semantic');
|
||||
const { extractAndStoreEntities } = require('../entities/extraction')
|
||||
|
||||
+// Common English function words that add noise to keyword matching. Kept short
|
||||
+// on purpose — FTS5's BM25 ranking already down-weights frequent terms; this is
|
||||
+// mainly to stop an all-stopword query ("how do I do this") from matching everything.
|
||||
+const FTS_STOPWORDS = new Set([
|
||||
+ 'the','a','an','and','or','but','if','then','of','to','in','on','at','for',
|
||||
+ 'with','is','are','was','were','be','been','being','do','does','did','how',
|
||||
+ 'what','why','when','where','who','which','this','that','these','those',
|
||||
+ 'i','you','it','we','they','me','my','your','can','could','would','should',
|
||||
+ 'will','about','as','by','from','so','just','get','got','have','has','had',
|
||||
+]);
|
||||
+
|
||||
+// Turn a free-text message into an FTS5 MATCH expression. The message is tokenized
|
||||
+// into significant terms, each wrapped in quotes (so any FTS5-significant token in
|
||||
+// the message — e.g. a literal "OR" or "*" — is treated as a search term, not an
|
||||
+// operator), and the terms are OR-joined for recall. Returns null when nothing
|
||||
+// usable remains, so the caller can skip keyword search rather than run a
|
||||
+// degenerate query.
|
||||
+//
|
||||
+// This replaces the previous behaviour of quoting the ENTIRE message as one phrase,
|
||||
+// which required the whole message to appear verbatim and made keyword recall ~nil.
|
||||
+function buildFtsQuery(query) {
|
||||
+ const tokens = query
|
||||
+ .toLowerCase()
|
||||
+ .split(/[^\p{L}\p{N}]+/u) // split on any non-letter/number, unicode-aware
|
||||
+ .filter(t => t.length >= 2 && !FTS_STOPWORDS.has(t));
|
||||
+
|
||||
+ if (tokens.length === 0) return null;
|
||||
+ return tokens.map(t => `"${t}"`).join(' OR ');
|
||||
+}
|
||||
+
|
||||
// --Sessions --------------------------------------------------
|
||||
|
||||
// Creates a new session with the given external ID and optional metadata
|
||||
@@ -199,7 +229,9 @@
|
||||
// Searches episodes using FTS5 full-text search, ordered by relevance, with a limit
|
||||
function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds = null) {
|
||||
const db = getDB();
|
||||
- const safeQuery = `"${query.replace(/"/g, '""')}"`;
|
||||
+ const match = buildFtsQuery(query);
|
||||
+ if (!match) return []; // no usable keyword terms — let semantic retrieval carry this turn
|
||||
+
|
||||
if (sessionIds && sessionIds.length > 0) {
|
||||
const ph = sessionIds.map(() => '?').join(',');
|
||||
return db.prepare(`
|
||||
@@ -209,7 +241,7 @@
|
||||
AND e.session_id IN (${ph})
|
||||
ORDER BY rank
|
||||
LIMIT ?
|
||||
- `).all(safeQuery, ...sessionIds, limit).map(parseRow);
|
||||
+ `).all(match, ...sessionIds, limit).map(parseRow);
|
||||
}
|
||||
return db.prepare(`
|
||||
SELECT e.* FROM episodes e
|
||||
@@ -217,7 +249,7 @@
|
||||
WHERE episodes_fts MATCH ?
|
||||
ORDER BY rank
|
||||
LIMIT ?
|
||||
- `).all(safeQuery, limit).map(parseRow);
|
||||
+ `).all(match, limit).map(parseRow);
|
||||
}
|
||||
|
||||
// Deletes an episode by its ID
|
||||
Reference in New Issue
Block a user