fts tokenization fix
This commit is contained in:
@@ -0,0 +1,69 @@
|
|||||||
|
diff -ruN nexusai-baseline/packages/memory-service/src/episodic/index.js nexusai/packages/memory-service/src/episodic/index.js
|
||||||
|
--- nexusai-baseline/packages/memory-service/src/episodic/index.js 2026-08-17 08:00:53.663975458 +0000
|
||||||
|
+++ nexusai/packages/memory-service/src/episodic/index.js 2026-08-17 08:02:22.352734961 +0000
|
||||||
|
@@ -3,6 +3,36 @@
|
||||||
|
const semantic = require('../semantic');
|
||||||
|
const { extractAndStoreEntities } = require('../entities/extraction')
|
||||||
|
|
||||||
|
+// Common English function words that add noise to keyword matching. Kept short
|
||||||
|
+// on purpose — FTS5's BM25 ranking already down-weights frequent terms; this is
|
||||||
|
+// mainly to stop an all-stopword query ("how do I do this") from matching everything.
|
||||||
|
+const FTS_STOPWORDS = new Set([
|
||||||
|
+ 'the','a','an','and','or','but','if','then','of','to','in','on','at','for',
|
||||||
|
+ 'with','is','are','was','were','be','been','being','do','does','did','how',
|
||||||
|
+ 'what','why','when','where','who','which','this','that','these','those',
|
||||||
|
+ 'i','you','it','we','they','me','my','your','can','could','would','should',
|
||||||
|
+ 'will','about','as','by','from','so','just','get','got','have','has','had',
|
||||||
|
+]);
|
||||||
|
+
|
||||||
|
+// Turn a free-text message into an FTS5 MATCH expression. The message is tokenized
|
||||||
|
+// into significant terms, each wrapped in quotes (so any FTS5-significant token in
|
||||||
|
+// the message — e.g. a literal "OR" or "*" — is treated as a search term, not an
|
||||||
|
+// operator), and the terms are OR-joined for recall. Returns null when nothing
|
||||||
|
+// usable remains, so the caller can skip keyword search rather than run a
|
||||||
|
+// degenerate query.
|
||||||
|
+//
|
||||||
|
+// This replaces the previous behaviour of quoting the ENTIRE message as one phrase,
|
||||||
|
+// which required the whole message to appear verbatim and made keyword recall ~nil.
|
||||||
|
+function buildFtsQuery(query) {
|
||||||
|
+ const tokens = query
|
||||||
|
+ .toLowerCase()
|
||||||
|
+ .split(/[^\p{L}\p{N}]+/u) // split on any non-letter/number, unicode-aware
|
||||||
|
+ .filter(t => t.length >= 2 && !FTS_STOPWORDS.has(t));
|
||||||
|
+
|
||||||
|
+ if (tokens.length === 0) return null;
|
||||||
|
+ return tokens.map(t => `"${t}"`).join(' OR ');
|
||||||
|
+}
|
||||||
|
+
|
||||||
|
// --Sessions --------------------------------------------------
|
||||||
|
|
||||||
|
// Creates a new session with the given external ID and optional metadata
|
||||||
|
@@ -199,7 +229,9 @@
|
||||||
|
// Searches episodes using FTS5 full-text search, ordered by relevance, with a limit
|
||||||
|
function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds = null) {
|
||||||
|
const db = getDB();
|
||||||
|
- const safeQuery = `"${query.replace(/"/g, '""')}"`;
|
||||||
|
+ const match = buildFtsQuery(query);
|
||||||
|
+ if (!match) return []; // no usable keyword terms — let semantic retrieval carry this turn
|
||||||
|
+
|
||||||
|
if (sessionIds && sessionIds.length > 0) {
|
||||||
|
const ph = sessionIds.map(() => '?').join(',');
|
||||||
|
return db.prepare(`
|
||||||
|
@@ -209,7 +241,7 @@
|
||||||
|
AND e.session_id IN (${ph})
|
||||||
|
ORDER BY rank
|
||||||
|
LIMIT ?
|
||||||
|
- `).all(safeQuery, ...sessionIds, limit).map(parseRow);
|
||||||
|
+ `).all(match, ...sessionIds, limit).map(parseRow);
|
||||||
|
}
|
||||||
|
return db.prepare(`
|
||||||
|
SELECT e.* FROM episodes e
|
||||||
|
@@ -217,7 +249,7 @@
|
||||||
|
WHERE episodes_fts MATCH ?
|
||||||
|
ORDER BY rank
|
||||||
|
LIMIT ?
|
||||||
|
- `).all(safeQuery, limit).map(parseRow);
|
||||||
|
+ `).all(match, limit).map(parseRow);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Deletes an episode by its ID
|
||||||
@@ -3,6 +3,36 @@ const { EPISODIC, getEnv, SERVICES, parseRow, formatEpisodeText, SUMMARIES, logg
|
|||||||
const semantic = require('../semantic');
|
const semantic = require('../semantic');
|
||||||
const { extractAndStoreEntities } = require('../entities/extraction')
|
const { extractAndStoreEntities } = require('../entities/extraction')
|
||||||
|
|
||||||
|
// Common English function words that add noise to keyword matching. Kept short
|
||||||
|
// on purpose — FTS5's BM25 ranking already down-weights frequent terms; this is
|
||||||
|
// mainly to stop an all-stopword query ("how do I do this") from matching everything.
|
||||||
|
const FTS_STOPWORDS = new Set([
|
||||||
|
'the','a','an','and','or','but','if','then','of','to','in','on','at','for',
|
||||||
|
'with','is','are','was','were','be','been','being','do','does','did','how',
|
||||||
|
'what','why','when','where','who','which','this','that','these','those',
|
||||||
|
'i','you','it','we','they','me','my','your','can','could','would','should',
|
||||||
|
'will','about','as','by','from','so','just','get','got','have','has','had',
|
||||||
|
]);
|
||||||
|
|
||||||
|
// Turn a free-text message into an FTS5 MATCH expression. The message is tokenized
|
||||||
|
// into significant terms, each wrapped in quotes (so any FTS5-significant token in
|
||||||
|
// the message — e.g. a literal "OR" or "*" — is treated as a search term, not an
|
||||||
|
// operator), and the terms are OR-joined for recall. Returns null when nothing
|
||||||
|
// usable remains, so the caller can skip keyword search rather than run a
|
||||||
|
// degenerate query.
|
||||||
|
//
|
||||||
|
// This replaces the previous behaviour of quoting the ENTIRE message as one phrase,
|
||||||
|
// which required the whole message to appear verbatim and made keyword recall ~nil.
|
||||||
|
function buildFtsQuery(query) {
|
||||||
|
const tokens = query
|
||||||
|
.toLowerCase()
|
||||||
|
.split(/[^\p{L}\p{N}]+/u) // split on any non-letter/number, unicode-aware
|
||||||
|
.filter(t => t.length >= 2 && !FTS_STOPWORDS.has(t));
|
||||||
|
|
||||||
|
if (tokens.length === 0) return null;
|
||||||
|
return tokens.map(t => `"${t}"`).join(' OR ');
|
||||||
|
}
|
||||||
|
|
||||||
// --Sessions --------------------------------------------------
|
// --Sessions --------------------------------------------------
|
||||||
|
|
||||||
// Creates a new session with the given external ID and optional metadata
|
// Creates a new session with the given external ID and optional metadata
|
||||||
@@ -199,7 +229,9 @@ function getEpisodesSince(sessionId, afterId) {
|
|||||||
// Searches episodes using FTS5 full-text search, ordered by relevance, with a limit
|
// Searches episodes using FTS5 full-text search, ordered by relevance, with a limit
|
||||||
function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds = null) {
|
function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds = null) {
|
||||||
const db = getDB();
|
const db = getDB();
|
||||||
const safeQuery = `"${query.replace(/"/g, '""')}"`;
|
const match = buildFtsQuery(query);
|
||||||
|
if (!match) return []; // no usable keyword terms — let semantic retrieval carry this turn
|
||||||
|
|
||||||
if (sessionIds && sessionIds.length > 0) {
|
if (sessionIds && sessionIds.length > 0) {
|
||||||
const ph = sessionIds.map(() => '?').join(',');
|
const ph = sessionIds.map(() => '?').join(',');
|
||||||
return db.prepare(`
|
return db.prepare(`
|
||||||
@@ -209,7 +241,7 @@ function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds
|
|||||||
AND e.session_id IN (${ph})
|
AND e.session_id IN (${ph})
|
||||||
ORDER BY rank
|
ORDER BY rank
|
||||||
LIMIT ?
|
LIMIT ?
|
||||||
`).all(safeQuery, ...sessionIds, limit).map(parseRow);
|
`).all(match, ...sessionIds, limit).map(parseRow);
|
||||||
}
|
}
|
||||||
return db.prepare(`
|
return db.prepare(`
|
||||||
SELECT e.* FROM episodes e
|
SELECT e.* FROM episodes e
|
||||||
@@ -217,7 +249,7 @@ function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds
|
|||||||
WHERE episodes_fts MATCH ?
|
WHERE episodes_fts MATCH ?
|
||||||
ORDER BY rank
|
ORDER BY rank
|
||||||
LIMIT ?
|
LIMIT ?
|
||||||
`).all(safeQuery, limit).map(parseRow);
|
`).all(match, limit).map(parseRow);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Deletes an episode by its ID
|
// Deletes an episode by its ID
|
||||||
|
|||||||
Reference in New Issue
Block a user