From b7788780fa97d45244692db06bfc42eeb5a9b10f Mon Sep 17 00:00:00 2001 From: Storme-bit Date: Mon, 17 Aug 2026 01:07:21 -0700 Subject: [PATCH] fts tokenization fix --- nexusai-fts-tokenizing.patch | 69 ------------------------------------ 1 file changed, 69 deletions(-) delete mode 100644 nexusai-fts-tokenizing.patch diff --git a/nexusai-fts-tokenizing.patch b/nexusai-fts-tokenizing.patch deleted file mode 100644 index 5ae66d4..0000000 --- a/nexusai-fts-tokenizing.patch +++ /dev/null @@ -1,69 +0,0 @@ -diff -ruN nexusai-baseline/packages/memory-service/src/episodic/index.js nexusai/packages/memory-service/src/episodic/index.js ---- nexusai-baseline/packages/memory-service/src/episodic/index.js 2026-08-17 08:00:53.663975458 +0000 -+++ nexusai/packages/memory-service/src/episodic/index.js 2026-08-17 08:02:22.352734961 +0000 -@@ -3,6 +3,36 @@ - const semantic = require('../semantic'); - const { extractAndStoreEntities } = require('../entities/extraction') - -+// Common English function words that add noise to keyword matching. Kept short -+// on purpose — FTS5's BM25 ranking already down-weights frequent terms; this is -+// mainly to stop an all-stopword query ("how do I do this") from matching everything. -+const FTS_STOPWORDS = new Set([ -+ 'the','a','an','and','or','but','if','then','of','to','in','on','at','for', -+ 'with','is','are','was','were','be','been','being','do','does','did','how', -+ 'what','why','when','where','who','which','this','that','these','those', -+ 'i','you','it','we','they','me','my','your','can','could','would','should', -+ 'will','about','as','by','from','so','just','get','got','have','has','had', -+]); -+ -+// Turn a free-text message into an FTS5 MATCH expression. The message is tokenized -+// into significant terms, each wrapped in quotes (so any FTS5-significant token in -+// the message — e.g. a literal "OR" or "*" — is treated as a search term, not an -+// operator), and the terms are OR-joined for recall. Returns null when nothing -+// usable remains, so the caller can skip keyword search rather than run a -+// degenerate query. -+// -+// This replaces the previous behaviour of quoting the ENTIRE message as one phrase, -+// which required the whole message to appear verbatim and made keyword recall ~nil. -+function buildFtsQuery(query) { -+ const tokens = query -+ .toLowerCase() -+ .split(/[^\p{L}\p{N}]+/u) // split on any non-letter/number, unicode-aware -+ .filter(t => t.length >= 2 && !FTS_STOPWORDS.has(t)); -+ -+ if (tokens.length === 0) return null; -+ return tokens.map(t => `"${t}"`).join(' OR '); -+} -+ - // --Sessions -------------------------------------------------- - - // Creates a new session with the given external ID and optional metadata -@@ -199,7 +229,9 @@ - // Searches episodes using FTS5 full-text search, ordered by relevance, with a limit - function searchEpisodes(query, limit = EPISODIC.DEFAULT_SEARCH_LIMIT, sessionIds = null) { - const db = getDB(); -- const safeQuery = `"${query.replace(/"/g, '""')}"`; -+ const match = buildFtsQuery(query); -+ if (!match) return []; // no usable keyword terms — let semantic retrieval carry this turn -+ - if (sessionIds && sessionIds.length > 0) { - const ph = sessionIds.map(() => '?').join(','); - return db.prepare(` -@@ -209,7 +241,7 @@ - AND e.session_id IN (${ph}) - ORDER BY rank - LIMIT ? -- `).all(safeQuery, ...sessionIds, limit).map(parseRow); -+ `).all(match, ...sessionIds, limit).map(parseRow); - } - return db.prepare(` - SELECT e.* FROM episodes e -@@ -217,7 +249,7 @@ - WHERE episodes_fts MATCH ? - ORDER BY rank - LIMIT ? -- `).all(safeQuery, limit).map(parseRow); -+ `).all(match, limit).map(parseRow); - } - - // Deletes an episode by its ID