diff --git a/nexusai-extraction-guard.patch b/nexusai-extraction-guard.patch deleted file mode 100644 index 00f231c..0000000 --- a/nexusai-extraction-guard.patch +++ /dev/null @@ -1,107 +0,0 @@ -diff -ruN nexusai-baseline/packages/memory-service/src/entities/extraction.js nexusai/packages/memory-service/src/entities/extraction.js ---- nexusai-baseline/packages/memory-service/src/entities/extraction.js 2026-08-17 08:30:14.134940899 +0000 -+++ nexusai/packages/memory-service/src/entities/extraction.js 2026-08-17 09:31:36.173324394 +0000 -@@ -18,6 +18,16 @@ - return IGNORED_NAMES.has(normalized); - } - -+// Guards against the extraction model regurgitating the "known entities" hint block -+// as if those entities appeared in the conversation. The prompt constrains names to -+// short proper nouns, so a genuinely-discussed entity appears verbatim in the -+// exchange; a regurgitated one generally does not. Whitespace/case-normalized -+// substring match. -+function mentionedIn(name, haystack) { -+ const norm = s => s.toLowerCase().replace(/\s+/g, ' ').trim(); -+ return norm(haystack).includes(norm(name)); -+} -+ - // NOTE: This prompt uses ChatML format (<|im_start|> / <|im_end|> tags), which is - // specific to qwen-family models. If EXTRACTION_MODEL is changed to a Llama-family - // or other model, this format will need to change — most alternatives use either -@@ -46,6 +56,7 @@ - ' "notes": one specific sentence about this entity based on the conversation', - 'For relationships, use snake_case verb labels (e.g. works_on, manages, uses, knows, located_in, part_of, created_by).', - 'Only include relationships between entities you have listed above.', -+ 'The known-entities list below is ONLY for consistent spelling and types. Do NOT output an entity unless it actually appears in the conversation.', - 'Return this exact JSON structure:', - '{ "entities": [{"name": "...", "type": "...", "notes": "..."}], "relationships": [{"from": "...", "fromType": "...", "to": "...", "toType": "...", "label": "...", "notes": "..."}] }', - '', -@@ -129,9 +140,17 @@ - const entityMap = new Map(); - let saved = 0; - -+ // Everything the model was actually shown from THIS exchange. Extracted names -+ // are verified against this to reject regurgitated known-entity hints. -+ const conversationText = `${userMessage} ${aiResponse}`; -+ - for (const { name, type, notes } of entities) { - if (!name || !type || !ENTITY_TYPES.includes(type)) continue; - if (isIgnoredName(name)) continue; -+ if (!mentionedIn(name, conversationText)) { -+ logger.debug(`[entities] Skipping "${name}" — not present in the exchange (likely hint regurgitation)`); -+ continue; -+ } - - const entity = upsertEntity(name, type, notes ?? null); - entityMap.set(`${name}::${type}`, entity); -@@ -178,4 +197,4 @@ - } - } - --module.exports = { extractAndStoreEntities }; -\ No newline at end of file -+module.exports = { extractAndStoreEntities, mentionedIn, isIgnoredName }; -\ No newline at end of file -diff -ruN nexusai-baseline/test/entity-extraction.test.js nexusai/test/entity-extraction.test.js ---- nexusai-baseline/test/entity-extraction.test.js 1970-01-01 00:00:00.000000000 +0000 -+++ nexusai/test/entity-extraction.test.js 2026-08-17 09:32:11.494207622 +0000 -@@ -0,0 +1,49 @@ -+// Entity extraction guards — tests the REAL mentionedIn and isIgnoredName filters -+// that keep the extractor from storing (a) greetings and (b) "known entities" hint -+// regurgitation. The mentionedIn cases are built from an actual observed failure: -+// a "good morning" turn where qwen echoed the hint list back as fake extractions. -+const { test } = require('node:test'); -+const assert = require('node:assert'); -+const { mentionedIn, isIgnoredName } = require('../packages/memory-service/src/entities/extraction'); -+ -+test('mentionedIn accepts names that actually appear in the exchange', () => { -+ const text = 'User: I love how Sanderson writes magic systems. Assistant: Brandon Sanderson is known for hard magic.'; -+ assert.ok(mentionedIn('Sanderson', text)); -+ assert.ok(mentionedIn('Brandon Sanderson', text)); -+}); -+ -+test('mentionedIn is case- and whitespace-insensitive', () => { -+ assert.ok(mentionedIn('digimon world next order', 'talking about Digimon World Next Order today')); -+ assert.ok(mentionedIn('Digimon World', 'the Digimon World game')); // collapsed whitespace -+}); -+ -+test('mentionedIn matches a name inside a possessive', () => { -+ assert.ok(mentionedIn('Sanderson', "Sanderson's latest book")); -+}); -+ -+test('regurgitated hint entities are rejected on a greeting turn (the bug)', () => { -+ // The exchange that triggered the bug — a bare greeting, no entities in it. -+ const exchange = 'good morning again Morning! Good timing — anything on your mind today?'; -+ // The names qwen wrongly "extracted" from the hint block, verbatim from the logs. -+ const regurgitated = [ -+ 'Sanderson', 'Baldree', 'Riordan', -+ 'Digimon World Next Order', -+ 'Digimon Adventure: Last Evolution Kizuna', -+ 'Digimon Adventure (2020 reboot)', -+ ]; -+ for (const name of regurgitated) { -+ assert.strictEqual(mentionedIn(name, exchange), false, `"${name}" should be rejected — not in the exchange`); -+ } -+}); -+ -+test('a real entity survives even when hint regurgitation is happening', () => { -+ const exchange = 'good morning — did you finish the Qdrant migration? Yes, the Qdrant collection is rebuilt.'; -+ assert.ok(mentionedIn('Qdrant', exchange), 'genuinely-discussed entity is kept'); -+}); -+ -+test('isIgnoredName catches greetings regardless of punctuation/case', () => { -+ assert.ok(isIgnoredName('Good morning')); -+ assert.ok(isIgnoredName('good morning!')); -+ assert.ok(isIgnoredName(' HELLO ')); -+ assert.strictEqual(isIgnoredName('Qdrant'), false); -+});