diff -ruN nexusai-baseline/packages/memory-service/src/entities/extraction.js nexusai/packages/memory-service/src/entities/extraction.js --- nexusai-baseline/packages/memory-service/src/entities/extraction.js 2026-08-17 08:30:14.134940899 +0000 +++ nexusai/packages/memory-service/src/entities/extraction.js 2026-08-17 09:31:36.173324394 +0000 @@ -18,6 +18,16 @@ return IGNORED_NAMES.has(normalized); } +// Guards against the extraction model regurgitating the "known entities" hint block +// as if those entities appeared in the conversation. The prompt constrains names to +// short proper nouns, so a genuinely-discussed entity appears verbatim in the +// exchange; a regurgitated one generally does not. Whitespace/case-normalized +// substring match. +function mentionedIn(name, haystack) { + const norm = s => s.toLowerCase().replace(/\s+/g, ' ').trim(); + return norm(haystack).includes(norm(name)); +} + // NOTE: This prompt uses ChatML format (<|im_start|> / <|im_end|> tags), which is // specific to qwen-family models. If EXTRACTION_MODEL is changed to a Llama-family // or other model, this format will need to change — most alternatives use either @@ -46,6 +56,7 @@ ' "notes": one specific sentence about this entity based on the conversation', 'For relationships, use snake_case verb labels (e.g. works_on, manages, uses, knows, located_in, part_of, created_by).', 'Only include relationships between entities you have listed above.', + 'The known-entities list below is ONLY for consistent spelling and types. Do NOT output an entity unless it actually appears in the conversation.', 'Return this exact JSON structure:', '{ "entities": [{"name": "...", "type": "...", "notes": "..."}], "relationships": [{"from": "...", "fromType": "...", "to": "...", "toType": "...", "label": "...", "notes": "..."}] }', '', @@ -129,9 +140,17 @@ const entityMap = new Map(); let saved = 0; + // Everything the model was actually shown from THIS exchange. Extracted names + // are verified against this to reject regurgitated known-entity hints. + const conversationText = `${userMessage} ${aiResponse}`; + for (const { name, type, notes } of entities) { if (!name || !type || !ENTITY_TYPES.includes(type)) continue; if (isIgnoredName(name)) continue; + if (!mentionedIn(name, conversationText)) { + logger.debug(`[entities] Skipping "${name}" — not present in the exchange (likely hint regurgitation)`); + continue; + } const entity = upsertEntity(name, type, notes ?? null); entityMap.set(`${name}::${type}`, entity); @@ -178,4 +197,4 @@ } } -module.exports = { extractAndStoreEntities }; \ No newline at end of file +module.exports = { extractAndStoreEntities, mentionedIn, isIgnoredName }; \ No newline at end of file diff -ruN nexusai-baseline/test/entity-extraction.test.js nexusai/test/entity-extraction.test.js --- nexusai-baseline/test/entity-extraction.test.js 1970-01-01 00:00:00.000000000 +0000 +++ nexusai/test/entity-extraction.test.js 2026-08-17 09:32:11.494207622 +0000 @@ -0,0 +1,49 @@ +// Entity extraction guards — tests the REAL mentionedIn and isIgnoredName filters +// that keep the extractor from storing (a) greetings and (b) "known entities" hint +// regurgitation. The mentionedIn cases are built from an actual observed failure: +// a "good morning" turn where qwen echoed the hint list back as fake extractions. +const { test } = require('node:test'); +const assert = require('node:assert'); +const { mentionedIn, isIgnoredName } = require('../packages/memory-service/src/entities/extraction'); + +test('mentionedIn accepts names that actually appear in the exchange', () => { + const text = 'User: I love how Sanderson writes magic systems. Assistant: Brandon Sanderson is known for hard magic.'; + assert.ok(mentionedIn('Sanderson', text)); + assert.ok(mentionedIn('Brandon Sanderson', text)); +}); + +test('mentionedIn is case- and whitespace-insensitive', () => { + assert.ok(mentionedIn('digimon world next order', 'talking about Digimon World Next Order today')); + assert.ok(mentionedIn('Digimon World', 'the Digimon World game')); // collapsed whitespace +}); + +test('mentionedIn matches a name inside a possessive', () => { + assert.ok(mentionedIn('Sanderson', "Sanderson's latest book")); +}); + +test('regurgitated hint entities are rejected on a greeting turn (the bug)', () => { + // The exchange that triggered the bug — a bare greeting, no entities in it. + const exchange = 'good morning again Morning! Good timing — anything on your mind today?'; + // The names qwen wrongly "extracted" from the hint block, verbatim from the logs. + const regurgitated = [ + 'Sanderson', 'Baldree', 'Riordan', + 'Digimon World Next Order', + 'Digimon Adventure: Last Evolution Kizuna', + 'Digimon Adventure (2020 reboot)', + ]; + for (const name of regurgitated) { + assert.strictEqual(mentionedIn(name, exchange), false, `"${name}" should be rejected — not in the exchange`); + } +}); + +test('a real entity survives even when hint regurgitation is happening', () => { + const exchange = 'good morning — did you finish the Qdrant migration? Yes, the Qdrant collection is rebuilt.'; + assert.ok(mentionedIn('Qdrant', exchange), 'genuinely-discussed entity is kept'); +}); + +test('isIgnoredName catches greetings regardless of punctuation/case', () => { + assert.ok(isIgnoredName('Good morning')); + assert.ok(isIgnoredName('good morning!')); + assert.ok(isIgnoredName(' HELLO ')); + assert.strictEqual(isIgnoredName('Qdrant'), false); +});