entities extraction fix
This commit is contained in:
@@ -18,6 +18,16 @@ function isIgnoredName(name) {
|
||||
return IGNORED_NAMES.has(normalized);
|
||||
}
|
||||
|
||||
// Guards against the extraction model regurgitating the "known entities" hint block
|
||||
// as if those entities appeared in the conversation. The prompt constrains names to
|
||||
// short proper nouns, so a genuinely-discussed entity appears verbatim in the
|
||||
// exchange; a regurgitated one generally does not. Whitespace/case-normalized
|
||||
// substring match.
|
||||
function mentionedIn(name, haystack) {
|
||||
const norm = s => s.toLowerCase().replace(/\s+/g, ' ').trim();
|
||||
return norm(haystack).includes(norm(name));
|
||||
}
|
||||
|
||||
// NOTE: This prompt uses ChatML format (<|im_start|> / <|im_end|> tags), which is
|
||||
// specific to qwen-family models. If EXTRACTION_MODEL is changed to a Llama-family
|
||||
// or other model, this format will need to change — most alternatives use either
|
||||
@@ -46,6 +56,7 @@ function buildExtractionPrompt(userMessage, aiResponse, knownEntities = []) {
|
||||
' "notes": one specific sentence about this entity based on the conversation',
|
||||
'For relationships, use snake_case verb labels (e.g. works_on, manages, uses, knows, located_in, part_of, created_by).',
|
||||
'Only include relationships between entities you have listed above.',
|
||||
'The known-entities list below is ONLY for consistent spelling and types. Do NOT output an entity unless it actually appears in the conversation.',
|
||||
'Return this exact JSON structure:',
|
||||
'{ "entities": [{"name": "...", "type": "...", "notes": "..."}], "relationships": [{"from": "...", "fromType": "...", "to": "...", "toType": "...", "label": "...", "notes": "..."}] }',
|
||||
'',
|
||||
@@ -129,9 +140,17 @@ async function extractAndStoreEntities(userMessage, aiResponse, episodeId=null,
|
||||
const entityMap = new Map();
|
||||
let saved = 0;
|
||||
|
||||
// Everything the model was actually shown from THIS exchange. Extracted names
|
||||
// are verified against this to reject regurgitated known-entity hints.
|
||||
const conversationText = `${userMessage} ${aiResponse}`;
|
||||
|
||||
for (const { name, type, notes } of entities) {
|
||||
if (!name || !type || !ENTITY_TYPES.includes(type)) continue;
|
||||
if (isIgnoredName(name)) continue;
|
||||
if (!mentionedIn(name, conversationText)) {
|
||||
logger.debug(`[entities] Skipping "${name}" — not present in the exchange (likely hint regurgitation)`);
|
||||
continue;
|
||||
}
|
||||
|
||||
const entity = upsertEntity(name, type, notes ?? null);
|
||||
entityMap.set(`${name}::${type}`, entity);
|
||||
@@ -178,4 +197,4 @@ async function extractAndStoreEntities(userMessage, aiResponse, episodeId=null,
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = { extractAndStoreEntities };
|
||||
module.exports = { extractAndStoreEntities, mentionedIn, isIgnoredName };
|
||||
Reference in New Issue
Block a user