utilityInference
This commit is contained in:
@@ -0,0 +1,44 @@
|
||||
const { getEnv, UTILITY } = require ('@nexusai/shared')
|
||||
|
||||
const UTILITY_URL = getEnv('UTILITY_URL', UTILITY.DEFAULT_URL);
|
||||
const UTILITY_MODEL = getEnv('UTILITY_MODEL', UTILITY.DEFAULT_MODEL);
|
||||
|
||||
// Background task inference (extraction/summarization). Uses ollama's /api/chat
|
||||
// so the model's own prompt template is server-side, no need for ChatML
|
||||
async function utilityComplete({
|
||||
system,
|
||||
user,
|
||||
json = false,
|
||||
temperature,
|
||||
maxTokens
|
||||
}) {
|
||||
const messages = [];
|
||||
if(system) messages.push({ role: 'system', content: system});
|
||||
messages.push ({role: 'user', content: user});
|
||||
|
||||
const res = await fetch (`${UTILITY_URL}/api/chat`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json'},
|
||||
body: JSON.stringify({
|
||||
model: UTILITY_MODEL,
|
||||
messages,
|
||||
stream: false,
|
||||
...UTILITY(json && {format: 'json' }),
|
||||
options: {
|
||||
temperature: temperature ?? UTILITY.TEMPERATURE,
|
||||
num_predict: maxTokens ?? UTILITY.MAX_TOKENS,
|
||||
},
|
||||
}),
|
||||
signal: AbortSignal.timeout(UTILITY.TIMEOUT_MS),
|
||||
});
|
||||
|
||||
if (!res.ok) throw new Error(`Utility backend responded ${res.status}`);
|
||||
const data = await res.json();
|
||||
|
||||
return {
|
||||
text: (data.message?.content ?? '').trim(),
|
||||
model: data.model,
|
||||
}
|
||||
}
|
||||
|
||||
module.exports = { utilityComplete};
|
||||
Reference in New Issue
Block a user