utility inference cleanup

This commit is contained in:
Storme-bit
2026-08-17 23:48:15 -07:00
parent a223b753e5
commit 551c7f03ec
7 changed files with 53 additions and 52 deletions
@@ -60,12 +60,20 @@ function buildProjectSummaryFromEpisodesPrompt(projectName, episodes) {
async function generateProjectSummaryFromEpisodes(projectName, episodes) {
const user = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
return utilityInference({
user,
temperature: SUMMARIES.TEMPERATURE,
maxTokens: SUMMARIES.PROJECT_GEN_MAX_TOKENS,
});
}
async function generateProjectSummary(projectName, sessionSummaries) {
const user = buildProjectSummaryPrompt(projectName, sessionSummaries);
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
return utilityInference({
user,
temperature: SUMMARIES.TEMPERATURE,
maxTokens: SUMMARIES.PROJECT_GEN_MAX_TOKENS,
});
}
// Main entry point — called by the route handler
@@ -43,8 +43,8 @@ async function generateSummary(episodes, existingSummary = null) {
const content = await utilityInference({
user,
temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
maxTokens: 500, // generous but bounded — keeps summaries from running long
temperature: SUMMARIES.TEMPERATURE,
maxTokens: SUMMARIES.SESSION_GEN_MAX_TOKENS,
});
return content;
+7
View File
@@ -78,6 +78,13 @@ const SUMMARIES = {
MIN_EPISODES_SINCE: 5, // don't resummarize until N new episodes since last summary
MAX_SUMMARY_CHARS: 8000, // max chars to include from recent episodes when generating summary (to control prompt size)
MAX_PROJECT_EPISODE_LIMIT: 200, // max number of episodes to consider from the entire project when generating summary (to control prompt size)
// Generation params for the utility model (passed to utilityInference).
// Distinct from MAX_SUMMARY_TOKENS above, which gates STORED summary size;
// these two cap GENERATION length (num_predict) per summary type.
TEMPERATURE: 0.2, // slightly higher than entities (0.1) — summaries benefit from some fluency
SESSION_GEN_MAX_TOKENS: 500, // num_predict for a session summary (3-5 sentences)
PROJECT_GEN_MAX_TOKENS: 1200, // num_predict for a project overview (multi-paragraph)
}
const ENTITIES = {