utility inference cleanup
This commit is contained in:
@@ -60,12 +60,20 @@ function buildProjectSummaryFromEpisodesPrompt(projectName, episodes) {
|
||||
|
||||
async function generateProjectSummaryFromEpisodes(projectName, episodes) {
|
||||
const user = buildProjectSummaryFromEpisodesPrompt(projectName, episodes);
|
||||
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
||||
return utilityInference({
|
||||
user,
|
||||
temperature: SUMMARIES.TEMPERATURE,
|
||||
maxTokens: SUMMARIES.PROJECT_GEN_MAX_TOKENS,
|
||||
});
|
||||
}
|
||||
|
||||
async function generateProjectSummary(projectName, sessionSummaries) {
|
||||
const user = buildProjectSummaryPrompt(projectName, sessionSummaries);
|
||||
return utilityInference({ user, temperature: 0.2, maxTokens: 1200 });
|
||||
return utilityInference({
|
||||
user,
|
||||
temperature: SUMMARIES.TEMPERATURE,
|
||||
maxTokens: SUMMARIES.PROJECT_GEN_MAX_TOKENS,
|
||||
});
|
||||
}
|
||||
|
||||
// Main entry point — called by the route handler
|
||||
|
||||
@@ -43,8 +43,8 @@ async function generateSummary(episodes, existingSummary = null) {
|
||||
|
||||
const content = await utilityInference({
|
||||
user,
|
||||
temperature: 0.2, // slightly higher than entities — summaries benefit from some fluency
|
||||
maxTokens: 500, // generous but bounded — keeps summaries from running long
|
||||
temperature: SUMMARIES.TEMPERATURE,
|
||||
maxTokens: SUMMARIES.SESSION_GEN_MAX_TOKENS,
|
||||
});
|
||||
|
||||
return content;
|
||||
|
||||
@@ -78,6 +78,13 @@ const SUMMARIES = {
|
||||
MIN_EPISODES_SINCE: 5, // don't resummarize until N new episodes since last summary
|
||||
MAX_SUMMARY_CHARS: 8000, // max chars to include from recent episodes when generating summary (to control prompt size)
|
||||
MAX_PROJECT_EPISODE_LIMIT: 200, // max number of episodes to consider from the entire project when generating summary (to control prompt size)
|
||||
|
||||
// Generation params for the utility model (passed to utilityInference).
|
||||
// Distinct from MAX_SUMMARY_TOKENS above, which gates STORED summary size;
|
||||
// these two cap GENERATION length (num_predict) per summary type.
|
||||
TEMPERATURE: 0.2, // slightly higher than entities (0.1) — summaries benefit from some fluency
|
||||
SESSION_GEN_MAX_TOKENS: 500, // num_predict for a session summary (3-5 sentences)
|
||||
PROJECT_GEN_MAX_TOKENS: 1200, // num_predict for a project overview (multi-paragraph)
|
||||
}
|
||||
|
||||
const ENTITIES = {
|
||||
|
||||
Reference in New Issue
Block a user