Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
38 changes: 36 additions & 2 deletions packages/runtime/src/__tests__/history-compact-summarizer.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,15 @@ import {
import { buildHistoryCompactCheckpoint } from '../history-compact-checkpoint.js';
import { SUMMARY_FORMAT_TEMPLATE } from '../history-compact-summary-validation.js';

// The summarization instruction rides as the request's trailing user message,
// never as a separate system-role field (#4634).
function trailingMessageText(options: Parameters<AiSdkGenerateTextLike>[0]): string {
const last = options.messages[options.messages.length - 1];
assert.ok(last && last.role === 'user');
const parts = Array.isArray(last.content) ? last.content : [];
return parts.map((part) => (part.type === 'text' ? part.text : '')).join('');
}

const ts = 1_700_000_000_000;
let __seq = 0;
function ev(overrides: Partial<RuntimeEvent> & { content?: RuntimeEventContent }): RuntimeEvent {
Expand Down Expand Up @@ -584,7 +593,7 @@ describe('buildLlmHistorySummarizer', () => {
const summarize = buildLlmHistorySummarizer({
resolveModel: () => 'fake-model',
generateText: async (options) => {
instructions.push(options.instructions);
instructions.push(trailingMessageText(options));
return {
text: VALID_SUMMARY,
finishReason: instructions.length === 1 ? 'length' : 'stop',
Expand Down Expand Up @@ -626,7 +635,7 @@ describe('buildLlmHistorySummarizer', () => {
const summarize = buildLlmHistorySummarizer({
resolveModel: () => 'fake-model',
generateText: async (options) => {
instructions.push(options.instructions);
instructions.push(trailingMessageText(options));
return {
text: instructions.length === 1 ? 'free-form incomplete summary' : VALID_SUMMARY,
finishReason: 'stop',
Expand All @@ -643,6 +652,31 @@ describe('buildLlmHistorySummarizer', () => {
assert.match(instructions[1] ?? '', /malformed_summary_missing_section/);
});

test('delivers the summarization instruction as the trailing user message (#4634)', async () => {
const seen: Parameters<AiSdkGenerateTextLike>[0][] = [];
const summarize = buildLlmHistorySummarizer({
resolveModel: () => 'fake-model',
generateText: async (options) => {
seen.push(options);
return { text: VALID_SUMMARY };
},
});

await summarize(
inputWith([ev({ role: 'user', author: 'user', content: { kind: 'text', text: 'hi' } })]),
);

assert.equal(seen.length, 1);
const request = seen[0];
assert.ok(request);
// A system-role instruction never reaches the wire: agentic models such
// as kimi k3-256k ignore it and keep answering the conversation, and a
// distinct system prompt forfeits the main loop's prefix cache.
assert.equal('instructions' in request, false);
assert.match(trailingMessageText(request), /context summarization assistant/);
assert.match(trailingMessageText(request), /Now write the structured summary/);
});

test('bounds a persistently malformed completion at two provider calls', async () => {
let calls = 0;
const summarize = buildLlmHistorySummarizer({
Expand Down
49 changes: 28 additions & 21 deletions packages/runtime/src/history-compact-summarizer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -41,7 +41,6 @@ export { HistoryCompactSummarizerError } from './history-compact-error.js';

export interface AiSdkGenerateTextOptions {
model: unknown;
instructions: string;
messages: ModelMessage[];
providerOptions?: Record<string, unknown>;
maxOutputTokens?: number;
Expand Down Expand Up @@ -69,7 +68,12 @@ export interface BuildLlmHistorySummarizerOptions {
// folded events are projected with the same policy the model would see them.
// The format block is the validation module's template, so the mandated
// format and the validation can never drift apart.
const SUMMARIZATION_SYSTEM_PROMPT = [
// It rides as the trailing user message, never as provider instructions
// (system role): agentic coding models such as kimi k3-256k ignore a system
// prompt in this shape and keep answering the conversation instead, so the
// fold fails open forever (#4634), and a distinct system prompt also forfeits
// the prefix cache the main-loop requests built.
const SUMMARIZATION_PROMPT = [
'You are a context summarization assistant.',
'Read the conversation between a user and an AI assistant, then produce a structured summary another LLM will use to continue the same task.',
'Do NOT continue the conversation. Do NOT answer questions in it. ONLY output the structured summary.',
Expand All @@ -84,17 +88,17 @@ const SUMMARIZATION_SYSTEM_PROMPT = [
const SUMMARY_REQUEST_INSTRUCTION =
'Now write the structured summary of the conversation above. Output only the summary.';

function shortenSummarizationSystemPrompt(): string {
function shortenSummarizationPrompt(): string {
return [
SUMMARIZATION_SYSTEM_PROMPT,
SUMMARIZATION_PROMPT,
'',
'Your previous attempt was cut off at the output limit. Produce the same summary in well under half the length: keep every section, drop detail rather than sections.',
].join('\n');
}

function repairSummarizationSystemPrompt(reason: string): string {
function repairSummarizationPrompt(reason: string): string {
return [
SUMMARIZATION_SYSTEM_PROMPT,
SUMMARIZATION_PROMPT,
'',
`A prior attempt was rejected as ${reason}.`,
'Produce one complete replacement summary from the source conversation.',
Expand Down Expand Up @@ -127,16 +131,6 @@ export function buildLlmHistorySummarizer(options: BuildLlmHistorySummarizerOpti
],
});
}
// The folded span usually ends on an assistant message. A chat-template
// model handed a conversation that already ends with its own turn emits
// an end-of-sequence token and nothing else (Ollama qwen2.5: finish
// `stop`, one output token, empty text), so the request must end with
// an instruction the model can answer. Hosted providers do not need the
// nudge and are not disturbed by it (#4559).
projectedMessages.push({
role: 'user',
content: [{ type: 'text', text: SUMMARY_REQUEST_INSTRUCTION }],
});
// Nothing is trimmed on a local estimate: whether this input fits the
// summarizer's window is its provider's answer (`input_too_large`, which
// the planner retreats on), and the output is capped outright (#4559).
Expand All @@ -162,8 +156,21 @@ export function buildLlmHistorySummarizer(options: BuildLlmHistorySummarizerOpti
step += 1;
const result = await generateText({
model,
instructions,
messages,
// The instruction rides as the trailing user message rather than the
// AI SDK's `instructions` (system role): the request must still end
// with an imperative the model can answer (#4559), and no
// system-role prompt may precede the conversation, which agentic
// models ignore and which forfeits the main loop's prefix cache
// (#4634).
messages: [
...messages,
{
role: 'user',
content: [
{ type: 'text', text: `${instructions}\n\n${SUMMARY_REQUEST_INSTRUCTION}` },
],
},
],
maxOutputTokens,
...(options.providerOptions !== undefined
? { providerOptions: options.providerOptions }
Expand All @@ -190,12 +197,12 @@ export function buildLlmHistorySummarizer(options: BuildLlmHistorySummarizerOpti
return { text: result.text, defect, truncated };
};

let initial = await generateSummary(SUMMARIZATION_SYSTEM_PROMPT, projectedMessages);
let initial = await generateSummary(SUMMARIZATION_PROMPT, projectedMessages);
if (initial.truncated) {
// The provider cut the summary at the output cap. One shorter attempt;
// a second cut is the provider saying this span will not summarize
// inside the cap, and the fold fails open.
initial = await generateSummary(shortenSummarizationSystemPrompt(), projectedMessages);
initial = await generateSummary(shortenSummarizationPrompt(), projectedMessages);
if (initial.truncated) throw new HistoryCompactSummarizerError('output_length');
}
if (!initial.defect) return initial.text;
Expand All @@ -206,7 +213,7 @@ export function buildLlmHistorySummarizer(options: BuildLlmHistorySummarizerOpti
// A malformed provider completion is often repairable, but retries must
// be bounded: one stricter attempt, then the caller's failure circuit
// records the stable defect for this compaction input.
const repairInstructions = repairSummarizationSystemPrompt(initial.defect);
const repairInstructions = repairSummarizationPrompt(initial.defect);
let repaired: Awaited<ReturnType<typeof generateSummary>>;
try {
repaired = await generateSummary(repairInstructions, projectedMessages);
Expand Down