diff --git a/packages/ai/src/models.generated.ts b/packages/ai/src/models.generated.ts index fb3d3447f..6a4372881 100644 --- a/packages/ai/src/models.generated.ts +++ b/packages/ai/src/models.generated.ts @@ -2006,6 +2006,23 @@ export const MODELS = { contextWindow: 1000000, maxTokens: 131072, } satisfies Model<"bedrock-converse-stream">, + "xai.grok-4.6": { + id: "xai.grok-4.6", + name: "Grok 4.6", + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + baseUrl: "https://bedrock-runtime.us-east-1.amazonaws.com", + reasoning: true, + input: ["text", "image"], + cost: { + input: 2.2, + output: 6.6, + cacheRead: 0.55, + cacheWrite: 0, + }, + contextWindow: 500000, + maxTokens: 500000, + } satisfies Model<"bedrock-converse-stream">, "zai.glm-4.7": { id: "zai.glm-4.7", name: "GLM-4.7", @@ -4296,6 +4313,24 @@ export const MODELS = { contextWindow: 32768, maxTokens: 32768, } satisfies Model<"openai-completions">, + "@cf/qwen/qwen3.8-27b": { + id: "@cf/qwen/qwen3.8-27b", + name: "Qwen3.8 27B", + api: "openai-completions", + provider: "cloudflare-workers-ai", + baseUrl: "https://api.cloudflare.com/client/v4/accounts/{CLOUDFLARE_ACCOUNT_ID}/ai/v1", + compat: {"sendSessionAffinityHeaders":true}, + reasoning: true, + input: ["text", "image"], + cost: { + input: 0.45, + output: 3.2, + cacheRead: 0.05, + cacheWrite: 0, + }, + contextWindow: 262144, + maxTokens: 262144, + } satisfies Model<"openai-completions">, "@cf/zai-org/glm-4.7-flash": { id: "@cf/zai-org/glm-4.7-flash", name: "GLM-4.7-Flash", @@ -6319,6 +6354,42 @@ export const MODELS = { contextWindow: 262144, maxTokens: 131072, } satisfies Model<"openai-completions">, + "Qwen/Qwen3-VL-235B-A22B-Instruct": { + id: "Qwen/Qwen3-VL-235B-A22B-Instruct", + name: "Qwen3 VL 235B A22B Instruct", + api: "openai-completions", + provider: "huggingface", + baseUrl: "https://router.huggingface.co/v1", + compat: {"supportsDeveloperRole":false}, + reasoning: false, + input: ["text", "image"], + cost: { + input: 0.3, + output: 1.5, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 131072, + maxTokens: 32768, + } satisfies Model<"openai-completions">, + "Qwen/Qwen3-VL-235B-A22B-Thinking": { + id: "Qwen/Qwen3-VL-235B-A22B-Thinking", + name: "Qwen3 VL 235B A22B Thinking", + api: "openai-completions", + provider: "huggingface", + baseUrl: "https://router.huggingface.co/v1", + compat: {"supportsDeveloperRole":false}, + reasoning: true, + input: ["text", "image"], + cost: { + input: 0.98, + output: 3.95, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 131072, + maxTokens: 32768, + } satisfies Model<"openai-completions">, "Qwen/Qwen3.5-122B-A10B": { id: "Qwen/Qwen3.5-122B-A10B", name: "Qwen3.5 122B-A10B", @@ -7094,6 +7165,24 @@ export const MODELS = { contextWindow: 204800, maxTokens: 131072, } satisfies Model<"openai-completions">, + "zai-org/GLM-4.6V-Flash": { + id: "zai-org/GLM-4.6V-Flash", + name: "GLM-4.6V-Flash", + api: "openai-completions", + provider: "huggingface", + baseUrl: "https://router.huggingface.co/v1", + compat: {"supportsDeveloperRole":false}, + reasoning: true, + input: ["text", "image"], + cost: { + input: 0.3, + output: 0.9, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 131072, + maxTokens: 32768, + } satisfies Model<"openai-completions">, "zai-org/GLM-4.7": { id: "zai-org/GLM-4.7", name: "GLM-4.7", @@ -9921,7 +10010,7 @@ export const MODELS = { } satisfies Model<"openai-responses">, "gpt-5.6-sol": { id: "gpt-5.6-sol", - name: "GPT-5.6 Sol", + name: "GPT-5.6 Sol (50% Off)", api: "openai-responses", provider: "opencode", baseUrl: "https://opencode.ai/zen/v1", @@ -9929,10 +10018,10 @@ export const MODELS = { thinkingLevelMap: {"off":null,"xhigh":"xhigh","minimal":null,"max":"max"}, input: ["text", "image"], cost: { - input: 5, - output: 30, - cacheRead: 0.5, - cacheWrite: 6.25, + input: 2.5, + output: 15, + cacheRead: 0.25, + cacheWrite: 3.125, }, contextWindow: 1050000, maxTokens: 128000, @@ -10092,23 +10181,6 @@ export const MODELS = { contextWindow: 1048576, maxTokens: 131072, } satisfies Model<"openai-completions">, - "laguna-s-2.1-free": { - id: "laguna-s-2.1-free", - name: "Laguna S 2.1 Free", - api: "openai-completions", - provider: "opencode", - baseUrl: "https://opencode.ai/zen/v1", - reasoning: true, - input: ["text"], - cost: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - }, - contextWindow: 256000, - maxTokens: 32000, - } satisfies Model<"openai-completions">, "mimo-v2.5-free": { id: "mimo-v2.5-free", name: "MiMo V2.5 Free", @@ -10194,6 +10266,23 @@ export const MODELS = { contextWindow: 1048576, maxTokens: 131072, } satisfies Model<"openai-responses">, + "muse-spark-1.2-contributor-free": { + id: "muse-spark-1.2-contributor-free", + name: "Muse Spark 1.2 Free", + api: "openai-responses", + provider: "opencode", + baseUrl: "https://opencode.ai/zen/v1", + reasoning: true, + input: ["text", "image"], + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 1048576, + maxTokens: 131072, + } satisfies Model<"openai-responses">, "nemotron-3-ultra-free": { id: "nemotron-3-ultra-free", name: "Nemotron 3 Ultra Free", @@ -10355,7 +10444,7 @@ export const MODELS = { } satisfies Model<"openai-completions">, "gpt-5.6-luna": { id: "gpt-5.6-luna", - name: "GPT-5.6 Luna (2x usage)", + name: "GPT-5.6 Luna", api: "openai-responses", provider: "opencode-go", baseUrl: "https://opencode.ai/zen/go/v1", @@ -10363,10 +10452,10 @@ export const MODELS = { thinkingLevelMap: {"off":null,"xhigh":"xhigh","minimal":null,"max":"max"}, input: ["text", "image"], cost: { - input: 0.1, - output: 0.6, - cacheRead: 0.01, - cacheWrite: 0.125, + input: 0.2, + output: 1.2, + cacheRead: 0.02, + cacheWrite: 0.25, }, contextWindow: 1050000, maxTokens: 128000, @@ -10390,16 +10479,16 @@ export const MODELS = { } satisfies Model<"openai-responses">, "hy3": { id: "hy3", - name: "Hy3", + name: "Hy3 (8x usage)", api: "openai-completions", provider: "opencode-go", baseUrl: "https://opencode.ai/zen/go/v1", reasoning: true, input: ["text"], cost: { - input: 0.14, - output: 0.58, - cacheRead: 0.035, + input: 0.0175, + output: 0.0725, + cacheRead: 0.004375, cacheWrite: 0, }, contextWindow: 256000, @@ -10525,6 +10614,23 @@ export const MODELS = { contextWindow: 1000000, maxTokens: 131072, } satisfies Model<"anthropic-messages">, + "muse-spark-1.2-contributor": { + id: "muse-spark-1.2-contributor", + name: "Muse Spark 1.2 Contributor", + api: "openai-responses", + provider: "opencode-go", + baseUrl: "https://opencode.ai/zen/go/v1", + reasoning: true, + input: ["text", "image"], + cost: { + input: 0.1, + output: 0.2, + cacheRead: 0.002, + cacheWrite: 0, + }, + contextWindow: 1048576, + maxTokens: 131072, + } satisfies Model<"openai-responses">, "qwen3.6-plus": { id: "qwen3.6-plus", name: "Qwen3.6 Plus", @@ -10596,23 +10702,6 @@ export const MODELS = { } satisfies Model<"openai-completions">, }, "openrouter": { - "ai21/jamba-large-1.7": { - id: "ai21/jamba-large-1.7", - name: "AI21: Jamba Large 1.7", - api: "openai-completions", - provider: "openrouter", - baseUrl: "https://openrouter.ai/api/v1", - reasoning: false, - input: ["text"], - cost: { - input: 2, - output: 8, - cacheRead: 0, - cacheWrite: 0, - }, - contextWindow: 256000, - maxTokens: 4096, - } satisfies Model<"openai-completions">, "aion-labs/aion-2.0": { id: "aion-labs/aion-2.0", name: "AionLabs: Aion-2.0", @@ -11311,13 +11400,13 @@ export const MODELS = { reasoning: false, input: ["text"], cost: { - input: 0.27, - output: 1.12, - cacheRead: 0.135, + input: 0.25, + output: 1, + cacheRead: 0, cacheWrite: 0, }, contextWindow: 163840, - maxTokens: 65536, + maxTokens: 163840, } satisfies Model<"openai-completions">, "deepseek/deepseek-chat-v3.1": { id: "deepseek/deepseek-chat-v3.1", @@ -11444,9 +11533,9 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":"max","max":null}, input: ["text"], cost: { - input: 0.0798, - output: 0.1596, - cacheRead: 0.01596, + input: 0.088606, + output: 0.177212, + cacheRead: 0.017721200000000003, cacheWrite: 0, }, contextWindow: 1048576, @@ -11482,13 +11571,13 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":"max","max":null}, input: ["text"], cost: { - input: 0.66, - output: 1.9800000000000002, - cacheRead: 0.022, + input: 1.5999999999999999, + output: 3.1999999999999997, + cacheRead: 0.135, cacheWrite: 0, }, contextWindow: 1048576, - maxTokens: 384000, + maxTokens: 393216, } satisfies Model<"openai-completions">, "deepseek/deepseek-v4-pro-0813": { id: "deepseek/deepseek-v4-pro-0813", @@ -11501,13 +11590,13 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":"max","max":null}, input: ["text"], cost: { - input: 0.66, - output: 1.9800000000000002, - cacheRead: 0.022, + input: 1.1880000000000002, + output: 3.564, + cacheRead: 0.039599999999999996, cacheWrite: 0, }, contextWindow: 1048576, - maxTokens: 384000, + maxTokens: 4096, } satisfies Model<"openai-completions">, "dots-studio/dots-3-note-preview:free": { id: "dots-studio/dots-3-note-preview:free", @@ -11887,13 +11976,13 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":null,"max":null}, input: ["text", "image"], cost: { - input: 0.09999999999999999, + input: 0.09, output: 0.33999999999999997, - cacheRead: 0.09999999999999999, + cacheRead: 0.049999999999999996, cacheWrite: 0, }, contextWindow: 262144, - maxTokens: 262144, + maxTokens: 16384, } satisfies Model<"openai-completions">, "google/gemma-4-31b-it:free": { id: "google/gemma-4-31b-it:free", @@ -12316,13 +12405,13 @@ export const MODELS = { thinkingLevelMap: {"off":null,"minimal":null,"low":null,"medium":null,"high":"high","xhigh":null,"max":null}, input: ["text"], cost: { - input: 0.22, + input: 0.22499999999999998, output: 0.8999999999999999, - cacheRead: 0.049999999999999996, + cacheRead: 0.06, cacheWrite: 0, }, contextWindow: 204800, - maxTokens: 196608, + maxTokens: 4096, } satisfies Model<"openai-completions">, "minimax/minimax-m2.7": { id: "minimax/minimax-m2.7", @@ -12719,9 +12808,9 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":null,"max":null}, input: ["text", "image"], cost: { - input: 0.5605, - output: 2.36, - cacheRead: 0.0944, + input: 0.95, + output: 4, + cacheRead: 0.16, cacheWrite: 0, }, contextWindow: 262144, @@ -13668,10 +13757,10 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":"low","medium":"medium","high":"high","xhigh":"xhigh","max":"max"}, input: ["text", "image"], cost: { - input: 5, - output: 30, - cacheRead: 0.5, - cacheWrite: 6.25, + input: 2.5, + output: 15, + cacheRead: 0.25, + cacheWrite: 3.125, }, contextWindow: 1050000, maxTokens: 128000, @@ -14478,13 +14567,13 @@ export const MODELS = { reasoning: false, input: ["text"], cost: { - input: 0.09999999999999999, + input: 0.09, output: 1.1, - cacheRead: 0.07, + cacheRead: 0, cacheWrite: 0, }, contextWindow: 262144, - maxTokens: 262144, + maxTokens: 16384, } satisfies Model<"openai-completions">, "qwen/qwen3-next-80b-a3b-thinking": { id: "qwen/qwen3-next-80b-a3b-thinking", @@ -14641,13 +14730,13 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":null,"max":null}, input: ["text", "image"], cost: { - input: 0.29, - output: 2.4, + input: 0.26, + output: 2.08, cacheRead: 0, cacheWrite: 0, }, contextWindow: 262144, - maxTokens: 81920, + maxTokens: 262144, } satisfies Model<"openai-completions">, "qwen/qwen3.5-27b": { id: "qwen/qwen3.5-27b", @@ -14679,13 +14768,13 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":null,"max":null}, input: ["text", "image"], cost: { - input: 0.22499999999999998, - output: 1.7999999999999998, - cacheRead: 0.22499999999999998, + input: 0.25, + output: 1.25, + cacheRead: 0.25, cacheWrite: 0, }, contextWindow: 262144, - maxTokens: 65536, + maxTokens: 262144, } satisfies Model<"openai-completions">, "qwen/qwen3.5-397b-a17b": { id: "qwen/qwen3.5-397b-a17b", @@ -14793,13 +14882,13 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":null,"max":null}, input: ["text", "image"], cost: { - input: 0.28900000000000003, - output: 2.4, - cacheRead: 0, + input: 0.6, + output: 3.5999999999999996, + cacheRead: 0.12, cacheWrite: 0, }, contextWindow: 262144, - maxTokens: 131072, + maxTokens: 262144, } satisfies Model<"openai-completions">, "qwen/qwen3.6-35b-a3b": { id: "qwen/qwen3.6-35b-a3b", @@ -14967,7 +15056,7 @@ export const MODELS = { cacheRead: 0.049999999999999996, cacheWrite: 0, }, - contextWindow: 262144, + contextWindow: 1000000, maxTokens: 131072, } satisfies Model<"openai-completions">, "qwen/qwen3.8-max": { @@ -15569,9 +15658,45 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":"xhigh","max":null}, input: ["text"], cost: { - input: 0.49, - output: 1.54, - cacheRead: 0.091, + input: 0.966, + output: 3.036, + cacheRead: 0.1932, + cacheWrite: 0, + }, + contextWindow: 1048576, + maxTokens: 131072, + } satisfies Model<"openai-completions">, + "z-ai/glm-5.2:free": { + id: "z-ai/glm-5.2:free", + name: "Z.ai: GLM 5.2 (free)", + api: "openai-completions", + provider: "openrouter", + baseUrl: "https://openrouter.ai/api/v1", + reasoning: true, + thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":"xhigh","max":null}, + input: ["text"], + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + }, + contextWindow: 256000, + maxTokens: 256000, + } satisfies Model<"openai-completions">, + "z-ai/glm-5.3": { + id: "z-ai/glm-5.3", + name: "Z.ai: GLM 5.3", + api: "openai-completions", + provider: "openrouter", + baseUrl: "https://openrouter.ai/api/v1", + reasoning: true, + thinkingLevelMap: {"off":null,"minimal":null,"low":"low","medium":null,"high":"high","xhigh":null,"max":"max"}, + input: ["text"], + cost: { + input: 1.4, + output: 4.4, + cacheRead: 0.26, cacheWrite: 0, }, contextWindow: 1048576, @@ -15680,13 +15805,13 @@ export const MODELS = { thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":"max","max":null}, input: ["text"], cost: { - input: 0.078596, - output: 0.157192, - cacheRead: 0.015719200000000003, + input: 0.065, + output: 0.14, + cacheRead: 0.014, cacheWrite: 0, }, contextWindow: 1310720, - maxTokens: 384000, + maxTokens: 262144, } satisfies Model<"openai-completions">, "~google/gemini-flash-latest": { id: "~google/gemini-flash-latest", @@ -15796,6 +15921,24 @@ export const MODELS = { contextWindow: 500000, maxTokens: 4096, } satisfies Model<"openai-completions">, + "~z-ai/glm-latest": { + id: "~z-ai/glm-latest", + name: "Z.ai: GLM Latest", + api: "openai-completions", + provider: "openrouter", + baseUrl: "https://openrouter.ai/api/v1", + reasoning: true, + thinkingLevelMap: {"off":null,"minimal":null,"low":"low","medium":null,"high":"high","xhigh":null,"max":"max"}, + input: ["text"], + cost: { + input: 1.4, + output: 4.4, + cacheRead: 0.26, + cacheWrite: 0, + }, + contextWindow: 1048576, + maxTokens: 131072, + } satisfies Model<"openai-completions">, }, "prime-inference": { "anthropic/claude-fable-5": { @@ -16068,7 +16211,7 @@ export const MODELS = { cacheWrite: 0, }, contextWindow: 163840, - maxTokens: 65536, + maxTokens: 163840, } satisfies Model<"openai-completions">, "deepseek/deepseek-chat-v3.1": { id: "deepseek/deepseek-chat-v3.1", @@ -16203,7 +16346,7 @@ export const MODELS = { cacheWrite: 0, }, contextWindow: 1048576, - maxTokens: 384000, + maxTokens: 393216, featured: true, } satisfies Model<"openai-completions">, "google/gemini-2.5-flash": { @@ -16446,7 +16589,7 @@ export const MODELS = { cacheWrite: 0, }, contextWindow: 204800, - maxTokens: 196608, + maxTokens: 8192, } satisfies Model<"openai-completions">, "minimax/minimax-m2.7": { id: "minimax/minimax-m2.7", @@ -17450,7 +17593,7 @@ export const MODELS = { cacheWrite: 0, }, contextWindow: 262144, - maxTokens: 65536, + maxTokens: 262144, } satisfies Model<"openai-completions">, "qwen/qwen3.5-397b-a17b": { id: "qwen/qwen3.5-397b-a17b", @@ -17488,7 +17631,7 @@ export const MODELS = { cacheWrite: 0, }, contextWindow: 262144, - maxTokens: 131072, + maxTokens: 262144, } satisfies Model<"openai-completions">, "qwen/qwen3.6-35b-a3b": { id: "qwen/qwen3.6-35b-a3b", @@ -18234,13 +18377,13 @@ export const MODELS = { reasoning: true, input: ["text", "image"], cost: { - input: 0, - output: 0, - cacheRead: 0, + input: 0.55, + output: 3.3000000000000003, + cacheRead: 0.11, cacheWrite: 0, }, - contextWindow: 262144, - maxTokens: 262133, + contextWindow: 1000000, + maxTokens: 131072, } satisfies Model<"anthropic-messages">, "alibaba/qwen3.8-max": { id: "alibaba/qwen3.8-max", @@ -19911,9 +20054,9 @@ export const MODELS = { reasoning: true, input: ["text"], cost: { - input: 0, - output: 0, - cacheRead: 0, + input: 0.049999999999999996, + output: 0.19999999999999998, + cacheRead: 0.01, cacheWrite: 0, }, contextWindow: 262144, @@ -20004,6 +20147,23 @@ export const MODELS = { contextWindow: 1047576, maxTokens: 32768, } satisfies Model<"anthropic-messages">, + "openai/gpt-4.1-fast": { + id: "openai/gpt-4.1-fast", + name: "GPT-4.1 (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: false, + input: ["text", "image"], + cost: { + input: 3.5, + output: 14, + cacheRead: 0.875, + cacheWrite: 0, + }, + contextWindow: 1047576, + maxTokens: 32768, + } satisfies Model<"anthropic-messages">, "openai/gpt-4.1-mini": { id: "openai/gpt-4.1-mini", name: "GPT-4.1 mini", @@ -20021,6 +20181,23 @@ export const MODELS = { contextWindow: 1047576, maxTokens: 32768, } satisfies Model<"anthropic-messages">, + "openai/gpt-4.1-mini-fast": { + id: "openai/gpt-4.1-mini-fast", + name: "GPT-4.1 mini (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: false, + input: ["text", "image"], + cost: { + input: 0.7, + output: 2.8, + cacheRead: 0.175, + cacheWrite: 0, + }, + contextWindow: 1047576, + maxTokens: 32768, + } satisfies Model<"anthropic-messages">, "openai/gpt-4.1-nano": { id: "openai/gpt-4.1-nano", name: "GPT-4.1 nano", @@ -20038,6 +20215,23 @@ export const MODELS = { contextWindow: 1047576, maxTokens: 32768, } satisfies Model<"anthropic-messages">, + "openai/gpt-4.1-nano-fast": { + id: "openai/gpt-4.1-nano-fast", + name: "GPT-4.1 nano (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: false, + input: ["text", "image"], + cost: { + input: 0.19999999999999998, + output: 0.7999999999999999, + cacheRead: 0.049999999999999996, + cacheWrite: 0, + }, + contextWindow: 1047576, + maxTokens: 32768, + } satisfies Model<"anthropic-messages">, "openai/gpt-4o": { id: "openai/gpt-4o", name: "GPT-4o", @@ -20055,6 +20249,23 @@ export const MODELS = { contextWindow: 128000, maxTokens: 16384, } satisfies Model<"anthropic-messages">, + "openai/gpt-4o-fast": { + id: "openai/gpt-4o-fast", + name: "GPT-4o (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: false, + input: ["text", "image"], + cost: { + input: 4.25, + output: 17, + cacheRead: 2.125, + cacheWrite: 0, + }, + contextWindow: 128000, + maxTokens: 16384, + } satisfies Model<"anthropic-messages">, "openai/gpt-4o-mini": { id: "openai/gpt-4o-mini", name: "GPT-4o mini", @@ -20072,6 +20283,23 @@ export const MODELS = { contextWindow: 128000, maxTokens: 16384, } satisfies Model<"anthropic-messages">, + "openai/gpt-4o-mini-fast": { + id: "openai/gpt-4o-mini-fast", + name: "GPT-4o mini (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: false, + input: ["text", "image"], + cost: { + input: 0.25, + output: 1, + cacheRead: 0.125, + cacheWrite: 0, + }, + contextWindow: 128000, + maxTokens: 16384, + } satisfies Model<"anthropic-messages">, "openai/gpt-5": { id: "openai/gpt-5", name: "GPT-5", @@ -20106,6 +20334,23 @@ export const MODELS = { contextWindow: 400000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5-fast": { + id: "openai/gpt-5-fast", + name: "GPT-5 (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + input: ["text", "image"], + cost: { + input: 2.5, + output: 20, + cacheRead: 0.25, + cacheWrite: 0, + }, + contextWindow: 400000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5-mini": { id: "openai/gpt-5-mini", name: "GPT-5 mini", @@ -20123,6 +20368,23 @@ export const MODELS = { contextWindow: 400000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5-mini-fast": { + id: "openai/gpt-5-mini-fast", + name: "GPT-5 mini (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + input: ["text", "image"], + cost: { + input: 0.44999999999999996, + output: 3.5999999999999996, + cacheRead: 0.045, + cacheWrite: 0, + }, + contextWindow: 400000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5-nano": { id: "openai/gpt-5-nano", name: "GPT-5 nano", @@ -20225,6 +20487,23 @@ export const MODELS = { contextWindow: 400000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5.1-thinking-fast": { + id: "openai/gpt-5.1-thinking-fast", + name: "GPT 5.1 Thinking (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + input: ["text", "image"], + cost: { + input: 2.5, + output: 20, + cacheRead: 0.25, + cacheWrite: 0, + }, + contextWindow: 400000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5.2": { id: "openai/gpt-5.2", name: "GPT 5.2", @@ -20261,6 +20540,24 @@ export const MODELS = { contextWindow: 400000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5.2-fast": { + id: "openai/gpt-5.2-fast", + name: "GPT 5.2 (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + thinkingLevelMap: {"xhigh":"xhigh"}, + input: ["text", "image"], + cost: { + input: 3.5, + output: 28, + cacheRead: 0.35, + cacheWrite: 0, + }, + contextWindow: 400000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5.2-pro": { id: "openai/gpt-5.2-pro", name: "GPT 5.2 ", @@ -20297,6 +20594,24 @@ export const MODELS = { contextWindow: 400000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5.3-codex-fast": { + id: "openai/gpt-5.3-codex-fast", + name: "GPT 5.3 Codex (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + thinkingLevelMap: {"xhigh":"xhigh"}, + input: ["text", "image"], + cost: { + input: 3.5, + output: 28, + cacheRead: 0.35, + cacheWrite: 0, + }, + contextWindow: 400000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5.4": { id: "openai/gpt-5.4", name: "GPT 5.4", @@ -20315,6 +20630,24 @@ export const MODELS = { contextWindow: 1050000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5.4-fast": { + id: "openai/gpt-5.4-fast", + name: "GPT 5.4 (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + thinkingLevelMap: {"xhigh":"xhigh"}, + input: ["text", "image"], + cost: { + input: 5, + output: 30, + cacheRead: 0.5, + cacheWrite: 0, + }, + contextWindow: 1050000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5.4-mini": { id: "openai/gpt-5.4-mini", name: "GPT 5.4 Mini", @@ -20333,6 +20666,24 @@ export const MODELS = { contextWindow: 400000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5.4-mini-fast": { + id: "openai/gpt-5.4-mini-fast", + name: "GPT 5.4 Mini (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + thinkingLevelMap: {"xhigh":"xhigh"}, + input: ["text", "image"], + cost: { + input: 1.5, + output: 9, + cacheRead: 0.15, + cacheWrite: 0, + }, + contextWindow: 400000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5.4-nano": { id: "openai/gpt-5.4-nano", name: "GPT 5.4 Nano", @@ -20387,6 +20738,24 @@ export const MODELS = { contextWindow: 1000000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5.5-fast": { + id: "openai/gpt-5.5-fast", + name: "GPT 5.5 (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + thinkingLevelMap: {"xhigh":"xhigh"}, + input: ["text", "image"], + cost: { + input: 12.5, + output: 75, + cacheRead: 1.25, + cacheWrite: 0, + }, + contextWindow: 1000000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5.5-pro": { id: "openai/gpt-5.5-pro", name: "GPT 5.5 Pro", @@ -20423,6 +20792,24 @@ export const MODELS = { contextWindow: 1050000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5.6-luna-fast": { + id: "openai/gpt-5.6-luna-fast", + name: "GPT 5.6 Luna (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + thinkingLevelMap: {"xhigh":"xhigh","minimal":null,"max":"max"}, + input: ["text", "image"], + cost: { + input: 0.39999999999999997, + output: 2.4, + cacheRead: 0.04, + cacheWrite: 0.25, + }, + contextWindow: 1050000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5.6-sol": { id: "openai/gpt-5.6-sol", name: "GPT 5.6 Sol", @@ -20441,6 +20828,24 @@ export const MODELS = { contextWindow: 1050000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5.6-sol-fast": { + id: "openai/gpt-5.6-sol-fast", + name: "GPT 5.6 Sol (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + thinkingLevelMap: {"xhigh":"xhigh","minimal":null,"max":"max"}, + input: ["text", "image"], + cost: { + input: 5, + output: 30, + cacheRead: 0.5, + cacheWrite: 3.125, + }, + contextWindow: 1050000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-5.6-terra": { id: "openai/gpt-5.6-terra", name: "GPT 5.6 Terra", @@ -20459,6 +20864,24 @@ export const MODELS = { contextWindow: 1050000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "openai/gpt-5.6-terra-fast": { + id: "openai/gpt-5.6-terra-fast", + name: "GPT 5.6 Terra (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + thinkingLevelMap: {"xhigh":"xhigh","minimal":null,"max":"max"}, + input: ["text", "image"], + cost: { + input: 4, + output: 24, + cacheRead: 0.39999999999999997, + cacheWrite: 2.5, + }, + contextWindow: 1050000, + maxTokens: 128000, + } satisfies Model<"anthropic-messages">, "openai/gpt-oss-120b": { id: "openai/gpt-oss-120b", name: "GPT OSS 120B", @@ -20561,6 +20984,23 @@ export const MODELS = { contextWindow: 200000, maxTokens: 100000, } satisfies Model<"anthropic-messages">, + "openai/o3-fast": { + id: "openai/o3-fast", + name: "o3 (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + input: ["text", "image"], + cost: { + input: 3.5, + output: 14, + cacheRead: 0.875, + cacheWrite: 0, + }, + contextWindow: 200000, + maxTokens: 100000, + } satisfies Model<"anthropic-messages">, "openai/o3-mini": { id: "openai/o3-mini", name: "o3-mini", @@ -20612,6 +21052,23 @@ export const MODELS = { contextWindow: 200000, maxTokens: 100000, } satisfies Model<"anthropic-messages">, + "openai/o4-mini-fast": { + id: "openai/o4-mini-fast", + name: "o4-mini (Fast)", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + input: ["text", "image"], + cost: { + input: 2, + output: 8, + cacheRead: 0.5, + cacheWrite: 0, + }, + contextWindow: 200000, + maxTokens: 100000, + } satisfies Model<"anthropic-messages">, "poolside/laguna-s-2.1": { id: "poolside/laguna-s-2.1", name: "Laguna S 2.1", @@ -21071,40 +21528,6 @@ export const MODELS = { contextWindow: 200000, maxTokens: 96000, } satisfies Model<"anthropic-messages">, - "zai/glm-4.6v": { - id: "zai/glm-4.6v", - name: "GLM-4.6V", - api: "anthropic-messages", - provider: "vercel-ai-gateway", - baseUrl: "https://ai-gateway.vercel.sh", - reasoning: true, - input: ["text", "image"], - cost: { - input: 0.3, - output: 0.8999999999999999, - cacheRead: 0.049999999999999996, - cacheWrite: 0, - }, - contextWindow: 128000, - maxTokens: 24000, - } satisfies Model<"anthropic-messages">, - "zai/glm-4.6v-flash": { - id: "zai/glm-4.6v-flash", - name: "GLM-4.6V-Flash", - api: "anthropic-messages", - provider: "vercel-ai-gateway", - baseUrl: "https://ai-gateway.vercel.sh", - reasoning: true, - input: ["text", "image"], - cost: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - }, - contextWindow: 128000, - maxTokens: 24000, - } satisfies Model<"anthropic-messages">, "zai/glm-4.7": { id: "zai/glm-4.7", name: "GLM 4.7", @@ -21216,9 +21639,9 @@ export const MODELS = { reasoning: true, input: ["text"], cost: { - input: 1.1, - output: 3.851, - cacheRead: 0.275, + input: 0.7999999999999999, + output: 2.5500000000000003, + cacheRead: 0.16, cacheWrite: 0, }, contextWindow: 1000000, @@ -21241,6 +21664,23 @@ export const MODELS = { contextWindow: 1000000, maxTokens: 128000, } satisfies Model<"anthropic-messages">, + "zai/glm-5.3": { + id: "zai/glm-5.3", + name: "GLM 5.3", + api: "anthropic-messages", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh", + reasoning: true, + input: ["text"], + cost: { + input: 1.4, + output: 4.4, + cacheRead: 0.26, + cacheWrite: 0, + }, + contextWindow: 1000000, + maxTokens: 12800, + } satisfies Model<"anthropic-messages">, "zai/glm-5v-turbo": { id: "zai/glm-5v-turbo", name: "GLM 5V Turbo", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 82246c2a5..aab2e32ad 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,7 @@ ## [Unreleased] +- Fixed ACP rejecting an immediate follow-up prompt when injected work restarted the session; follow-ups now queue behind in-flight work, and cancellation drops queued follow-ups before they start. - Added correlated ACP terminal-quiescence metadata, resident session settlement, and fail-closed daemon input fencing; prevented recovery state from persisting runtime credentials or model configuration. - Fixed explicit RLM child deletion leaving hidden unsettled work after runtime teardown, including reporting cleanup failures and notifying the parent when deletion completes. diff --git a/packages/coding-agent/src/core/agent-session.ts b/packages/coding-agent/src/core/agent-session.ts index f3d7408ec..42aa249f2 100644 --- a/packages/coding-agent/src/core/agent-session.ts +++ b/packages/coding-agent/src/core/agent-session.ts @@ -4387,12 +4387,32 @@ export class AgentSession { const outcome = this._agentMessageOutcome(agentMessageId); outcome.completion = createAgentMessageDeferred(); const completion = outcome.completion.promise; + const signal = options?.signal; + let cancelQueuedPrompt: (() => void) | undefined; try { await this.promptUntilAccepted(text, { ...options, agentMessageId }); + if (signal) { + cancelQueuedPrompt = () => { + const error = new Error("Prompt was cancelled before it started."); + const cancelled = this._cancelSessionActions( + (action) => action.agentMessageId === agentMessageId && action.payload.kind === "turn", + error, + ); + if (cancelled.length > 0) { + this._settleAgentMessage(agentMessageId, "completion", error); + } + }; + signal.addEventListener("abort", cancelQueuedPrompt, { once: true }); + if (signal.aborted) cancelQueuedPrompt(); + } await completion; } catch (error) { this._settleAgentMessage(agentMessageId, "completion", this._asError(error)); throw error; + } finally { + if (signal && cancelQueuedPrompt) { + signal.removeEventListener("abort", cancelQueuedPrompt); + } } } diff --git a/packages/coding-agent/src/modes/acp/acp-mode.ts b/packages/coding-agent/src/modes/acp/acp-mode.ts index 3c1b76820..988dac322 100644 --- a/packages/coding-agent/src/modes/acp/acp-mode.ts +++ b/packages/coding-agent/src/modes/acp/acp-mode.ts @@ -799,7 +799,10 @@ export async function runAcpModeWithConnection( await entry.pendingTerminal?.task; if (session !== entry) throw new Error(`Unknown ACP session: ${params.sessionId}`); if (sessionCloseInFlight) throw new Error(`ACP session is closing: ${params.sessionId}`); - if (entry.cancelling) throw new Error(`ACP session is cancelling: ${params.sessionId}`); + // This prompt was admitted before the cancellation started; it is dropped + // by the cancel rather than malformed, so report the protocol stop reason + // instead of a request error. + if (entry.cancelling) return { stopReason: "cancelled" satisfies AcpStopReason }; if (entry.stopFailure) throw new Error(`ACP session stop failed: ${entry.stopFailure}`); if (entry.pendingTerminal?.failure) { throw new Error(`ACP lifecycle reconciliation failed: ${entry.pendingTerminal.failure}`); @@ -825,7 +828,16 @@ export async function runAcpModeWithConnection( await entry.producer.drain(); return { stopReason: "cancelled" satisfies AcpStopReason }; } - await connection.promptAndWait(text, images.length > 0 ? { images } : undefined); + // A follow-up prompt can arrive while injected work (subagent replies, + // heartbeats) keeps the resident session busy. ACP has no native queue + // field, so queue the host turn behind that work with follow-up + // semantics instead of rejecting it as "Agent is already processing". + await connection.promptAndWait(text, { + ...(images.length > 0 ? { images } : {}), + streamingBehavior: "followUp", + queueIfBusy: true, + signal: abort.signal, + }); if (abort.signal.aborted) { await entry.producer.drain(); return { stopReason: "cancelled" satisfies AcpStopReason }; diff --git a/packages/coding-agent/src/modes/agent-connection/daemon-agent-connection.ts b/packages/coding-agent/src/modes/agent-connection/daemon-agent-connection.ts index e9d9c8354..b0ef8ffc4 100644 --- a/packages/coding-agent/src/modes/agent-connection/daemon-agent-connection.ts +++ b/packages/coding-agent/src/modes/agent-connection/daemon-agent-connection.ts @@ -931,6 +931,7 @@ export class DaemonAgentConnection implements AgentConnection { type: "cancel_prompt_admission", activeSessionId: this.activeSessionId, admissionId, + ...(this.client.supportsServerCapability("owned_prompt_cancellation") ? { cancelOwned: true } : {}), }); status = result.status; } catch { diff --git a/packages/coding-agent/src/modes/daemon/daemon-mode.ts b/packages/coding-agent/src/modes/daemon/daemon-mode.ts index 9e84c54a5..01350cf38 100644 --- a/packages/coding-agent/src/modes/daemon/daemon-mode.ts +++ b/packages/coding-agent/src/modes/daemon/daemon-mode.ts @@ -4059,6 +4059,7 @@ export class AgentDaemon { }); } if (admission.status === "owned") { + if (command.cancelOwned) admission.controller?.abort(); return success(command.id, command.type, { status: "owned" as const, }); diff --git a/packages/coding-agent/src/modes/daemon/daemon-protocol.ts b/packages/coding-agent/src/modes/daemon/daemon-protocol.ts index e0f815392..6f77aeb04 100644 --- a/packages/coding-agent/src/modes/daemon/daemon-protocol.ts +++ b/packages/coding-agent/src/modes/daemon/daemon-protocol.ts @@ -62,8 +62,10 @@ export const DAEMON_COMMAND_ENVELOPE_MIN_PROTOCOL_VERSION = 7; // Revision 16 adds the "stopping" workerState and stops reporting disconnected workers as "ready". // Revision 17 gates authoritative child rosters and transient owned-session recovery context. // Revision 18 adds the opt-in RLM quiescence barrier to headless completion. -export const DAEMON_SCHEMA_REVISION = 19; -export const DAEMON_SCHEMA_ID = "protocol-7-schema-19-29b4f87f83e7"; +// Revision 19 adds daemon-held session input pauses. +// Revision 20 lets cancellation target a prompt the session owns but has not started. +export const DAEMON_SCHEMA_REVISION = 20; +export const DAEMON_SCHEMA_ID = "protocol-7-schema-20-ed994cc39507"; export type DaemonProtocolName = typeof DAEMON_PROTOCOL_NAME; export type DaemonProtocolVersion = number; @@ -106,7 +108,8 @@ export type DaemonServerCapability = | "authoritative_child_roster" | "owned_session_recovery_context" | "rlm_quiescence_barrier" - | "session_input_pause"; + | "session_input_pause" + | "owned_prompt_cancellation"; export type DaemonReplayStatus = "complete" | "partial" | "unavailable"; @@ -144,6 +147,7 @@ export const DAEMON_DEFAULT_SERVER_CAPABILITIES: readonly DaemonServerCapability "transient_bash", "session_input_admission", "prompt_admission_cancellation", + "owned_prompt_cancellation", "queue_message_mutation", "authoritative_child_roster", "owned_session_recovery_context", @@ -426,6 +430,8 @@ export type DaemonCommand = type: "cancel_prompt_admission"; activeSessionId: string; admissionId: string; + /** Cancel session-owned work too when it has not started delivery. */ + cancelOwned?: boolean; } | { id?: string; @@ -659,6 +665,11 @@ const PROMPT_ADMISSION_CANCELLATION_COMMAND = { minSchemaRevision: 8, capability: "prompt_admission_cancellation", } as const; +const OWNED_PROMPT_CANCELLATION_COMMAND = { + minProtocol: 7, + minSchemaRevision: 20, + capability: "owned_prompt_cancellation", +} as const; const CLIENT_OWNED_DAEMON_COMMAND = { minProtocol: 7, capability: "client_owned_sessions", @@ -809,6 +820,9 @@ export function getDaemonCommandCompatibilities(command: DaemonCommand): readonl if (command.type === "wait_for_headless_completion" && command.waitForRlmQuiescence === true) { requirements.push(RLM_QUIESCENCE_BARRIER_COMMAND); } + if (command.type === "cancel_prompt_admission" && command.cancelOwned === true) { + requirements.push(OWNED_PROMPT_CANCELLATION_COMMAND); + } return [...requirements, DAEMON_COMMAND_COMPATIBILITY[command.type]]; } diff --git a/packages/coding-agent/test/agent-connection-daemon.test.ts b/packages/coding-agent/test/agent-connection-daemon.test.ts index 256da26b4..3426fc6dd 100644 --- a/packages/coding-agent/test/agent-connection-daemon.test.ts +++ b/packages/coding-agent/test/agent-connection-daemon.test.ts @@ -874,11 +874,36 @@ describe("DaemonAgentConnection", () => { await vi.waitFor(() => expect(fakeClient.requests.map((request) => request.type)).toContain("cancel_prompt_admission"), ); + expect(fakeClient.requests.find((request) => request.type === "cancel_prompt_admission")).not.toHaveProperty( + "cancelOwned", + ); releasePrompt(); await expect(prompt).resolves.toBeUndefined(); }); + it("requests owned prompt cancellation when the daemon advertises it", async () => { + const fakeClient = new FakeDaemonClient(); + fakeClient.serverCapabilities.add("owned_prompt_cancellation"); + let releasePrompt = () => {}; + fakeClient.promptGate = new Promise((resolve) => { + releasePrompt = resolve; + }); + const connection = new DaemonAgentConnection(asDaemonClient(fakeClient), "active-1"); + const abort = new AbortController(); + + const prompt = connection.prompt("startup", { signal: abort.signal }); + abort.abort(); + await vi.waitFor(() => + expect(fakeClient.requests.map((request) => request.type)).toContain("cancel_prompt_admission"), + ); + expect(fakeClient.requests.find((request) => request.type === "cancel_prompt_admission")).toMatchObject({ + cancelOwned: true, + }); + releasePrompt(); + await expect(prompt).resolves.toBeUndefined(); + }); + it("preserves a definitive prompt rejection when cancellation reports owned", async () => { const fakeClient = new FakeDaemonClient(); let releasePrompt = () => {}; diff --git a/packages/coding-agent/test/daemon-mode.test.ts b/packages/coding-agent/test/daemon-mode.test.ts index 000443c74..58facb9cf 100644 --- a/packages/coding-agent/test/daemon-mode.test.ts +++ b/packages/coding-agent/test/daemon-mode.test.ts @@ -9053,7 +9053,7 @@ describe("daemon mode helpers", () => { }, ); - it("cancels only pre-ownership prompt admission and cleans up its controller", async () => { + it("capability-gates cancellation after prompt ownership", async () => { const daemon = new AgentDaemon("/tmp/prime-agent-test.sock", { defaultSessionConfig: { agentDir: "/tmp/prime-agent-test-agent", cwd: "/tmp" }, createRuntime: async () => { @@ -9112,7 +9112,7 @@ describe("daemon mode helpers", () => { ).resolves.toMatchObject({ success: true, data: { status: "cancelled" } }); await vi.waitFor(() => expect(internals.promptAdmissions.size).toBe(0)); - // Once ownership commits the same cancellation is a no-op. + // Old clients retain the pre-ownership-only behavior. internals.parseCommandAndRegisterPromptAdmission( client, JSON.stringify({ @@ -9139,6 +9139,19 @@ describe("daemon mode helpers", () => { admissionId: "admission-2", }), ).resolves.toMatchObject({ success: true, data: { status: "owned" } }); + expect(promptOptions?.signal?.aborted).toBe(false); + + // New clients request the capability-gated session-owned cancellation. + await expect( + internals.handleCommand(client, { + id: "cancel-3", + type: "cancel_prompt_admission", + activeSessionId: state.activeSessionId, + admissionId: "admission-2", + cancelOwned: true, + }), + ).resolves.toMatchObject({ success: true, data: { status: "owned" } }); + expect(promptOptions?.signal?.aborted).toBe(true); rejectPrompt?.(new Error("test cleanup")); }); diff --git a/packages/coding-agent/test/daemon-protocol.test.ts b/packages/coding-agent/test/daemon-protocol.test.ts index 809d0c50f..615008c90 100644 --- a/packages/coding-agent/test/daemon-protocol.test.ts +++ b/packages/coding-agent/test/daemon-protocol.test.ts @@ -234,6 +234,16 @@ describe("daemon protocol helpers", () => { expect(DAEMON_DEFAULT_SERVER_CAPABILITIES).toContain("prompt_admission_cancellation"); }); + it("capability-gates cancellation after prompt ownership", () => { + const legacy = { type: "cancel_prompt_admission", activeSessionId: "active-1", admissionId: "a-1" } as const; + expect(getDaemonCommandCompatibilities(legacy)).toEqual([DAEMON_COMMAND_COMPATIBILITY.cancel_prompt_admission]); + expect(getDaemonCommandCompatibilities({ ...legacy, cancelOwned: true })).toEqual([ + { minProtocol: 7, minSchemaRevision: 20, capability: "owned_prompt_cancellation" }, + DAEMON_COMMAND_COMPATIBILITY.cancel_prompt_admission, + ]); + expect(DAEMON_DEFAULT_SERVER_CAPABILITIES).toContain("owned_prompt_cancellation"); + }); + it("gates honest worker-state reporting at its introducing schema revision", () => { // Revision 16 adds the "stopping" workerState and stops reporting // disconnected workers as "ready". The field is optional and old clients diff --git a/packages/coding-agent/test/suite/acp-mode.test.ts b/packages/coding-agent/test/suite/acp-mode.test.ts index c62520cbe..aa43cfda1 100644 --- a/packages/coding-agent/test/suite/acp-mode.test.ts +++ b/packages/coding-agent/test/suite/acp-mode.test.ts @@ -1,6 +1,7 @@ import * as acp from "@agentclientprotocol/sdk"; -import { fauxAssistantMessage } from "@earendil-works/pi-ai"; +import { type AssistantMessage, fauxAssistantMessage } from "@earendil-works/pi-ai"; import { describe, expect, it, vi } from "vitest"; +import type { AgentSession } from "../../src/core/agent-session.js"; import type { AgentSessionRuntime } from "../../src/core/agent-session-runtime.js"; import { PRIME_AGENT_META_NAMESPACE } from "../../src/modes/acp/acp-meta.js"; import { runAcpModeWithConnection } from "../../src/modes/acp/index.js"; @@ -29,6 +30,29 @@ interface ClientHarness { close: () => void; } +function injectWorkAfterHeadlessCompletion( + connection: InProcessAgentConnection, + session: AgentSession, + text: string, +): () => boolean { + const waitForHeadlessCompletion = connection.waitForHeadlessCompletion.bind(connection); + let injected = false; + connection.waitForHeadlessCompletion = async () => { + const status = await waitForHeadlessCompletion(); + if (!injected) { + injected = true; + void session.prompt(text); + const deadline = Date.now() + 5_000; + while (!session.isStreaming && Date.now() < deadline) { + await new Promise((resolve) => setTimeout(resolve, 1)); + } + expect(session.isStreaming).toBe(true); + } + return status; + }; + return () => injected; +} + interface ClientHarnessOptions { beforeAcpUpdatePublish?: (update: Record) => void | Promise; } @@ -208,6 +232,171 @@ describe("ACP mode end to end", () => { harness.cleanup(); }, 30_000); + it("queues a follow-up prompt behind injected work instead of rejecting it", async () => { + const harness = await createHarness(); + let releaseInjected!: () => void; + const injectedHeld = new Promise((resolve) => { + releaseInjected = () => resolve(fauxAssistantMessage("injected work done")); + }); + harness.setResponses([ + fauxAssistantMessage("turn one done"), + () => injectedHeld, + fauxAssistantMessage("turn two done"), + ]); + const connection = new InProcessAgentConnection(runtimeHostFor(harness.session)); + const injected = injectWorkAfterHeadlessCompletion(connection, harness.session, "injected work"); + const { client, updates } = connectAcpClient(connection); + await client.request("initialize", { protocolVersion: acp.PROTOCOL_VERSION, clientCapabilities: {} }); + const session = await client.request("session/new", { cwd: harness.tempDir, mcpServers: [] }); + + const first = await client.request("session/prompt", { + sessionId: session.sessionId, + prompt: [{ type: "text", text: "First turn" }], + }); + expect(first.stopReason).toBe("end_turn"); + expect(injected()).toBe(true); + expect(harness.session.isStreaming).toBe(true); + + let secondSettled = false; + const second = client + .request("session/prompt", { + sessionId: session.sessionId, + prompt: [{ type: "text", text: "Second turn" }], + }) + .finally(() => { + secondSettled = true; + }); + await new Promise((resolve) => setTimeout(resolve, 50)); + expect(secondSettled).toBe(false); + releaseInjected(); + await expect(second).resolves.toMatchObject({ stopReason: "end_turn" }); + + const text = updates + .filter((update) => update.update?.sessionUpdate === "agent_message_chunk") + .map((update) => update.update.content.text) + .join(""); + expect(text).toContain("turn two done"); + harness.cleanup(); + }, 5_000); + + it("reports the queued turn's stop reason from a fresh autonomous status", async () => { + const harness = await createHarness({ + autonomous: { + enabled: true, + maxTurns: 2, + maxContinuations: 3, + maxTokens: 80_000, + gates: { commands: ["true"], maxRetries: 3 }, + }, + }); + let releaseInjected!: () => void; + const injectedHeld = new Promise((resolve) => { + releaseInjected = () => resolve(fauxAssistantMessage("injected work done")); + }); + harness.setResponses([ + fauxAssistantMessage("turn one done"), + () => injectedHeld, + fauxAssistantMessage("turn two done"), + ]); + const connection = new InProcessAgentConnection(runtimeHostFor(harness.session)); + injectWorkAfterHeadlessCompletion(connection, harness.session, "injected work"); + const { client } = connectAcpClient(connection); + await client.request("initialize", { protocolVersion: acp.PROTOCOL_VERSION, clientCapabilities: {} }); + const session = await client.request("session/new", { cwd: harness.tempDir, mcpServers: [] }); + + const first = await client.request("session/prompt", { + sessionId: session.sessionId, + prompt: [{ type: "text", text: "First turn" }], + }); + expect(first.stopReason).toBe("end_turn"); + expect(harness.session.getAutonomousStatus().turnsUsed).toBe(1); + + const second = client.request("session/prompt", { + sessionId: session.sessionId, + prompt: [{ type: "text", text: "Second turn" }], + }); + await new Promise((resolve) => setTimeout(resolve, 50)); + releaseInjected(); + await expect(second).resolves.toMatchObject({ stopReason: "max_turn_requests" }); + expect(harness.session.getAutonomousStatus().turnsUsed).toBeGreaterThanOrEqual( + harness.session.getAutonomousStatus().limits.maxTurns, + ); + harness.cleanup(); + }, 5_000); + + it("does not hold the prompt response open for detached work", async () => { + const harness = await createHarness(); + let releaseInjected!: () => void; + const injectedHeld = new Promise((resolve) => { + releaseInjected = () => resolve(fauxAssistantMessage("injected work done")); + }); + harness.setResponses([fauxAssistantMessage("turn one done"), () => injectedHeld]); + const connection = new InProcessAgentConnection(runtimeHostFor(harness.session)); + const injected = injectWorkAfterHeadlessCompletion(connection, harness.session, "injected work"); + const { client } = connectAcpClient(connection); + await client.request("initialize", { protocolVersion: acp.PROTOCOL_VERSION, clientCapabilities: {} }); + const session = await client.request("session/new", { cwd: harness.tempDir, mcpServers: [] }); + + let settled = false; + const prompt = client + .request("session/prompt", { + sessionId: session.sessionId, + prompt: [{ type: "text", text: "First turn" }], + }) + .finally(() => { + settled = true; + }); + const deadline = Date.now() + 5_000; + while (!injected() && Date.now() < deadline) { + await new Promise((resolve) => setTimeout(resolve, 1)); + } + expect(injected()).toBe(true); + await new Promise((resolve) => setTimeout(resolve, 100)); + expect(settled).toBe(true); + expect(harness.session.isStreaming).toBe(true); + releaseInjected(); + await expect(prompt).resolves.toMatchObject({ stopReason: "end_turn" }); + harness.cleanup(); + }, 5_000); + + it("cancels a prompt that is still queued behind busy work", async () => { + const harness = await createHarness(); + let releaseInjected!: () => void; + const injectedHeld = new Promise((resolve) => { + releaseInjected = () => resolve(fauxAssistantMessage("injected work done")); + }); + harness.setResponses([ + fauxAssistantMessage("turn one done"), + () => injectedHeld, + fauxAssistantMessage("queued turn done"), + ]); + const connection = new InProcessAgentConnection(runtimeHostFor(harness.session)); + injectWorkAfterHeadlessCompletion(connection, harness.session, "injected work"); + const { client } = connectAcpClient(connection); + await client.request("initialize", { protocolVersion: acp.PROTOCOL_VERSION, clientCapabilities: {} }); + const session = await client.request("session/new", { cwd: harness.tempDir, mcpServers: [] }); + await client.request("session/prompt", { + sessionId: session.sessionId, + prompt: [{ type: "text", text: "First turn" }], + }); + + const queued = client.request("session/prompt", { + sessionId: session.sessionId, + prompt: [{ type: "text", text: "Second turn" }], + }); + await new Promise((resolve) => setTimeout(resolve, 50)); + void client.notify("session/cancel", { sessionId: session.sessionId }); + await expect(queued).resolves.toMatchObject({ stopReason: "cancelled" }); + releaseInjected(); + await new Promise((resolve) => setTimeout(resolve, 300)); + const assistantText = harness.session.messages + .filter((message) => message.role === "assistant") + .map((message) => JSON.stringify(message.content)) + .join("|"); + expect(assistantText).not.toContain("queued turn done"); + harness.cleanup(); + }, 5_000); + it("emits score-safe quiescence metadata with outstanding work and budget", async () => { const harness = await createHarness(); harness.setResponses([fauxAssistantMessage("done")]);