Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,13 @@ for reviewers.

Add user-visible changes here before running `bun run release:prepare <version>`.

### Removed

- **The "Extended Context (1M)" toggle is gone.** It opted Anthropic requests
into the 1M-token context beta, which bills over-200K requests at higher
rates and fails outright on lower-tier API keys. Models now always run with
the standard 200K context window.

## [0.13.0] - 2026-09-13

### Added
Expand Down
15 changes: 0 additions & 15 deletions apps/electron/src/renderer/pages/settings/AiSettingsPage.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -639,7 +639,6 @@ export default function AiSettingsPage() {
// Default settings state (app-level)
const [defaultThinking, setDefaultThinking] = useState<ThinkingLevel>(DEFAULT_THINKING_LEVEL)
const [extendedPromptCache, setExtendedPromptCache] = useState(false)
const [enable1MContext, setEnable1MContext] = useState(false)
const [rtkEnabled, setRtkEnabled] = useState(false)
const [rtkStatus, setRtkStatus] = useState<{ installed: boolean; path: string | null; version: string | null } | null>(null)
const [rtkRechecking, setRtkRechecking] = useState(false)
Expand Down Expand Up @@ -673,9 +672,6 @@ export default function AiSettingsPage() {
const extendedCache = await window.electronAPI.getExtendedPromptCache()
setExtendedPromptCache(extendedCache)

const enable1M = await window.electronAPI.getEnable1MContext()
setEnable1MContext(enable1M)

const rtkOn = await window.electronAPI.getRtkEnabled()
setRtkEnabled(rtkOn)

Expand Down Expand Up @@ -979,11 +975,6 @@ export default function AiSettingsPage() {
await window.electronAPI?.setExtendedPromptCache(enabled)
}, [])

const handleEnable1MContextChange = useCallback(async (enabled: boolean) => {
setEnable1MContext(enabled)
await window.electronAPI?.setEnable1MContext(enabled)
}, [])

const handleRtkToggle = useCallback(async (enabled: boolean) => {
setRtkEnabled(enabled)
await window.electronAPI?.setRtkEnabled(enabled)
Expand Down Expand Up @@ -1125,12 +1116,6 @@ export default function AiSettingsPage() {
{/* Performance */}
<SettingsSection title={t("settings.ai.performance")} description={t("settings.ai.performanceDesc")}>
<SettingsCard>
<SettingsToggle
label={t("settings.ai.extendedContext")}
description={t("settings.ai.extendedContextDesc")}
checked={enable1MContext}
onCheckedChange={handleEnable1MContextChange}
/>
<SettingsToggle
label={t("settings.ai.extendedPromptCache")}
description={t("settings.ai.extendedPromptCacheDesc")}
Expand Down
4 changes: 1 addition & 3 deletions apps/electron/src/shared/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -544,11 +544,9 @@ export interface ElectronAPI {
getRichToolDescriptions(): Promise<boolean>
setRichToolDescriptions(enabled: boolean): Promise<void>

// Prompt caching & context
// Prompt caching
getExtendedPromptCache(): Promise<boolean>
setExtendedPromptCache(enabled: boolean): Promise<void>
getEnable1MContext(): Promise<boolean>
setEnable1MContext(enabled: boolean): Promise<void>

// RTK token optimization
getRtkEnabled(): Promise<boolean>
Expand Down
2 changes: 0 additions & 2 deletions apps/electron/src/transport/channel-map.ts
Original file line number Diff line number Diff line change
Expand Up @@ -203,8 +203,6 @@ export const CHANNEL_MAP = {
setRichToolDescriptions: invoke(RPC_CHANNELS.appearance.SET_RICH_TOOL_DESCRIPTIONS),
getExtendedPromptCache: invoke(RPC_CHANNELS.caching.GET_EXTENDED_PROMPT_CACHE),
setExtendedPromptCache: invoke(RPC_CHANNELS.caching.SET_EXTENDED_PROMPT_CACHE),
getEnable1MContext: invoke(RPC_CHANNELS.caching.GET_ENABLE_1M_CONTEXT),
setEnable1MContext: invoke(RPC_CHANNELS.caching.SET_ENABLE_1M_CONTEXT),
getRtkEnabled: invoke(RPC_CHANNELS.rtk.GET_ENABLED),
setRtkEnabled: invoke(RPC_CHANNELS.rtk.SET_ENABLED),
getRtkStatus: invoke(RPC_CHANNELS.rtk.GET_STATUS),
Expand Down
14 changes: 0 additions & 14 deletions packages/server-core/src/handlers/rpc/settings.ts
Original file line number Diff line number Diff line change
Expand Up @@ -31,8 +31,6 @@ export const HANDLED_CHANNELS = [
RPC_CHANNELS.appearance.SET_RICH_TOOL_DESCRIPTIONS,
RPC_CHANNELS.caching.GET_EXTENDED_PROMPT_CACHE,
RPC_CHANNELS.caching.SET_EXTENDED_PROMPT_CACHE,
RPC_CHANNELS.caching.GET_ENABLE_1M_CONTEXT,
RPC_CHANNELS.caching.SET_ENABLE_1M_CONTEXT,
RPC_CHANNELS.sessions.GET_MODEL,
RPC_CHANNELS.sessions.SET_MODEL,
RPC_CHANNELS.settings.GET_DEFAULT_THINKING_LEVEL,
Expand Down Expand Up @@ -296,18 +294,6 @@ export function registerSettingsHandlers(server: RpcServer, deps: HandlerDeps):
setExtendedPromptCache(enabled)
})

// Get 1M context window setting
server.handle(RPC_CHANNELS.caching.GET_ENABLE_1M_CONTEXT, async () => {
const { getEnable1MContext } = await import('@bitlab/shared/config/storage')
return getEnable1MContext()
})

// Set 1M context window setting
server.handle(RPC_CHANNELS.caching.SET_ENABLE_1M_CONTEXT, async (_ctx, enabled: boolean) => {
const { setEnable1MContext } = await import('@bitlab/shared/config/storage')
setEnable1MContext(enabled)
})

// ============================================================
// RTK Token-Optimization Settings
// ============================================================
Expand Down
3 changes: 0 additions & 3 deletions packages/shared/src/agent/backend/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -203,9 +203,6 @@ export interface CoreBackendConfig {
*/
onImageResize?: (filePath: string, maxSizeBytes: number) => Promise<string | null>;

/** Enable 1M context window for current Opus models. Default: true. Set false to use 200K and conserve usage limits. */
enable1MContext?: boolean;

}

// ============================================================
Expand Down
25 changes: 1 addition & 24 deletions packages/shared/src/config/storage.ts
Original file line number Diff line number Diff line change
Expand Up @@ -93,9 +93,8 @@ export interface StoredConfig {
/** Whether skills may pre-approve tools for their turn. Absent means enabled. */
skillToolApproval?: boolean;
mcpSettings?: BitlabMcpSettings;
// Prompt caching & context
// Prompt caching
extendedPromptCache?: boolean; // Use 1h prompt cache TTL instead of 5m (default: false)
enable1MContext?: boolean; // Enable 1M context window for supported models (default: false — opt-in; requires Anthropic Tier 4+)
// Token optimization
rtkEnabled?: boolean; // Route Bash commands through rtk to compress tool output (default: false). https://github.com/rtk-ai/rtk
// Network proxy
Expand Down Expand Up @@ -588,28 +587,6 @@ export function setMcpSettings(settings: BitlabMcpSettings): void {
saveConfig(config);
}

/**
* Get whether 1M context window is enabled.
* When disabled, models use 200K context and the interceptor strips the context-1m beta header.
* Defaults to false — the 1M beta requires Anthropic Tier 4+, and enabling it by default
* causes 400 "Invalid Request" for lower-tier API keys on large contexts (issue #567).
* Users opt in via AI Settings → Performance → Extended Context (1M).
*/
export function getEnable1MContext(): boolean {
const config = loadStoredConfig();
return config?.enable1MContext === true;
}

/**
* Set whether 1M context window is enabled.
*/
export function setEnable1MContext(enabled: boolean): void {
const config = loadStoredConfig();
if (!config) return;
config.enable1MContext = enabled;
saveConfig(config);
}

/**
* Get whether rtk Bash-output compression is enabled.
* When enabled, the PreToolUse pipeline rewrites Bash commands to their `rtk` equivalents
Expand Down
2 changes: 0 additions & 2 deletions packages/shared/src/i18n/locales/en.json
Original file line number Diff line number Diff line change
Expand Up @@ -584,8 +584,6 @@
"settings.ai.defaultSectionDesc": "Settings for new chats when no workspace override is set.",
"settings.ai.description": "Model, thinking, connections",
"settings.ai.enterConnectionName": "Enter connection name...",
"settings.ai.extendedContext": "Extended Context (1M)",
"settings.ai.extendedContextDesc": "Use 1M token context window for current Opus models. Disable to use 200K and conserve usage limits.",
"settings.ai.extendedPromptCache": "Extended prompt cache (1 hour)",
"settings.ai.extendedPromptCacheDesc": "Cache prompts for 1 hour instead of 5 minutes. Only applies to Claude models via Anthropic API. Reduces cost for long sessions but increases cache write cost.",
"settings.ai.inheritFromApp": "Inherit from app settings",
Expand Down
2 changes: 0 additions & 2 deletions packages/shared/src/i18n/locales/zh-Hans.json
Original file line number Diff line number Diff line change
Expand Up @@ -584,8 +584,6 @@
"settings.ai.defaultSectionDesc": "未设置 Workspace 覆盖时新聊天的设置。",
"settings.ai.description": "模型、思考、连接",
"settings.ai.enterConnectionName": "输入连接名称...",
"settings.ai.extendedContext": "扩展上下文 (1M)",
"settings.ai.extendedContextDesc": "为 Opus 4.7 使用 1M token 上下文窗口。禁用则使用 200K 以节省用量配额。",
"settings.ai.extendedPromptCache": "扩展提示缓存 (1 小时)",
"settings.ai.extendedPromptCacheDesc": "将提示缓存从 5 分钟延长到 1 小时。仅适用于通过 Anthropic API 的 Claude 模型。可降低长会话成本,但会增加缓存写入成本。",
"settings.ai.inheritFromApp": "继承应用设置",
Expand Down
12 changes: 0 additions & 12 deletions packages/shared/src/interceptor-common.ts
Original file line number Diff line number Diff line change
Expand Up @@ -138,18 +138,6 @@ export function isExtendedPromptCacheEnabled(): boolean {
return config?.extendedPromptCache === true;
}

/**
* Check if 1M context window is enabled.
* When disabled, the interceptor strips the context-1m beta header.
* Defaults to false — the 1M beta requires Anthropic Tier 4+, so it's opt-in
* to avoid 400 "Invalid Request" on lower-tier API keys (issue #567).
* Must stay in sync with getEnable1MContext() in config/storage.ts.
*/
export function is1MContextEnabled(): boolean {
const config = getInterceptorConfig();
return config?.enable1MContext === true;
}

// ============================================================================
// LAST API ERROR
// ============================================================================
Expand Down
2 changes: 0 additions & 2 deletions packages/shared/src/protocol/channels.ts
Original file line number Diff line number Diff line change
Expand Up @@ -284,8 +284,6 @@ export const RPC_CHANNELS = {
caching: {
GET_EXTENDED_PROMPT_CACHE: 'caching:getExtendedPromptCache',
SET_EXTENDED_PROMPT_CACHE: 'caching:setExtendedPromptCache',
GET_ENABLE_1M_CONTEXT: 'caching:getEnable1MContext',
SET_ENABLE_1M_CONTEXT: 'caching:setEnable1MContext',
},
rtk: {
GET_ENABLED: 'rtk:getEnabled',
Expand Down
19 changes: 8 additions & 11 deletions packages/shared/src/unified-network-interceptor.ts
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,6 @@ import {
debugLog,
isRichToolDescriptionsEnabled,
isExtendedPromptCacheEnabled,
is1MContextEnabled,
setStoredError,
toolMetadataStore,
displayNameSchema,
Expand Down Expand Up @@ -790,16 +789,14 @@ const anthropicAdapter: ApiAdapter = {
sanitizeEmptyTextCacheControl(body);
upgradePromptCacheTtl(body);

// Strip SDK-injected 1M context beta when setting disables it.
// The SDK adds this header automatically for Opus/Sonnet 4.6 models,
// but the user may want 200K context to conserve usage limits.
if (!is1MContextEnabled()) {
debugLog('[Anthropic] Stripping context-1m beta header (enable1MContext=false)');
init = {
...init,
headers: stripBetaHeader(init?.headers as HeadersInitType | undefined, 'context-1m-2025-08-07'),
};
}
// Always strip the SDK-injected 1M context beta. The SDK adds this header
// automatically for Opus/Sonnet 4.6 models, but it 400s on lower-tier
// Anthropic API keys (issue #567) and burns usage limits faster, so all
// requests stay on the 200K context window.
init = {
...init,
headers: stripBetaHeader(init?.headers as HeadersInitType | undefined, 'context-1m-2025-08-07'),
};

const fastMode = shouldEnableFastMode(body.model);
if (fastMode) {
Expand Down
Loading