diff --git a/docs/advanced-setup.md b/docs/advanced-setup.md
index 534b7078f..59a426e35 100644
--- a/docs/advanced-setup.md
+++ b/docs/advanced-setup.md
@@ -665,21 +665,22 @@ would otherwise prevent it. Set `OPENCLAUDE_MAX_ACTIVE_MESSAGES_HARD_CAP=0`
only when you need to suppress that safety cap for diagnostics.
If you frequently resume long sessions that accumulate hundreds of small
-tool-result messages with negligible token cost, you can opt in to message-count
+tool-result messages with negligible token cost, adjust message-count
compaction via the in-app `/config` command:
```text
/config
```
-Select **Message-count compaction** and choose a threshold (`100`, `200`, `500`,
-or `1000`). Setting it to `off` (default) leaves only the built-in hard cap.
+Message-count compaction defaults to `200` messages. Select
+**Message-count compaction** to choose a different threshold (`100`, `500`, or
+`1000`), or set it to `off` to disable the setting's proactive guard. The
+built-in hard cap remains, and an `OPENCLAUDE_MAX_ACTIVE_MESSAGES` override
+remains active when configured.
-This setting is intended for power users debugging specific edge cases. Most
-users should leave it at `off`.
-
-The legacy `OPENCLAUDE_MAX_ACTIVE_MESSAGES` environment variable is still
-honored when the setting is `off`. `OPENCLAUDE_MAX_ACTIVE_MESSAGES_HARD_CAP`
+The legacy `OPENCLAUDE_MAX_ACTIVE_MESSAGES` environment variable is honored
+when the setting is unset or `off`. An explicit numeric setting takes
+precedence over that legacy value. `OPENCLAUDE_MAX_ACTIVE_MESSAGES_HARD_CAP`
can override the safety cap; set it to `0` only for diagnostics.
### Long-session memory guard validation
diff --git a/scripts/system-check.test.ts b/scripts/system-check.test.ts
index ebea4b1eb..562979420 100644
--- a/scripts/system-check.test.ts
+++ b/scripts/system-check.test.ts
@@ -658,7 +658,21 @@ describe('system-check WebSearch diagnostics', () => {
})
describe('system-check memory guard diagnostics', () => {
- test('reports safe default auto-compact and hard-cap guards', () => {
+ test('reports explicit off without a legacy message-count override', () => {
+ const results = buildMemoryGuardChecks({
+ autoCompactEnabled: true,
+ maxMessagesCompactionThreshold: 'off',
+ env: {},
+ })
+
+ expect(results).toContainEqual({
+ ok: true,
+ label: 'Auto-compact guard',
+ detail: `Enabled; message-count threshold off; hard cap ${DEFAULT_MAX_ACTIVE_MESSAGES_HARD_CAP}.`,
+ })
+ })
+
+ test('reports the effective default auto-compact and hard-cap guards', () => {
const results = buildMemoryGuardChecks({
autoCompactEnabled: true,
maxMessagesCompactionThreshold: undefined,
@@ -668,7 +682,7 @@ describe('system-check memory guard diagnostics', () => {
expect(results).toContainEqual({
ok: true,
label: 'Auto-compact guard',
- detail: `Enabled; message-count threshold off; hard cap ${DEFAULT_MAX_ACTIVE_MESSAGES_HARD_CAP}.`,
+ detail: `Enabled; message-count threshold 200; hard cap ${DEFAULT_MAX_ACTIVE_MESSAGES_HARD_CAP}.`,
})
expect(results).toContainEqual({
ok: true,
@@ -679,6 +693,36 @@ describe('system-check memory guard diagnostics', () => {
.toMatchObject({ ok: true })
})
+ test('reports the legacy message-count override when the setting is unset or off', () => {
+ for (const maxMessagesCompactionThreshold of [undefined, 'off']) {
+ const results = buildMemoryGuardChecks({
+ autoCompactEnabled: true,
+ maxMessagesCompactionThreshold,
+ env: { OPENCLAUDE_MAX_ACTIVE_MESSAGES: '500' },
+ })
+
+ expect(results).toContainEqual({
+ ok: true,
+ label: 'Auto-compact guard',
+ detail: `Enabled; message-count threshold 500; hard cap ${DEFAULT_MAX_ACTIVE_MESSAGES_HARD_CAP}.`,
+ })
+ }
+ })
+
+ test('reports an explicit message-count setting ahead of the legacy override', () => {
+ const results = buildMemoryGuardChecks({
+ autoCompactEnabled: true,
+ maxMessagesCompactionThreshold: '100',
+ env: { OPENCLAUDE_MAX_ACTIVE_MESSAGES: '500' },
+ })
+
+ expect(results).toContainEqual({
+ ok: true,
+ label: 'Auto-compact guard',
+ detail: `Enabled; message-count threshold 100; hard cap ${DEFAULT_MAX_ACTIVE_MESSAGES_HARD_CAP}.`,
+ })
+ })
+
test('falls back to the default hard cap when the override is malformed', () => {
const results = buildMemoryGuardChecks({
autoCompactEnabled: true,
@@ -725,7 +769,35 @@ describe('system-check memory guard diagnostics', () => {
ok: false,
label: 'Auto-compact guard',
detail:
- 'settings disabled; DISABLE_COMPACT is set; DISABLE_AUTO_COMPACT is set',
+ 'settings disabled; DISABLE_COMPACT is set; DISABLE_AUTO_COMPACT is set; message-count threshold 500 remains active.',
+ })
+ })
+
+ test('reports an explicit message-count guard when token auto-compact is disabled', () => {
+ const results = buildMemoryGuardChecks({
+ autoCompactEnabled: true,
+ maxMessagesCompactionThreshold: '100',
+ env: { DISABLE_AUTO_COMPACT: '1' },
+ })
+
+ expect(results[0]).toEqual({
+ ok: false,
+ label: 'Auto-compact guard',
+ detail: 'DISABLE_AUTO_COMPACT is set; message-count threshold 100 remains active.',
+ })
+ })
+
+ test('does not report the unset default as active when auto-compact is disabled', () => {
+ const results = buildMemoryGuardChecks({
+ autoCompactEnabled: true,
+ maxMessagesCompactionThreshold: undefined,
+ env: { DISABLE_AUTO_COMPACT: '1' },
+ })
+
+ expect(results[0]).toEqual({
+ ok: false,
+ label: 'Auto-compact guard',
+ detail: 'DISABLE_AUTO_COMPACT is set',
})
})
diff --git a/scripts/system-check.ts b/scripts/system-check.ts
index 9e67d2799..a19a38ee2 100644
--- a/scripts/system-check.ts
+++ b/scripts/system-check.ts
@@ -36,6 +36,10 @@ import {
DEFAULT_MAX_ACTIVE_MESSAGES_HARD_CAP,
getMaxActiveMessagesHardCap,
} from '../src/utils/maxActiveMessages.js'
+import {
+ isValidMaxMessagesCompactionThreshold,
+ normalizeMaxMessagesCompactionThreshold,
+} from '../src/utils/config.js'
import {
getAvailableProviders,
getProviderChain,
@@ -120,28 +124,48 @@ export function buildMemoryGuardChecks(
input.autoCompactEnabled && !disableCompact && !disableAutoCompact
const hardCapOverride = env.OPENCLAUDE_MAX_ACTIVE_MESSAGES_HARD_CAP
const hardCap = getMaxActiveMessagesHardCap(env)
- const configuredLimit =
- input.maxMessagesCompactionThreshold &&
- input.maxMessagesCompactionThreshold !== 'off'
- ? input.maxMessagesCompactionThreshold
- : undefined
const legacyLimit = parsePositiveInteger(env.OPENCLAUDE_MAX_ACTIVE_MESSAGES)
+ const configuredLimit =
+ ((input.maxMessagesCompactionThreshold === undefined ||
+ input.maxMessagesCompactionThreshold === 'off') &&
+ legacyLimit > 0)
+ ? legacyLimit
+ : normalizeMaxMessagesCompactionThreshold(
+ input.maxMessagesCompactionThreshold,
+ )
+ const hasExplicitMessageCountGuard =
+ input.maxMessagesCompactionThreshold !== undefined &&
+ isValidMaxMessagesCompactionThreshold(
+ input.maxMessagesCompactionThreshold,
+ ) &&
+ input.maxMessagesCompactionThreshold !== 'off'
+ const hasLegacyMessageCountGuard =
+ (input.maxMessagesCompactionThreshold === undefined ||
+ input.maxMessagesCompactionThreshold === 'off') &&
+ legacyLimit > 0
const memoryBudget = parsePositiveInteger(env.OPENCLAUDE_MAX_MEMORY_MB) || 1536
+ const hasIndependentMessageCountGuard =
+ configuredLimit !== 'off' &&
+ (hasExplicitMessageCountGuard || hasLegacyMessageCountGuard)
+ const autoCompactDisabledReason =
+ [
+ input.autoCompactEnabled ? undefined : 'settings disabled',
+ disableCompact ? 'DISABLE_COMPACT is set' : undefined,
+ disableAutoCompact ? 'DISABLE_AUTO_COMPACT is set' : undefined,
+ ].filter(Boolean).join('; ') || 'Disabled by configuration.'
results.push(
autoCompactAvailable
? pass(
'Auto-compact guard',
- `Enabled; message-count threshold ${configuredLimit ?? (legacyLimit > 0 ? legacyLimit : 'off')}; hard cap ${hardCap === 0 ? 'disabled' : hardCap}.`,
+ `Enabled; message-count threshold ${configuredLimit}; hard cap ${hardCap === 0 ? 'disabled' : hardCap}.`,
)
: fail(
'Auto-compact guard',
- [
- input.autoCompactEnabled ? undefined : 'settings disabled',
- disableCompact ? 'DISABLE_COMPACT is set' : undefined,
- disableAutoCompact ? 'DISABLE_AUTO_COMPACT is set' : undefined,
- ].filter(Boolean).join('; ') ||
- 'Disabled by configuration.',
+ autoCompactDisabledReason +
+ (hasIndependentMessageCountGuard
+ ? `; message-count threshold ${configuredLimit} remains active.`
+ : ''),
),
)
diff --git a/src/components/Settings/Config.tsx b/src/components/Settings/Config.tsx
index 14f47d134..37ce2f8df 100644
--- a/src/components/Settings/Config.tsx
+++ b/src/components/Settings/Config.tsx
@@ -294,7 +294,7 @@ export function Config({
}, {
id: 'maxMessagesCompactionThreshold',
label: 'Message-count compaction',
- value: globalConfig.maxMessagesCompactionThreshold ?? 'off',
+ value: normalizeMaxMessagesCompactionThreshold(globalConfig.maxMessagesCompactionThreshold),
options: [...MAX_MESSAGES_COMPACTION_THRESHOLDS],
type: 'enum' as const,
onChange(maxMessagesCompactionThreshold: string) {
@@ -1242,7 +1242,7 @@ export function Config({
formattedChanges.push(`${globalConfig.autoCompactEnabled ? 'Enabled' : 'Disabled'} auto-compact`);
}
if (globalConfig.maxMessagesCompactionThreshold !== initialConfig.current.maxMessagesCompactionThreshold) {
- const threshold = globalConfig.maxMessagesCompactionThreshold ?? 'off';
+ const threshold = normalizeMaxMessagesCompactionThreshold(globalConfig.maxMessagesCompactionThreshold);
formattedChanges.push(threshold === 'off' ? 'Disabled message-count compaction' : `Set message-count compaction to ${threshold}`);
}
if (globalConfig.toolHistoryCompressionEnabled !== initialConfig.current.toolHistoryCompressionEnabled) {
diff --git a/src/components/messages/AttachmentMessage.tsx b/src/components/messages/AttachmentMessage.tsx
index d19b2b22f..8f42783e3 100644
--- a/src/components/messages/AttachmentMessage.tsx
+++ b/src/components/messages/AttachmentMessage.tsx
@@ -121,6 +121,10 @@ export function AttachmentMessage({
// eslint-disable-next-line @typescript-eslint/switch-exhaustiveness-check -- teammate_mailbox/skill_discovery handled before switch
switch (attachment.type) {
+ case 'max_turns_reached':
+ return
+ Reached the maximum number of turns ({attachment.maxTurns}).
+ ;
case 'directory':
return
Listed directory {attachment.displayPath + sep}
diff --git a/src/components/messages/nullRenderingAttachments.ts b/src/components/messages/nullRenderingAttachments.ts
index a6f1bca62..f7c6f053d 100644
--- a/src/components/messages/nullRenderingAttachments.ts
+++ b/src/components/messages/nullRenderingAttachments.ts
@@ -43,7 +43,6 @@ const NULL_RENDERING_TYPES = [
'token_usage',
'ultrathink_effort',
'ultracode_mode',
- 'max_turns_reached',
'task_reminder',
'auto_mode',
'auto_mode_exit',
diff --git a/src/main.tsx b/src/main.tsx
index ac07178e4..bfb214e03 100644
--- a/src/main.tsx
+++ b/src/main.tsx
@@ -3045,7 +3045,7 @@ async function run(): Promise {
strictMcpConfig,
systemPrompt,
appendSystemPrompt,
- thinkingConfig
+ thinkingConfig,
};
// Shared context for processResumedConversation calls
diff --git a/src/query.ts b/src/query.ts
index b2db34350..ff7c69a34 100644
--- a/src/query.ts
+++ b/src/query.ts
@@ -74,7 +74,9 @@ import {
startRelevantMemoryPrefetch,
} from './utils/attachments.js'
import {
+ getMaxActiveMessagesHardCap,
isAboveMaxActiveMessagesLimit,
+ parseMaxActiveMessagesLimit,
resolveMaxActiveMessagesLimit,
} from './utils/maxActiveMessages.js'
/* eslint-disable @typescript-eslint/no-require-imports */
@@ -125,7 +127,9 @@ import {
import { AGENT_STEP_LIMIT_TOOL_RESULT_PREFIX } from './query/agentStepLimit.js'
import { buildQueryConfig } from './query/config.js'
import {
+ MAX_MESSAGES_COMPACTION_THRESHOLDS,
getGlobalConfig,
+ isValidMaxMessagesCompactionThreshold,
normalizeMaxMessagesCompactionThreshold,
} from './utils/config.js'
import { productionDeps, type QueryDeps } from './query/deps.js'
@@ -813,18 +817,39 @@ async function* queryLoop(
// compaction and forcing would deadlock via recursive autocompaction.
const canForceCompact =
querySource !== 'compact' && querySource !== 'session_memory'
+ // An unset UI setting keeps the legacy environment override. Without that
+ // override, enforce the new effective 200-message default.
+ const hasValidLegacyActiveMessageLimit =
+ parseMaxActiveMessagesLimit(process.env.OPENCLAUDE_MAX_ACTIVE_MESSAGES) > 0
+ const maxMessagesLimitSetting =
+ configuredMaxMessagesCompactionThreshold === undefined &&
+ hasValidLegacyActiveMessageLimit
+ ? undefined
+ : maxMessagesCompactionThreshold
+ const hasExplicitMessageCountThreshold =
+ configuredMaxMessagesCompactionThreshold !== undefined &&
+ isValidMaxMessagesCompactionThreshold(configuredMaxMessagesCompactionThreshold) &&
+ configuredMaxMessagesCompactionThreshold !== 'off'
+ const hasActiveMessageLimitOverride =
+ hasExplicitMessageCountThreshold ||
+ ((configuredMaxMessagesCompactionThreshold === undefined ||
+ configuredMaxMessagesCompactionThreshold === 'off') &&
+ hasValidLegacyActiveMessageLimit)
const activeMessageLimit = canForceCompact
? resolveMaxActiveMessagesLimit(
- maxMessagesCompactionThreshold,
+ maxMessagesLimitSetting,
process.env.OPENCLAUDE_MAX_ACTIVE_MESSAGES,
)
: 0
if (canForceCompact) {
if (
- isAboveMaxActiveMessagesLimit(
- messagesForQuery.length,
- activeMessageLimit,
- )
+ isAboveMaxActiveMessagesLimit(messagesForQuery.length, activeMessageLimit) &&
+ (isAutoCompactEnabled() ||
+ hasActiveMessageLimitOverride ||
+ isAboveMaxActiveMessagesLimit(
+ messagesForQuery.length,
+ getMaxActiveMessagesHardCap(),
+ ))
) {
tracking = {
...(tracking ?? { compacted: false, turnId: '', turnCounter: 0 }),
@@ -1167,6 +1192,14 @@ async function* queryLoop(
// cooling down or otherwise exhausted and context or message count is still
// over the safety threshold, block immediately with a clear message instead
// of burning an oversized API call.
+ const isAboveActiveMessageHardCap = isAboveMaxActiveMessagesLimit(
+ messagesForQuery.length,
+ getMaxActiveMessagesHardCap(),
+ )
+ const shouldEnforceActiveMessageLimit =
+ (!collapseOwnsIt && isAutoCompactEnabled()) ||
+ hasActiveMessageLimitOverride ||
+ isAboveActiveMessageHardCap
if (
tracking?.consecutiveFailures !== undefined &&
tracking.consecutiveFailures >=
@@ -1181,14 +1214,16 @@ async function* queryLoop(
tokenUsage,
model,
)
+ const isAboveActiveMessageSafetyLimit =
+ isAboveMaxActiveMessagesLimit(
+ messagesForQuery.length,
+ activeMessageLimit,
+ ) && shouldEnforceActiveMessageLimit
const isAboveBreakerThreshold =
isAboveAutoCompactThreshold ||
((circuitBreakerActive === true || circuitBreakerTripped === true) &&
tokenUsage >= getAutoCompactThreshold(model)) ||
- isAboveMaxActiveMessagesLimit(
- messagesForQuery.length,
- activeMessageLimit,
- )
+ isAboveActiveMessageSafetyLimit
if (isAboveBreakerThreshold) {
const nowMs = Date.now()
const retryDelayMs =
@@ -1211,7 +1246,11 @@ async function* queryLoop(
}
if (
- isAboveMaxActiveMessagesLimit(messagesForQuery.length, activeMessageLimit)
+ shouldEnforceActiveMessageLimit &&
+ isAboveMaxActiveMessagesLimit(
+ messagesForQuery.length,
+ activeMessageLimit,
+ )
) {
yield createAssistantAPIErrorMessage({
content:
diff --git a/src/query/autoCompactCooldown.test.ts b/src/query/autoCompactCooldown.test.ts
index 9a91385c4..91af999a5 100644
--- a/src/query/autoCompactCooldown.test.ts
+++ b/src/query/autoCompactCooldown.test.ts
@@ -368,6 +368,96 @@ test('default active-message hard cap forces compaction', async () => {
expect(seenTracking[0]?.forceReason).toBe('message-count')
})
+test('unset message threshold forces compaction at the 200-message default', async () => {
+ const { terminal, callModel, seenTracking } =
+ await runMessageCountHardCapQuery(manySmallMessages(201))
+
+ expect(terminal.reason).toBe('max_turns')
+ expect(callModel).toHaveBeenCalledTimes(1)
+ expect(seenTracking[0]?.forceReason).toBe('message-count')
+})
+
+test('invalid legacy message threshold keeps the 200-message default', async () => {
+ process.env.OPENCLAUDE_MAX_ACTIVE_MESSAGES = 'not-a-number'
+
+ const { terminal, callModel, seenTracking } =
+ await runMessageCountHardCapQuery(manySmallMessages(201))
+
+ expect(terminal.reason).toBe('max_turns')
+ expect(callModel).toHaveBeenCalledTimes(1)
+ expect(seenTracking[0]?.forceReason).toBe('message-count')
+})
+
+test('disabled auto-compact leaves the default message threshold inactive', async () => {
+ process.env.DISABLE_AUTO_COMPACT = '1'
+
+ const { terminal, callModel, seenTracking } =
+ await runMessageCountHardCapQuery(manySmallMessages(201))
+
+ expect(terminal.reason).toBe('max_turns')
+ expect(callModel).toHaveBeenCalledTimes(1)
+ expect(seenTracking[0]?.forceReason).toBeUndefined()
+})
+
+test('disabled auto-compact ignores an invalid persisted message threshold', async () => {
+ saveGlobalConfig(current => ({
+ ...current,
+ maxMessagesCompactionThreshold:
+ 'not-a-threshold' as MaxMessagesCompactionThreshold,
+ }))
+ process.env.DISABLE_AUTO_COMPACT = '1'
+
+ const { terminal, callModel, seenTracking } =
+ await runMessageCountHardCapQuery(manySmallMessages(201))
+
+ expect(terminal.reason).toBe('max_turns')
+ expect(callModel).toHaveBeenCalledTimes(1)
+ expect(seenTracking[0]?.forceReason).toBeUndefined()
+})
+
+test('disabled auto-compact preserves an explicit message threshold', async () => {
+ saveGlobalConfig(current => ({
+ ...current,
+ maxMessagesCompactionThreshold: '100',
+ }))
+ process.env.DISABLE_AUTO_COMPACT = '1'
+
+ const { terminal, callModel, seenTracking } =
+ await runMessageCountHardCapQuery(manySmallMessages(101))
+
+ expect(terminal.reason).toBe('max_turns')
+ expect(callModel).toHaveBeenCalledTimes(1)
+ expect(seenTracking[0]?.forceReason).toBe('message-count')
+})
+
+test('disabled auto-compact preserves a legacy message threshold', async () => {
+ process.env.DISABLE_AUTO_COMPACT = '1'
+ process.env.OPENCLAUDE_MAX_ACTIVE_MESSAGES = '100'
+
+ const { terminal, callModel, seenTracking } =
+ await runMessageCountHardCapQuery(manySmallMessages(101))
+
+ expect(terminal.reason).toBe('max_turns')
+ expect(callModel).toHaveBeenCalledTimes(1)
+ expect(seenTracking[0]?.forceReason).toBe('message-count')
+})
+
+test('explicit off preserves a legacy message threshold', async () => {
+ saveGlobalConfig(current => ({
+ ...current,
+ maxMessagesCompactionThreshold: 'off',
+ }))
+ process.env.DISABLE_AUTO_COMPACT = '1'
+ process.env.OPENCLAUDE_MAX_ACTIVE_MESSAGES = '100'
+
+ const { terminal, callModel, seenTracking } =
+ await runMessageCountHardCapQuery(manySmallMessages(101))
+
+ expect(terminal.reason).toBe('max_turns')
+ expect(callModel).toHaveBeenCalledTimes(1)
+ expect(seenTracking[0]?.forceReason).toBe('message-count')
+})
+
test('long-session smoke keeps repeated over-cap turns bounded before provider calls', async () => {
const seenProviderMessageCounts: number[] = []
const seenTracking: Array = []
@@ -440,6 +530,13 @@ test('invalid active-message hard cap override keeps default safety cap', async
})
test('explicit zero active-message hard cap override disables safety cap', async () => {
+ // Isolate the hard-cap override: with the 200-message-count default active,
+ // a 1001-message history would otherwise force message-count compaction, so
+ // disable message-count compaction explicitly to test only the hard cap.
+ saveGlobalConfig(current => ({
+ ...current,
+ maxMessagesCompactionThreshold: 'off',
+ }))
process.env.OPENCLAUDE_MAX_ACTIVE_MESSAGES_HARD_CAP = '0'
const { terminal, callModel, seenTracking } =
diff --git a/src/screens/REPL.tsx b/src/screens/REPL.tsx
index e5456496a..4d2596cb0 100644
--- a/src/screens/REPL.tsx
+++ b/src/screens/REPL.tsx
@@ -38,6 +38,7 @@ import { logForDebugging } from '../utils/debug.js';
import { QueryGuard } from '../utils/QueryGuard.js';
import { getQueryGuardOptionsFromEnv } from '../utils/queryGuardConfig.js';
import { QueryLifecycleOperationTracker, formatQueryLifecycleAbortSignalReason, formatQueryLifecycleLogMessage, getQueryTerminalReason, type QueryActiveOperationSnapshot, type QueryGuardTimeoutInfo, type QueryLifecycleContext, type QueryTerminalReason } from '../utils/queryLifecycle.js';
+import { resolveReplMaxTurns } from './replMaxTurns.js';
import { createCombinedAbortSignal } from '../utils/combinedAbortSignal.js';
import { isEnvTruthy } from '../utils/envUtils.js';
import { formatTokens, truncateToWidth } from '../utils/format.js';
@@ -569,6 +570,8 @@ function summarizeActiveOperations(snapshot: QueryActiveOperationSnapshot): stri
function logQueryLifecycle(event: string, context: QueryLifecycleContext, extras = ''): void {
logForDebugging(formatQueryLifecycleLogMessage(event, context, extras));
}
+// Default per-prompt cap for every local interactive REPL entrypoint. Headless
+// and SDK callers retain their explicit maxTurns contracts.
export type Props = {
commands: Command[];
debug: boolean;
@@ -617,6 +620,8 @@ export type Props = {
thinkingConfig: ThinkingConfig;
// Model to fallback to when primary model returns overloaded errors (529)
fallbackModel?: string;
+ // Bound a single interactive prompt's sequential tool-use turns.
+ maxTurns?: number;
};
export type Screen = 'prompt' | 'transcript';
export function REPL({
@@ -646,8 +651,10 @@ export function REPL({
directConnectConfig,
sshSession,
thinkingConfig,
- fallbackModel
+ fallbackModel,
+ maxTurns: maxTurnsProp
}: Props): React.ReactNode {
+ const maxTurns = resolveReplMaxTurns(maxTurnsProp)
const isRemoteSession = !!remoteSessionConfig;
// Env-var gates hoisted to mount-time — isEnvTruthy does toLowerCase+trim+
@@ -2809,6 +2816,7 @@ export function REPL({
canUseTool,
toolUseContext,
fallbackModel,
+ maxTurns,
querySource: getQuerySourceForREPL(),
autoCompactTracking: getAutoCompactTrackingForSession(backgroundSessionId),
onAutoCompactTrackingChange: tracking => {
@@ -2820,7 +2828,7 @@ export function REPL({
agentDefinition: mainThreadAgentDefinition
});
})();
- }, [abortController, mainLoopModel, toolPermissionContext, mainThreadAgentDefinition, getToolUseContext, customSystemPrompt, appendSystemPrompt, canUseTool, setAppState, getAutoCompactTrackingForSession, setAutoCompactTrackingForSession, fallbackModel]);
+ }, [abortController, mainLoopModel, toolPermissionContext, mainThreadAgentDefinition, getToolUseContext, customSystemPrompt, appendSystemPrompt, canUseTool, setAppState, getAutoCompactTrackingForSession, setAutoCompactTrackingForSession, fallbackModel, maxTurns]);
const {
handleBackgroundSession
} = useSessionBackgrounding({
@@ -3041,6 +3049,7 @@ export function REPL({
toolUseContext,
querySource: getQuerySourceForREPL(),
fallbackModel,
+ maxTurns,
autoCompactTracking: queryAutoCompactTracking,
onAutoCompactTrackingChange: tracking => {
if (setAutoCompactTrackingForSessionIfUnchanged(querySessionId, expectedAutoCompactTracking, tracking)) {
@@ -3066,7 +3075,7 @@ export function REPL({
// Signal that a query turn has completed successfully
await onTurnComplete?.(messagesRef.current);
- }, [initialMcpClients, resetLoadingState, getToolUseContext, toolPermissionContext, setAppState, customSystemPrompt, onTurnComplete, appendSystemPrompt, canUseTool, mainThreadAgentDefinition, onQueryEvent, sessionTitle, titleDisabled, getAutoCompactTrackingForSession, setAutoCompactTrackingForSession, setAutoCompactTrackingForSessionIfUnchanged, queryGuard]);
+ }, [initialMcpClients, resetLoadingState, getToolUseContext, toolPermissionContext, setAppState, customSystemPrompt, onTurnComplete, appendSystemPrompt, canUseTool, mainThreadAgentDefinition, onQueryEvent, sessionTitle, titleDisabled, maxTurns, getAutoCompactTrackingForSession, setAutoCompactTrackingForSession, setAutoCompactTrackingForSessionIfUnchanged, queryGuard]);
const onQuery = useCallback(async (newMessages: MessageType[], abortController: AbortController, shouldQuery: boolean, additionalAllowedTools: string[], mainLoopModelParam: string, onBeforeQueryCallback?: (input: string, newMessages: MessageType[]) => Promise, input?: string, effort?: EffortValue): Promise => {
// If this is a teammate, mark them as active when starting a turn
if (isAgentSwarmsEnabled()) {
diff --git a/src/screens/ResumeConversation.tsx b/src/screens/ResumeConversation.tsx
index af56b80d6..f5c8bb178 100644
--- a/src/screens/ResumeConversation.tsx
+++ b/src/screens/ResumeConversation.tsx
@@ -56,6 +56,7 @@ type Props = {
filterByPr?: boolean | number | string;
thinkingConfig: ThinkingConfig;
fallbackModel?: string;
+ maxTurns?: number;
onTurnComplete?: (messages: Message[]) => void | Promise;
};
export function ResumeConversation({
@@ -78,6 +79,7 @@ export function ResumeConversation({
filterByPr,
thinkingConfig,
fallbackModel,
+ maxTurns,
onTurnComplete
}: Props): React.ReactNode {
const {
@@ -286,7 +288,7 @@ export function ResumeConversation({
return ;
}
if (resumeData) {
- return ;
+ return ;
}
if (loading) {
return
diff --git a/src/screens/replMaxTurns.ts b/src/screens/replMaxTurns.ts
new file mode 100644
index 000000000..c60c8ab86
--- /dev/null
+++ b/src/screens/replMaxTurns.ts
@@ -0,0 +1,5 @@
+export const DEFAULT_REPL_MAX_TURNS = 50
+
+export function resolveReplMaxTurns(maxTurns?: number): number {
+ return maxTurns ?? DEFAULT_REPL_MAX_TURNS
+}
diff --git a/src/screens/replMaxTurnsProp.test.ts b/src/screens/replMaxTurnsProp.test.ts
new file mode 100644
index 000000000..c125ba721
--- /dev/null
+++ b/src/screens/replMaxTurnsProp.test.ts
@@ -0,0 +1,52 @@
+import { describe, expect, test } from 'bun:test'
+import { readFileSync } from 'node:fs'
+import { join } from 'node:path'
+import { DEFAULT_REPL_MAX_TURNS, resolveReplMaxTurns } from './replMaxTurns.js'
+
+const screenDir = import.meta.dirname
+
+function readScreen(name: string): string {
+ return readFileSync(join(screenDir, name), 'utf8')
+}
+
+function objectBody(source: string, marker: RegExp): string {
+ const match = source.match(marker)
+ expect(match).not.toBeNull()
+ const start = match!.index! + match![0].length - 1
+ let depth = 0
+ for (let index = start; index < source.length; index++) {
+ if (source[index] === '{') depth++
+ if (source[index] === '}') {
+ depth--
+ if (depth === 0) return source.slice(start, index + 1)
+ }
+ }
+ throw new Error(`Unclosed object after ${marker}`)
+}
+
+describe('interactive REPL max-turn cap', () => {
+ test('supplies the local interactive default at runtime', () => {
+ expect(DEFAULT_REPL_MAX_TURNS).toBe(50)
+ expect(resolveReplMaxTurns()).toBe(50)
+ })
+
+ test('preserves an explicit interactive cap at runtime', () => {
+ expect(resolveReplMaxTurns(7)).toBe(7)
+ })
+
+ test('passes the resolved cap to foreground and background queries', () => {
+ const source = readScreen('REPL.tsx')
+ const foreground = objectBody(source, /for await \(const event of query\(\{/)
+ const background = objectBody(source, /queryParams:\s*\{/)
+
+ expect(foreground).toContain('maxTurns,')
+ expect(background).toContain('maxTurns,')
+ })
+
+ test('passes the cap from the resume selector into REPL', () => {
+ const source = readScreen('ResumeConversation.tsx')
+ const repl = source.slice(source.indexOf('', source.indexOf(' {
try {
const effective = getEffectiveContextWindowSize('some-unknown-3p-model')
expect(effective).toBeGreaterThan(0)
- // 21k = CAPPED_DEFAULT_MAX_TOKENS (8k) + AUTOCOMPACT_BUFFER_TOKENS (13k).
+ // 21k = CAPPED_DEFAULT_MAX_TOKENS (8k) + AUTOCOMPACT_FLOOR_BUFFER_TOKENS (13k).
// Covers the anti-regression intent of issue #635 without assuming
// the GrowthBook flag state.
expect(effective).toBeGreaterThanOrEqual(21_000)
@@ -243,6 +243,40 @@ describe('getAutoCompactThreshold', () => {
restoreEnv()
}
})
+
+ test('keeps the floor buffer for constrained context windows', async () => {
+ process.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = '30000'
+ process.env.CLAUDE_CODE_MAX_OUTPUT_TOKENS = '20000'
+ const { getAutoCompactThreshold } = await importAutoCompact()
+
+ // The effective window is floor-raised to 33k in this configuration.
+ // Selecting the 30k buffer here would compact after only 3k tokens.
+ expect(getAutoCompactThreshold('claude-sonnet-4')).toBe(20_000)
+ })
+
+ test('keeps compaction and warning thresholds usable across mid-sized windows', async () => {
+ process.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = '64000'
+ const { calculateTokenWarningState, getAutoCompactThreshold } =
+ await importAutoCompact()
+
+ // The effective window is 44k. Do not consume so much headroom that the
+ // 20k warning/error buffer makes a fresh conversation immediately warn.
+ expect(getAutoCompactThreshold('claude-sonnet-4')).toBe(30_000)
+ expect(
+ calculateTokenWarningState(0, 'claude-sonnet-4').isAboveWarningThreshold,
+ ).toBe(false)
+ })
+
+ test('does not lower the threshold when a configured window grows', async () => {
+ const { getAutoCompactThreshold } = await importAutoCompact()
+
+ process.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = '62999'
+ const smallerWindowThreshold = getAutoCompactThreshold('claude-sonnet-4')
+ process.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = '63000'
+ const largerWindowThreshold = getAutoCompactThreshold('claude-sonnet-4')
+
+ expect(largerWindowThreshold).toBeGreaterThanOrEqual(smallerWindowThreshold)
+ })
})
describe('getAutoCompactFailureCooldownMs', () => {
diff --git a/src/services/compact/autoCompact.ts b/src/services/compact/autoCompact.ts
index e93e76092..08c2149a3 100644
--- a/src/services/compact/autoCompact.ts
+++ b/src/services/compact/autoCompact.ts
@@ -49,8 +49,12 @@ export function getEffectiveContextWindowSize(model: string): number {
// Floor: effective context must be at least the summary reservation plus a
// usable buffer. If it goes lower, the auto-compact threshold becomes
- // negative and fires on every message (issue #635).
- const autocompactBuffer = 13_000 // must match AUTOCOMPACT_BUFFER_TOKENS
+ // negative and fires on every message (issue #635). This floor buffer is
+ // intentionally decoupled from AUTOCOMPACT_BUFFER_TOKENS: the latter is the
+ // (larger) threshold buffer used by getAutoCompactThreshold(), while this
+ // stays at the conservative 13k so getEffectiveContextWindowSize() —
+ // also consumed by tool-history compression — is unchanged (issue #1949).
+ const autocompactBuffer = AUTOCOMPACT_FLOOR_BUFFER_TOKENS
const effectiveContext = contextWindow - reservedTokensForSummary
return Math.max(effectiveContext, reservedTokensForSummary + autocompactBuffer)
}
@@ -74,7 +78,20 @@ export type AutoCompactTrackingState = {
forceReason?: 'memory-pressure' | 'message-count'
}
-export const AUTOCOMPACT_BUFFER_TOKENS = 13_000
+// Threshold buffer: auto-compact fires when token usage reaches this far below
+// the effective context window. Bumped from 13_000 -> 30_000 so compaction runs
+// earlier and with less accumulated history, bounding per-turn latency growth
+// in a single session (issue #1949). Kept below the effective-context floor
+// (AUTOCOMPACT_FLOOR_BUFFER_TOKENS) for large-context models; for small-context
+// models getAutoCompactThreshold() falls back to the floor buffer so the
+// threshold can never go negative (issue #635).
+export const AUTOCOMPACT_BUFFER_TOKENS = 30_000
+
+// Conservative floor buffer for getEffectiveContextWindowSize(). Must guarantee
+// a non-negative auto-compact threshold for small-context models, so it stays at
+// the pre-#1949 value of 13_000 and is decoupled from AUTOCOMPACT_BUFFER_TOKENS.
+const AUTOCOMPACT_FLOOR_BUFFER_TOKENS = 13_000
+
export const WARNING_THRESHOLD_BUFFER_TOKENS = 20_000
export const ERROR_THRESHOLD_BUFFER_TOKENS = 20_000
export const MANUAL_COMPACT_BUFFER_TOKENS = 3_000
@@ -170,8 +187,19 @@ export function resolveAutoCompactCircuitBreakerState(args: {
export function getAutoCompactThreshold(model: string): number {
const effectiveContextWindow = getEffectiveContextWindowSize(model)
- const autocompactThreshold =
- effectiveContextWindow - AUTOCOMPACT_BUFFER_TOKENS
+ // Increase the buffer gradually between the old 13k and new 30k values.
+ // This keeps the threshold monotonic and preserves the 20k warning/error
+ // headroom consumed by calculateTokenWarningState(). A direct switch would
+ // make a one-token window increase cause an earlier compact, and can make
+ // the warning threshold negative for mid-sized context windows.
+ const buffer = Math.min(
+ AUTOCOMPACT_BUFFER_TOKENS,
+ Math.max(
+ AUTOCOMPACT_FLOOR_BUFFER_TOKENS,
+ effectiveContextWindow - AUTOCOMPACT_BUFFER_TOKENS,
+ ),
+ )
+ const autocompactThreshold = effectiveContextWindow - buffer
// Override for easier testing of autocompact
const envPercent = process.env.CLAUDE_AUTOCOMPACT_PCT_OVERRIDE
diff --git a/src/utils/analyzeContext.ts b/src/utils/analyzeContext.ts
index e1073758e..a89a2a9c1 100644
--- a/src/utils/analyzeContext.ts
+++ b/src/utils/analyzeContext.ts
@@ -10,7 +10,7 @@ import { getCommandName } from '../commands.js'
import { getSystemContext } from '../context.js'
import { getFeatureValue_CACHED_MAY_BE_STALE } from '../services/analytics/growthbook.js'
import {
- AUTOCOMPACT_BUFFER_TOKENS,
+ getAutoCompactThreshold,
getEffectiveContextWindowSize,
isAutoCompactEnabled,
MANUAL_COMPACT_BUFFER_TOKENS,
@@ -1056,7 +1056,7 @@ export async function analyzeContextUsage(
// Check if autocompact is enabled and calculate threshold
const isAutoCompact = isAutoCompactEnabled()
const autoCompactThreshold = isAutoCompact
- ? getEffectiveContextWindowSize(model) - AUTOCOMPACT_BUFFER_TOKENS
+ ? getAutoCompactThreshold(model)
: undefined
// Create categories
diff --git a/src/utils/config.ts b/src/utils/config.ts
index 73ab4dfb6..80eb0c411 100644
--- a/src/utils/config.ts
+++ b/src/utils/config.ts
@@ -197,14 +197,20 @@ export const MAX_MESSAGES_COMPACTION_THRESHOLDS = [
export type MaxMessagesCompactionThreshold =
(typeof MAX_MESSAGES_COMPACTION_THRESHOLDS)[number]
-export function normalizeMaxMessagesCompactionThreshold(
+export function isValidMaxMessagesCompactionThreshold(
value: unknown,
-): MaxMessagesCompactionThreshold {
+): value is MaxMessagesCompactionThreshold {
return MAX_MESSAGES_COMPACTION_THRESHOLDS.includes(
value as MaxMessagesCompactionThreshold,
)
+}
+
+export function normalizeMaxMessagesCompactionThreshold(
+ value: unknown,
+): MaxMessagesCompactionThreshold {
+ return isValidMaxMessagesCompactionThreshold(value)
? (value as MaxMessagesCompactionThreshold)
- : 'off'
+ : '200'
}
export type OutputStyle = string
@@ -681,7 +687,7 @@ export type GlobalConfig = {
logoColor?: string
// Message-count-based compaction threshold. Set via /config.
- // 'off' = disabled (default). Otherwise, one of '100', '200', '500', '1000'.
+ // 'off' = disabled. Otherwise, one of '100', '200', '500', '1000'.
// When enabled, triggers forced compaction if the message count exceeds the
// chosen threshold, regardless of token usage.
maxMessagesCompactionThreshold?: MaxMessagesCompactionThreshold
@@ -741,8 +747,9 @@ function createDefaultGlobalConfig(): GlobalConfig {
openaiAdditionalModelOptionsCacheByProfile: {},
knowledgeGraphEnabled: true,
// Omitted by default so callers can distinguish "unset" from an explicit
- // persisted "off"; normalizeMaxMessagesCompactionThreshold keeps the
- // effective default disabled.
+ // persisted "off"; normalizeMaxMessagesCompactionThreshold resolves an
+ // unset value to the effective default of '200' (message-count compaction
+ // enabled at 200 messages) to bound per-turn latency growth (issue #1949).
}
return config
}
@@ -1153,9 +1160,12 @@ registerCleanup(async () => {
*/
function migrateConfigFields(config: GlobalConfig): GlobalConfig {
const { maxMessagesCompactionThreshold, ...restConfig } = config
+ const hasValidMaxMessagesCompactionThreshold =
+ maxMessagesCompactionThreshold !== undefined &&
+ isValidMaxMessagesCompactionThreshold(maxMessagesCompactionThreshold)
const normalizedConfig = {
...restConfig,
- ...(maxMessagesCompactionThreshold === undefined
+ ...(!hasValidMaxMessagesCompactionThreshold
? {}
: {
maxMessagesCompactionThreshold:
diff --git a/src/utils/maxActiveMessages.test.ts b/src/utils/maxActiveMessages.test.ts
index 0fa5e61f8..3f67108c5 100644
--- a/src/utils/maxActiveMessages.test.ts
+++ b/src/utils/maxActiveMessages.test.ts
@@ -36,6 +36,7 @@ test('explicit zero hard cap disables only the hard cap', () => {
expect(isAboveMaxActiveMessagesLimit(1001)).toBe(false)
expect(resolveMaxActiveMessagesLimit('100', undefined)).toBe(100)
expect(resolveMaxActiveMessagesLimit('off', '5')).toBe(5)
+ expect(resolveMaxActiveMessagesLimit(undefined, '5')).toBe(5)
})
test('configured and hard cap combine by choosing the tighter positive limit', () => {
diff --git a/src/utils/maxActiveMessages.ts b/src/utils/maxActiveMessages.ts
index 4a683c9f5..3294f99f0 100644
--- a/src/utils/maxActiveMessages.ts
+++ b/src/utils/maxActiveMessages.ts
@@ -31,11 +31,11 @@ export function getMaxActiveMessagesHardCap(
}
export function resolveMaxActiveMessagesLimit(
- configSetting: string,
+ configSetting: string | undefined,
envSetting: string | undefined,
): number {
const configuredLimit =
- configSetting !== 'off'
+ configSetting !== undefined && configSetting !== 'off'
? parseMaxActiveMessagesLimit(configSetting)
: parseMaxActiveMessagesLimit(envSetting)
const hardCap = getMaxActiveMessagesHardCap()
diff --git a/src/utils/messages.ts b/src/utils/messages.ts
index 55a66d61a..17f98a768 100644
--- a/src/utils/messages.ts
+++ b/src/utils/messages.ts
@@ -3123,6 +3123,7 @@ You have exited auto mode. The user may now want to interact more directly. You
case 'hook_system_message':
case 'structured_output':
case 'hook_permission_decision':
+ case 'max_turns_reached':
return []
}
diff --git a/src/utils/swarm/inProcessRunner.ts b/src/utils/swarm/inProcessRunner.ts
index 2e64dc1c3..32f909c63 100644
--- a/src/utils/swarm/inProcessRunner.ts
+++ b/src/utils/swarm/inProcessRunner.ts
@@ -23,7 +23,10 @@ import {
type AnalyticsMetadata_I_VERIFIED_THIS_IS_NOT_CODE_OR_FILEPATHS,
logEvent,
} from '../../services/analytics/index.js'
-import { getAutoCompactThreshold } from '../../services/compact/autoCompact.js'
+import {
+ getAutoCompactThreshold,
+ isAutoCompactEnabled,
+} from '../../services/compact/autoCompact.js'
import {
buildPostCompactMessages,
compactConversation,
@@ -71,8 +74,15 @@ import { logForDebugging } from '../debug.js'
import { cloneFileStateCache } from '../fileStateCache.js'
import {
getMaxActiveMessagesHardCap,
- shouldCompactActiveMessageHistory,
+ isAboveMaxActiveMessagesLimit,
+ parseMaxActiveMessagesLimit,
+ resolveMaxActiveMessagesLimit,
} from '../maxActiveMessages.js'
+import {
+ getGlobalConfig,
+ isValidMaxMessagesCompactionThreshold,
+ normalizeMaxMessagesCompactionThreshold,
+} from '../config.js'
import {
SUBAGENT_REJECT_MESSAGE,
SUBAGENT_REJECT_MESSAGE_WITH_REASON_PREFIX,
@@ -1104,18 +1114,41 @@ export async function runInProcessTeammate(
// Check if compaction is needed before building context
let contextMessages = allMessages
const tokenCount = tokenCountWithEstimation(allMessages)
- const activeMessageHardCap = getMaxActiveMessagesHardCap()
+ const configuredMessageThreshold =
+ getGlobalConfig().maxMessagesCompactionThreshold
+ const legacyMessageThreshold = parseMaxActiveMessagesLimit(
+ process.env.OPENCLAUDE_MAX_ACTIVE_MESSAGES,
+ )
+ const hasExplicitMessageCountThreshold =
+ configuredMessageThreshold !== undefined &&
+ isValidMaxMessagesCompactionThreshold(configuredMessageThreshold) &&
+ configuredMessageThreshold !== 'off'
+ const hasLegacyMessageCountThreshold =
+ (configuredMessageThreshold === undefined ||
+ configuredMessageThreshold === 'off') &&
+ legacyMessageThreshold > 0
+ const shouldApplyMessageCountThreshold =
+ isAutoCompactEnabled() ||
+ hasExplicitMessageCountThreshold ||
+ hasLegacyMessageCountThreshold
+ const activeMessageLimit = shouldApplyMessageCountThreshold
+ ? resolveMaxActiveMessagesLimit(
+ configuredMessageThreshold === undefined && legacyMessageThreshold > 0
+ ? undefined
+ : normalizeMaxMessagesCompactionThreshold(configuredMessageThreshold),
+ process.env.OPENCLAUDE_MAX_ACTIVE_MESSAGES,
+ )
+ : getMaxActiveMessagesHardCap()
const tokenThreshold = getAutoCompactThreshold(
toolUseContext.options.mainLoopModel,
)
- if (
- shouldCompactActiveMessageHistory({
- messageCount: allMessages.length,
- tokenCount,
- tokenThreshold,
- activeMessageLimit: activeMessageHardCap,
- })
- ) {
+ const shouldCompactForTokens =
+ isAutoCompactEnabled() && tokenCount > tokenThreshold
+ const shouldCompactForMessages = isAboveMaxActiveMessagesLimit(
+ allMessages.length,
+ activeMessageLimit,
+ )
+ if (shouldCompactForTokens || shouldCompactForMessages) {
logForDebugging(
`[inProcessRunner] ${identity.agentId} compacting history (${tokenCount} tokens, ${allMessages.length} messages)`,
)