From a3cae91b0333cb7a744d454539f717df3b48babd Mon Sep 17 00:00:00 2001 From: Justin Mclean Date: Thu, 3 Sep 2026 03:42:54 +0000 Subject: [PATCH 1/2] Make max_tool_steps configurable, and actually wire it up The tool-calling loop in openai_agent.py ran on the hardcoded MAX_TOOL_STEPS = 20. That is enough for a report-writing subtask and not for one that has to extract, compile and then summarise: such an item hit the cap repeatedly, each time doing most of the work and returning no answer at all. max_tool_steps is now a scalar key at [harness], inside a stage table and inside a profile, resolved like every other. Two details that made the first cut of this silently ineffective: - _build() has to pass max_tool_steps into the top-level HarnessConfig. The key already passed the _HARNESS_SCALAR_KEYS check, so omitting it there raised no error; the field stayed None and the loop fell back to 20, producing runs that reported a cap of 20 while the file said 35 -- exactly the silent misconfiguration that key check exists to prevent. - _finalise() has to coerce it to int. _scalar() stringifies every value during layering, so a per-stage max_tool_steps arrived as "35" and range("35") raises TypeError. max_tokens was already coerced; the loop now covers both integer keys. The step-cap error message also names the setting and the alternative (split the subtask), since the number in it is now configurable. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01S7Y1ASxt4ox8riauVcYCsF --- src/jumar/config.py | 71 +++++++++++++++++++++++++++------------ src/jumar/openai_agent.py | 14 ++++++-- 2 files changed, 62 insertions(+), 23 deletions(-) diff --git a/src/jumar/config.py b/src/jumar/config.py index d1f5bfb..f336d73 100644 --- a/src/jumar/config.py +++ b/src/jumar/config.py @@ -178,7 +178,8 @@ def _system_timezone() -> str: # Scalar keys valid at [harness], inside a stage table, and inside a profile. _HARNESS_SCALAR_KEYS: frozenset[str] = frozenset( - {"agent", "model", "base_url", "api_key_env", "reasoning_effort", "max_tokens"} + {"agent", "model", "base_url", "api_key_env", "reasoning_effort", "max_tokens", + "max_tool_steps"} ) # Sub-table under [harness] holding named alternative harnesses: @@ -227,6 +228,12 @@ class HarnessConfig: context window, so one runaway response can consume an entire subtask deadline and take the whole attempt down with it. + ``max_tool_steps`` bounds how many tool calls one execute attempt may make + before the harness gives up. The default of 20 was set for report-writing + items; a task that has to extract, compile and then summarise runs out of + steps mid-way and returns no answer at all, having done most of the work. + Raise it for items that genuinely need a longer tool loop. + ``None`` for a per-stage field means "inherit the top-level value". A ``HarnessConfig`` is always fully resolved by the time it reaches this @@ -241,6 +248,7 @@ class HarnessConfig: api_key_env: str | None = None reasoning_effort: str | None = None max_tokens: int | None = None + max_tool_steps: int | None = None # Per-stage overrides — None inherits the top-level value. decompose_agent: str | None = None decompose_model: str | None = None @@ -248,18 +256,21 @@ class HarnessConfig: decompose_api_key_env: str | None = None decompose_reasoning_effort: str | None = None decompose_max_tokens: int | None = None + decompose_max_tool_steps: int | None = None execute_agent: str | None = None execute_model: str | None = None execute_base_url: str | None = None execute_api_key_env: str | None = None execute_reasoning_effort: str | None = None execute_max_tokens: int | None = None + execute_max_tool_steps: int | None = None judge_agent: str | None = None judge_model: str | None = None judge_base_url: str | None = None judge_api_key_env: str | None = None judge_reasoning_effort: str | None = None judge_max_tokens: int | None = None + judge_max_tool_steps: int | None = None def for_stage(self, stage: str) -> HarnessConfig: """Return the resolved ``HarnessConfig`` for *stage*. @@ -278,6 +289,7 @@ def for_stage(self, stage: str) -> HarnessConfig: k: str | None = getattr(self, f"{stage}_api_key_env") r: str | None = getattr(self, f"{stage}_reasoning_effort") x: int | None = getattr(self, f"{stage}_max_tokens") + n: int | None = getattr(self, f"{stage}_max_tool_steps") return HarnessConfig( agent=a if a is not None else self.agent, model=m if m is not None else self.model, @@ -285,6 +297,7 @@ def for_stage(self, stage: str) -> HarnessConfig: api_key_env=k if k is not None else self.api_key_env, reasoning_effort=r if r is not None else self.reasoning_effort, max_tokens=x if x is not None else self.max_tokens, + max_tool_steps=n if n is not None else self.max_tool_steps, ) @@ -548,9 +561,10 @@ def _finalise(resolved: HarnessConfig, label: str) -> HarnessConfig: just the selected one, so a typo in an unused profile fails the next load rather than the next scheduled run that happens to select it. - max_tokens arrives as a string because the layering above resolves - every scalar uniformly; it is coerced back to int here, which is also - where a non-numeric or non-positive value is rejected. + max_tokens and max_tool_steps arrive as strings because the layering + above resolves every scalar uniformly; both are coerced back to int + here, which is also where a non-numeric or non-positive value is + rejected. """ stage_attrs = (f"{s}_reasoning_effort" for s in sorted(_VALID_HARNESS_STAGES)) for attr in ("reasoning_effort", *stage_attrs): @@ -564,24 +578,30 @@ def _finalise(resolved: HarnessConfig, label: str) -> HarnessConfig: # Any, not int: the values land in fields the dataclass types # int | None, and **kwargs into dc_replace cannot be narrowed per key. + # + # Both integer keys are coerced here, not just max_tokens. A stage + # table's max_tool_steps reached the agent loop as the string "35" and + # `range("35")` raises TypeError, so the per-stage path was as broken + # as the top-level one, in a noisier way. coerced: dict[str, Any] = {} - token_attrs = (f"{s}_max_tokens" for s in sorted(_VALID_HARNESS_STAGES)) - for attr in ("max_tokens", *token_attrs): - value = getattr(resolved, attr) - if value is None: - continue - where = label if attr == "max_tokens" else f"{label}.{attr.split('_')[0]}" - try: - as_int = int(str(value)) - except ValueError: - raise ConfigError( - f"Invalid max_tokens {value!r} under [{where}]: expected a positive integer." - ) from None - if as_int <= 0: - raise ConfigError( - f"Invalid max_tokens {as_int} under [{where}]: expected a positive integer." - ) - coerced[attr] = as_int + for key in ("max_tokens", "max_tool_steps"): + int_attrs = (f"{s}_{key}" for s in sorted(_VALID_HARNESS_STAGES)) + for attr in (key, *int_attrs): + value = getattr(resolved, attr) + if value is None: + continue + where = label if attr == key else f"{label}.{attr.split('_')[0]}" + try: + as_int = int(str(value)) + except ValueError: + raise ConfigError( + f"Invalid {key} {value!r} under [{where}]: expected a positive integer." + ) from None + if as_int <= 0: + raise ConfigError( + f"Invalid {key} {as_int} under [{where}]: expected a positive integer." + ) + coerced[attr] = as_int return dc_replace(resolved, **coerced) if coerced else resolved # Resolution order, highest first: @@ -629,6 +649,15 @@ def _build(profile_table: dict[str, Any], profile_name: str | None) -> HarnessCo _scalar(profile_table, "max_tokens"), _scalar(harness_raw, "max_tokens"), ), # type: ignore[arg-type] + # Same treatment as max_tokens. Omitting this line is why a + # top-level `max_tool_steps` was accepted by the key check above, + # then silently dropped: the field stayed None and the agent loop + # fell back to MAX_TOOL_STEPS, so a run reported a cap of 20 while + # the file said 35. + max_tool_steps=_first( + _scalar(profile_table, "max_tool_steps"), + _scalar(harness_raw, "max_tool_steps"), + ), # type: ignore[arg-type] **stage_kwargs, ) diff --git a/src/jumar/openai_agent.py b/src/jumar/openai_agent.py index 5615681..fd373b4 100644 --- a/src/jumar/openai_agent.py +++ b/src/jumar/openai_agent.py @@ -51,6 +51,11 @@ # stops calling tools (or loops between two of them) must not hang the run — # it trips this cap and the subtask is recorded as a failed attempt, exactly # like a subprocess harness that exits non-zero. +# Default ceiling on tool calls in one attempt. Overridable per stage with +# `max_tool_steps` in jumar.toml: on 2026-08-29 a compile-and-report item hit +# this cap three times, each attempt extracting all 14 snippets and running +# cargo before running out of steps with no report written — the work was +# done and thrown away for want of a few more turns. MAX_TOOL_STEPS = 20 # Every request gets at least this many seconds even if the overall deadline @@ -372,7 +377,8 @@ def _rate() -> dict[str, Any]: "generation_seconds": time.monotonic() - started, } - for _step in range(MAX_TOOL_STEPS): + step_cap = getattr(harness, "max_tool_steps", None) or MAX_TOOL_STEPS + for _step in range(step_cap): remaining = deadline - time.monotonic() if remaining <= 0: return AgentResult( @@ -518,7 +524,11 @@ def _rate() -> dict[str, Any]: return AgentResult( exit_status=1, stdout="\n".join(transcript), - stderr=f"tool-call step cap ({MAX_TOOL_STEPS}) exceeded without a final answer", + stderr=( + f"tool-call step cap ({step_cap}) exceeded without a final answer; " + "raise [harness] max_tool_steps, or split the subtask so each half " + "gets its own budget" + ), timed_out=False, agent_claim=None, **_rate(), From ce1dcb063b59d8e78cd8a3409809cc73b270e519 Mon Sep 17 00:00:00 2001 From: Justin Mclean Date: Thu, 3 Sep 2026 14:00:40 +1000 Subject: [PATCH 2/2] ruff fix --- src/jumar/config.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/src/jumar/config.py b/src/jumar/config.py index f336d73..d2ce086 100644 --- a/src/jumar/config.py +++ b/src/jumar/config.py @@ -178,8 +178,15 @@ def _system_timezone() -> str: # Scalar keys valid at [harness], inside a stage table, and inside a profile. _HARNESS_SCALAR_KEYS: frozenset[str] = frozenset( - {"agent", "model", "base_url", "api_key_env", "reasoning_effort", "max_tokens", - "max_tool_steps"} + { + "agent", + "model", + "base_url", + "api_key_env", + "reasoning_effort", + "max_tokens", + "max_tool_steps", + } ) # Sub-table under [harness] holding named alternative harnesses: