Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
78 changes: 57 additions & 21 deletions src/jumar/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -178,7 +178,15 @@ def _system_timezone() -> str:

# Scalar keys valid at [harness], inside a stage table, and inside a profile.
_HARNESS_SCALAR_KEYS: frozenset[str] = frozenset(
{"agent", "model", "base_url", "api_key_env", "reasoning_effort", "max_tokens"}
{
"agent",
"model",
"base_url",
"api_key_env",
"reasoning_effort",
"max_tokens",
"max_tool_steps",
}
)

# Sub-table under [harness] holding named alternative harnesses:
Expand Down Expand Up @@ -227,6 +235,12 @@ class HarnessConfig:
context window, so one runaway response can consume an entire subtask
deadline and take the whole attempt down with it.

``max_tool_steps`` bounds how many tool calls one execute attempt may make
before the harness gives up. The default of 20 was set for report-writing
items; a task that has to extract, compile and then summarise runs out of
steps mid-way and returns no answer at all, having done most of the work.
Raise it for items that genuinely need a longer tool loop.

``None`` for a per-stage field means "inherit the top-level value".

A ``HarnessConfig`` is always fully resolved by the time it reaches this
Expand All @@ -241,25 +255,29 @@ class HarnessConfig:
api_key_env: str | None = None
reasoning_effort: str | None = None
max_tokens: int | None = None
max_tool_steps: int | None = None
# Per-stage overrides — None inherits the top-level value.
decompose_agent: str | None = None
decompose_model: str | None = None
decompose_base_url: str | None = None
decompose_api_key_env: str | None = None
decompose_reasoning_effort: str | None = None
decompose_max_tokens: int | None = None
decompose_max_tool_steps: int | None = None
execute_agent: str | None = None
execute_model: str | None = None
execute_base_url: str | None = None
execute_api_key_env: str | None = None
execute_reasoning_effort: str | None = None
execute_max_tokens: int | None = None
execute_max_tool_steps: int | None = None
judge_agent: str | None = None
judge_model: str | None = None
judge_base_url: str | None = None
judge_api_key_env: str | None = None
judge_reasoning_effort: str | None = None
judge_max_tokens: int | None = None
judge_max_tool_steps: int | None = None

def for_stage(self, stage: str) -> HarnessConfig:
"""Return the resolved ``HarnessConfig`` for *stage*.
Expand All @@ -278,13 +296,15 @@ def for_stage(self, stage: str) -> HarnessConfig:
k: str | None = getattr(self, f"{stage}_api_key_env")
r: str | None = getattr(self, f"{stage}_reasoning_effort")
x: int | None = getattr(self, f"{stage}_max_tokens")
n: int | None = getattr(self, f"{stage}_max_tool_steps")
return HarnessConfig(
agent=a if a is not None else self.agent,
model=m if m is not None else self.model,
base_url=b if b is not None else self.base_url,
api_key_env=k if k is not None else self.api_key_env,
reasoning_effort=r if r is not None else self.reasoning_effort,
max_tokens=x if x is not None else self.max_tokens,
max_tool_steps=n if n is not None else self.max_tool_steps,
)


Expand Down Expand Up @@ -548,9 +568,10 @@ def _finalise(resolved: HarnessConfig, label: str) -> HarnessConfig:
just the selected one, so a typo in an unused profile fails the next
load rather than the next scheduled run that happens to select it.

max_tokens arrives as a string because the layering above resolves
every scalar uniformly; it is coerced back to int here, which is also
where a non-numeric or non-positive value is rejected.
max_tokens and max_tool_steps arrive as strings because the layering
above resolves every scalar uniformly; both are coerced back to int
here, which is also where a non-numeric or non-positive value is
rejected.
"""
stage_attrs = (f"{s}_reasoning_effort" for s in sorted(_VALID_HARNESS_STAGES))
for attr in ("reasoning_effort", *stage_attrs):
Expand All @@ -564,24 +585,30 @@ def _finalise(resolved: HarnessConfig, label: str) -> HarnessConfig:

# Any, not int: the values land in fields the dataclass types
# int | None, and **kwargs into dc_replace cannot be narrowed per key.
#
# Both integer keys are coerced here, not just max_tokens. A stage
# table's max_tool_steps reached the agent loop as the string "35" and
# `range("35")` raises TypeError, so the per-stage path was as broken
# as the top-level one, in a noisier way.
coerced: dict[str, Any] = {}
token_attrs = (f"{s}_max_tokens" for s in sorted(_VALID_HARNESS_STAGES))
for attr in ("max_tokens", *token_attrs):
value = getattr(resolved, attr)
if value is None:
continue
where = label if attr == "max_tokens" else f"{label}.{attr.split('_')[0]}"
try:
as_int = int(str(value))
except ValueError:
raise ConfigError(
f"Invalid max_tokens {value!r} under [{where}]: expected a positive integer."
) from None
if as_int <= 0:
raise ConfigError(
f"Invalid max_tokens {as_int} under [{where}]: expected a positive integer."
)
coerced[attr] = as_int
for key in ("max_tokens", "max_tool_steps"):
int_attrs = (f"{s}_{key}" for s in sorted(_VALID_HARNESS_STAGES))
for attr in (key, *int_attrs):
value = getattr(resolved, attr)
if value is None:
continue
where = label if attr == key else f"{label}.{attr.split('_')[0]}"
try:
as_int = int(str(value))
except ValueError:
raise ConfigError(
f"Invalid {key} {value!r} under [{where}]: expected a positive integer."
) from None
if as_int <= 0:
raise ConfigError(
f"Invalid {key} {as_int} under [{where}]: expected a positive integer."
)
coerced[attr] = as_int
return dc_replace(resolved, **coerced) if coerced else resolved

# Resolution order, highest first:
Expand Down Expand Up @@ -629,6 +656,15 @@ def _build(profile_table: dict[str, Any], profile_name: str | None) -> HarnessCo
_scalar(profile_table, "max_tokens"),
_scalar(harness_raw, "max_tokens"),
), # type: ignore[arg-type]
# Same treatment as max_tokens. Omitting this line is why a
# top-level `max_tool_steps` was accepted by the key check above,
# then silently dropped: the field stayed None and the agent loop
# fell back to MAX_TOOL_STEPS, so a run reported a cap of 20 while
# the file said 35.
max_tool_steps=_first(
_scalar(profile_table, "max_tool_steps"),
_scalar(harness_raw, "max_tool_steps"),
), # type: ignore[arg-type]
**stage_kwargs,
)

Expand Down
14 changes: 12 additions & 2 deletions src/jumar/openai_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,11 @@
# stops calling tools (or loops between two of them) must not hang the run —
# it trips this cap and the subtask is recorded as a failed attempt, exactly
# like a subprocess harness that exits non-zero.
# Default ceiling on tool calls in one attempt. Overridable per stage with
# `max_tool_steps` in jumar.toml: on 2026-08-29 a compile-and-report item hit
# this cap three times, each attempt extracting all 14 snippets and running
# cargo before running out of steps with no report written — the work was
# done and thrown away for want of a few more turns.
MAX_TOOL_STEPS = 20

# Every request gets at least this many seconds even if the overall deadline
Expand Down Expand Up @@ -372,7 +377,8 @@ def _rate() -> dict[str, Any]:
"generation_seconds": time.monotonic() - started,
}

for _step in range(MAX_TOOL_STEPS):
step_cap = getattr(harness, "max_tool_steps", None) or MAX_TOOL_STEPS
for _step in range(step_cap):
remaining = deadline - time.monotonic()
if remaining <= 0:
return AgentResult(
Expand Down Expand Up @@ -518,7 +524,11 @@ def _rate() -> dict[str, Any]:
return AgentResult(
exit_status=1,
stdout="\n".join(transcript),
stderr=f"tool-call step cap ({MAX_TOOL_STEPS}) exceeded without a final answer",
stderr=(
f"tool-call step cap ({step_cap}) exceeded without a final answer; "
"raise [harness] max_tool_steps, or split the subtask so each half "
"gets its own budget"
),
timed_out=False,
agent_claim=None,
**_rate(),
Expand Down
Loading