From 9f2077ee4c014e7ce4c9adf3a5e04501496d8ed6 Mon Sep 17 00:00:00 2001 From: AstroHan Date: Sun, 2 Aug 2026 21:22:20 +0800 Subject: [PATCH 1/2] docs: publish Ollama DeepSeek Flash harness comparison --- ...eepseek-v4-flash-0731-maka-vs-opencode.csv | 90 ++++++++++ ...deepseek-v4-flash-0731-maka-vs-opencode.md | 162 ++++++++++++++++++ 2 files changed, 252 insertions(+) create mode 100644 docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.csv create mode 100644 docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.md diff --git a/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.csv b/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.csv new file mode 100644 index 0000000000..4d9ec7678e --- /dev/null +++ b/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.csv @@ -0,0 +1,90 @@ +task_id,maka,opencode,maka_failure_class,opencode_failure_class +adaptive-rejection-sampler,failed,failed,budget_exhausted,verification_failed +bn-fit-modify,passed,failed,,verification_failed +break-filter-js-from-html,passed,failed,,verification_failed +build-cython-ext,passed,passed,, +build-pmars,passed,passed,, +build-pov-ray,passed,passed,, +caffe-cifar-10,passed,failed,,budget_exhausted +cancel-async-tasks,passed,passed,, +chess-best-move,failed,passed,budget_exhausted, +circuit-fibsqrt,passed,failed,,verification_failed +cobol-modernization,failed,failed,budget_exhausted,budget_exhausted +code-from-image,passed,passed,, +compile-compcert,passed,passed,, +configure-git-webserver,passed,failed,,verification_failed +constraints-scheduling,passed,passed,, +count-dataset-tokens,passed,failed,,verification_failed +crack-7z-hash,passed,passed,, +custom-memory-heap-crash,passed,passed,, +db-wal-recovery,failed,failed,budget_exhausted,budget_exhausted +distribution-search,passed,passed,, +dna-assembly,failed,failed,budget_exhausted,verification_failed +dna-insert,failed,failed,verification_failed,verification_failed +extract-elf,failed,failed,budget_exhausted,budget_exhausted +extract-moves-from-video,failed,failed,budget_exhausted,budget_exhausted +feal-differential-cryptanalysis,passed,passed,, +feal-linear-cryptanalysis,failed,failed,budget_exhausted,verification_failed +filter-js-from-html,failed,failed,verification_failed,verification_failed +financial-document-processor,passed,passed,, +fix-code-vulnerability,passed,passed,, +fix-git,passed,passed,, +fix-ocaml-gc,passed,passed,, +gcode-to-text,failed,failed,budget_exhausted,budget_exhausted +git-leak-recovery,passed,passed,, +git-multibranch,passed,passed,, +gpt2-codegolf,failed,failed,budget_exhausted,verification_failed +headless-terminal,passed,failed,,verification_failed +hf-model-inference,passed,passed,, +install-windows-3.11,failed,failed,verification_failed,budget_exhausted +kv-store-grpc,failed,passed,verification_failed, +large-scale-text-editing,passed,passed,, +largest-eigenval,failed,failed,budget_exhausted,verification_failed +llm-inference-batching-scheduler,failed,failed,budget_exhausted,budget_exhausted +log-summary-date-ranges,passed,passed,, +mailman,failed,passed,budget_exhausted, +make-doom-for-mips,failed,failed,budget_exhausted,budget_exhausted +make-mips-interpreter,failed,failed,budget_exhausted,budget_exhausted +mcmc-sampling-stan,passed,failed,,budget_exhausted +merge-diff-arc-agi-task,failed,passed,verification_failed, +model-extraction-relu-logits,failed,failed,budget_exhausted,verification_failed +modernize-scientific-stack,passed,passed,, +mteb-leaderboard,passed,passed,, +mteb-retrieve,passed,failed,,verification_failed +multi-source-data-merger,passed,passed,, +nginx-request-logging,passed,passed,, +openssl-selfsigned-cert,passed,passed,, +overfull-hbox,passed,failed,,verification_failed +password-recovery,passed,passed,, +path-tracing,failed,failed,budget_exhausted,verification_failed +path-tracing-reverse,failed,failed,budget_exhausted,verification_failed +polyglot-c-py,passed,failed,,budget_exhausted +polyglot-rust-c,passed,failed,,verification_failed +portfolio-optimization,passed,passed,, +protein-assembly,failed,failed,verification_failed,verification_failed +prove-plus-comm,passed,passed,, +pypi-server,passed,passed,, +pytorch-model-cli,passed,passed,, +pytorch-model-recovery,passed,passed,, +qemu-alpine-ssh,failed,passed,budget_exhausted, +qemu-startup,passed,failed,,budget_exhausted +query-optimize,passed,failed,,verification_failed +raman-fitting,failed,failed,budget_exhausted,budget_exhausted +regex-chess,failed,failed,max_tokens,verification_failed +regex-log,passed,passed,, +reshard-c4-data,passed,failed,,verification_failed +rstan-to-pystan,failed,failed,budget_exhausted,budget_exhausted +sam-cell-seg,passed,passed,, +sanitize-git-repo,failed,passed,verification_failed, +schemelike-metacircular-eval,failed,failed,budget_exhausted,verification_failed +sparql-university,passed,passed,, +sqlite-db-truncate,passed,failed,,verification_failed +sqlite-with-gcov,failed,passed,verification_failed, +torch-pipeline-parallelism,failed,failed,verification_failed,verification_failed +torch-tensor-parallelism,failed,passed,verification_failed, +train-fasttext,failed,failed,budget_exhausted,budget_exhausted +tune-mjcf,passed,failed,,budget_exhausted +video-processing,failed,failed,budget_exhausted,budget_exhausted +vulnerable-secret,failed,passed,verification_failed, +winning-avg-corewars,passed,failed,,verification_failed +write-compressor,failed,failed,budget_exhausted,verification_failed diff --git a/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.md b/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.md new file mode 100644 index 0000000000..e7c4339417 --- /dev/null +++ b/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.md @@ -0,0 +1,162 @@ +# Terminal-Bench 2.1 — Ollama Cloud DeepSeek V4 Flash 0731: Maka vs OpenCode + +This report compares Maka and OpenCode on all 89 Terminal-Bench 2.1 tasks using Ollama Cloud's `deepseek-v4-flash:0731` model. It also combines the paired outcomes with the [previous DeepSeek V4 Flash run](./terminal-bench-2.1-deepseek-v4-flash-maka-vs-opencode.md) while treating the benchmark task, rather than each repeated observation, as the independent unit. + +**Run id:** `deepseek-v4-flash-0731-maka-vs-opencode-tbench-2.1-full-v1` + +**Local artifacts (git-excluded):** `~/.maka/eval/runs/deepseek-v4-flash-0731-maka-vs-opencode-tbench-2.1-full-v1/` + +**Metric:** end-to-end pass@1 by the official task verifier + +**Score status:** complete — 178/178 cells model-scored, with no accepted infrastructure failure + +**Metering status:** one Maka timeout has a durable usage checkpoint but no exact final usage, so economic totals use the 88 fully metered pairs + +**Per-task outcomes:** [`terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.csv`](./terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.csv) + +## TL;DR + +- **Maka passed 52/89 tasks (58.43%); OpenCode passed 43/89 (48.31%).** Maka led by 9 tasks, or 10.11 percentage points. +- This run alone has 18 Maka-only passes and 9 OpenCode-only passes. Its exact two-sided McNemar p-value is `0.1221`, so this repetition alone does not clear a 5% significance threshold. +- Across this run and the previous run, Maka passed 113/178 observations (63.48%) and OpenCode 92/178 (51.69%), a 21-observation or 11.80-point lead. A task-clustered exact sign-flip test gives **p = 0.00416**; the approximate task-clustered 95% interval for the mean lead is **+4.18 to +19.42 percentage points**. +- The account plan recorded $0 of incremental API cost for both arms. On the 88 fully metered pairs, Maka used 109.98M tokens and OpenCode 79.20M. Maka was more effective, while OpenCode used fewer measured tokens per cell and per successful task. +- Budget exhaustion did not disappear. It affected 25 Maka cells and 18 OpenCode cells in this run, compared with 15 and 24 respectively in the previous run. Provider speed alone therefore did not determine the deadline outcome. + +## Current run + +Budget-exhausted cells remain scored failures in the primary pass@1 denominator. + +| Primary result | Maka | OpenCode | Maka − OpenCode | +| --- | ---: | ---: | ---: | +| End-to-end pass@1 | **52/89 (58.43%)** | **43/89 (48.31%)** | **+9 tasks (+10.11 pp)** | + +The paired outcome table is: + +| | OpenCode pass | OpenCode fail | Total | +| --- | ---: | ---: | ---: | +| Maka pass | 34 | 18 | 52 | +| Maka fail | 9 | 28 | 37 | +| Total | 43 | 46 | 89 | + +For the exact McNemar test, the null assigns equal probability to either direction among the 27 discordant pairs. With 18 Maka-only and 9 OpenCode-only outcomes, the exact two-sided probability is `0.1220781`. The point estimate favors Maka, but this repetition by itself is not statistically conclusive at the conventional 5% threshold. + +## Two-run evidence + +The two runs used the same frozen 89-task suite, task fingerprint, system prompt, reasoning effort, deadline policy, concurrency, and OpenCode version. They used different provider/model identities: the first run used unversioned `deepseek-v4-flash` through DeepSeek, while this run used versioned `deepseek-v4-flash:0731` through Ollama Cloud. The combined result is evidence of repeatability across these two observed conditions, not a claim that the observations are identically distributed. + +| Run | Maka | OpenCode | Maka − OpenCode | Exact paired p | +| --- | ---: | ---: | ---: | ---: | +| DeepSeek V4 Flash | 61/89 (68.54%) | 49/89 (55.06%) | +12 (+13.48 pp) | 0.0118 | +| Ollama Cloud 0731 | 52/89 (58.43%) | 43/89 (48.31%) | +9 (+10.11 pp) | 0.1221 | +| Combined observations | **113/178 (63.48%)** | **92/178 (51.69%)** | **+21 (+11.80 pp)** | — | + +The repeated observations for one task are correlated, so treating all 178 observations as independent pairs would overstate the effective sample size. The combined test instead swaps the Maka/OpenCode labels jointly for both runs within each of the 89 tasks. Its exact two-sided sign-flip probability is `0.00416393`. A task-clustered t interval around the 11.80-point mean difference is approximately `+4.18` to `+19.42` percentage points. + +The direction is also stable at the task level: + +| Outcome across the two runs | Tasks | +| --- | ---: | +| Maka-only pass in both runs | 5 | +| OpenCode-only pass in both runs | 0 | +| Maka-only pass in one run, tie in the other | 21 | +| OpenCode-only pass in one run, tie in the other | 10 | +| Direction reversed between runs | 3 | +| Neither run had a one-sided pass | 50 | + +The two-run result supports the narrower claim that Maka maintained an advantage over OpenCode on this fixed suite under both tested DeepSeek V4 Flash conditions. It does not establish a universal harness ranking. + +## Budget and non-budget diagnostics + +The conditional denominator excludes the entire pair whenever either arm exhausted its budget. It is diagnostic, not an alternate headline score or an unlimited-time counterfactual. + +| Diagnostic | Maka | OpenCode | Maka − OpenCode | +| --- | ---: | ---: | ---: | +| Non-budget Conditional Pass Rate | 47/58 (81.03%) | 40/58 (68.97%) | +12.07 pp | +| Budget Exhaustion Rate | 25/89 (28.09%) | 18/89 (20.22%) | +7.87 pp | + +The previous run's corresponding conditional result was 52/61 versus 46/61, a 6-task lead; this run's lead is 7 tasks. The non-budget gap was therefore similar even though both absolute scores fell. Budget exhaustion shifted in the opposite direction across arms: Maka increased from 15 to 25 exhausted cells, while OpenCode decreased from 24 to 18. + +This run also had 11 ordinary verification failures and one `max_tokens` failure for Maka, versus 28 ordinary verification failures for OpenCode. The lower OpenCode budget-exhaustion count did not translate into a higher pass rate because more of its completed candidates failed the official verifier. + +## Economics + +Ollama Cloud was used through an account plan whose frozen pricing identity records zero incremental USD cost. The result therefore supports a resource-footprint comparison, not a dollar cost-equivalence claim. + +| Fully metered result (88 paired tasks) | Maka | OpenCode | +| --- | ---: | ---: | +| Input tokens | 103,865,919 | 75,546,644 | +| Output tokens | 6,117,410 | 3,651,993 | +| Total tokens | **109,983,329** | **79,198,637** | +| Tokens per metered cell | 1,249,811 | 899,985 | +| Tokens per successful task | 2,115,064 | 1,841,829 | +| Recorded incremental cost | $0 | $0 | + +On this basis, Maka used 38.87% more measured tokens in aggregate and 14.83% more tokens per successful task. That is the tradeoff observed here: Maka produced 20.93% more passes (`52` versus `43`), while OpenCode had the lower token footprint. + +The missing pair is `largest-eigenval`. Its Maka cell reached the 900-second deadline with a durable checkpoint of 217,648 input and 33,796 output tokens (251,444 total), but the outer timeout could have interrupted one additional in-flight request. The harness intentionally does not promote that checkpoint to exact final usage. Both arms failed this task, so the metering gap does not change either pass count or the primary result. + +The previous run's API-equivalent cost per pass was $0.031718 for Maka and $0.031712 for OpenCode. Those dollar estimates should not be pooled with this account-plan run, and token totals should not be compared as if provider cache reporting were identical. + +## Frozen setup + +| Dimension | Value | +| --- | --- | +| Benchmark | Terminal-Bench 2.1, revision `d49e28f1e4ddd13d289e85a5f312a66750951932`; all 89 tasks | +| Task-tree fingerprint | `sha256:456826aa4c47ed309716c964c96d2a3acc998764ebc84f3e8449c807d74bd4e7` | +| Run fingerprint | `sha256:cf4e14cc32fa95bb2c39ce791ca450d01c9b25fdcc39ccc3bcf756638179ff94` | +| Model | `deepseek-v4-flash:0731` through Ollama Cloud on both arms | +| Reasoning effort | `max` on both arms | +| Repetitions | 1 | +| Metric | Paired pass@1 | +| Attempt policy | One accepted model attempt per arm/task; only pre-execution infrastructure-invalid admissions may be replaced | +| Deadline policy | Task-native agent timeout ×1; 900-second outer setup and teardown grace | +| Pair execution | Up to four task pairs concurrently; Maka and OpenCode start in parallel within a pair; at most eight cells concurrently | +| External system prompt | Empty on both arms | +| Maka arm | `maka_agent:MakaAgent`; continuation off; active and stale tool-result pruning enabled at a 2,048 estimated-token threshold; semantic compact off | +| OpenCode arm | `opencode_agent:MakaOpenCodeAgent` 1.17.18; pure mode; automatic permissions; `max` variant | +| Billing mode | Account plan; frozen incremental token prices are zero | + +This is a same-model harness comparison, not a same-system/same-tool ablation. The two arms retain their native instructions, tools, context management, and execution loops. The observed difference belongs to the compared harness systems as a whole. + +## Outcome and infrastructure audit + +The controller WAL contains 188 admissions. The first 10 were the five pilot task pairs attempted while Docker was unavailable; they failed before model execution, were superseded once under the authorized infrastructure-retry policy, and do not enter pass@1. The accepted dataset contains exactly one model-scored outcome for every arm/task cell. + +All 178 final Harbor trials contain an official verifier reward: 95 pass and 83 fail. Harbor recorded 21 `AgentTimeoutError` exceptions and no other final trial exception type; every timeout still reached the official verifier, and two timeout trials passed. The accepted controller projection contains 43 budget-exhausted failures because it also recognizes agent-runtime deadline evidence that does not surface as a Harbor exception. + +There are 176 CTRF reports containing 608 verifier test cases: 424 passed and 184 failed, with no skipped, pending, `other`, or error-status test. The two trials without CTRF are Maka's `merge-diff-arc-agi-task` and `sqlite-with-gcov`. In both traces, the agent manually installed cached packages with `dpkg`, left the package database in a broken dependency state, and the unchanged verifier then failed to install its prerequisites. The corresponding OpenCode trial in the same task image reached the official tests and passed. These are agent-caused end-to-end failures, not external infrastructure failures. + +Maka's `write-compressor` verifier had one setup-classified attempt followed automatically by a second verifier attempt with a determinate failure; its final reward is 0. Other apparent infrastructure strings in verifier logs were checked against the task and agent traces. They resolve to missing candidate services or files, candidate correctness failures, expected browser/test behavior, or warnings after successful downloads rather than external setup failures. + +OpenCode's proxy telemetry contains 2,392 requests: 2,382 completed, 8 aborted at task deadlines, and 2 failed with HTTP 410. Both 410s occurred on the first request. `compile-compcert` then completed 29 later requests and passed; `rstan-to-pystan` completed 41 later requests before exhausting its task budget. Neither transient provider response invalidates the accepted outcome. + +The generated harness report is `completed_with_gaps` and the background wrapper exits 1 only because the strict completion assertion requires exact final usage for every cell. This is a metering gap, not a score or infrastructure gap. + +## Caveats + +- Each run is one repetition over a fixed suite. The exact tests describe outcome asymmetry on these tasks and do not guarantee performance on another benchmark or task distribution. +- The task-clustered combined test preserves dependence between repeated outcomes for the same task, but the two provider/model conditions are not identical. The combined inference is about these two observed runs. +- Non-budget Conditional Pass Rate is selection-conditional. It cannot be interpreted as an unlimited-time result. +- Budget exhaustion does not isolate provider latency, generation length, tool time, or agent policy. Ollama Cloud's high generation throughput did not guarantee shorter end-to-end trajectories. +- Account-plan $0 is the recorded incremental price identity, not the subscription's total cost. The report does not assign a hypothetical API-equivalent price to this model. +- No external Oracle registry snapshot was configured. The official Terminal-Bench verifier remains the scoring authority. + +## Integrity + +SHA-256 hashes of the frozen local evidence and committed outcome projection: + +| Source | SHA-256 | +| --- | --- | +| `harness-ab-manifest.json` | `f9360431fe82fd95cf61d23d8011cb790054f90335f4ce635b9c687aa22bb591` | +| `harness-ab-report.json` | `e4f48bf81d3300a3841d0c3a98e36462f679130d9d960f295605af357203ef57` | +| `controller/results.jsonl` | `f6a4b6ea073b3b558fac1f376f3383b7e057bf11fb23a7c8833612e05ceb087b` | +| `controller/results.jsonl.attempts.jsonl` | `ad6932e5f1eeef7fee9127aea051efe64ad243a008da816eac4305174604fb1c` | +| Committed outcome CSV | `7cf3abe68c23406228be7ff59b447d6842355ed8a0e92d54379ffba8ba79f2ba` | + +## Artifact pointers + +| Artifact | Local path | +| --- | --- | +| Generated report | `~/.maka/eval/runs/deepseek-v4-flash-0731-maka-vs-opencode-tbench-2.1-full-v1/harness-ab-report.{json,csv,md}` | +| Immutable manifest | `~/.maka/eval/runs/deepseek-v4-flash-0731-maka-vs-opencode-tbench-2.1-full-v1/harness-ab-manifest.json` | +| Controller WAL | `.../controller/results.jsonl` and `.../controller/results.jsonl.attempts.jsonl` | From c03e993e022d5a6207c77c81389c0cf96f83833d Mon Sep 17 00:00:00 2001 From: AstroHan Date: Fri, 4 Sep 2026 15:07:00 +0800 Subject: [PATCH 2/2] docs: add the ASF license header to the Ollama comparison report Every other report under docs/eval carries the header; this one was published without it. Generated-by: Claude Code --- ...deepseek-v4-flash-0731-maka-vs-opencode.md | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.md b/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.md index e7c4339417..4676cbd25c 100644 --- a/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.md +++ b/docs/eval/terminal-bench-2.1-ollama-deepseek-v4-flash-0731-maka-vs-opencode.md @@ -1,3 +1,22 @@ + + # Terminal-Bench 2.1 — Ollama Cloud DeepSeek V4 Flash 0731: Maka vs OpenCode This report compares Maka and OpenCode on all 89 Terminal-Bench 2.1 tasks using Ollama Cloud's `deepseek-v4-flash:0731` model. It also combines the paired outcomes with the [previous DeepSeek V4 Flash run](./terminal-bench-2.1-deepseek-v4-flash-maka-vs-opencode.md) while treating the benchmark task, rather than each repeated observation, as the independent unit.