From 9a1f6c52d0ac8a20f259ac959d7f874b05070c1a Mon Sep 17 00:00:00 2001 From: shangzh0 <2586756592@qq.com> Date: Sat, 19 Sep 2026 00:20:45 +0800 Subject: [PATCH] fix(site): publish the LHTB brief route Signed-off-by: shangzh0 <2586756592@qq.com> --- README.md | 8 ++++-- README.zh-CN.md | 7 +++-- apps/presentation/site/src/App.tsx | 3 ++ benchmark/README.md | 3 ++ benchmark/tests/test_publication_scope.py | 28 +++++++++++++++++++ ...board-frontstage-design-baseline-smoke.mjs | 2 +- examples/export-frontstage-share-bundle.mjs | 7 +++++ examples/frontstage-share-bundle-smoke.mjs | 11 ++++++++ 8 files changed, 64 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 8c21799ea7..7ff1792700 100644 --- a/README.md +++ b/README.md @@ -229,12 +229,16 @@ creator dogfooding, reproducible demos, and explicit evidence-strength labels. - **[SWE-Marathon](https://huangruiteng.github.io/loopx/benchmarks/swe-marathon/):** Five execution modes on 15 matched tasks compare self-verification, scores, and cost. More self-verification did not consistently yield higher scores. +- **[LHTB × LoopX](https://huangruiteng.github.io/loopx/benchmarks/lhtb/):** + Five execution mechanisms on 46 long-horizon terminal tasks compare durable + state, bounded Todos, replanning, and fresh executor sessions. - **[DeepSWE behavior analysis](https://huangruiteng.github.io/loopx/benchmarks/deepswe/behavior-discovery/)** (Chinese): Selected cases examine how domain hints relate to requirement retention and verification choices, offering mechanism hypotheses for further testing. -SWE-Marathon has one trial per task and mode; DeepSWE uses selected cases and -post-hoc analysis. Neither establishes a general performance gain. +SWE-Marathon and LHTB have one effective trial per task and mode; DeepSWE uses +selected cases and post-hoc analysis. None establishes a general performance +gain. More inspectable surfaces: diff --git a/README.zh-CN.md b/README.zh-CN.md index 254e5a0670..13133bd1ce 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -200,11 +200,14 @@ creator dogfooding、reproducible demo 和证据强度标签见 - **[SWE-Marathon](https://huangruiteng.github.io/loopx/benchmarks/swe-marathon/?lang=zh)**: 在相同的 15 个任务上对照 5 种执行模式,比较自验证行为、得分与成本。 更多自验证并未稳定转化为更高得分。 +- **[LHTB × LoopX](https://huangruiteng.github.io/loopx/benchmarks/lhtb/?lang=zh)**: + 在 46 个长程终端任务上对比 5 种执行机制,研究持久状态、Todo、replan + 与 fresh executor session。 - **[DeepSWE 行为分析](https://huangruiteng.github.io/loopx/benchmarks/deepswe/behavior-discovery/)**: 通过精选案例观察领域提示、需求保留与验证选择之间的关系,提出有待复验的机制假设。 -SWE-Marathon 每个任务、每种模式仅运行一次;DeepSWE 包含精选案例与事后分析。 -目前两者均不足以证明普遍的性能提升。 +SWE-Marathon 与 LHTB 每个任务、每种模式仅保留一条有效轨迹;DeepSWE 包含精选案例与 +事后分析。目前三者均不足以证明普遍的性能提升。 更多可检查入口: diff --git a/apps/presentation/site/src/App.tsx b/apps/presentation/site/src/App.tsx index cde5a94e9f..6d8f988407 100644 --- a/apps/presentation/site/src/App.tsx +++ b/apps/presentation/site/src/App.tsx @@ -140,6 +140,7 @@ const content = { cards: [ ["Product demo", "Personal Workspace", "A shared view of Goals, Tasks, Chat, and outputs. Watch the demo and start your own local workspace.", "Watch demo & read guide"], ["Research & evaluation", "SWE-Marathon", "Compare three retained execution modes across 15 matched tasks, with results, costs, and study limitations.", "Read the study"], + ["Research & evaluation", "LHTB × LoopX", "Compare five execution mechanisms across 46 long-horizon terminal tasks and inspect where recoverable state helps.", "Read the study"], ["Research · Chinese", "DeepSWE behavior discoveries", "Explore how domain hints affect implementation and validation in individual cases.", "Read the behavior analysis"], ["Research · Chinese", "DeepSWE × Sol", "Read the historical 113-task results, continuation mechanisms, and patch-delivery checks.", "Read the research brief"], ["Cases", "All LoopX showcases", "Browse public cases, interactive walkthroughs, and their evidence boundaries.", "Browse all cases"], @@ -254,6 +255,7 @@ const content = { cards: [ ["产品演示", "Personal Workspace", "在一个工作区查看目标、任务、对话与产出。观看演示,开始使用本地工作区。", "观看演示与使用指南"], ["研究与评测", "SWE-Marathon", "查看三种保留模式在 15 个匹配任务上的结果、成本与研究局限。", "阅读研究简报"], + ["研究与评测", "LHTB × LoopX", "对比五种执行机制在 46 个长程终端任务上的结果,观察可恢复状态在什么场景有效。", "阅读研究简报"], ["研究与评测", "DeepSWE 行为分析", "从具体案例观察领域提示如何影响实现选择与验证行为。", "阅读行为分析"], ["研究与评测", "DeepSWE × Sol", "解读 113 任务的历史结果、持续执行机制,以及补丁如何进入最终验收。", "阅读研究简报"], ["案例", "完整案例目录", "浏览公开案例、交互式讲解及其证据边界。", "浏览全部案例"], @@ -1011,6 +1013,7 @@ export function App() { const paths = [ "docs/guides/personal-workspace-user-guide/", `benchmarks/swe-marathon/${language === "zh" ? "?lang=zh" : ""}`, + `benchmarks/lhtb/${language === "zh" ? "?lang=zh" : ""}`, "benchmarks/deepswe/behavior-discovery/", "benchmarks/deepswe-sol/", `docs/showcases/index${language === "en" ? ".en" : ""}.html`, diff --git a/benchmark/README.md b/benchmark/README.md index 10bb3271e2..ab592c5f70 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -48,6 +48,9 @@ by itself establish a C2 uplift claim. - [`swe-marathon/README.md`](swe-marathon/README.md) links the published [SWE-Marathon research brief](https://huangruiteng.github.io/loopx/benchmarks/swe-marathon/). +- [`LHTB/studies/five-arm-gpt56sol-max/README.md`](LHTB/studies/five-arm-gpt56sol-max/README.md) + documents the public-safe aggregate behind the bilingual + [LHTB research brief](https://huangruiteng.github.io/loopx/benchmarks/lhtb/). - [`deepswe/behavior-discovery/README.md`](deepswe/behavior-discovery/README.md) links the standalone [DeepSWE behavior-discovery article](https://huangruiteng.github.io/loopx/benchmarks/deepswe/behavior-discovery/). diff --git a/benchmark/tests/test_publication_scope.py b/benchmark/tests/test_publication_scope.py index af4c41eaf9..67fed33749 100644 --- a/benchmark/tests/test_publication_scope.py +++ b/benchmark/tests/test_publication_scope.py @@ -80,3 +80,31 @@ def test_published_data_and_bilingual_tables_share_scope(exporters): for localized in copy.values(): assert {row[0] for row in localized["armRows"]} == allowed assert set(localized["executiveReads"]) == allowed + + +def test_lhtb_published_data_and_bilingual_copy_share_scope(): + study = STUDY.parent / "LHTB" / "studies" / "five-arm-gpt56sol-max" + data = json.loads((study / "data.json").read_text()) + arms = { + "plain", + "native_goal", + "ssh_goal", + "legacy_heartbeat", + "new_heartbeat", + } + assert set(data["arms"]) == arms + assert len(data["tasks"]) == len({row["task"] for row in data["tasks"]}) == 46 + assert all(set(row) == arms | {"task"} for row in data["tasks"]) + + for arm, summary in data["arms"].items(): + rewards = [row[arm] for row in data["tasks"]] + assert summary["mean_reward"] == pytest.approx(sum(rewards) / 46, rel=0, abs=1e-12) + assert summary["pass_095"] == sum(reward >= 0.95 for reward in rewards) + + site = STUDY.parents[1] / "apps/presentation/site/src" + localized_copy = json.loads((site / "lhtb-copy.json").read_text()) + assert set(localized_copy) == {"en", "zh"} + for localized in localized_copy.values(): + assert set(localized["armLabels"]) == arms + assert set(localized["armKinds"]) == arms + assert {row[0] for row in localized["mechanismRows"]} == arms diff --git a/examples/dashboard-frontstage-design-baseline-smoke.mjs b/examples/dashboard-frontstage-design-baseline-smoke.mjs index 03a2a1edae..8a96bcaae7 100644 --- a/examples/dashboard-frontstage-design-baseline-smoke.mjs +++ b/examples/dashboard-frontstage-design-baseline-smoke.mjs @@ -6,7 +6,7 @@ const read = (path) => readFileSync(fileURLToPath(new URL(`../${path}`, import.m const home = read("apps/presentation/site/src/App.tsx"); const styles = read("apps/presentation/site/src/styles.css"); assert.doesNotMatch(home, /href=.[^\n]*(?:deprecated|frontstage\/)/, "homepage must not promote retired surfaces"); -for (const destination of ["docs/guides/personal-workspace-user-guide/", "benchmarks/swe-marathon/", "benchmarks/deepswe/behavior-discovery/", "benchmarks/deepswe-sol/", "docs/showcases/index", "developers/projections/"]) { +for (const destination of ["docs/guides/personal-workspace-user-guide/", "benchmarks/swe-marathon/", "benchmarks/lhtb/", "benchmarks/deepswe/behavior-discovery/", "benchmarks/deepswe-sol/", "docs/showcases/index", "developers/projections/"]) { assert.ok(home.includes(destination), `missing public destination: ${destination}`); } assert.ok(styles.includes("prefers-reduced-motion")); diff --git a/examples/export-frontstage-share-bundle.mjs b/examples/export-frontstage-share-bundle.mjs index e01d694da9..020db0dfae 100644 --- a/examples/export-frontstage-share-bundle.mjs +++ b/examples/export-frontstage-share-bundle.mjs @@ -268,6 +268,7 @@ async function writeShareReadme(outDir, base, interactivePages) { const homepageUrl = base; const frontstageUrl = `${base}frontstage/`; const sweMarathonBriefUrl = `${base}benchmarks/swe-marathon/`; + const lhtbBriefUrl = `${base}benchmarks/lhtb/`; const deepSweBehaviorArticleUrl = `${base}benchmarks/deepswe/behavior-discovery/`; const previewBlock = base === "/" ? `## Try It Locally @@ -283,6 +284,7 @@ Then open the homepage or showcase: http://127.0.0.1:8080${homepageUrl} http://127.0.0.1:8080${frontstageUrl} http://127.0.0.1:8080${sweMarathonBriefUrl} +http://127.0.0.1:8080${lhtbBriefUrl} http://127.0.0.1:8080${deepSweBehaviorArticleUrl} \`\`\` ` @@ -298,6 +300,7 @@ Hosted entries: ${homepageUrl} ${frontstageUrl} ${sweMarathonBriefUrl} +${lhtbBriefUrl} ${deepSweBehaviorArticleUrl} \`\`\` `; @@ -322,6 +325,7 @@ ${previewBlock} catalog-declared interactive case pages. - Homepage source: \`apps/presentation/site\`. - SWE-Marathon research brief: \`${sweMarathonBriefUrl}\`, built from the pinned public-safe aggregate and case-insight projection under \`benchmark/swe-marathon/\`. +- LHTB research brief: \`${lhtbBriefUrl}\`, built from the public-safe five-arm aggregate under \`benchmark/LHTB/studies/five-arm-gpt56sol-max/\`. - DeepSWE behavior discoveries: \`${deepSweBehaviorArticleUrl}\`, copied byte-for-byte from the reviewed standalone article at \`${deepSweBehaviorArticlePath}\`. - DeepSWE × Sol research brief: \`${base}benchmarks/deepswe-sol/\`, a static historical-study interpretation from \`${deepSweSolArticlePath}\`. - Homepage evidence assets: ${homepageEvidenceAssets.map((path) => `\`${path}\``).join(", ")}. @@ -345,6 +349,7 @@ async function writeManifest(outDir, base, interactivePages) { status_fixture: `site/${statusFileName}`, homepage_entry: "site/index.html", swe_marathon_brief_entry: "site/benchmarks/swe-marathon/index.html", + lhtb_brief_entry: "site/benchmarks/lhtb/index.html", deepswe_behavior_article_entry: "site/benchmarks/deepswe/behavior-discovery/index.html", deepswe_sol_article_entry: "site/benchmarks/deepswe-sol/index.html", installer_entry: "site/install.sh", @@ -354,6 +359,7 @@ async function writeManifest(outDir, base, interactivePages) { content_sources: { public_homepage: "apps/presentation/site", swe_marathon_brief: "benchmark/swe-marathon", + lhtb_brief: "benchmark/LHTB/studies/five-arm-gpt56sol-max", deepswe_behavior_article: deepSweBehaviorArticlePath, deepswe_sol_article: deepSweSolArticlePath, installer_script: installerScriptPath, @@ -474,6 +480,7 @@ async function main() { site_dir: siteDir, homepage_url: args.base, swe_marathon_brief_url: `${args.base}benchmarks/swe-marathon/`, + lhtb_brief_url: `${args.base}benchmarks/lhtb/`, deepswe_behavior_article_url: `${args.base}benchmarks/deepswe/behavior-discovery/`, frontstage_url: `${args.base}frontstage/`, status_fixture: `site/${statusFileName}`, diff --git a/examples/frontstage-share-bundle-smoke.mjs b/examples/frontstage-share-bundle-smoke.mjs index 5d3a0c642f..b9789a2ce1 100644 --- a/examples/frontstage-share-bundle-smoke.mjs +++ b/examples/frontstage-share-bundle-smoke.mjs @@ -234,6 +234,10 @@ const benchmarkHtml = await readFile(resolve(siteDir, "benchmarks/swe-marathon/i if (benchmarkHtml !== homepageHtml) { throw new Error("SWE-Marathon static route must reuse the compiled public-site entry"); } +const lhtbHtml = await readFile(resolve(siteDir, "benchmarks/lhtb/index.html"), "utf8"); +if (lhtbHtml !== homepageHtml) { + throw new Error("LHTB static route must reuse the compiled public-site entry"); +} const deepSweBehaviorHtml = await readFile( resolve(siteDir, "benchmarks/deepswe/behavior-discovery/index.html"), "utf8", @@ -367,6 +371,7 @@ if (manifest.base !== "/loopx/") { if ( manifest.homepage_entry !== "site/index.html" || manifest.swe_marathon_brief_entry !== "site/benchmarks/swe-marathon/index.html" || + manifest.lhtb_brief_entry !== "site/benchmarks/lhtb/index.html" || manifest.deepswe_behavior_article_entry !== "site/benchmarks/deepswe/behavior-discovery/index.html" || manifest.frontstage_entry !== "site/frontstage/index.html" || manifest.installer_entry !== "site/install.sh" @@ -379,6 +384,9 @@ if (manifest.content_sources?.public_homepage !== "apps/presentation/site") { if (manifest.content_sources?.swe_marathon_brief !== "benchmark/swe-marathon") { throw new Error(`manifest benchmark brief source mismatch: ${JSON.stringify(manifest.content_sources)}`); } +if (manifest.content_sources?.lhtb_brief !== "benchmark/LHTB/studies/five-arm-gpt56sol-max") { + throw new Error(`manifest LHTB brief source mismatch: ${JSON.stringify(manifest.content_sources)}`); +} if ( manifest.content_sources?.deepswe_behavior_article !== "benchmark/deepswe/behavior-discovery/index.html" @@ -458,6 +466,9 @@ if (!readmeText.includes("frontstage/")) { if (!readmeText.includes("benchmarks/swe-marathon/")) { throw new Error("share bundle README must publish the SWE-Marathon research brief entry"); } +if (!readmeText.includes("benchmarks/lhtb/")) { + throw new Error("share bundle README must publish the LHTB research brief entry"); +} if (!readmeText.includes("benchmarks/deepswe/behavior-discovery/")) { throw new Error("share bundle README must publish the DeepSWE behavior article entry"); }