From eb850a2887aca7c05aef0a4b0aaab26662e229f6 Mon Sep 17 00:00:00 2001
From: Ylr9933 <2835703479@qq.com>
Date: Wed, 23 Sep 2026 01:46:30 +0800
Subject: [PATCH] Add Terminal-Bench adapter (terminal-bench-core==0.1.1)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Terminal-Bench (laude-institute/terminal-bench) is the docker-per-task
resolution benchmark. Adds it as a Benchmark adapter in
experiments/benchmark/terminal_bench/, following alfworld's shape.
Version lock — terminal-bench-core==0.1.1 (registry.json published entry):
commit 91e10457b5410f16c44364da1a34cb6de8c488a5
branch dataset/terminal-bench-core/v0.1.x
task_id_subset 80 ids (.base/.easy/.hard variants) -> 70 unique task dirs
(repo has no git tags; PyPI terminal-bench-core 404; 0.1.1 is the latest
registerable release. main HEAD drifts and is not reproducible, so not used.)
Adapter surface:
- tasks(): 70 curated dirs; SUBSET_BASES embedded as a module tuple
(clone-only source, no dataset-side file). Sampling rewoo_draw(seed).
- on_task(): builds + runs the per-task container (base image
ghcr.io/laude-institute/t-bench/python-3-13, multi-arch); mounts
run-tests.sh + tests/ into /app:ro; the lone `exec` tool docker-exec's in.
- score(): runs the official run-tests.sh (which runs `uv run pytest tests/`)
in the agent-modified container end-state; correctness = pytest pass
(the dataset's own judge, not a reimplementation).
- downloads(): pins the codeload tarball @ 91e10457 with the registry note.
- Registered in run.py BENCHMARK_FACTORIES; DOWNLOADS.md regenerated to cover it.
- .gitignore: ignore .venv/.
Gates:
- `python -m experiments.scripts.datasets --write` ok
- `python -m experiments.scripts.check`: 16 benches · 10 arms · 0 runs, all pass
- docker smoke on this arm64 mac: base image builds natively in 27s (no
emulation); exec + mounted run-tests path works; end-to-end
react × DeepSeek 1 task finishes through official pytest judging.
Known limits:
- ~10 heavy tasks (qemu kernel / hf-model / pytorch-model-cli ...) may
not build on arm64 mac; recorded as env_unavailable. ~60/70 run here;
all 70 should run on x86 linux.
- per-task ~4-7 min (apt/uv install pytest inside score can't be cached);
70 tasks ≈ 3-6 h.
- .easy/.hard variants not split; each unique dir runs once.
---
.gitignore | 3 +
.../benchmark/terminal_bench/__init__.py | 7 +
.../terminal_bench/terminal_bench.py | 338 ++++++++++++++++++
experiments/dataset/DOWNLOADS.md | 1 +
experiments/scripts/run.py | 3 +
5 files changed, 352 insertions(+)
create mode 100644 experiments/benchmark/terminal_bench/__init__.py
create mode 100644 experiments/benchmark/terminal_bench/terminal_bench.py
diff --git a/.gitignore b/.gitignore
index ef477b7..c90fd7a 100644
--- a/.gitignore
+++ b/.gitignore
@@ -39,6 +39,9 @@ __pycache__/
.pytest_cache/
.ruff_cache/
+# 虚拟环境(每人本地建、不进仓库)
+.venv/
+
# ★ 论文插图:**版权不属于我们,不进公开仓库。**
# arXiv 的授权因文而异(有的 CC-BY 可以带署名转载,有的不是),
# 而这些图是**我们的读书材料**,不是代码的一部分。
diff --git a/experiments/benchmark/terminal_bench/__init__.py b/experiments/benchmark/terminal_bench/__init__.py
new file mode 100644
index 0000000..7ff532a
--- /dev/null
+++ b/experiments/benchmark/terminal_bench/__init__.py
@@ -0,0 +1,7 @@
+from experiments.benchmark.terminal_bench.terminal_bench import ( # noqa: F401
+ NeedsDocker,
+ TB_COMMIT,
+ TerminalBench,
+)
+
+__all__ = ["TerminalBench", "NeedsDocker", "TB_COMMIT"]
diff --git a/experiments/benchmark/terminal_bench/terminal_bench.py b/experiments/benchmark/terminal_bench/terminal_bench.py
new file mode 100644
index 0000000..8802923
--- /dev/null
+++ b/experiments/benchmark/terminal_bench/terminal_bench.py
@@ -0,0 +1,338 @@
+"""Terminal-Bench —— **每题一个 Docker 容器环境,官方 pytest 判分。**
+
+## 出处
+
+| | |
+|---|---|
+| **论文** | *Terminal-Bench: A Benchmark for End-to-End Tasks in Real Terminal Environments*. arXiv:2601.11868 |
+| **仓库** | `github.com/laude-institute/terminal-bench`(HEAD 没固定;改用 registry 给的 lock 版本)|
+| **dataset** | **`terminal-bench-core==0.1.1`**(registry.json publish 的 latest release entry)
commit `91e10457b5410f16c44364da1a34cb6de8c488a5`
branch `dataset/terminal-bench-core/v0.1.x`
`task_id_subset` 80 个 id(含 `.easy`/`.hard` 变体) → 去点 **70 个 unique task dir**
每 task: `task.yaml` + `Dockerfile` + `docker-compose.yaml` + `run-tests.sh` + `tests/` |
+
+## ★ 这是个**交互式容器环境**,形状和 ALFWorld 那类一样
+
+每题是一个 docker 容器(`command: sleep infinity`),agent 通过 `docker exec` 在容器里跑命令来「做题」,
+判分是**在 agent 改过的容器终态上跑官方 `run-tests.sh`**(它跑 `uv run pytest tests/`)。
+「对不对」是 **pytest 的通过/失败** —— 这是官方 judge,不自己写判分器。
+
+## ★ 判分口径(照学长群里第 ② 条:调官方别自己写)
+
+`parser_name: pytest` 的 task 判据 = **`pytest` 的 pass/fail**。我们不重写 parser:
+跑 `docker exec sh -c 'bash ./run-tests.sh'`,读 pytest 的退出码与 LS 输出里的
+`failed` 计数。correctness = `exit==0 且无 failed`。该任务的官方判据就是 pytest,所以这一步
+不是「自己抄 parser」,是「跑官方给的 pytest 并读它的结论」。
+
+## ⚠️ 必须记住的坑(读原仓库 + CLAUDE.md 读出来)
+
+- **镜像每题现场 build**(`Dockerfile` 在 task 目录里)。100 题里大部分 task 各自一个镜像,
+ 这是跑这一项的主要时间成本(不是 deepseek 的钱,是 docker build)。
+- **arm64 mac 上部分 task 跑不了**(如 pin 了 win32/x86 的 `3d-model-format-legacy`)。
+ `docker build` 失败的那题,判分里如实记 `env_unavailable`,不静默当对。
+- **不依赖 `terminal-bench` pip 包**(它要 Python 3.13 + postgres/supabase,太重,且它带自己的
+ agent,我们要用自己的 `react` 臂)。这里直接用 `docker` CLI 管 build/exec/run-tests。
+
+## ★ 抽样用 `rewoo_draw`(群里第 ③ 条点名要的)
+
+ --seed 决定我们从 79 题里抽哪一批,全项目同一种做法。
+"""
+
+from __future__ import annotations
+
+import re
+import shutil
+import subprocess
+import uuid
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import Any, Iterator, Sequence
+
+import yaml
+
+from experiments.core.download import DATASET_DIR, DownloadSpec
+from experiments.core.types import Judgment, Task, Tool, Trajectory
+
+#: Terminal-Bench **published dataset** `terminal-bench-core==0.1.1`
+#: —— 不跟 main(main 一夜会换 task 集,不可复现)。registry.json 里登记的
+#: 最 release 的 lock 版本: commit + branch + task_id_subset 三件全。
+TB_COMMIT = "91e10457b5410f16c44364da1a34cb6de8c488a5"
+TB_VERSION = "0.1.1"
+TB_BRANCH = "dataset/terminal-bench-core/v0.1.x"
+
+#: `v0.1.1` 分支下 task 定义在 `tasks//`(main 后来才改名 `original-tasks/`)。
+TB_ROOT = DATASET_DIR / "terminal_bench" / "v0.1.1"
+TASKS_DIR = TB_ROOT / "tasks"
+#: v0.1.1 里 `datasets/terminal-bench-core-v0.yaml` 的 `task_ids`(80 个,含
+#: `.base`/`.easy`/`.hard` 变体)**去后缀**的 70 个 unique task dir ——
+#: 圈定的 published subset。不跑分支全集 86 个,只跑这 70。
+#: (变体 .easy/.hard 是同一 task dir 不同难度,paper 一同报;我们这版每道跑一次,
+#: 70 是 published subset 的 unique base,可复现。)
+SUBSET_BASES: tuple[str, ...] = (
+ 'blind-maze-explorer-5x5', 'blind-maze-explorer-algorithm', 'build-initramfs-qemu', 'build-linux-kernel-qemu',
+ 'build-tcc-qemu', 'cartpole-rl-training', 'chess-best-move', 'conda-env-conflict-resolution',
+ 'configure-git-webserver', 'count-dataset-tokens', 'crack-7z-hash', 'create-bucket',
+ 'cron-broken-network', 'csv-to-parquet', 'decommissioning-service-with-sensitive-data', 'download-youtube',
+ 'eval-mteb', 'extract-moves-from-video', 'extract-safely', 'fibonacci-server',
+ 'fix-git', 'fix-pandas-version', 'fix-permissions', 'get-bitcoin-nodes',
+ 'git-multibranch', 'git-workflow-hack', 'gpt2-codegolf', 'grid-pattern-transform',
+ 'hello-world', 'heterogeneous-dates', 'hf-model-inference', 'incompatible-python-fasttext',
+ 'intrusion-detection', 'jupyter-notebook-server', 'modernize-fortran-build', 'new-encrypt-command',
+ 'nginx-request-logging', 'oom', 'openssl-selfsigned-cert', 'organization-json-generator',
+ 'password-recovery', 'path-tracing', 'path-tracing-reverse', 'play-zork',
+ 'polyglot-c-py', 'polyglot-rust-c', 'processing-pipeline', 'prove-plus-comm',
+ 'pytorch-model-cli', 'qemu-alpine-ssh', 'qemu-startup', 'raman-fitting',
+ 'reshard-c4-data', 'run-pdp11-code', 'sanitize-git-repo', 'security-vulhub-minio',
+ 'simple-sheets-put', 'simple-web-scraper', 'solana-data', 'sqlite-db-truncate',
+ 'sqlite-with-gcov', 'super-benchmark-upet', 'swe-bench-astropy-1', 'swe-bench-astropy-2',
+ 'swe-bench-fsspec', 'swe-bench-langcodes', 'tmux-advanced-workflow', 'train-fasttext',
+ 'vim-terminal-task', 'write-compressor',
+)
+IMAGE_PREFIX = "jevloop-tb"
+
+EXEC_TOOL = Tool(
+ name="exec",
+ description=("Run a shell command in the task's container. The command runs in the task's "
+ "working directory; you see its stdout/stderr. Use this to inspect and modify "
+ "the environment to complete the task."),
+ parameters={"type": "object",
+ "properties": {"command": {"type": "string"}},
+ "required": ["command"]},
+)
+
+
+class NeedsDocker(RuntimeError):
+ """要用容器(exec/score)但 docker 不可用,或这一题的镜像在本机 build 不起来。"""
+
+
+def _run(cmd: list[str], *, timeout: int = 1800, **kw) -> subprocess.CompletedProcess:
+ """跑一条命令,超时和失败都抛清楚(不把 stderr 当 stdout 吞)。"""
+ return subprocess.run(cmd, capture_output=True, text=True, timeout=timeout, **kw)
+
+
+@dataclass
+class TerminalBench:
+ """Terminal-Bench v0.1.1 published dataset(70 unique task dirs)。`split` 只有 `test`。"""
+
+ name: str = "terminal-bench"
+ dataset_version: str = f"terminal-bench-core=={TB_VERSION}"
+ data_dir: Path = field(default_factory=lambda: DATASET_DIR / "terminal_bench")
+ split: str = "test"
+ limit: int | None = None
+ seed: int = 0
+ #: agent 单题最多步数。TB task 有的要很久,runner 的 `--max-steps` 是默认 20,可经命令行覆盖。
+ max_steps: int = 40
+ #: docker build 单题超时(秒)。legacy build 慢,给宽一点。
+ build_timeout_s: int = 1800
+ #: run-tests.sh 超时(秒)。
+ test_timeout_s: int = 600
+ exec_timeout_s: int = 120
+ #: 当前题。`score()` 在 agent 改过的**同一容器**终态上跑官方 `run-tests.sh`——
+ # 不同于 ALFWorld 的「回放」,TB 的判据是容器状态被 agent 改成什么样了。
+ _current: Task | None = field(default=None, init=False, repr=False)
+ #: 这一题起出来的容器名。每题一个独立容器。
+ _container: str | None = field(default=None, init=False, repr=False)
+
+ # ── 数据 ────────────────────────────────────────────────
+
+ def _task_dirs(self) -> list[Path]:
+ if not TASKS_DIR.is_dir():
+ raise FileNotFoundError(
+ f"{self.name}: 缺 {TASKS_DIR}\n"
+ f" 跑 `python3 -m experiments.scripts.datasets --fetch terminal-bench`"
+ )
+ all_dirs = sorted(d for d in TASKS_DIR.iterdir()
+ if d.is_dir() and (d / "task.yaml").exists())
+ want = set(SUBSET_BASES)
+ kept = [d for d in all_dirs if d.name in want]
+ if len(kept) != len(SUBSET_BASES):
+ have = {d.name for d in kept}
+ missing = sorted(want - have)[:8]
+ raise ValueError(
+ f"{self.name}: subset 里有 {missing} 等 task 不在 {TASKS_DIR}。"
+ f" → 数据版本和 subset 不一致?"
+ )
+ return kept
+
+ def tasks(self, *, split: str, limit: int | None, seed: int) -> Iterator[Task]:
+ """出题。只取 0.1.1 `task_id_subset` 的 70 个 unique base dir;抽样 `rewoo_draw(seed)`。"""
+ from experiments.benchmark.rewoo_port import rewoo_draw
+
+ _ = split or self.split # 官方只有 `test`;不分 train/test
+ dirs = self._task_dirs()
+ for i in rewoo_draw(len(dirs), limit if limit is not None else self.limit, seed):
+ d = dirs[i]
+ meta = yaml.safe_load((d / "task.yaml").read_text(encoding="utf-8")) or {}
+ instruction = str(meta.get("instruction") or "").strip()
+ if not instruction:
+ continue
+ yield Task(
+ task_id=f"terminal-bench/{d.name}",
+ prompt=(instruction
+ + "\n\nYou have a shell tool (`exec`). Inspect and modify the environment "
+ "to complete the task; stop calling tools when finished."),
+ gold=None, # TB 没有字符串金标——判据是 pytest 在容器终态的结果
+ oracle_context=None,
+ meta={
+ "task_name": d.name,
+ "difficulty": meta.get("difficulty"),
+ "category": meta.get("category"),
+ "parser": meta.get("parser_name"),
+ "estimated_duration_sec": meta.get("estimated_duration_sec"),
+ "task_dir": str(d),
+ "answer_kind": "environment",
+ },
+ )
+
+ # ── 工具与环境 ────────────────────────────────────────────
+
+ def on_task(self, task: Task) -> None:
+ """runner 每题调一次。停下上一题的容器(每题独立容器),记这一题。"""
+ self._stop()
+ self._current = task
+ self._container = None
+
+ def _has_image(self, task_dir: Path) -> bool:
+ r = _run(["docker", "image", "inspect", f"{IMAGE_PREFIX}:{task_dir.name}"], timeout=10)
+ return r.returncode == 0
+
+ def _ensure_container(self) -> str:
+ """懒建这一题的镜像 + 容器。第一次 `tool_impls()` / `score()` 调时触发。"""
+ if self._container is not None:
+ return self._container
+ if self._current is None:
+ raise NeedsDocker(f"{self.name}: 没绑任务,先调 on_task(task)")
+ if not shutil.which("docker"):
+ raise NeedsDocker(f"{self.name}: 需要 `docker`(build/exec/run-tests)——没装")
+
+ task_dir = Path(self._current.meta["task_dir"])
+ image = f"{IMAGE_PREFIX}:{task_dir.name}"
+
+ if not self._has_image(task_dir):
+ br = _run(["docker", "build", "-t", image, "-f", "Dockerfile", "."],
+ cwd=str(task_dir), timeout=self.build_timeout_s)
+ if br.returncode != 0:
+ # ★ 多半是这题在本机 build 不动(arm64 起不来 x86/win32 镜像)。
+ # 不是「task 不可解」,是「环境跑不了这题」——分开记。
+ tail = br.stderr[-800:] or br.stdout[-800:]
+ raise NeedsDocker(
+ f"{self.name}: docker build 失败 ({task_dir.name})\n{tail}"
+ )
+ cont = f"jevloop-tb-{uuid.uuid4().hex[:10]}"
+ # 把 task dir 的 `run-tests.sh` + `tests/` 挂进容器 `/app`(base image 默认 WORKDIR)。
+ # ★ 大部分 Dockerfile **不 COPY 这两样**(broken-python 等只 `FROM base + RUN ...`),
+ # Terminal-Bench 官方 harness 是靠绑定挂载把它们弄进容器的——我们手动 run 时
+ # 不挂就缺,判分找不到 `run-tests.sh`(实测)。`:ro` 让判分脚本和测试不被 agent 改动,
+ # 但镜像 /app 里的业务文件 agent 仍可写。
+ td = str(task_dir.resolve())
+ rr = _run(["docker", "run", "-d", "--name", cont, "--workdir", "/app",
+ "-v", f"{td}/run-tests.sh:/app/run-tests.sh:ro",
+ "-v", f"{td}/tests:/app/tests:ro",
+ "-e", "TEST_DIR=./tests", image, "sleep", "infinity"],
+ timeout=60)
+ if rr.returncode != 0:
+ raise NeedsDocker(f"{self.name}: docker run 失败\n{rr.stderr[-800:]}")
+ self._container = cont
+ return cont
+
+ def tools(self) -> Sequence[Tool]:
+ self._ensure_container()
+ return (EXEC_TOOL,)
+
+ def tool_impls(self) -> dict[str, Any]:
+ self._ensure_container()
+
+ def exec_(command: str = "", **kwargs: Any) -> str:
+ cmd = str(command or kwargs.get("arg") or "").strip()
+ if not cmd:
+ return "Error: empty command."
+ r = _run(["docker", "exec", self._container, "sh", "-c", cmd],
+ timeout=self.exec_timeout_s)
+ out = r.stdout
+ if r.stderr:
+ out += (("\n[stderr]\n" + r.stderr) if r.stderr.strip() else "")
+ if r.returncode != 0:
+ out += f"\n[exit={r.returncode}]"
+ # 截一下,免得一条 ls -R 把 context 撑爆
+ return out[-8000:] if len(out) > 8000 else out
+
+ return {"exec": exec_}
+
+ # ── 判分 ────────────────────────────────────────────────
+
+ def score(self, task: Task, trajectory: Trajectory) -> Judgment:
+ """在 agent 改过的同一容器里跑官方 `run-tests.sh`,读 pytest 的结果。
+
+ correctness = run-tests.sh 退出 0 **且** 输出里没有 `failed` 计数 > 0。
+ pytest 自身就是这一题的官方判据,所以这里**不重写 parser**——
+ 跑它给的脚本、读它给的结论。
+ """
+ # 即使 agent 一条 exec 没调也得能判分(必然失败,但要走完记账)
+ try:
+ self._ensure_container()
+ except NeedsDocker as exc:
+ return Judgment(correct=False, score=0.0, detail=str(exc),
+ failure_class="env_unavailable")
+
+ task_dir = Path(task.meta["task_dir"])
+ r = _run(["docker", "exec", self._container, "sh", "-c", "bash ./run-tests.sh"],
+ timeout=self.test_timeout_s)
+ out = (r.stdout or "") + (("\n[stderr]\n" + r.stderr) if (r.stderr or "").strip() else "")
+ failed = _count_failed(out)
+ passed = (r.returncode == 0) and failed == 0
+
+ detail = f"run-tests.sh exit={r.returncode}; pytest failed={failed}"
+ if not r.stdout and r.stderr:
+ detail = f"只输出到 stderr(exit={r.returncode}):{r.stderr[-300:]}"
+ return Judgment(
+ correct=passed, score=1.0 if passed else 0.0,
+ detail=detail,
+ failure_class=None if passed else "task_not_completed",
+ )
+
+ def check(self, task: Task, answer: str) -> bool:
+ """Reflexion 的 Evaluator 要的成败信号 —— TB 的成败在容器状态,不在答案字符串。"""
+ raise NeedsDocker(
+ f"{self.name}: 成败由容器里的官方 pytest 判定,不是对答案字符串打分。"
+ f"用 score()(在容器跑 run-tests.sh),或让臂通过 session.check_answer 接上。"
+ )
+
+ # ── 下载 ────────────────────────────────────────────────
+
+ def downloads(self) -> Sequence[DownloadSpec]:
+ return [
+ DownloadSpec(
+ dataset=self.name,
+ kind="http",
+ # ★ registry.json 里登记的 v0.1.1: commit + branch 都写明
+ locator=(f"https://codeload.github.com/laude-institute/terminal-bench"
+ f"/tar.gz/{TB_COMMIT}"),
+ files=("tasks/", "docker/", "registry.json"),
+ revision=TB_COMMIT,
+ size_hint="~14 MB(repo @ v0.1.1,不含 docker 镜像)",
+ note=(f"★ **terminal-bench-core=={TB_VERSION}** —— registry.json 里 publish "
+ f"的 lock entry:commit {TB_COMMIT[:8]} / branch {TB_BRANCH}。"
+ "解压到 `dataset/terminal_bench/v0.1.1/`(顶层名是 codeloud 的前缀)。"
+ "docker 镜像**不在**这份里,每题 Dockerfile 现场 build。"
+ "task_id_subset 80 个 id (含 .easy/.hard 变体) 的 unique base = 70 dir。"),
+ ),
+ ]
+
+ # ── 清理 ────────────────────────────────────────────────
+
+ def _stop(self) -> None:
+ if self._container:
+ _run(["docker", "stop", self._container], timeout=30)
+ _run(["docker", "rm", "-f", self._container], timeout=30)
+ self._container = None
+
+
+_FAILED_RE = re.compile(r"(\d+)\s+failed", re.IGNORECASE)
+
+
+def _count_failed(test_output: str) -> int:
+ """从 pytest 输出里抠 `failed` 计数;抠不到返回 -1(未知,但不当作 0 —— 见 gsm8k 的口径教训)。"""
+ m = _FAILED_RE.search(test_output)
+ if not m:
+ return 0 if test_output.strip() else -1
+ return int(m.group(1))
+
+
+__all__ = ["TerminalBench", "NeedsDocker", "TB_COMMIT", "TASKS_DIR", "EXEC_TOOL"]
diff --git a/experiments/dataset/DOWNLOADS.md b/experiments/dataset/DOWNLOADS.md
index f409801..45a4181 100644
--- a/experiments/dataset/DOWNLOADS.md
+++ b/experiments/dataset/DOWNLOADS.md
@@ -32,6 +32,7 @@
| `tau2-bench` | http | `https://codeload.github.com/sierra-research/tau2-bench/tar.gz/refs/tags/v0.1.0` | `9` 个 | 37199f36924c | ~56 MB(整个仓库;本数据集约 25 MB) | ★ **必须钉 v0.1.0** —— 仓库 HEAD 已经是 τ³-bench v1.0.1,而它自己的 README 写着 `<1.0.1` 的结果不能和 `>=1.0.1` 比。
★ 而 v0.1.0 的 `tasks.json` **就是 base 划分**(278 条);HEAD 上多了 `split_tasks.json`,base 要从里面挑 —— 换版本连哪 278 条都会变。
★ `test` 那一列在排行榜口径下要跑 **pass^k = 4 次/任务**(`comb(success,k)/comb(trials,k)`) |
+| `terminal-bench` | http | `https://codeload.github.com/laude-institute/terminal-bench/tar.gz/91e10457b5410f16c44364da1a34cb6de8c488a5` | `3` 个 | 91e10457b5410f16c44364da1a34cb6de8c488a5 | ~14 MB(repo @ v0.1.1,不含 docker 镜像) | ★ **terminal-bench-core==0.1.1** —— registry.json 里 publish 的 lock entry:commit 91e10457 / branch dataset/terminal-bench-core/v0.1.x。解压到 `dataset/terminal_bench/v0.1.1/`(顶层名是 codeloud 的前缀)。docker 镜像**不在**这份里,每题 Dockerfile 现场 build。task_id_subset 80 个 id (含 .easy/.hard 变体) 的 unique base = 70 dir。 |
| `triviaqa` | hf-dataset | `trivia_qa:rc.nocontext` | — | ⚠️ **main**(未钉) | ~700 MB(138,384 + 17,944 + 17,210) | ★ config 必须是 `rc.nocontext` —— ReWOO 用的就是它。★ 答案带**别名集**,官方口径与 ReWOO 口径不同,见本文件头部。`revision` 未钉,已用 fingerprint 记进 dataset_version |
## 怎么用
diff --git a/experiments/scripts/run.py b/experiments/scripts/run.py
index b058d11..8e6eb1e 100644
--- a/experiments/scripts/run.py
+++ b/experiments/scripts/run.py
@@ -39,6 +39,7 @@
# ★ 三个 BigBench 任务**共用一份 ReWOO tarball** —— 一次下载抽三个 CSV。
from experiments.benchmark import bigbench as bigbench_bench # noqa: F401
from experiments.benchmark import alfworld as alfworld_bench # noqa: F401
+from experiments.benchmark import terminal_bench as terminal_bench_bench # noqa: F401
from experiments.benchmark import fever as fever_bench # noqa: F401
from experiments.benchmark import gsm8k as gsm8k_bench # noqa: F401
from experiments.benchmark import hotpotqa as hotpotqa_bench # noqa: F401
@@ -153,6 +154,8 @@ def build_model(args: argparse.Namespace) -> ModelClient:
sotuqa_bench.SotuQa.name: sotuqa_bench.SotuQa,
# ★ 交互式环境:需要 textworld,而且要用 `on_task` 绑定每题的环境
alfworld_bench.AlfWorld.name: alfworld_bench.AlfWorld,
+ # ★ Terminal-Bench: 每题一个 docker 容器,判分跑官方 pytest(见 benchmark/terminal_bench/terminal_bench.py 头部)
+ terminal_bench_bench.TerminalBench.name: terminal_bench_bench.TerminalBench,
# ★ 两方任务:还需要一个 LLM 用户模拟器(见它文件头的说明)
tau2_bench.Tau2Bench.name: tau2_bench.Tau2Bench,
}