Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 3 additions & 2 deletions benchmarks/utils/critics.py
Original file line number Diff line number Diff line change
Expand Up @@ -147,8 +147,7 @@ def evaluate_output(critic: CriticBase, eval_output: EvalOutput) -> bool:

def get_completed_instances(output_file: str) -> Set[EvalInstanceID]:
"""
Get all instance IDs present in output file
(completed, regardless of success/failure).
Get instance IDs that produced output without a runner error.

Reads ``instance_id`` directly from each JSON line WITHOUT validating the
full ``EvalOutput`` model. Resume must recognise prior completion across
Expand Down Expand Up @@ -184,6 +183,8 @@ def get_completed_instances(output_file: str) -> Set[EvalInstanceID]:
f"Missing 'instance_id' on line {line_num} in {output_file}"
)
continue
if isinstance(data, dict) and data.get("error") is not None:
continue
completed_instances.add(instance_id)

except Exception as e:
Expand Down
21 changes: 17 additions & 4 deletions tests/test_iterative_resume.py
Original file line number Diff line number Diff line change
Expand Up @@ -445,7 +445,7 @@ def _make_output(instance_id, has_patch):
)


def test_get_completed_instances_tolerates_stale_archive_schema():
def test_completed_instances_tolerate_stale_schema_and_skip_runner_errors():
"""
Resume must skip instances that were completed by an older SDK whose
EvalOutput schema has since drifted (e.g. browser tool/observation
Expand All @@ -454,6 +454,8 @@ def test_get_completed_instances_tolerates_stale_archive_schema():
Reading instance_id from raw JSON avoids treating those rows as
"never completed" and re-running them from scratch. Regression test
for evaluation#515.

Rows with a top-level runner error must remain eligible for retry.
"""
from benchmarks.utils.critics import get_completed_instances

Expand Down Expand Up @@ -492,13 +494,24 @@ def test_get_completed_instances_tolerates_stale_archive_schema():
)
f.write(fresh.model_dump_json() + "\n")

# 3) Blank line: must be tolerated, not counted.
# 3) Runner error: must be retried rather than counted as complete.
errored = EvalOutput(
instance_id="instance_error",
test_result={},
instruction=None,
error="Runtime timed out",
history=[],
instance={"x": 3},
)
f.write(errored.model_dump_json() + "\n")

# 4) Blank line: must be tolerated, not counted.
f.write("\n")

# 4) Malformed JSON: must be skipped with a warning, not raise.
# 5) Malformed JSON: must be skipped with a warning, not raise.
f.write("{not valid json}\n")

# 5) JSON object missing instance_id: must be skipped.
# 6) JSON object missing instance_id: must be skipped.
f.write(json.dumps({"foo": "bar"}) + "\n")

completed = get_completed_instances(path)
Expand Down