Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 8 additions & 4 deletions benchmarks/swebench/eval_infer.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@
from benchmarks.utils.laminar import LaminarService
from benchmarks.utils.patch_utils import remove_files_from_patch
from benchmarks.utils.report_costs import generate_cost_report
from benchmarks.utils.swebench_reports import ensure_swebench_run_report
from openhands.sdk import get_logger


Expand Down Expand Up @@ -349,10 +350,13 @@ def main() -> None:
timeout=args.timeout,
)

# Move report file to input file directory with .report.json extension
# SWE-Bench creates: {MODEL_NAME_OR_PATH}.{run_id}.json
report_filename = f"{MODEL_NAME_OR_PATH}.{args.run_id}.json"
report_path = output_file.parent / report_filename
report_path = ensure_swebench_run_report(
output_file,
dataset=args.dataset,
split=args.split,
run_id=args.run_id,
modal=args.modal,
)

shutil.move(str(report_path), str(dest_report_path))
logger.info(f"Moved report file to: {dest_report_path}")
Expand Down
12 changes: 8 additions & 4 deletions benchmarks/swebenchmultilingual/eval_infer.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@
from benchmarks.utils.laminar import LaminarService
from benchmarks.utils.patch_utils import remove_files_from_patch
from benchmarks.utils.report_costs import generate_cost_report
from benchmarks.utils.swebench_reports import ensure_swebench_run_report
from openhands.sdk import get_logger


Expand Down Expand Up @@ -302,10 +303,13 @@ def main() -> None:
timeout=args.timeout,
)

# Move report file to input file directory with .report.json extension
# SWE-Bench creates: {MODEL_NAME_OR_PATH}.{run_id}.json
report_filename = f"{MODEL_NAME_OR_PATH}.{args.run_id}.json"
report_path = output_file.parent / report_filename
report_path = ensure_swebench_run_report(
output_file,
dataset=args.dataset,
split=args.split,
run_id=args.run_id,
modal=args.modal,
)
dest_report_path = input_file.with_suffix(".report.json")

shutil.move(str(report_path), str(dest_report_path))
Expand Down
32 changes: 32 additions & 0 deletions benchmarks/utils/swebench_reports.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,32 @@
from contextlib import chdir
from pathlib import Path

from swebench.harness.constants import KEY_INSTANCE_ID
from swebench.harness.reporting import make_run_report
from swebench.harness.utils import get_predictions_from_file, load_swebench_dataset

from benchmarks.utils.constants import MODEL_NAME_OR_PATH


def ensure_swebench_run_report(
predictions_file: Path,
dataset: str,
split: str,
run_id: str,
modal: bool,
) -> Path:
"""Return the aggregate report, building it after a Modal run if needed."""
predictions_file = predictions_file.resolve()
report_path = predictions_file.parent / f"{MODEL_NAME_OR_PATH}.{run_id}.json"
if report_path.exists() or not modal:
return report_path

prediction_rows = get_predictions_from_file(str(predictions_file), dataset, split)
predictions = {row[KEY_INSTANCE_ID]: row for row in prediction_rows}
full_dataset = load_swebench_dataset(dataset, split)
with chdir(predictions_file.parent):
generated_path = make_run_report(predictions, full_dataset, run_id)

if generated_path.is_absolute():
return generated_path
return predictions_file.parent / generated_path
69 changes: 69 additions & 0 deletions tests/test_swebench_eval_infer.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@

import json
import tempfile
from pathlib import Path
from types import SimpleNamespace

import pytest
Expand All @@ -14,6 +15,7 @@

from benchmarks.swebench import apptainer_eval
from benchmarks.swebench.eval_infer import convert_to_swebench_format
from benchmarks.utils import swebench_reports
from benchmarks.utils.constants import MODEL_NAME_OR_PATH


Expand Down Expand Up @@ -72,6 +74,73 @@ def test_model_name_or_path_uses_constant(self):
assert result["model_name_or_path"] == MODEL_NAME_OR_PATH


def test_modal_report_is_built_when_upstream_does_not_write_one(tmp_path, monkeypatch):
predictions_file = tmp_path / "output.swebench.jsonl"
predictions_file.write_text("{}\n")
prediction_rows = [
{
"instance_id": "django__django-1",
"model_name_or_path": MODEL_NAME_OR_PATH,
}
]
predictions = {"django__django-1": prediction_rows[0]}
dataset = [{"instance_id": "django__django-1"}]
original_cwd = Path.cwd()

monkeypatch.setattr(
swebench_reports,
"get_predictions_from_file",
lambda predictions_path, dataset_name, split: prediction_rows,
)
monkeypatch.setattr(
swebench_reports,
"load_swebench_dataset",
lambda dataset_name, split: dataset,
)

def fake_make_run_report(actual_predictions, full_dataset, run_id):
assert actual_predictions == predictions
assert full_dataset == dataset
assert run_id == "modal-run"
assert Path.cwd() == tmp_path
report_path = Path(f"{MODEL_NAME_OR_PATH}.{run_id}.json")
report_path.write_text("{}")
return report_path

monkeypatch.setattr(swebench_reports, "make_run_report", fake_make_run_report)

report_path = swebench_reports.ensure_swebench_run_report(
predictions_file,
dataset="princeton-nlp/SWE-bench_Verified",
split="test",
run_id="modal-run",
modal=True,
)

assert report_path == tmp_path / "OpenHands.modal-run.json"
assert report_path.read_text() == "{}"
assert Path.cwd() == original_cwd


def test_non_modal_report_keeps_the_upstream_output_path(tmp_path, monkeypatch):
predictions_file = tmp_path / "output.swebench.jsonl"

def fail_if_called(*args, **kwargs):
raise AssertionError("non-Modal runs should use the upstream report")

monkeypatch.setattr(swebench_reports, "make_run_report", fail_if_called)

report_path = swebench_reports.ensure_swebench_run_report(
predictions_file,
dataset="princeton-nlp/SWE-bench_Verified",
split="test",
run_id="docker-run",
modal=False,
)

assert report_path == tmp_path / "OpenHands.docker-run.json"


class TestApptainerEvaluation:
"""Tests for Apptainer SWE-bench evaluation helpers."""

Expand Down