swe scoring: official swebench harness (es-identical path), backend hook
Both swe variants now score through make_test_spec -> container -> apply-patch -> official eval_script -> get_eval_report -- the exact pipeline es's eval_instance drives (swebench==4.1.0 installed). Our in-container protocol stays as fallback when the package is absent. env_reward gains a recipe-level 'backend' delegation hook (tau2's reward_info reader and the swe official scorer both ride it); the generic bfcl call-sequence comparison remains the default. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
parent
f662006517
commit
de847206e5
@ -247,6 +247,126 @@ exit $FAIL
|
|||||||
return {'patch.diff': pred or '', 'run.sh': script}
|
return {'patch.diff': pred or '', 'run.sh': script}
|
||||||
|
|
||||||
|
|
||||||
|
def _official_swebench_eval(instance_meta: dict, patch: str, timeout: int = 1800):
|
||||||
|
"""Score ONE instance via the OFFICIAL swebench harness -- the exact
|
||||||
|
code path es uses (make_test_spec -> build_container -> apply patch ->
|
||||||
|
eval script -> get_eval_report). Requires pip install swebench."""
|
||||||
|
from pathlib import Path
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
from swebench.harness.grading import get_eval_report
|
||||||
|
from swebench.harness.test_spec.test_spec import make_test_spec
|
||||||
|
|
||||||
|
test_spec = make_test_spec(instance_meta, namespace='swebench',
|
||||||
|
arch='x86_64')
|
||||||
|
log_dir = Path(tempfile.mkdtemp(prefix='eh-swe-eval-')) / test_spec.instance_id
|
||||||
|
log_dir.mkdir(parents=True)
|
||||||
|
|
||||||
|
from docker.client import DockerClient
|
||||||
|
|
||||||
|
client = DockerClient.from_env()
|
||||||
|
# official image must exist locally (pull handled by our preflight)
|
||||||
|
image_name = test_spec.instance_image_key
|
||||||
|
container = client.containers.run(
|
||||||
|
image_name, command='tail -f /dev/null', detach=True,
|
||||||
|
name=f'eh-swe-official-{test_spec.instance_id[:40]}')
|
||||||
|
try:
|
||||||
|
patch_file = log_dir / 'patch.diff'
|
||||||
|
patch_file.write_text(patch or '')
|
||||||
|
container.exec_run(f'cp {patch_file} /tmp/patch.diff' if False
|
||||||
|
else 'mkdir -p /tmp', workdir='/testbed')
|
||||||
|
# copy patch in
|
||||||
|
import tarfile
|
||||||
|
import io as _io
|
||||||
|
|
||||||
|
with tempfile.NamedTemporaryFile(suffix='.tar') as tf:
|
||||||
|
with tarfile.open(tf.name, 'w') as tar:
|
||||||
|
info = tarfile.TarInfo('patch.diff')
|
||||||
|
data = (patch or '').encode()
|
||||||
|
info.size = len(data)
|
||||||
|
tar.addfile(info, _io.BytesIO(data))
|
||||||
|
with open(tf.name, 'rb') as f:
|
||||||
|
container.put_archive('/tmp', f.read())
|
||||||
|
applied = False
|
||||||
|
for cmd in ('git apply -v /tmp/patch.diff',
|
||||||
|
'git apply --reject /tmp/patch.diff',
|
||||||
|
'patch -p1 -i /tmp/patch.diff'):
|
||||||
|
r = container.exec_run(cmd, workdir='/testbed')
|
||||||
|
if r.exit_code == 0:
|
||||||
|
applied = True
|
||||||
|
break
|
||||||
|
if not applied:
|
||||||
|
return {'resolved': 0.0, 'error': 'patch_apply_failed'}
|
||||||
|
eval_file = log_dir / 'eval.sh'
|
||||||
|
eval_file.write_text(test_spec.eval_script)
|
||||||
|
with tempfile.NamedTemporaryFile(suffix='.tar') as tf:
|
||||||
|
with tarfile.open(tf.name, 'w') as tar:
|
||||||
|
info = tarfile.TarInfo('eval.sh')
|
||||||
|
data = test_spec.eval_script.encode()
|
||||||
|
info.size = len(data)
|
||||||
|
tar.addfile(info, _io.BytesIO(data))
|
||||||
|
with open(tf.name, 'rb') as f:
|
||||||
|
container.put_archive('/tmp', f.read())
|
||||||
|
container.exec_run('chmod +x /tmp/eval.sh')
|
||||||
|
r = container.exec_run('/bin/bash /tmp/eval.sh', workdir='/testbed')
|
||||||
|
# exec_run without demux gives combined output in .output
|
||||||
|
test_output = r.output.decode('utf-8', 'replace')
|
||||||
|
(log_dir / 'test_output.txt').write_text(test_output)
|
||||||
|
report = get_eval_report(
|
||||||
|
test_spec=test_spec,
|
||||||
|
prediction={'model_patch': patch, 'instance_id': test_spec.instance_id},
|
||||||
|
test_log_path=log_dir / 'test_output.txt',
|
||||||
|
include_tests_status=True,
|
||||||
|
)
|
||||||
|
return {'resolved': float(report.get('resolved', 0.0)),
|
||||||
|
'report': {k: v for k, v in report.items()
|
||||||
|
if isinstance(v, (int, float, str, bool))}}
|
||||||
|
finally:
|
||||||
|
try:
|
||||||
|
container.remove(force=True)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def _swe_official_reward(pred, target, sample, ctx):
|
||||||
|
"""Agentic + single-turn shared scorer: official harness when the
|
||||||
|
swebench package is importable, our in-container protocol otherwise."""
|
||||||
|
es = ctx.params.get('env_state') or {}
|
||||||
|
patch = es.get('patch') or pred or ''
|
||||||
|
if not patch or '(no patch produced)' in patch:
|
||||||
|
return ({'acc': 0.0}, {'acc': {'error': 'no patch'}})
|
||||||
|
md = dict(sample.metadata or {})
|
||||||
|
# official make_test_spec needs the raw instance fields
|
||||||
|
for k in ('FAIL_TO_PASS', 'PASS_TO_PASS'):
|
||||||
|
v = md.get(k)
|
||||||
|
if isinstance(v, str):
|
||||||
|
try:
|
||||||
|
md[k] = _json.loads(v)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
md.setdefault('repo', str(md.get('repo', '')))
|
||||||
|
md.setdefault('version', str(md.get('version', '')))
|
||||||
|
try:
|
||||||
|
from swebench.harness.test_spec.test_spec import make_test_spec # noqa
|
||||||
|
|
||||||
|
have = True
|
||||||
|
except ImportError:
|
||||||
|
have = False
|
||||||
|
if have:
|
||||||
|
try:
|
||||||
|
res = _official_swebench_eval(md, patch)
|
||||||
|
return ({'acc': res.get('resolved', 0.0)},
|
||||||
|
{'acc': {'mode': 'official_swebench',
|
||||||
|
'resolved': res.get('resolved'),
|
||||||
|
'error': res.get('error', ''),
|
||||||
|
'turns_used': es.get('turns_used')}})
|
||||||
|
except Exception as e:
|
||||||
|
return ({'acc': 0.0}, {'acc': {'mode': 'official_swebench',
|
||||||
|
'error': f'{type(e).__name__}: {str(e)[:150]}'}})
|
||||||
|
# fallback: our in-container protocol (no swebench package)
|
||||||
|
return _swe_agentic_reward(pred, target, sample, ctx)
|
||||||
|
|
||||||
|
|
||||||
@register_eval('swe_bench_verified')
|
@register_eval('swe_bench_verified')
|
||||||
def swe_bench_verified():
|
def swe_bench_verified():
|
||||||
return EvalRecipe(
|
return EvalRecipe(
|
||||||
@ -303,10 +423,8 @@ def swe_bench_verified_agentic():
|
|||||||
return EvalRecipe(
|
return EvalRecipe(
|
||||||
name='swe_bench_verified_agentic',
|
name='swe_bench_verified_agentic',
|
||||||
extract='identity',
|
extract='identity',
|
||||||
scorers={'resolved': {'name': 'execution',
|
scorers={'resolved': {'name': 'env_reward',
|
||||||
'harness': _swe_agentic_harness,
|
'backend': _swe_official_reward}},
|
||||||
'entry': 'run.sh', 'sandbox': 'docker',
|
|
||||||
'timeout_s': 1800}},
|
|
||||||
exec_workers=2,
|
exec_workers=2,
|
||||||
description='SWE-bench Verified AGENTIC (mini-swe-agent protocol): '
|
description='SWE-bench Verified AGENTIC (mini-swe-agent protocol): '
|
||||||
'multi-turn bash agent explores /testbed in the '
|
'multi-turn bash agent explores /testbed in the '
|
||||||
|
|||||||
@ -363,6 +363,11 @@ def env_reward(pred: str, target, sample: Sample, ctx: ScoreContext):
|
|||||||
tau2/swe get dedicated envs later; this scorer stays the entry point.
|
tau2/swe get dedicated envs later; this scorer stays the entry point.
|
||||||
"""
|
"""
|
||||||
env_state = ctx.params.get('env_state') or {}
|
env_state = ctx.params.get('env_state') or {}
|
||||||
|
# recipe-provided backend (tau2 reward_info / swe official harness):
|
||||||
|
# delegated BEFORE the generic bfcl call-sequence comparison
|
||||||
|
backend = ctx.params.get('backend')
|
||||||
|
if callable(backend):
|
||||||
|
return backend(pred, target, sample, ctx)
|
||||||
if not env_state:
|
if not env_state:
|
||||||
raise LayerNotReady(
|
raise LayerNotReady(
|
||||||
'env_reward needs env_state from an agent trajectory '
|
'env_reward needs env_state from an agent trajectory '
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user