From ee8631252670bdbe9a83c3c6189fdad4024623da Mon Sep 17 00:00:00 2001 From: Dhaval Patel Date: Mon, 5 Oct 2026 17:39:14 -0400 Subject: [PATCH] Add pass@k, temperature and turn-budget controls to harbor runs run.sh gains -k (repeat each task n times, the basis for pass@k and pass^k), -t (sampling temperature) and -u (agent turn budget), with validation on each. passk.py computes pass@k and pass^k from a leaderboard directory. to_reward.py also publishes exact_f1 and strict_passed. Signed-off-by: Dhaval Patel --- benchmarks/harbor/passk.py | 111 +++++ benchmarks/harbor/run.sh | 469 ++++++++++++++++++ benchmarks/harbor/template/tests/to_reward.py | 58 +++ 3 files changed, 638 insertions(+) create mode 100644 benchmarks/harbor/passk.py create mode 100755 benchmarks/harbor/run.sh create mode 100644 benchmarks/harbor/template/tests/to_reward.py diff --git a/benchmarks/harbor/passk.py b/benchmarks/harbor/passk.py new file mode 100644 index 000000000..329e63fd9 --- /dev/null +++ b/benchmarks/harbor/passk.py @@ -0,0 +1,111 @@ +#!/usr/bin/env python3 +"""Consistency metrics for a Harbor job run with -k repeats. + +Harbor's own pass_at_k is unusable for this suite: harbor/utils/pass_at_k.py +returns {} unless reward.json holds exactly one key whose value is 0 or 1, and +its eligible k values are 2, 4, 8, ... and 5, 10, ..., so k=3 is never computed. +This reads the trial results directly instead. + + python3 benchmarks/harbor/passk.py [/] +""" +from __future__ import annotations + +import argparse +import json +import statistics as st +from collections import defaultdict +from pathlib import Path + +KEYS = (("passed", "mode"), ("strict_passed", "strict")) + + +def is_job(path: Path) -> bool: + """A job holds trial directories; a trial holds result.json beside verifier/.""" + return any( + (child / "result.json").exists() and (child / "verifier").is_dir() + for child in path.iterdir() + if child.is_dir() + ) + + +def iter_jobs(root: Path): + if is_job(root): + yield root + return + for child in sorted(root.iterdir()): + if child.is_dir() and is_job(child): + yield child + + +def collect(job: Path) -> dict[str, list[dict]]: + """task name -> one entry per attempt.""" + by_task: dict[str, list[dict]] = defaultdict(list) + for trial in sorted(job.glob("*/result.json")): + result = json.loads(trial.read_text()) + task = result.get("task_name") or trial.parent.name.split("__")[0] + rewards = ((result.get("verifier_result") or {}).get("rewards")) or {} + by_task[task].append( + { + "reward": rewards.get("reward"), + "passed": bool(rewards.get("passed")), + "exact_f1": rewards.get("exact_f1"), + "strict_passed": bool(rewards.get("strict_passed")), + "errored": bool(result.get("exception_info")), + } + ) + return by_task + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("path", type=Path) + ap.add_argument("--detail", action="store_true", help="per-scenario pass pattern") + args = ap.parse_args() + + for job in iter_jobs(args.path): + by_task = collect(job) + if not by_task: + continue + counts = {len(v) for v in by_task.values()} + k = min(counts) + print(f"\n=== {job.name}") + print(f" tasks {len(by_task)} attempts per task {sorted(counts)}" + f"{' RAGGED, using k=' + str(k) if len(counts) > 1 else ''}") + + errored = sum(a["errored"] for v in by_task.values() for a in v) + if errored: + print(f" {errored} attempt(s) errored and count as a failure") + + rewards = [a["reward"] for v in by_task.values() for a in v if isinstance(a["reward"], (int, float))] + exacts = [a["exact_f1"] for v in by_task.values() for a in v if isinstance(a["exact_f1"], (int, float))] + if rewards: + print(f" mean reward (mode_f1) {st.mean(rewards):.3f}" + + (f" mean exact_f1 {st.mean(exacts):.3f}" if exacts else + " exact_f1 absent: run predates the four-key reward")) + + for field, label in KEYS: + per_task = [[a[field] for a in v[:k]] for v in by_task.values()] + any_pass = sum(any(p) for p in per_task) / len(per_task) + all_pass = sum(all(p) for p in per_task) / len(per_task) + per_attempt = [ + sum(p[i] for p in per_task) / len(per_task) for i in range(k) + ] + spread = (f" per-run {', '.join(f'{x:.3f}' for x in per_attempt)}" + f" mean {st.mean(per_attempt):.3f}" + + (f" sd {st.stdev(per_attempt):.3f}" if k > 1 else "")) + print(f" {label:<7} pass@{k} {any_pass:.3f} pass^{k} {all_pass:.3f}{spread}") + + if args.detail: + print(f" {'task':<22}{'pattern':<10}reward per attempt") + for task, attempts in sorted(by_task.items()): + pattern = "".join("Y" if a["passed"] else "." for a in attempts) + vals = " ".join( + f"{a['reward']:.3f}" if isinstance(a["reward"], (int, float)) else " - " + for a in attempts + ) + print(f" {task.split('/')[-1]:<22}{pattern:<10}{vals}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/harbor/run.sh b/benchmarks/harbor/run.sh new file mode 100755 index 000000000..0b962359b --- /dev/null +++ b/benchmarks/harbor/run.sh @@ -0,0 +1,469 @@ +#!/usr/bin/env bash +# Harbor counterpart of benchmarks/run.sh: the same scenarios, stirrup-agent and +# code sandbox, but each scenario is a Harbor trial with its own CouchDB, so +# scenarios run concurrently. See benchmarks/harbor/README.md. +# +# bash benchmarks/harbor/run.sh -s SCENARIO_DIR -l LEADERBOARD_DIR \ +# [-n N_CONCURRENT] [-p PROFILE] [-r RUNTIME_IMAGE] \ +# [-k N_ATTEMPTS] [-t TEMPERATURE] [-u MAX_TURNS] \ +# [-m "MODEL_ID REASONING_EFFORT"]... +# +# -k repeats every task that many times inside one job, which is what pass@k and +# pass^k are computed from. -t sets the sampling temperature and -u the agent +# turn budget; both are forwarded to stirrup-agent as agent kwargs. +# +# Needs Docker, `uv sync --extra harbor` and the runtime image, built with +# scripts/build-runtime-image.sh or published and passed as -r (or +# AOB_RUNTIME_IMAGE). Credentials come from ENV_FILE (default: the repo's .env), +# the only file read. Relative paths are relative to the caller's directory. +# +# One Harbor job per profile, model and effort, at +# LEADERBOARD_DIR/harbor-jobs/stirrup_agent____[__]. +# Re-running resumes it with its original settings. Exits non-zero when a +# model's job could not start or resume, or the model was skipped. + +set -euo pipefail + +usage() { + printf 'Usage: %s -s SCENARIO_DIR -l LEADERBOARD_DIR [-n N_CONCURRENT] [-p PROFILE] [-r RUNTIME_IMAGE] [-k N_ATTEMPTS] [-t TEMPERATURE] [-u MAX_TURNS] [-m "MODEL_ID EFFORT"]...\n' "$0" >&2 +} + +scenario_dir="${SCENARIO_DIR:-}" +leaderboard_dir="${LEADERBOARD_DIR:-}" +n_concurrent="${N_CONCURRENT:-4}" +profile="${PROFILE:-}" +env_file="${ENV_FILE:-}" +runtime_image="${AOB_RUNTIME_IMAGE:-}" +n_attempts="${N_ATTEMPTS:-}" +temperature="${TEMPERATURE:-}" +max_turns="${MAX_TURNS:-}" +model_configs=() + +while getopts ':s:l:n:p:r:m:k:t:u:' option; do + case "$option" in + s) scenario_dir="$OPTARG" ;; + l) leaderboard_dir="$OPTARG" ;; + n) n_concurrent="$OPTARG" ;; + p) profile="$OPTARG" ;; + r) runtime_image="$OPTARG" ;; + m) model_configs+=("$OPTARG") ;; + k) n_attempts="$OPTARG" ;; + t) temperature="$OPTARG" ;; + u) max_turns="$OPTARG" ;; + :) printf 'Option -%s requires an argument.\n' "$OPTARG" >&2; usage; exit 2 ;; + \?) printf 'Unknown option: -%s\n' "$OPTARG" >&2; usage; exit 2 ;; + esac +done + +if [[ -z "$scenario_dir" || -z "$leaderboard_dir" ]]; then + usage + exit 2 +fi + +for numeric in n_attempts:"$n_attempts" max_turns:"$max_turns"; do + name="${numeric%%:*}" + value="${numeric#*:}" + if [[ -n "$value" && ! "$value" =~ ^[1-9][0-9]*$ ]]; then + printf '%s must be a positive integer, got %s\n' "$name" "$value" >&2 + exit 2 + fi +done +if [[ -n "$temperature" && ! "$temperature" =~ ^[0-9]+(\.[0-9]+)?$ ]]; then + printf 'temperature must be a non-negative number, got %s\n' "$temperature" >&2 + exit 2 +fi + +if [[ -z "${model_configs[*]+set}" ]]; then + model_configs=( + "litellm_proxy/gcp/gemini-3.6-flash high" + "litellm_proxy/azure/gpt-5.6-sol max" + "litellm_proxy/aws/claude-opus-5 high" + "litellm_proxy/aws/claude-sonnet-5 max" + "tokenrouter/MiniMax-M3 high" + "tokenrouter/moonshotai/kimi-k3 max" + "tokenrouter/z-ai/glm-5.3 max" + "tokenrouter/deepseek/deepseek-v4-flash max" + ) +fi + +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" + +# Resolve caller paths before the cd below. +caller_dir="$PWD" +absolute() { + case "$1" in + /*) printf '%s' "$1" ;; + *) printf '%s/%s' "$caller_dir" "$1" ;; + esac +} +scenario_dir="$(absolute "$scenario_dir")" +leaderboard_dir="$(absolute "$leaderboard_dir")" +if [[ -n "$profile" ]]; then + profile="$(absolute "$profile")" +else + profile="$repo_root/benchmarks/scenario_suite/all.yaml" +fi +if [[ -n "$env_file" ]]; then + env_file="$(absolute "$env_file")" +else + env_file="$repo_root/.env" +fi +code_tar_dir="$(absolute "${AOB_CODE_TAR_DIR:-$HOME/.cache/assetopsbench}")" + +cd "$repo_root" + +if [[ ! -d "$scenario_dir" ]]; then + printf 'Scenario directory not found: %s\n' "$scenario_dir" >&2 + exit 2 +fi +scenario_dir="$(cd "$scenario_dir" && pwd)" +if [[ ! -f "$profile" ]]; then + printf 'Profile not found: %s\n' "$profile" >&2 + exit 2 +fi +mkdir -p "$leaderboard_dir" +leaderboard_dir="$(cd "$leaderboard_dir" && pwd)" +jobs_dir="$leaderboard_dir/harbor-jobs" +mkdir -p "$jobs_dir" + +if [[ ! -f "$env_file" ]]; then + printf 'Credentials file not found: %s (set ENV_FILE)\n' "$env_file" >&2 + exit 2 +fi +# StirrupAgent reads this file instead of the nearest .env. +export AOB_ENV_FILE="$env_file" + +# Each job gets its own copy of the tasks, generated once when it starts: +# Harbor refuses to resume a job whose tasks changed, and a shared folder could +# be rewritten under a running job. The copies hold answers, so they stay in the +# gitignored datasets/ rather than beside the results. +tasks_root="$repo_root/benchmarks/harbor/datasets/jobs" + +# Removed on exit: the runtime image pin, a partial code tar, the job lock. +runtime_pin="" +code_tar_partial="" +job_lock="" +cleanup() { + if [[ -n "$runtime_pin" ]]; then docker rmi "$runtime_pin" >/dev/null 2>&1 || true; fi + if [[ -n "$code_tar_partial" ]]; then rm -f "$code_tar_partial"; fi + if [[ -n "$job_lock" ]]; then rm -rf "$job_lock"; fi +} +trap cleanup EXIT + +# overlays/private-data.yaml mounts $AOB_PRIVATE_DIR/shared into each trial. +if [[ ! -d "$scenario_dir/shared" ]]; then + printf "No shared/ directory in %s; is -s the suite's scenarios_data?\n" "$scenario_dir" >&2 + exit 2 +fi +export AOB_PRIVATE_DIR="$scenario_dir" + +# -r wins, then the shell's AOB_RUNTIME_IMAGE, then ENV_FILE's, then the local +# default. +if [[ -z "$runtime_image" ]]; then + runtime_image="$(uv run --env-file "$env_file" python -c \ + 'import os; print(os.environ.get("AOB_RUNTIME_IMAGE", ""))')" +fi +runtime_image="${runtime_image:-assetopsbench/runtime:dev}" + +# The reference without its tag or digest, spelled as .RepoDigests spells it. +image_repo() { + local ref="${1%@*}" + if [[ "${ref##*/}" == *:* ]]; then ref="${ref%:*}"; fi + ref="${ref#docker.io/}" + printf '%s' "${ref#library/}" +} + +# True when the local copy of $1 came from (or went to) that same repository, +# i.e. it is a published image rather than a local build. +from_registry() { + local repo digest + repo="$(image_repo "$1")" + while read -r digest; do + [[ "${digest%@*}" == "$repo" ]] && return 0 + done < <(docker image inspect --format '{{range .RepoDigests}}{{println .}}{{end}}' "$1") + return 1 +} + +# A build never refreshes a published image it already has, so pull it here. A +# local build (e.g. the default assetopsbench/runtime:dev) is used as is. +if ! docker image inspect "$runtime_image" >/dev/null 2>&1; then + if ! docker pull "$runtime_image"; then + printf 'Runtime image %s is not local and could not be pulled. Build it with\n' "$runtime_image" >&2 + printf ' bash benchmarks/harbor/scripts/build-runtime-image.sh\n' >&2 + exit 1 + fi +elif from_registry "$runtime_image" && ! docker pull "$runtime_image"; then + printf 'warning: could not pull %s; using the local copy, which may be stale\n' \ + "$runtime_image" >&2 +fi + +# Pin the base for the whole run: a tag such as :dev can move mid-run, and each +# trial resolves FROM when it builds. FROM cannot name an image id, so tag it +# under a name private to this process. +runtime_id="$(docker image inspect --format '{{.Id}}' "$runtime_image")" +runtime_id="${runtime_id#sha256:}" +runtime_pin="aob-runtime-pin:${runtime_id:0:12}-$$" +docker tag "$runtime_image" "$runtime_pin" +export AOB_RUNTIME_IMAGE="$runtime_pin" +printf 'Runtime image: %s (%s)\n' "$runtime_image" "${runtime_id:0:12}" + +# The code sandbox image, as a tar each trial's dind loads. Rebuilt every run +# (cheap when cached) and saved once per image id, so a running job's tar is +# never rewritten. Old tars stay in AOB_CODE_TAR_DIR until removed. +code_image=assetops-code:dev +docker build -q -t "$code_image" \ + -f src/agent/stirrup_agent/Dockerfile.code src/agent/stirrup_agent >/dev/null +code_id="$(docker image inspect --format '{{.Id}}' "$code_image")" +code_id="${code_id#sha256:}" +code_tar="$code_tar_dir/assetops-code-${code_id:0:12}.tar" +if [[ ! -s "$code_tar" ]]; then + mkdir -p "$code_tar_dir" + echo "Saving $code_image to $code_tar" + code_tar_partial="$code_tar.partial.$$" + docker save "$code_image" -o "$code_tar_partial" + chmod 644 "$code_tar_partial" + mv "$code_tar_partial" "$code_tar" + code_tar_partial="" +fi +export AOB_CODE_IMAGE="$code_image" +printf 'Code image: %s (%s)\n' "$code_image" "${code_id:0:12}" + +# A name safe for a directory: anything but letters, digits and ._- becomes a +# single dash, and a trailing dash is dropped. +slug() { + local name + name="$(printf '%s' "$1" | tr -c 'A-Za-z0-9._-' '-' | tr -s '-')" + printf '%s' "${name%-}" +} + +profile_name="$(basename "$profile")" +profile_slug="$(slug "${profile_name%.*}")" + +# Fail fast when a model cannot be served, rather than a job of failed trials. +# The model's router must answer GET /models without a 401 or 403. +check_model() { + uv run --env-file "$env_file" python - "$1" <<'PY' +import os +import sys +import urllib.error +import urllib.request + +# src/llm/routers.py PROXY_ROUTERS; src/assetops_harbor/tests checks they match. +ROUTERS = { + "litellm_proxy/": ("LITELLM_BASE_URL", "LITELLM_API_KEY"), + "tokenrouter/": ("TOKENROUTER_BASE_URL", "TOKENROUTER_API_KEY"), +} + + +def router(model): + return next((prefix for prefix in ROUTERS if model.startswith(prefix)), None) + + +prefix = router(sys.argv[1]) +if prefix is None: + sys.exit(0) + +base_var, key_var = ROUTERS[prefix] +base, key = os.environ.get(base_var, ""), os.environ.get(key_var, "") +missing = [name for name, value in ((base_var, base), (key_var, key)) if not value] +if missing: + sys.exit(f"{' and '.join(missing)} not set for {prefix} models") +request = urllib.request.Request( + base.rstrip("/") + "/models", headers={"Authorization": f"Bearer {key}"} +) +try: + urllib.request.urlopen(request, timeout=15) +except urllib.error.HTTPError as exc: + if exc.code in (401, 403): + sys.exit(f"{base_var} rejected {key_var} (HTTP {exc.code})") +except Exception as exc: + sys.exit(f"cannot reach {base_var} ({exc}); check the VPN or network") +PY +} + +# One run.sh per job at a time. mkdir is atomic; the lock holds its owner's +# PID, so a lock left by a dead run.sh is taken over. +lock_job() { + local lock="$1" owner + if mkdir "$lock" 2>/dev/null; then + printf '%s\n' "$$" >"$lock/pid" + return 0 + fi + owner="$(cat "$lock/pid" 2>/dev/null || true)" + if [[ -z "$owner" ]] || kill -0 "$owner" 2>/dev/null; then + return 1 + fi + rm -rf "$lock" + mkdir "$lock" 2>/dev/null || return 1 + printf '%s\n' "$$" >"$lock/pid" +} + +# Trials a resume reruns: failures not caused by the model's own work (API, +# network, environment, verifier, Ctrl-C). Harbor matches exact class names; +# src/assetops_harbor/tests/test_run_sh.py checks them against Harbor. Timeouts, +# context/output overruns and safety refusals are kept as results. +retry_error_types=( + CancelledError + NonZeroAgentExitCodeError + ApiError + ApiRateLimitError + ApiUsageLimitError + ApiInternalServerError + ApiOverloadedError + ApiConnectionClosedError + ApiResponseStalledError + UnknownApiError + ApiProviderResourceNotFoundError + AgentAuthenticationError + ModelNotFoundError + NetworkConnectionError + AgentSetupTimeoutError + EnvironmentStartTimeoutError + HealthcheckError + VerifierTimeoutError + RewardFileNotFoundError + RewardFileEmptyError + VerifierOutputParseError + AddTestsDirError + DownloadVerifierDirError +) +retry_filters=() +for error_type in "${retry_error_types[@]}"; do + retry_filters+=(--filter-error-type "$error_type") +done + +# Non-zero when a model's job could not start or resume, or was skipped. +status=0 + +for model_config in "${model_configs[@]}"; do + read -r model_id reasoning_effort <<< "$model_config" + [[ -z "${model_id:-}" ]] && continue + + # Profile and effort are in the name so each gets its own job. + job_name="stirrup_agent__${profile_slug}__$(slug "$model_id")" + if [[ -n "${reasoning_effort:-}" ]]; then + job_name+="__$(slug "$reasoning_effort")" + fi + job_path="$jobs_dir/$job_name" + # Keyed by the job's full path, so another LEADERBOARD_DIR gets its own copy. + # No "__": Harbor names the dataset after this folder and splits its + # agent__model__dataset keys on "__", failing the run once trials finish. + tasks_dir="$tasks_root/${job_name//__/--}-$(printf '%s' "$job_path" | cksum | cut -d' ' -f1)" + # What the job started on, which Harbor's own resume check does not cover: the + # runtime image, the code tar, and the suite whose shared/ the mount supplies. + image_record="$jobs_dir/$job_name.runtime-image" + code_record="$jobs_dir/$job_name.code-tar" + suite_record="$jobs_dir/$job_name.suite" + + if ! check_model "$model_id"; then + echo "Skipping $model_id" >&2 + status=1 + continue + fi + + if ! lock_job "$job_path.lock"; then + printf 'Skipping %s: another run.sh (PID %s) is working on it.\n' \ + "$job_path" "$(cat "$job_path.lock/pid" 2>/dev/null || echo unknown)" >&2 + printf 'If none is, remove %s.\n' "$job_path.lock" >&2 + status=1 + continue + fi + job_lock="$job_path.lock" + + echo "Running $model_id with reasoning effort ${reasoning_effort:-default} -> $job_path" + + if [[ -f "$job_path/config.json" ]]; then + job_code_tar="$code_tar" + [[ -f "$code_record" ]] && job_code_tar="$(cat "$code_record")" + if [[ -f "$image_record" ]] && [[ "$(cut -f1 "$image_record")" != "$runtime_id" ]]; then + printf '%s started on runtime image %s, not %s (%s).\n' \ + "$job_path" "$(cut -f2 "$image_record")" "$runtime_image" "${runtime_id:0:12}" >&2 + printf 'Pass that image as -r to finish it, or move the job aside to rerun %s.\n' \ + "$model_id" >&2 + status=1 + elif [[ -f "$suite_record" ]] && [[ "$(cat "$suite_record")" != "$scenario_dir" ]]; then + printf '%s started on the suite in %s, not %s.\n' \ + "$job_path" "$(cat "$suite_record")" "$scenario_dir" >&2 + printf 'Pass that directory as -s to finish it, or move the job aside to rerun %s.\n' \ + "$model_id" >&2 + status=1 + elif [[ ! -d "$tasks_dir" ]]; then + printf 'The tasks %s started with are gone (%s).\n' "$job_path" "$tasks_dir" >&2 + printf 'Move the job aside to rerun %s from scratch.\n' "$model_id" >&2 + status=1 + elif [[ ! -s "$job_code_tar" ]]; then + printf 'The code image %s started with is gone (%s).\n' "$job_path" "$job_code_tar" >&2 + printf 'Move the job aside to rerun %s from scratch.\n' "$model_id" >&2 + status=1 + # Rerun the trials in retry_error_types; scored trials are kept. + elif ! AOB_CODE_TAR="$job_code_tar" uv run --env-file "$env_file" \ + harbor jobs resume -p "$job_path" "${retry_filters[@]}"; then + printf 'Could not resume %s. If its overlays changed since it started,\n' "$job_path" >&2 + printf 'move it aside to rerun %s from scratch.\n' "$model_id" >&2 + status=1 + fi + rm -rf "$job_lock" + job_lock="" + continue + fi + + # The generator never removes tasks, so clear any left by a failed start. + rm -rf "$tasks_dir" + uv run python benchmarks/harbor/adapter/generate_tasks.py \ + --scenario-root "$scenario_dir" \ + --profile "$profile" \ + --output-dir "$tasks_dir" \ + --dataset-name assetopsbench/suite \ + --skip-missing \ + --overwrite >/dev/null + + printf '%s\t%s\n' "$runtime_id" "$runtime_image" >"$image_record" + printf '%s\n' "$code_tar" >"$code_record" + printf '%s\n' "$scenario_dir" >"$suite_record" + + effort_args=() + if [[ -n "${reasoning_effort:-}" ]]; then + effort_args=(--ak "reasoning_effort=$reasoning_effort") + fi + + # Sampling and budget overrides reach stirrup-agent through agent kwargs. + sampling_args=() + if [[ -n "$temperature" ]]; then + sampling_args+=(--ak "temperature=$temperature") + fi + if [[ -n "$max_turns" ]]; then + sampling_args+=(--ak "max_turns=$max_turns") + fi + + # Repeats live inside one job so Harbor keeps them under the same eval key. + attempt_args=() + if [[ -n "$n_attempts" ]]; then + attempt_args=(-k "$n_attempts") + fi + + # harbor run exits 0 when trials fail; non-zero means the job itself could not + # run, e.g. a rejected config or missing credentials. + if ! AOB_CODE_TAR="$code_tar" uv run --env-file "$env_file" harbor run -y \ + -p "$tasks_dir" \ + --agent assetops_harbor.stirrup:StirrupAgent \ + --model "$model_id" \ + --ak code_enabled=true \ + --ak code_backend=docker \ + --ak allow_docker_backend=true \ + --ak workspace_dir=/workspace-share \ + ${effort_args[@]+"${effort_args[@]}"} \ + ${sampling_args[@]+"${sampling_args[@]}"} \ + ${attempt_args[@]+"${attempt_args[@]}"} \ + --extra-docker-compose benchmarks/harbor/overlays/private-data.yaml \ + --extra-docker-compose benchmarks/harbor/overlays/code-sandbox.yaml \ + --n-concurrent "$n_concurrent" \ + --job-name "$job_name" \ + -o "$jobs_dir"; then + printf 'Harbor could not run %s; see the error above.\n' "$job_path" >&2 + status=1 + fi + rm -rf "$job_lock" + job_lock="" +done + +exit "$status" diff --git a/benchmarks/harbor/template/tests/to_reward.py b/benchmarks/harbor/template/tests/to_reward.py new file mode 100644 index 000000000..3975ce580 --- /dev/null +++ b/benchmarks/harbor/template/tests/to_reward.py @@ -0,0 +1,58 @@ +"""Map an AssetOpsBench EvalReport onto Harbor's reward.json. + +Harbor averages each key across trials, so only scores belong here; token +counts and cost reach Harbor through the ATIF trajectory. +""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--report", type=Path, required=True) + parser.add_argument("--out", type=Path, required=True) + parser.add_argument("--eval-status", type=int, default=0) + args = parser.parse_args() + + rewards: dict[str, float | int] = {"reward": 0.0, "passed": 0} + + if not args.report.exists(): + print( + f"evaluation produced no report at {args.report} " + f"(exit status {args.eval_status}); scoring 0", + file=sys.stderr, + ) + else: + report = json.loads(args.report.read_text(encoding="utf-8")) + results = report.get("results") or [] + if not results: + print( + "evaluation report contains no scored results; scoring 0", + file=sys.stderr, + ) + else: + # One task is one scenario, so there is exactly one result. + score = results[0].get("score") or {} + details = score.get("details") or {} + rewards["reward"] = float(score.get("score") or 0.0) + rewards["passed"] = int(bool(score.get("passed"))) + # The strict exact-match baseline, carried alongside the mode-aware + # reward so a run reports both without re-scoring. + rewards["exact_f1"] = float(details.get("exact_f1") or 0.0) + rewards["strict_passed"] = int( + (details.get("strict_exact_match_accuracy") or 0.0) == 1.0 + ) + + args.out.parent.mkdir(parents=True, exist_ok=True) + args.out.write_text(json.dumps(rewards), encoding="utf-8") + print(json.dumps(rewards)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())