Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
111 changes: 111 additions & 0 deletions benchmarks/harbor/passk.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
#!/usr/bin/env python3
"""Consistency metrics for a Harbor job run with -k repeats.

Harbor's own pass_at_k is unusable for this suite: harbor/utils/pass_at_k.py
returns {} unless reward.json holds exactly one key whose value is 0 or 1, and
its eligible k values are 2, 4, 8, ... and 5, 10, ..., so k=3 is never computed.
This reads the trial results directly instead.

python3 benchmarks/harbor/passk.py <harbor-jobs>[/<job>]
"""
from __future__ import annotations

import argparse
import json
import statistics as st
from collections import defaultdict
from pathlib import Path

KEYS = (("passed", "mode"), ("strict_passed", "strict"))


def is_job(path: Path) -> bool:
"""A job holds trial directories; a trial holds result.json beside verifier/."""
return any(
(child / "result.json").exists() and (child / "verifier").is_dir()
for child in path.iterdir()
if child.is_dir()
)


def iter_jobs(root: Path):
if is_job(root):
yield root
return
for child in sorted(root.iterdir()):
if child.is_dir() and is_job(child):
yield child


def collect(job: Path) -> dict[str, list[dict]]:
"""task name -> one entry per attempt."""
by_task: dict[str, list[dict]] = defaultdict(list)
for trial in sorted(job.glob("*/result.json")):
result = json.loads(trial.read_text())
task = result.get("task_name") or trial.parent.name.split("__")[0]
rewards = ((result.get("verifier_result") or {}).get("rewards")) or {}
by_task[task].append(
{
"reward": rewards.get("reward"),
"passed": bool(rewards.get("passed")),
"exact_f1": rewards.get("exact_f1"),
"strict_passed": bool(rewards.get("strict_passed")),
"errored": bool(result.get("exception_info")),
}
)
return by_task


def main() -> int:
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("path", type=Path)
ap.add_argument("--detail", action="store_true", help="per-scenario pass pattern")
args = ap.parse_args()

for job in iter_jobs(args.path):
by_task = collect(job)
if not by_task:
continue
counts = {len(v) for v in by_task.values()}
k = min(counts)
print(f"\n=== {job.name}")
print(f" tasks {len(by_task)} attempts per task {sorted(counts)}"
f"{' RAGGED, using k=' + str(k) if len(counts) > 1 else ''}")

errored = sum(a["errored"] for v in by_task.values() for a in v)
if errored:
print(f" {errored} attempt(s) errored and count as a failure")

rewards = [a["reward"] for v in by_task.values() for a in v if isinstance(a["reward"], (int, float))]
exacts = [a["exact_f1"] for v in by_task.values() for a in v if isinstance(a["exact_f1"], (int, float))]
if rewards:
print(f" mean reward (mode_f1) {st.mean(rewards):.3f}"
+ (f" mean exact_f1 {st.mean(exacts):.3f}" if exacts else
" exact_f1 absent: run predates the four-key reward"))

for field, label in KEYS:
per_task = [[a[field] for a in v[:k]] for v in by_task.values()]
any_pass = sum(any(p) for p in per_task) / len(per_task)
all_pass = sum(all(p) for p in per_task) / len(per_task)
per_attempt = [
sum(p[i] for p in per_task) / len(per_task) for i in range(k)
]
spread = (f" per-run {', '.join(f'{x:.3f}' for x in per_attempt)}"
f" mean {st.mean(per_attempt):.3f}"
+ (f" sd {st.stdev(per_attempt):.3f}" if k > 1 else ""))
print(f" {label:<7} pass@{k} {any_pass:.3f} pass^{k} {all_pass:.3f}{spread}")

if args.detail:
print(f" {'task':<22}{'pattern':<10}reward per attempt")
for task, attempts in sorted(by_task.items()):
pattern = "".join("Y" if a["passed"] else "." for a in attempts)
vals = " ".join(
f"{a['reward']:.3f}" if isinstance(a["reward"], (int, float)) else " - "
for a in attempts
)
print(f" {task.split('/')[-1]:<22}{pattern:<10}{vals}")
return 0


if __name__ == "__main__":
raise SystemExit(main())
45 changes: 43 additions & 2 deletions benchmarks/harbor/run.sh
Original file line number Diff line number Diff line change
Expand Up @@ -5,8 +5,13 @@
#
# bash benchmarks/harbor/run.sh -s SCENARIO_DIR -l LEADERBOARD_DIR \
# [-n N_CONCURRENT] [-p PROFILE] [-r RUNTIME_IMAGE] \
# [-k N_ATTEMPTS] [-t TEMPERATURE] [-u MAX_TURNS] \
# [-m "MODEL_ID REASONING_EFFORT"]...
#
# -k repeats every task that many times inside one job, which is what pass@k and
# pass^k are computed from. -t sets the sampling temperature and -u the agent
# turn budget; both are forwarded to stirrup-agent as agent kwargs.
#
# Needs Docker, `uv sync --extra harbor` and the runtime image, built with
# scripts/build-runtime-image.sh or published and passed as -r (or
# AOB_RUNTIME_IMAGE). Credentials come from ENV_FILE (default: the repo's .env),
Expand All @@ -20,7 +25,7 @@
set -euo pipefail

usage() {
printf 'Usage: %s -s SCENARIO_DIR -l LEADERBOARD_DIR [-n N_CONCURRENT] [-p PROFILE] [-r RUNTIME_IMAGE] [-m "MODEL_ID EFFORT"]...\n' "$0" >&2
printf 'Usage: %s -s SCENARIO_DIR -l LEADERBOARD_DIR [-n N_CONCURRENT] [-p PROFILE] [-r RUNTIME_IMAGE] [-k N_ATTEMPTS] [-t TEMPERATURE] [-u MAX_TURNS] [-m "MODEL_ID EFFORT"]...\n' "$0" >&2
}

scenario_dir="${SCENARIO_DIR:-}"
Expand All @@ -29,16 +34,22 @@ n_concurrent="${N_CONCURRENT:-4}"
profile="${PROFILE:-}"
env_file="${ENV_FILE:-}"
runtime_image="${AOB_RUNTIME_IMAGE:-}"
n_attempts="${N_ATTEMPTS:-}"
temperature="${TEMPERATURE:-}"
max_turns="${MAX_TURNS:-}"
model_configs=()

while getopts ':s:l:n:p:r:m:' option; do
while getopts ':s:l:n:p:r:m:k:t:u:' option; do
case "$option" in
s) scenario_dir="$OPTARG" ;;
l) leaderboard_dir="$OPTARG" ;;
n) n_concurrent="$OPTARG" ;;
p) profile="$OPTARG" ;;
r) runtime_image="$OPTARG" ;;
m) model_configs+=("$OPTARG") ;;
k) n_attempts="$OPTARG" ;;
t) temperature="$OPTARG" ;;
u) max_turns="$OPTARG" ;;
:) printf 'Option -%s requires an argument.\n' "$OPTARG" >&2; usage; exit 2 ;;
\?) printf 'Unknown option: -%s\n' "$OPTARG" >&2; usage; exit 2 ;;
esac
Expand All @@ -49,6 +60,19 @@ if [[ -z "$scenario_dir" || -z "$leaderboard_dir" ]]; then
exit 2
fi

for numeric in n_attempts:"$n_attempts" max_turns:"$max_turns"; do
name="${numeric%%:*}"
value="${numeric#*:}"
if [[ -n "$value" && ! "$value" =~ ^[1-9][0-9]*$ ]]; then
printf '%s must be a positive integer, got %s\n' "$name" "$value" >&2
exit 2
fi
done
if [[ -n "$temperature" && ! "$temperature" =~ ^[0-9]+(\.[0-9]+)?$ ]]; then
printf 'temperature must be a non-negative number, got %s\n' "$temperature" >&2
exit 2
fi

if [[ -z "${model_configs[*]+set}" ]]; then
model_configs=(
"litellm_proxy/gcp/gemini-3.6-flash high"
Expand Down Expand Up @@ -402,6 +426,21 @@ for model_config in "${model_configs[@]}"; do
effort_args=(--ak "reasoning_effort=$reasoning_effort")
fi

# Sampling and budget overrides reach stirrup-agent through agent kwargs.
sampling_args=()
if [[ -n "$temperature" ]]; then
sampling_args+=(--ak "temperature=$temperature")
fi
if [[ -n "$max_turns" ]]; then
sampling_args+=(--ak "max_turns=$max_turns")
fi

# Repeats live inside one job so Harbor keeps them under the same eval key.
attempt_args=()
if [[ -n "$n_attempts" ]]; then
attempt_args=(-k "$n_attempts")
fi

# harbor run exits 0 when trials fail; non-zero means the job itself could not
# run, e.g. a rejected config or missing credentials.
if ! AOB_CODE_TAR="$code_tar" uv run --env-file "$env_file" harbor run -y \
Expand All @@ -413,6 +452,8 @@ for model_config in "${model_configs[@]}"; do
--ak allow_docker_backend=true \
--ak workspace_dir=/workspace-share \
${effort_args[@]+"${effort_args[@]}"} \
${sampling_args[@]+"${sampling_args[@]}"} \
${attempt_args[@]+"${attempt_args[@]}"} \
--extra-docker-compose benchmarks/harbor/overlays/private-data.yaml \
--extra-docker-compose benchmarks/harbor/overlays/code-sandbox.yaml \
--n-concurrent "$n_concurrent" \
Expand Down
7 changes: 7 additions & 0 deletions benchmarks/harbor/template/tests/to_reward.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,8 +38,15 @@ def main() -> int:
else:
# One task is one scenario, so there is exactly one result.
score = results[0].get("score") or {}
details = score.get("details") or {}
rewards["reward"] = float(score.get("score") or 0.0)
rewards["passed"] = int(bool(score.get("passed")))
# The strict exact-match baseline, carried alongside the mode-aware
# reward so a run reports both without re-scoring.
rewards["exact_f1"] = float(details.get("exact_f1") or 0.0)
rewards["strict_passed"] = int(
(details.get("strict_exact_match_accuracy") or 0.0) == 1.0
)

args.out.parent.mkdir(parents=True, exist_ok=True)
args.out.write_text(json.dumps(rewards), encoding="utf-8")
Expand Down
Loading