diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index 89bcbcce3..49de5cf50 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -67,6 +67,39 @@ uv run harbor run \ --n-concurrent 16 ``` +## Running a full scenario suite (the `benchmarks/run.sh` equivalent) + +`benchmarks/harbor/run.sh` runs what `benchmarks/run.sh` runs, the same +profile, the same `stirrup-agent` and the same Docker code sandbox, with the +scenarios in parallel: + +```bash +bash benchmarks/harbor/run.sh \ + -s /AssetOpsBenchScenarioGeneration/scenarios_data \ + -l ~/AssetOpsBenchRuns/leaderboard \ + -n 4 \ + -m "litellm_proxy/aws/claude-opus-5 high" +``` + +`-m` may repeat; without it the script runs `benchmarks/run.sh`'s model list. +`-p` picks the profile (default `benchmarks/scenario_suite/all.yaml`), and `-n` +the number of concurrent trials. The script: + +1. layers the suite onto the runtime image as `assetopsbench/runtime:suite` + (`suite-image/Dockerfile`). The suite lives at `/opt/suite/scenarios_data` + and only `init_data.py` reads it, as in `scenario_suite_runner`; +2. builds the code sandbox image and saves it to `~/assetops-code.tar` for the + per-trial Docker-in-Docker daemon (`overlays/code-sandbox.yaml`); +3. generates one task per scenario with `--runtime-image`, `--data-dir` and + `--skip-missing`, skipping profile entries the suite lacks; +4. runs one Harbor job per model at + `/harbor-jobs/stirrup_agent__`, with credentials loaded + from `.env` by `uv run --env-file` into the Harbor process only. + +Re-running resumes an existing job and finishes only its incomplete trials, +the equivalent of `--skip-existing`. Each trial with the code sandbox runs a +privileged `dind` sidecar, so keep `-n` around 4 on a laptop-sized Docker VM. + ## Why the MCP servers need an explicit env `StirrupAgentRunner._build_mcp_config` sets `env` on every stdio server. It has diff --git a/benchmarks/harbor/adapter/generate_tasks.py b/benchmarks/harbor/adapter/generate_tasks.py index caf07c2e0..f5083d021 100644 --- a/benchmarks/harbor/adapter/generate_tasks.py +++ b/benchmarks/harbor/adapter/generate_tasks.py @@ -21,7 +21,9 @@ import argparse import json +import re import shutil +import sys from pathlib import Path import yaml @@ -79,6 +81,8 @@ def generate( template: Path, output_dir: Path, overwrite: bool, + runtime_image: str | None = None, + data_dir: str | None = None, ) -> Path: source = scenario_root / f"scenario_{scenario_id}" if not source.is_dir(): @@ -108,6 +112,38 @@ def generate( f'scoring_method = "{scoring_method_for(source)}"', ) text = text.replace("init_data.py 1", f"init_data.py {scenario_id}") + # The template's description and keywords are scenario 1's; replace + # them wholesale rather than leak that text into every task. + text = re.sub( + r'^description = ".*"$', + f'description = "AssetOpsBench scenario {scenario_id}, ' + f'{category} category."', + text, + flags=re.MULTILINE, + ) + text = text.replace( + '"assetopsbench", "wosr",', f'"assetopsbench", "{category}",' + ) + if runtime_image: + text = re.sub( + r"^ARG AOB_RUNTIME_IMAGE=.*$", + f"ARG AOB_RUNTIME_IMAGE={runtime_image}", + text, + flags=re.MULTILINE, + ) + if data_dir: + # A suite baked in at data_dir: the per-task layer copies the + # scenario there, and only init_data.py reads it. The agent's own + # SCENARIOS_DATA_DIR stays on the repo copy, as in + # scenario_suite_runner, which sets it for the data load alone. + text = text.replace( + "/opt/aob/src/couchdb/scenarios_data/", f"{data_dir.rstrip('/')}/" + ) + text = text.replace( + 'command = "uv run python src/couchdb/init_data.py', + f'command = "SCENARIOS_DATA_DIR={data_dir} ' + "uv run python src/couchdb/init_data.py", + ) target.write_text(text, encoding="utf-8") # The question the agent sees. @@ -202,11 +238,34 @@ def main() -> int: help="Harbor dataset name written into dataset.toml.", ) parser.add_argument("--overwrite", action="store_true") + parser.add_argument( + "--runtime-image", + help="Image each task builds FROM (default: the template's " + "assetopsbench/runtime:dev). Use the suite image for an external suite.", + ) + parser.add_argument( + "--data-dir", + help="Path of the scenario suite INSIDE the runtime image, e.g. " + "/opt/suite/scenarios_data. The data load reads it; the agent does not.", + ) + parser.add_argument( + "--skip-missing", + action="store_true", + help="Warn and skip profile scenarios with no folder under " + "--scenario-root, instead of failing.", + ) args = parser.parse_args() args.output_dir.mkdir(parents=True, exist_ok=True) written = [] + skipped = [] for category, scenario_id in scenario_ids_by_category(args.profile): + if ( + args.skip_missing + and not (args.scenario_root / f"scenario_{scenario_id}").is_dir() + ): + skipped.append(f"{category}-{scenario_id}") + continue written.append( generate( category=category, @@ -215,8 +274,16 @@ def main() -> int: template=args.template, output_dir=args.output_dir, overwrite=args.overwrite, + runtime_image=args.runtime_image, + data_dir=args.data_dir, ) ) + if skipped: + print( + f"skipped {len(skipped)} scenario(s) missing from " + f"{args.scenario_root}: {', '.join(skipped)}", + file=sys.stderr, + ) write_dataset_files( output_dir=args.output_dir, dataset_name=args.dataset_name, task_dirs=written diff --git a/benchmarks/harbor/run.sh b/benchmarks/harbor/run.sh new file mode 100755 index 000000000..d5ed15f77 --- /dev/null +++ b/benchmarks/harbor/run.sh @@ -0,0 +1,160 @@ +#!/usr/bin/env bash +# Harbor counterpart of benchmarks/run.sh. +# +# Same scenarios, same stirrup-agent CLI and the same code-execution sandbox, +# but each scenario runs as a Harbor trial with its own Compose project and its +# own CouchDB, so scenarios run concurrently instead of one at a time against a +# shared database. +# +# bash benchmarks/harbor/run.sh -s SCENARIO_DIR -l LEADERBOARD_DIR \ +# [-n N_CONCURRENT] [-p PROFILE] [-m "MODEL_ID REASONING_EFFORT"]... +# +# Prerequisites: Docker running, `uv sync --extra harbor`, and the runtime image +# +# docker build -t assetopsbench/runtime:dev \ +# -f benchmarks/harbor/base-image/Dockerfile . +# +# Credentials are read from ENV_FILE (default .env) by `uv run --env-file`, into +# the Harbor process only; StirrupAgent forwards them to the agent phase. They +# never enter an image. +# +# One Harbor job per model, at LEADERBOARD_DIR/harbor-jobs/stirrup_agent__. +# Re-running resumes that job, finishing only the trials it has not completed, +# which is the equivalent of run.sh's --skip-existing. + +set -euo pipefail + +usage() { + printf 'Usage: %s -s SCENARIO_DIR -l LEADERBOARD_DIR [-n N_CONCURRENT] [-p PROFILE] [-m "MODEL_ID EFFORT"]...\n' "$0" >&2 +} + +scenario_dir="${SCENARIO_DIR:-}" +leaderboard_dir="${LEADERBOARD_DIR:-}" +n_concurrent="${N_CONCURRENT:-4}" +profile="${PROFILE:-benchmarks/scenario_suite/all.yaml}" +env_file="${ENV_FILE:-.env}" +model_configs=() + +while getopts ':s:l:n:p:m:' option; do + case "$option" in + s) scenario_dir="$OPTARG" ;; + l) leaderboard_dir="$OPTARG" ;; + n) n_concurrent="$OPTARG" ;; + p) profile="$OPTARG" ;; + m) model_configs+=("$OPTARG") ;; + :) printf 'Option -%s requires an argument.\n' "$OPTARG" >&2; usage; exit 2 ;; + \?) printf 'Unknown option: -%s\n' "$OPTARG" >&2; usage; exit 2 ;; + esac +done + +if [[ -z "$scenario_dir" || -z "$leaderboard_dir" ]]; then + usage + exit 2 +fi + +if [[ -z "${model_configs[*]+set}" ]]; then + model_configs=( + "litellm_proxy/gcp/gemini-3.6-flash high" + "litellm_proxy/azure/gpt-5.6-sol max" + "litellm_proxy/aws/claude-opus-5 high" + "litellm_proxy/aws/claude-sonnet-5 max" + "tokenrouter/MiniMax-M3 high" + "tokenrouter/moonshotai/kimi-k3 max" + "tokenrouter/z-ai/glm-5.3 max" + "tokenrouter/deepseek/deepseek-v4-flash max" + ) +fi + +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +cd "$repo_root" + +scenario_dir="$(cd "$scenario_dir" && pwd)" +mkdir -p "$leaderboard_dir" +leaderboard_dir="$(cd "$leaderboard_dir" && pwd)" +jobs_dir="$leaderboard_dir/harbor-jobs" + +if [[ ! -f "$env_file" ]]; then + printf 'Credentials file not found: %s (set ENV_FILE)\n' "$env_file" >&2 + exit 2 +fi + +runtime_image=assetopsbench/runtime:dev +suite_image=assetopsbench/runtime:suite +suite_data_dir=/opt/suite/scenarios_data +code_image=assetops-code:dev +code_tar="${AOB_CODE_TAR:-$HOME/assetops-code.tar}" +dataset_dir=benchmarks/harbor/datasets/assetopsbench-suite + +if ! docker image inspect "$runtime_image" >/dev/null 2>&1; then + printf 'Runtime image %s not found. Build it first:\n' "$runtime_image" >&2 + printf ' docker build -t %s -f benchmarks/harbor/base-image/Dockerfile .\n' "$runtime_image" >&2 + exit 1 +fi + +# The suite layer. Docker's cache makes this a no-op when neither the runtime +# image nor the suite changed. +echo "Building $suite_image from $scenario_dir" +docker build -q -t "$suite_image" \ + --build-arg "AOB_RUNTIME_IMAGE=$runtime_image" \ + -f benchmarks/harbor/suite-image/Dockerfile "$scenario_dir" + +# The code sandbox image, as a tar each trial's Docker-in-Docker daemon loads +# (benchmarks/harbor/overlays/code-sandbox.yaml). +if [[ ! -s "$code_tar" ]]; then + echo "Building $code_image and saving it to $code_tar" + docker build -q -t "$code_image" \ + -f src/agent/stirrup_agent/Dockerfile.code src/agent/stirrup_agent + docker save "$code_image" -o "$code_tar" + chmod 644 "$code_tar" +fi +export AOB_CODE_TAR="$code_tar" AOB_CODE_IMAGE="$code_image" + +# Regenerate from scratch: the generator overwrites tasks but never removes +# them, so a folder left over from a larger profile would join this run. +rm -rf "$dataset_dir" +uv run python benchmarks/harbor/adapter/generate_tasks.py \ + --scenario-root "$scenario_dir" \ + --profile "$profile" \ + --output-dir "$dataset_dir" \ + --dataset-name assetopsbench/suite \ + --runtime-image "$suite_image" \ + --data-dir "$suite_data_dir" \ + --skip-missing \ + --overwrite >/dev/null + +for model_config in "${model_configs[@]}"; do + read -r model_id reasoning_effort <<< "$model_config" + [[ -z "${model_id:-}" ]] && continue + + model_slug="$(printf '%s' "$model_id" | tr -c 'A-Za-z0-9._-' '-' | tr -s '-')" + job_name="stirrup_agent__${model_slug%-}" + job_path="$jobs_dir/$job_name" + + echo "Running $model_id with reasoning effort ${reasoning_effort:-default} -> $job_path" + + if [[ -f "$job_path/config.json" ]]; then + uv run --env-file "$env_file" harbor jobs resume -p "$job_path" || true + continue + fi + + effort_args=() + if [[ -n "${reasoning_effort:-}" ]]; then + effort_args=(--ak "reasoning_effort=$reasoning_effort") + fi + + # --continue-on-error equivalent: a failed trial is recorded in the job and + # the loop moves on to the next model. + uv run --env-file "$env_file" harbor run -y \ + -p "$dataset_dir" \ + --agent assetops_harbor.stirrup:StirrupAgent \ + --model "$model_id" \ + --ak code_enabled=true \ + --ak code_backend=docker \ + --ak allow_docker_backend=true \ + --ak workspace_dir=/workspace-share \ + ${effort_args[@]+"${effort_args[@]}"} \ + --extra-docker-compose benchmarks/harbor/overlays/code-sandbox.yaml \ + --n-concurrent "$n_concurrent" \ + --job-name "$job_name" \ + -o "$jobs_dir" || true +done diff --git a/benchmarks/harbor/suite-image/Dockerfile b/benchmarks/harbor/suite-image/Dockerfile new file mode 100644 index 000000000..aeca8224d --- /dev/null +++ b/benchmarks/harbor/suite-image/Dockerfile @@ -0,0 +1,24 @@ +# Runtime image plus an external scenario suite, for running a suite that +# does not ship in the repo (e.g. AssetOpsBenchScenarioGeneration/scenarios_data). +# +# The suite's shared/ data is large (GBs of IoT history) and every scenario's +# manifest resolves against it, so it goes in one layer that all tasks share, +# rather than into each task's build context. The build context is the suite +# folder itself: +# +# docker build -t assetopsbench/runtime:suite \ +# -f benchmarks/harbor/suite-image/Dockerfile /scenarios_data +# +# Generate tasks against it with +# +# generate_tasks.py --scenario-root /scenarios_data \ +# --runtime-image assetopsbench/runtime:suite \ +# --data-dir /opt/suite/scenarios_data +# +# The suite sits beside the repo copy, not over it: only init_data.py reads it +# (through SCENARIOS_DATA_DIR on the healthcheck), matching +# scenario_suite_runner, which sets that variable for the data load alone. +ARG AOB_RUNTIME_IMAGE=assetopsbench/runtime:dev +FROM ${AOB_RUNTIME_IMAGE} + +COPY . /opt/suite/scenarios_data/