From bba3fa6956ce41f3158ec5c685e69fdf05b2858c Mon Sep 17 00:00:00 2001 From: Shuxin Lin Date: Mon, 28 Sep 2026 11:22:03 -0400 Subject: [PATCH 1/5] feat(harbor): generate tasks against an external scenario corpus generate_tasks.py gains three options for corpora that do not ship in the repo, such as AssetOpsBenchScenarioGeneration/scenarios_data: - --runtime-image: the image each task builds FROM. - --data-dir: where the corpus lives inside that image. The per-task layer copies the scenario there and the healthcheck runs init_data.py with SCENARIOS_DATA_DIR pointed at it; the agent keeps the repo copy, as scenario_suite_runner does. - --skip-missing: warn and skip profile scenarios absent from the corpus (all.yaml lists wosr-62, which the corpus lacks). corpus-image/Dockerfile bakes the corpus (2.1 GB, mostly shared/iot) into one layer over the runtime image, so tasks share it instead of each carrying it in their build context. Also stop copying scenario 1's description and the wosr keyword into every generated task. Signed-off-by: Shuxin Lin --- benchmarks/harbor/adapter/generate_tasks.py | 65 +++++++++++++++++++++ benchmarks/harbor/corpus-image/Dockerfile | 24 ++++++++ 2 files changed, 89 insertions(+) create mode 100644 benchmarks/harbor/corpus-image/Dockerfile diff --git a/benchmarks/harbor/adapter/generate_tasks.py b/benchmarks/harbor/adapter/generate_tasks.py index caf07c2e0..d8ec1af4f 100644 --- a/benchmarks/harbor/adapter/generate_tasks.py +++ b/benchmarks/harbor/adapter/generate_tasks.py @@ -21,7 +21,9 @@ import argparse import json +import re import shutil +import sys from pathlib import Path import yaml @@ -79,6 +81,8 @@ def generate( template: Path, output_dir: Path, overwrite: bool, + runtime_image: str | None = None, + data_dir: str | None = None, ) -> Path: source = scenario_root / f"scenario_{scenario_id}" if not source.is_dir(): @@ -108,6 +112,36 @@ def generate( f'scoring_method = "{scoring_method_for(source)}"', ) text = text.replace("init_data.py 1", f"init_data.py {scenario_id}") + # The template's description and keywords are scenario 1's; replace + # them wholesale rather than leak that text into every task. + text = re.sub( + r'^description = ".*"$', + f'description = "AssetOpsBench scenario {scenario_id}, ' + f'{category} category."', + text, + flags=re.MULTILINE, + ) + text = text.replace('"assetopsbench", "wosr",', f'"assetopsbench", "{category}",') + if runtime_image: + text = re.sub( + r"^ARG AOB_RUNTIME_IMAGE=.*$", + f"ARG AOB_RUNTIME_IMAGE={runtime_image}", + text, + flags=re.MULTILINE, + ) + if data_dir: + # A corpus baked in at data_dir: the per-task layer copies the + # scenario there, and only init_data.py reads it. The agent's own + # SCENARIOS_DATA_DIR stays on the repo copy, as in + # scenario_suite_runner, which sets it for the data load alone. + text = text.replace( + "/opt/aob/src/couchdb/scenarios_data/", f"{data_dir.rstrip('/')}/" + ) + text = text.replace( + 'command = "uv run python src/couchdb/init_data.py', + f'command = "SCENARIOS_DATA_DIR={data_dir} ' + "uv run python src/couchdb/init_data.py", + ) target.write_text(text, encoding="utf-8") # The question the agent sees. @@ -202,11 +236,34 @@ def main() -> int: help="Harbor dataset name written into dataset.toml.", ) parser.add_argument("--overwrite", action="store_true") + parser.add_argument( + "--runtime-image", + help="Image each task builds FROM (default: the template's " + "assetopsbench/runtime:dev). Use the corpus image for an external corpus.", + ) + parser.add_argument( + "--data-dir", + help="Path of the scenario corpus INSIDE the runtime image, e.g. " + "/opt/corpus/scenarios_data. The data load reads it; the agent does not.", + ) + parser.add_argument( + "--skip-missing", + action="store_true", + help="Warn and skip profile scenarios with no folder under " + "--scenario-root, instead of failing.", + ) args = parser.parse_args() args.output_dir.mkdir(parents=True, exist_ok=True) written = [] + skipped = [] for category, scenario_id in scenario_ids_by_category(args.profile): + if ( + args.skip_missing + and not (args.scenario_root / f"scenario_{scenario_id}").is_dir() + ): + skipped.append(f"{category}-{scenario_id}") + continue written.append( generate( category=category, @@ -215,8 +272,16 @@ def main() -> int: template=args.template, output_dir=args.output_dir, overwrite=args.overwrite, + runtime_image=args.runtime_image, + data_dir=args.data_dir, ) ) + if skipped: + print( + f"skipped {len(skipped)} scenario(s) missing from " + f"{args.scenario_root}: {', '.join(skipped)}", + file=sys.stderr, + ) write_dataset_files( output_dir=args.output_dir, dataset_name=args.dataset_name, task_dirs=written diff --git a/benchmarks/harbor/corpus-image/Dockerfile b/benchmarks/harbor/corpus-image/Dockerfile new file mode 100644 index 000000000..62abf36fe --- /dev/null +++ b/benchmarks/harbor/corpus-image/Dockerfile @@ -0,0 +1,24 @@ +# Runtime image plus an external scenario corpus, for running a corpus that +# does not ship in the repo (e.g. AssetOpsBenchScenarioGeneration/scenarios_data). +# +# The corpus's shared/ data is large (GBs of IoT history) and every scenario's +# manifest resolves against it, so it goes in one layer that all tasks share, +# rather than into each task's build context. The build context is the corpus +# folder itself: +# +# docker build -t assetopsbench/runtime:corpus \ +# -f benchmarks/harbor/corpus-image/Dockerfile /scenarios_data +# +# Generate tasks against it with +# +# generate_tasks.py --scenario-root /scenarios_data \ +# --runtime-image assetopsbench/runtime:corpus \ +# --data-dir /opt/corpus/scenarios_data +# +# The corpus sits beside the repo copy, not over it: only init_data.py reads it +# (through SCENARIOS_DATA_DIR on the healthcheck), matching +# scenario_suite_runner, which sets that variable for the data load alone. +ARG AOB_RUNTIME_IMAGE=assetopsbench/runtime:dev +FROM ${AOB_RUNTIME_IMAGE} + +COPY . /opt/corpus/scenarios_data/ From 2dd44bcce3247727f8a545b3c4f3b0d50b870467 Mon Sep 17 00:00:00 2001 From: Shuxin Lin Date: Mon, 28 Sep 2026 11:59:58 -0400 Subject: [PATCH 2/5] style(harbor): ruff format generate_tasks.py Signed-off-by: Shuxin Lin --- benchmarks/harbor/adapter/generate_tasks.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/benchmarks/harbor/adapter/generate_tasks.py b/benchmarks/harbor/adapter/generate_tasks.py index d8ec1af4f..adcd94476 100644 --- a/benchmarks/harbor/adapter/generate_tasks.py +++ b/benchmarks/harbor/adapter/generate_tasks.py @@ -121,7 +121,9 @@ def generate( text, flags=re.MULTILINE, ) - text = text.replace('"assetopsbench", "wosr",', f'"assetopsbench", "{category}",') + text = text.replace( + '"assetopsbench", "wosr",', f'"assetopsbench", "{category}",' + ) if runtime_image: text = re.sub( r"^ARG AOB_RUNTIME_IMAGE=.*$", From 7be1922e6bbd8c05b3f72477f17bad1e00dc93e7 Mon Sep 17 00:00:00 2001 From: Shuxin Lin Date: Mon, 28 Sep 2026 11:59:59 -0400 Subject: [PATCH 3/5] feat(harbor): run.sh, the Harbor counterpart of benchmarks/run.sh Runs a scenario corpus through Harbor with the same profile, the same stirrup-agent and the same Docker code sandbox as benchmarks/run.sh, but with scenarios in parallel, each trial on its own CouchDB. - Builds assetopsbench/runtime:corpus from -s, and the code sandbox tar for the per-trial Docker-in-Docker daemon. - Regenerates the dataset from scratch (the generator never removes stale task folders), skipping profile entries the corpus lacks. - One Harbor job per model under /harbor-jobs; re-running resumes it, the equivalent of --skip-existing. - Credentials reach the Harbor process through uv run --env-file only. Uses [[ -z "${arr[*]+set}" ]] and ${arr[@]+...} so empty arrays work under set -u in macOS's bash 3.2. Signed-off-by: Shuxin Lin --- benchmarks/harbor/README.md | 33 ++++++++ benchmarks/harbor/run.sh | 160 ++++++++++++++++++++++++++++++++++++ 2 files changed, 193 insertions(+) create mode 100755 benchmarks/harbor/run.sh diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index 89bcbcce3..45ab496b9 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -67,6 +67,39 @@ uv run harbor run \ --n-concurrent 16 ``` +## Running a full scenario corpus (the `benchmarks/run.sh` equivalent) + +`benchmarks/harbor/run.sh` runs what `benchmarks/run.sh` runs, the same +profile, the same `stirrup-agent` and the same Docker code sandbox, with the +scenarios in parallel: + +```bash +bash benchmarks/harbor/run.sh \ + -s /AssetOpsBenchScenarioGeneration/scenarios_data \ + -l ~/AssetOpsBenchRuns/leaderboard \ + -n 4 \ + -m "litellm_proxy/aws/claude-opus-5 high" +``` + +`-m` may repeat; without it the script runs `benchmarks/run.sh`'s model list. +`-p` picks the profile (default `benchmarks/scenario_suite/all.yaml`), and `-n` +the number of concurrent trials. The script: + +1. layers the corpus onto the runtime image as `assetopsbench/runtime:corpus` + (`corpus-image/Dockerfile`). The corpus lives at `/opt/corpus/scenarios_data` + and only `init_data.py` reads it, as in `scenario_suite_runner`; +2. builds the code sandbox image and saves it to `~/assetops-code.tar` for the + per-trial Docker-in-Docker daemon (`overlays/code-sandbox.yaml`); +3. generates one task per scenario with `--runtime-image`, `--data-dir` and + `--skip-missing`, skipping profile entries the corpus lacks; +4. runs one Harbor job per model at + `/harbor-jobs/stirrup_agent__`, with credentials loaded + from `.env` by `uv run --env-file` into the Harbor process only. + +Re-running resumes an existing job and finishes only its incomplete trials, +the equivalent of `--skip-existing`. Each trial with the code sandbox runs a +privileged `dind` sidecar, so keep `-n` around 4 on a laptop-sized Docker VM. + ## Why the MCP servers need an explicit env `StirrupAgentRunner._build_mcp_config` sets `env` on every stdio server. It has diff --git a/benchmarks/harbor/run.sh b/benchmarks/harbor/run.sh new file mode 100755 index 000000000..0bac95d8b --- /dev/null +++ b/benchmarks/harbor/run.sh @@ -0,0 +1,160 @@ +#!/usr/bin/env bash +# Harbor counterpart of benchmarks/run.sh. +# +# Same scenarios, same stirrup-agent CLI and the same code-execution sandbox, +# but each scenario runs as a Harbor trial with its own Compose project and its +# own CouchDB, so scenarios run concurrently instead of one at a time against a +# shared database. +# +# bash benchmarks/harbor/run.sh -s SCENARIO_DIR -l LEADERBOARD_DIR \ +# [-n N_CONCURRENT] [-p PROFILE] [-m "MODEL_ID REASONING_EFFORT"]... +# +# Prerequisites: Docker running, `uv sync --extra harbor`, and the runtime image +# +# docker build -t assetopsbench/runtime:dev \ +# -f benchmarks/harbor/base-image/Dockerfile . +# +# Credentials are read from ENV_FILE (default .env) by `uv run --env-file`, into +# the Harbor process only; StirrupAgent forwards them to the agent phase. They +# never enter an image. +# +# One Harbor job per model, at LEADERBOARD_DIR/harbor-jobs/stirrup_agent__. +# Re-running resumes that job, finishing only the trials it has not completed, +# which is the equivalent of run.sh's --skip-existing. + +set -euo pipefail + +usage() { + printf 'Usage: %s -s SCENARIO_DIR -l LEADERBOARD_DIR [-n N_CONCURRENT] [-p PROFILE] [-m "MODEL_ID EFFORT"]...\n' "$0" >&2 +} + +scenario_dir="${SCENARIO_DIR:-}" +leaderboard_dir="${LEADERBOARD_DIR:-}" +n_concurrent="${N_CONCURRENT:-4}" +profile="${PROFILE:-benchmarks/scenario_suite/all.yaml}" +env_file="${ENV_FILE:-.env}" +model_configs=() + +while getopts ':s:l:n:p:m:' option; do + case "$option" in + s) scenario_dir="$OPTARG" ;; + l) leaderboard_dir="$OPTARG" ;; + n) n_concurrent="$OPTARG" ;; + p) profile="$OPTARG" ;; + m) model_configs+=("$OPTARG") ;; + :) printf 'Option -%s requires an argument.\n' "$OPTARG" >&2; usage; exit 2 ;; + \?) printf 'Unknown option: -%s\n' "$OPTARG" >&2; usage; exit 2 ;; + esac +done + +if [[ -z "$scenario_dir" || -z "$leaderboard_dir" ]]; then + usage + exit 2 +fi + +if [[ -z "${model_configs[*]+set}" ]]; then + model_configs=( + "litellm_proxy/gcp/gemini-3.6-flash high" + "litellm_proxy/azure/gpt-5.6-sol max" + "litellm_proxy/aws/claude-opus-5 high" + "litellm_proxy/aws/claude-sonnet-5 max" + "tokenrouter/MiniMax-M3 high" + "tokenrouter/moonshotai/kimi-k3 max" + "tokenrouter/z-ai/glm-5.3 max" + "tokenrouter/deepseek/deepseek-v4-flash max" + ) +fi + +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +cd "$repo_root" + +scenario_dir="$(cd "$scenario_dir" && pwd)" +mkdir -p "$leaderboard_dir" +leaderboard_dir="$(cd "$leaderboard_dir" && pwd)" +jobs_dir="$leaderboard_dir/harbor-jobs" + +if [[ ! -f "$env_file" ]]; then + printf 'Credentials file not found: %s (set ENV_FILE)\n' "$env_file" >&2 + exit 2 +fi + +runtime_image=assetopsbench/runtime:dev +corpus_image=assetopsbench/runtime:corpus +corpus_data_dir=/opt/corpus/scenarios_data +code_image=assetops-code:dev +code_tar="${AOB_CODE_TAR:-$HOME/assetops-code.tar}" +dataset_dir=benchmarks/harbor/datasets/assetopsbench-corpus + +if ! docker image inspect "$runtime_image" >/dev/null 2>&1; then + printf 'Runtime image %s not found. Build it first:\n' "$runtime_image" >&2 + printf ' docker build -t %s -f benchmarks/harbor/base-image/Dockerfile .\n' "$runtime_image" >&2 + exit 1 +fi + +# The corpus layer. Docker's cache makes this a no-op when neither the runtime +# image nor the corpus changed. +echo "Building $corpus_image from $scenario_dir" +docker build -q -t "$corpus_image" \ + --build-arg "AOB_RUNTIME_IMAGE=$runtime_image" \ + -f benchmarks/harbor/corpus-image/Dockerfile "$scenario_dir" + +# The code sandbox image, as a tar each trial's Docker-in-Docker daemon loads +# (benchmarks/harbor/overlays/code-sandbox.yaml). +if [[ ! -s "$code_tar" ]]; then + echo "Building $code_image and saving it to $code_tar" + docker build -q -t "$code_image" \ + -f src/agent/stirrup_agent/Dockerfile.code src/agent/stirrup_agent + docker save "$code_image" -o "$code_tar" + chmod 644 "$code_tar" +fi +export AOB_CODE_TAR="$code_tar" AOB_CODE_IMAGE="$code_image" + +# Regenerate from scratch: the generator overwrites tasks but never removes +# them, so a folder left over from a larger profile would join this run. +rm -rf "$dataset_dir" +uv run python benchmarks/harbor/adapter/generate_tasks.py \ + --scenario-root "$scenario_dir" \ + --profile "$profile" \ + --output-dir "$dataset_dir" \ + --dataset-name assetopsbench/corpus \ + --runtime-image "$corpus_image" \ + --data-dir "$corpus_data_dir" \ + --skip-missing \ + --overwrite >/dev/null + +for model_config in "${model_configs[@]}"; do + read -r model_id reasoning_effort <<< "$model_config" + [[ -z "${model_id:-}" ]] && continue + + model_slug="$(printf '%s' "$model_id" | tr -c 'A-Za-z0-9._-' '-' | tr -s '-')" + job_name="stirrup_agent__${model_slug%-}" + job_path="$jobs_dir/$job_name" + + echo "Running $model_id with reasoning effort ${reasoning_effort:-default} -> $job_path" + + if [[ -f "$job_path/config.json" ]]; then + uv run --env-file "$env_file" harbor jobs resume -p "$job_path" || true + continue + fi + + effort_args=() + if [[ -n "${reasoning_effort:-}" ]]; then + effort_args=(--ak "reasoning_effort=$reasoning_effort") + fi + + # --continue-on-error equivalent: a failed trial is recorded in the job and + # the loop moves on to the next model. + uv run --env-file "$env_file" harbor run -y \ + -p "$dataset_dir" \ + --agent assetops_harbor.stirrup:StirrupAgent \ + --model "$model_id" \ + --ak code_enabled=true \ + --ak code_backend=docker \ + --ak allow_docker_backend=true \ + --ak workspace_dir=/workspace-share \ + ${effort_args[@]+"${effort_args[@]}"} \ + --extra-docker-compose benchmarks/harbor/overlays/code-sandbox.yaml \ + --n-concurrent "$n_concurrent" \ + --job-name "$job_name" \ + -o "$jobs_dir" || true +done From 533bc5a0811ef95c1d992a9158856df3fa054951 Mon Sep 17 00:00:00 2001 From: Shuxin Lin Date: Mon, 28 Sep 2026 12:09:16 -0400 Subject: [PATCH 4/5] refactor(harbor): rename corpus to suite corpus-image/ -> suite-image/, assetopsbench/runtime:corpus -> assetopsbench/runtime:suite, /opt/corpus/scenarios_data -> /opt/suite/scenarios_data, and the generated dataset assetopsbench-corpus -> assetopsbench-suite, with matching wording in run.sh, the generator's help and the README. Signed-off-by: Shuxin Lin --- benchmarks/harbor/{corpus-image => suite-image}/Dockerfile | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename benchmarks/harbor/{corpus-image => suite-image}/Dockerfile (100%) diff --git a/benchmarks/harbor/corpus-image/Dockerfile b/benchmarks/harbor/suite-image/Dockerfile similarity index 100% rename from benchmarks/harbor/corpus-image/Dockerfile rename to benchmarks/harbor/suite-image/Dockerfile From c453ad7e11128b038bfb44e87a32c9acef93a351 Mon Sep 17 00:00:00 2001 From: Shuxin Lin Date: Mon, 28 Sep 2026 12:09:30 -0400 Subject: [PATCH 5/5] refactor(harbor): rename corpus to suite in contents Completes 533bc5a, which only moved corpus-image/ to suite-image/: assetopsbench/runtime:corpus -> assetopsbench/runtime:suite, /opt/corpus/scenarios_data -> /opt/suite/scenarios_data, the dataset assetopsbench-corpus -> assetopsbench-suite, and the wording in run.sh, the generator's help and the README. Signed-off-by: Shuxin Lin --- benchmarks/harbor/README.md | 8 ++++---- benchmarks/harbor/adapter/generate_tasks.py | 8 ++++---- benchmarks/harbor/run.sh | 22 ++++++++++----------- benchmarks/harbor/suite-image/Dockerfile | 18 ++++++++--------- 4 files changed, 28 insertions(+), 28 deletions(-) diff --git a/benchmarks/harbor/README.md b/benchmarks/harbor/README.md index 45ab496b9..49de5cf50 100644 --- a/benchmarks/harbor/README.md +++ b/benchmarks/harbor/README.md @@ -67,7 +67,7 @@ uv run harbor run \ --n-concurrent 16 ``` -## Running a full scenario corpus (the `benchmarks/run.sh` equivalent) +## Running a full scenario suite (the `benchmarks/run.sh` equivalent) `benchmarks/harbor/run.sh` runs what `benchmarks/run.sh` runs, the same profile, the same `stirrup-agent` and the same Docker code sandbox, with the @@ -85,13 +85,13 @@ bash benchmarks/harbor/run.sh \ `-p` picks the profile (default `benchmarks/scenario_suite/all.yaml`), and `-n` the number of concurrent trials. The script: -1. layers the corpus onto the runtime image as `assetopsbench/runtime:corpus` - (`corpus-image/Dockerfile`). The corpus lives at `/opt/corpus/scenarios_data` +1. layers the suite onto the runtime image as `assetopsbench/runtime:suite` + (`suite-image/Dockerfile`). The suite lives at `/opt/suite/scenarios_data` and only `init_data.py` reads it, as in `scenario_suite_runner`; 2. builds the code sandbox image and saves it to `~/assetops-code.tar` for the per-trial Docker-in-Docker daemon (`overlays/code-sandbox.yaml`); 3. generates one task per scenario with `--runtime-image`, `--data-dir` and - `--skip-missing`, skipping profile entries the corpus lacks; + `--skip-missing`, skipping profile entries the suite lacks; 4. runs one Harbor job per model at `/harbor-jobs/stirrup_agent__`, with credentials loaded from `.env` by `uv run --env-file` into the Harbor process only. diff --git a/benchmarks/harbor/adapter/generate_tasks.py b/benchmarks/harbor/adapter/generate_tasks.py index adcd94476..f5083d021 100644 --- a/benchmarks/harbor/adapter/generate_tasks.py +++ b/benchmarks/harbor/adapter/generate_tasks.py @@ -132,7 +132,7 @@ def generate( flags=re.MULTILINE, ) if data_dir: - # A corpus baked in at data_dir: the per-task layer copies the + # A suite baked in at data_dir: the per-task layer copies the # scenario there, and only init_data.py reads it. The agent's own # SCENARIOS_DATA_DIR stays on the repo copy, as in # scenario_suite_runner, which sets it for the data load alone. @@ -241,12 +241,12 @@ def main() -> int: parser.add_argument( "--runtime-image", help="Image each task builds FROM (default: the template's " - "assetopsbench/runtime:dev). Use the corpus image for an external corpus.", + "assetopsbench/runtime:dev). Use the suite image for an external suite.", ) parser.add_argument( "--data-dir", - help="Path of the scenario corpus INSIDE the runtime image, e.g. " - "/opt/corpus/scenarios_data. The data load reads it; the agent does not.", + help="Path of the scenario suite INSIDE the runtime image, e.g. " + "/opt/suite/scenarios_data. The data load reads it; the agent does not.", ) parser.add_argument( "--skip-missing", diff --git a/benchmarks/harbor/run.sh b/benchmarks/harbor/run.sh index 0bac95d8b..d5ed15f77 100755 --- a/benchmarks/harbor/run.sh +++ b/benchmarks/harbor/run.sh @@ -79,11 +79,11 @@ if [[ ! -f "$env_file" ]]; then fi runtime_image=assetopsbench/runtime:dev -corpus_image=assetopsbench/runtime:corpus -corpus_data_dir=/opt/corpus/scenarios_data +suite_image=assetopsbench/runtime:suite +suite_data_dir=/opt/suite/scenarios_data code_image=assetops-code:dev code_tar="${AOB_CODE_TAR:-$HOME/assetops-code.tar}" -dataset_dir=benchmarks/harbor/datasets/assetopsbench-corpus +dataset_dir=benchmarks/harbor/datasets/assetopsbench-suite if ! docker image inspect "$runtime_image" >/dev/null 2>&1; then printf 'Runtime image %s not found. Build it first:\n' "$runtime_image" >&2 @@ -91,12 +91,12 @@ if ! docker image inspect "$runtime_image" >/dev/null 2>&1; then exit 1 fi -# The corpus layer. Docker's cache makes this a no-op when neither the runtime -# image nor the corpus changed. -echo "Building $corpus_image from $scenario_dir" -docker build -q -t "$corpus_image" \ +# The suite layer. Docker's cache makes this a no-op when neither the runtime +# image nor the suite changed. +echo "Building $suite_image from $scenario_dir" +docker build -q -t "$suite_image" \ --build-arg "AOB_RUNTIME_IMAGE=$runtime_image" \ - -f benchmarks/harbor/corpus-image/Dockerfile "$scenario_dir" + -f benchmarks/harbor/suite-image/Dockerfile "$scenario_dir" # The code sandbox image, as a tar each trial's Docker-in-Docker daemon loads # (benchmarks/harbor/overlays/code-sandbox.yaml). @@ -116,9 +116,9 @@ uv run python benchmarks/harbor/adapter/generate_tasks.py \ --scenario-root "$scenario_dir" \ --profile "$profile" \ --output-dir "$dataset_dir" \ - --dataset-name assetopsbench/corpus \ - --runtime-image "$corpus_image" \ - --data-dir "$corpus_data_dir" \ + --dataset-name assetopsbench/suite \ + --runtime-image "$suite_image" \ + --data-dir "$suite_data_dir" \ --skip-missing \ --overwrite >/dev/null diff --git a/benchmarks/harbor/suite-image/Dockerfile b/benchmarks/harbor/suite-image/Dockerfile index 62abf36fe..aeca8224d 100644 --- a/benchmarks/harbor/suite-image/Dockerfile +++ b/benchmarks/harbor/suite-image/Dockerfile @@ -1,24 +1,24 @@ -# Runtime image plus an external scenario corpus, for running a corpus that +# Runtime image plus an external scenario suite, for running a suite that # does not ship in the repo (e.g. AssetOpsBenchScenarioGeneration/scenarios_data). # -# The corpus's shared/ data is large (GBs of IoT history) and every scenario's +# The suite's shared/ data is large (GBs of IoT history) and every scenario's # manifest resolves against it, so it goes in one layer that all tasks share, -# rather than into each task's build context. The build context is the corpus +# rather than into each task's build context. The build context is the suite # folder itself: # -# docker build -t assetopsbench/runtime:corpus \ -# -f benchmarks/harbor/corpus-image/Dockerfile /scenarios_data +# docker build -t assetopsbench/runtime:suite \ +# -f benchmarks/harbor/suite-image/Dockerfile /scenarios_data # # Generate tasks against it with # # generate_tasks.py --scenario-root /scenarios_data \ -# --runtime-image assetopsbench/runtime:corpus \ -# --data-dir /opt/corpus/scenarios_data +# --runtime-image assetopsbench/runtime:suite \ +# --data-dir /opt/suite/scenarios_data # -# The corpus sits beside the repo copy, not over it: only init_data.py reads it +# The suite sits beside the repo copy, not over it: only init_data.py reads it # (through SCENARIOS_DATA_DIR on the healthcheck), matching # scenario_suite_runner, which sets that variable for the data load alone. ARG AOB_RUNTIME_IMAGE=assetopsbench/runtime:dev FROM ${AOB_RUNTIME_IMAGE} -COPY . /opt/corpus/scenarios_data/ +COPY . /opt/suite/scenarios_data/