diff --git a/.claude/skills/benchmark/SKILL.md b/.claude/skills/benchmark/SKILL.md index 8f6a2e78a..7bcb69942 100644 --- a/.claude/skills/benchmark/SKILL.md +++ b/.claude/skills/benchmark/SKILL.md @@ -55,6 +55,12 @@ go run . -benchnum 5 Multi-component snapshot with configurable parallelism: +```bash +make benchmark_stress +``` + +Or manually: + ```bash cd benchmark/stress ./prepare_data.sh @@ -69,7 +75,30 @@ EC_STRESS_COMPONENTS=50 EC_STRESS_WORKERS=20 go run . Defaults: 10 components, 35 workers. -## Step 5: Profile if needed +## Step 5: Compare against baseline + +The stress benchmark has regression detection. After running: + +```bash +cd benchmark/stress +./compare.sh benchmark-output.txt +``` + +This compares current results against `baseline.json` using thresholds from +`thresholds.json` (default: 15% RSS, 20% ns/op). Exits non-zero on regression. + +## Step 6: Regenerate baseline + +After intentional performance changes, update the stored baseline: + +```bash +make generate-baseline +``` + +This runs the stress benchmark, parses results, and writes `benchmark/stress/baseline.json` +with current metrics, commit SHA, date, and Go version. + +## Step 7: Profile if needed Use the CLI's built-in profiling: @@ -79,7 +108,7 @@ ec validate image --trace=cpu ... # pprof CPU profile ec validate image --trace=mem ... # heap profile ``` -## Step 6: Report results +## Step 8: Report results Output is in standard Go benchmark format (ns/op, memory stats). Summarize: - Benchmark type run (simple/stress) diff --git a/.github/workflows/benchmark.yaml b/.github/workflows/benchmark.yaml new file mode 100644 index 000000000..655982aa2 --- /dev/null +++ b/.github/workflows/benchmark.yaml @@ -0,0 +1,168 @@ +# Copyright The Conforma Contributors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# SPDX-License-Identifier: Apache-2.0 + +--- +name: Stress Benchmark + +"on": + pull_request: + branches: + - main + workflow_dispatch: + +permissions: + contents: read + +jobs: + + stress: + name: Stress Benchmark + runs-on: ubuntu-latest + timeout-minutes: 15 + env: + # Tuned for 4 vCPU / 16 GB CI runners to complete within 5 minutes. + # Code defaults are 10 components / 35 workers. + EC_STRESS_COMPONENTS: "10" + EC_STRESS_WORKERS: "10" + steps: + - name: Harden Runner + uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1 + with: + egress-policy: audit + disable-telemetry: true + + - name: Checkout repository + uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0 + + - name: Restore Cache + uses: actions/cache/restore@caa296126883cff596d87d8935842f9db880ef25 # v5.1.0 + with: + key: main + path: '**' + + - name: Setup Go environment + uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6.5.0 + with: + go-version-file: go.mod + cache: false + + - name: Setup ORAS + uses: oras-project/setup-oras@1d808f7d7f6995cc68b7bf507bfe5c5446e1dc9d # v2.0.1 + + - name: Prepare benchmark data + run: | + cd benchmark/stress + ./prepare_data.sh + + - name: Build stress benchmark + run: go build -o benchmark/stress/stress ./benchmark/stress + + - name: Run stress benchmark + id: bench + continue-on-error: true + run: | + set -o pipefail + cd benchmark/stress + ./stress 2>benchmark-stderr.txt | tee benchmark-output.txt + + - name: Compare against baseline + id: compare + if: steps.bench.outcome == 'success' + run: | + cd benchmark/stress + if [[ -f baseline.json ]]; then + ./compare.sh benchmark-output.txt + else + echo "No baseline found, skipping comparison." + fi + + - name: Write job summary + if: always() + run: | + if [[ "${{ steps.bench.outcome }}" == "failure" && -s benchmark/stress/benchmark-stderr.txt ]]; then + { + echo "### Stderr" + echo '```' + tail -50 benchmark/stress/benchmark-stderr.txt + echo '```' + } >> "$GITHUB_STEP_SUMMARY" + fi + + if [[ ! -f benchmark/stress/benchmark-output.txt ]]; then + echo "## Stress Benchmark" >> "$GITHUB_STEP_SUMMARY" + echo "Benchmark did not produce output." >> "$GITHUB_STEP_SUMMARY" + exit 0 + fi + + line=$(grep '^BenchmarkStress' benchmark/stress/benchmark-output.txt || true) + if [[ -z "$line" ]]; then + echo "## Stress Benchmark" >> "$GITHUB_STEP_SUMMARY" + echo "No benchmark results found in output." >> "$GITHUB_STEP_SUMMARY" + exit 0 + fi + + read -r ns_op peak_rss alloc heap < <(BENCH_LINE="$line" python3 -c " + import os, re + line = os.environ['BENCH_LINE'] + def val(p): + m = re.search(p, line) + return m.group(1) if m else '0' + print(val(r'([\d.]+)\s+ns/op'), val(r'([\d.]+)\s+peak-RSS-bytes'), val(r'([\d.]+)\s+allocated-bytes/op'), val(r'([\d.]+)\s+heap-bytes-from-system')) + ") + + secs=$(awk -v val="${ns_op:-0}" 'BEGIN {printf "%.1f", val / 1000000000}') + rss_mb=$(awk -v val="${peak_rss:-0}" 'BEGIN {printf "%.0f", val / 1048576}') + alloc_mb=$(awk -v val="${alloc:-0}" 'BEGIN {printf "%.0f", val / 1048576}') + heap_mb=$(awk -v val="${heap:-0}" 'BEGIN {printf "%.0f", val / 1048576}') + + has_baseline=false + if [[ -f benchmark/stress/baseline.json ]]; then + has_baseline=true + bl_rss=$(python3 -c "import json; print(json.load(open('benchmark/stress/baseline.json'))['peak_rss_bytes'])") + bl_ns=$(python3 -c "import json; print(json.load(open('benchmark/stress/baseline.json'))['ns_per_op'])") + bl_rss_mb=$(awk -v val="$bl_rss" 'BEGIN {printf "%.0f", val / 1048576}') + bl_secs=$(awk -v val="$bl_ns" 'BEGIN {printf "%.1f", val / 1000000000}') + rss_change=$(awk -v cur="$peak_rss" -v base="$bl_rss" 'BEGIN {printf "%+.1f", ((cur - base) / base) * 100}') + time_change=$(awk -v cur="$ns_op" -v base="$bl_ns" 'BEGIN {printf "%+.1f", ((cur - base) / base) * 100}') + fi + + { + echo "## Stress Benchmark" + echo "" + if [[ "$has_baseline" == "true" ]]; then + echo "| Metric | Current | Baseline | Change | Description |" + echo "|--------|---------|----------|--------|-------------|" + echo "| Components | ${EC_STRESS_COMPONENTS} | | | Snapshot components validated |" + echo "| Workers | ${EC_STRESS_WORKERS} | | | Parallel validation workers |" + echo "| Execution time | ${secs}s | ${bl_secs}s | ${time_change}% | Wall-clock time per iteration |" + echo "| Peak RSS | ${rss_mb} MB | ${bl_rss_mb} MB | ${rss_change}% | Max physical memory used |" + echo "| Allocated memory | ${alloc_mb} MB | | | Total Go heap allocations |" + echo "| Heap from system | ${heap_mb} MB | | | Heap memory requested from OS |" + else + echo "| Metric | Value | Description |" + echo "|--------|-------|-------------|" + echo "| Components | ${EC_STRESS_COMPONENTS} | Snapshot components validated |" + echo "| Workers | ${EC_STRESS_WORKERS} | Parallel validation workers |" + echo "| Execution time | ${secs}s | Wall-clock time per iteration |" + echo "| Peak RSS | ${rss_mb} MB | Max physical memory used |" + echo "| Allocated memory | ${alloc_mb} MB | Total Go heap allocations |" + echo "| Heap from system | ${heap_mb} MB | Heap memory requested from OS |" + fi + if [[ "${{ steps.compare.outcome }}" == "failure" ]]; then + echo "" + echo "> **⚠️ Performance regression detected.** Update the baseline with \`make generate-baseline\` if this is expected." + fi + } >> "$GITHUB_STEP_SUMMARY" diff --git a/AGENTS.md b/AGENTS.md index 1d1cbbd6a..79fadb343 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -59,6 +59,26 @@ Acceptance tests require `/etc/hosts` entries: 127.0.0.1 rekor.localhost ``` +## Benchmarks + +Two benchmarks live under `benchmark/`: + +- **simple/** — Single-component validation against the `@redhat` policy collection. +- **stress/** — Multi-component validation with configurable parallelism + (`EC_STRESS_COMPONENTS`, `EC_STRESS_WORKERS`). + +```bash +make benchmark # Run simple benchmark +make benchmark_stress # Run stress benchmark (via pattern rule) +make generate-baseline # Run stress benchmark and write baseline.json +``` + +The stress benchmark has regression detection: `benchmark/stress/baseline.json` stores +reference metrics (peak RSS, ns/op) and `benchmark/stress/thresholds.json` defines +percentage thresholds. CI runs `benchmark/stress/compare.sh` to compare each run +against the baseline and fails the check on regression. Update the baseline with +`make generate-baseline` after intentional performance changes. + ## Single-File Verification ```bash diff --git a/Makefile b/Makefile index 9f094e69e..efe3d34eb 100644 --- a/Makefile +++ b/Makefile @@ -197,6 +197,28 @@ benchmark_data: benchmark/simple/data.tar.gz ## Prepare data for benchmark .PHONY: benchmark benchmark: benchmark_simple ## Run benchmarks +.PHONY: generate-baseline +generate-baseline: benchmark/stress/data.tar.gz ## Generate stress benchmark baseline + @cd benchmark/stress && \ + go run . 2>benchmark-stderr.txt | tee benchmark-output.txt && \ + python3 -c "\ + import re, json, sys; \ + line = [l for l in open('benchmark-output.txt') if l.startswith('BenchmarkStress')]; \ + line or sys.exit('No BenchmarkStress results found'); \ + line = line[0]; \ + def val(p): \ + m = re.search(p, line); \ + return m.group(1) if m else ''; \ + ns = val(r'([\d.]+)\s+ns/op'); rss = val(r'([\d.]+)\s+peak-RSS-bytes'); \ + (ns and rss) or sys.exit('Failed to parse benchmark metrics'); \ + json.dump({'peak_rss_bytes': int(float(rss)), 'ns_per_op': int(float(ns)), \ + 'components': int('$${EC_STRESS_COMPONENTS:-10}'), 'workers': int('$${EC_STRESS_WORKERS:-10}'), \ + 'commit': '$(shell git rev-parse --short HEAD)', 'date': '$(shell date -u +%Y-%m-%d)', \ + 'go_version': '$(shell go env GOVERSION | sed "s/^go//")' \ + }, open('baseline.json','w'), indent=2); print()" && \ + rm -f benchmark-output.txt benchmark-stderr.txt && \ + echo "Baseline written to benchmark/stress/baseline.json" + .PHONY: tools-ci tools-ci: ## Ensure all tools build cleanly @echo "• tkn:" && \ diff --git a/benchmark/README.md b/benchmark/README.md index fe72b6d6e..733e72b53 100644 --- a/benchmark/README.md +++ b/benchmark/README.md @@ -17,3 +17,25 @@ times. - **stress/** — Multi-component validation with configurable parallelism. Set `EC_STRESS_COMPONENTS` (default 10) and `EC_STRESS_WORKERS` (default 35) to control the workload. + +## Baseline and regression detection + +The stress benchmark stores a performance baseline in +`stress/baseline.json` (peak RSS and ns/op) along with configurable +regression thresholds in `stress/thresholds.json`. The CI workflow +compares each run against the baseline and fails the check when a metric +exceeds its threshold. + +To regenerate the baseline after an intentional change: + +``` +make generate-baseline +``` + +This runs the stress benchmark locally, parses the results, and writes a +new `baseline.json` with the current commit SHA, date, Go version, and +worker/component counts. + +Thresholds are expressed as percentages (e.g., 15 means a 15% increase +triggers a failure). Adjust them in `stress/thresholds.json` as +optimizations land. diff --git a/benchmark/stress/baseline.json b/benchmark/stress/baseline.json new file mode 100644 index 000000000..9260dea7d --- /dev/null +++ b/benchmark/stress/baseline.json @@ -0,0 +1,9 @@ +{ + "peak_rss_bytes": 2250485760, + "ns_per_op": 2567888013, + "components": 10, + "workers": 10, + "commit": "fc37eb13", + "date": "2026-08-11", + "go_version": "1.26.3" +} diff --git a/benchmark/stress/compare.sh b/benchmark/stress/compare.sh new file mode 100755 index 000000000..f0462703d --- /dev/null +++ b/benchmark/stress/compare.sh @@ -0,0 +1,108 @@ +#!/bin/bash +# Copyright The Conforma Contributors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# SPDX-License-Identifier: Apache-2.0 + +# Compares current benchmark results against a stored baseline and exits +# non-zero if any metric regresses beyond the configured threshold. +set -o errexit +set -o nounset +set -o pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +BASELINE="${SCRIPT_DIR}/baseline.json" +THRESHOLDS="${SCRIPT_DIR}/thresholds.json" +BENCHMARK_OUTPUT="${1:-${SCRIPT_DIR}/benchmark-output.txt}" + +if [[ ! -f "$BASELINE" ]]; then + echo "No baseline found, skipping comparison." + exit 0 +fi + +if [[ ! -f "$THRESHOLDS" ]]; then + echo "No thresholds file found, skipping comparison." + exit 0 +fi + +if [[ ! -f "$BENCHMARK_OUTPUT" ]]; then + echo "No benchmark output found at ${BENCHMARK_OUTPUT}" + exit 1 +fi + +line=$(grep '^BenchmarkStress' "$BENCHMARK_OUTPUT" || true) +if [[ -z "$line" ]]; then + echo "No BenchmarkStress results found in output." + exit 1 +fi + +read -r current_ns current_rss baseline_ns baseline_rss threshold_rss threshold_time < <( + BENCH_LINE="${line}" BASELINE_PATH="${BASELINE}" THRESHOLDS_PATH="${THRESHOLDS}" python3 -c " +import json, os, re, sys +line = os.environ['BENCH_LINE'] +def extract(pattern): + m = re.search(pattern, line) + return m.group(1) if m else '' +ns = extract(r'([\d.]+)\s+ns/op') +rss = extract(r'([\d.]+)\s+peak-RSS-bytes') +if not ns or not rss: + print('Failed to parse benchmark metrics from output.', file=sys.stderr) + sys.exit(1) +b = json.load(open(os.environ['BASELINE_PATH'])) +t = json.load(open(os.environ['THRESHOLDS_PATH'])) +print(ns, rss, b['ns_per_op'], b['peak_rss_bytes'], t['peak_rss_percent'], t['ns_per_op_percent']) +" +) + +if awk -v b="$baseline_rss" -v t="$baseline_ns" 'BEGIN {exit !(b==0 || t==0)}'; then + echo "Baseline contains zero values, cannot compute regression." + exit 1 +fi + +rss_change=$(awk -v cur="$current_rss" -v base="$baseline_rss" 'BEGIN {printf "%.1f", ((cur - base) / base) * 100}') +time_change=$(awk -v cur="$current_ns" -v base="$baseline_ns" 'BEGIN {printf "%.1f", ((cur - base) / base) * 100}') + +baseline_rss_mb=$(awk -v val="$baseline_rss" 'BEGIN {printf "%.0f", val / 1048576}') +current_rss_mb=$(awk -v val="$current_rss" 'BEGIN {printf "%.0f", val / 1048576}') +baseline_secs=$(awk -v val="$baseline_ns" 'BEGIN {printf "%.1f", val / 1000000000}') +current_secs=$(awk -v val="$current_ns" 'BEGIN {printf "%.1f", val / 1000000000}') + +echo "" +echo "=== Benchmark Comparison ===" +echo "" +printf "%-20s %10s %10s %10s %10s\n" "Metric" "Baseline" "Current" "Change" "Threshold" +printf "%-20s %10s %10s %9s%% %9s%%\n" "Peak RSS" "${baseline_rss_mb} MB" "${current_rss_mb} MB" "$rss_change" "$threshold_rss" +printf "%-20s %10s %10s %9s%% %9s%%\n" "Execution time" "${baseline_secs}s" "${current_secs}s" "$time_change" "$threshold_time" +echo "" + +failed=0 + +rss_exceeded=$(awk -v change="$rss_change" -v thresh="$threshold_rss" 'BEGIN {print (change > thresh) ? 1 : 0}') +time_exceeded=$(awk -v change="$time_change" -v thresh="$threshold_time" 'BEGIN {print (change > thresh) ? 1 : 0}') + +if [[ "$rss_exceeded" == "1" ]]; then + echo "FAIL: Peak RSS regressed by ${rss_change}% (threshold: ${threshold_rss}%)" + failed=1 +fi + +if [[ "$time_exceeded" == "1" ]]; then + echo "FAIL: Execution time regressed by ${time_change}% (threshold: ${threshold_time}%)" + failed=1 +fi + +if [[ "$failed" == "0" ]]; then + echo "PASS: No regressions detected." +fi + +exit "$failed" diff --git a/benchmark/stress/stress.go b/benchmark/stress/stress.go index da7538484..ddb5d1fc7 100644 --- a/benchmark/stress/stress.go +++ b/benchmark/stress/stress.go @@ -165,6 +165,7 @@ func ec(dir string, components, workers int) func() { strconv.Itoa(workers), "--effective-time", "2024-12-10T00:00:00Z", + "--allow-past-effective-time", }); err != nil { panic(err) } diff --git a/benchmark/stress/thresholds.json b/benchmark/stress/thresholds.json new file mode 100644 index 000000000..660f6f93a --- /dev/null +++ b/benchmark/stress/thresholds.json @@ -0,0 +1,4 @@ +{ + "peak_rss_percent": 15, + "ns_per_op_percent": 20 +}