diff --git a/.github/workflows/rust-benchmark.yml b/.github/workflows/rust-benchmark.yml index 3494136..5d5659f 100644 --- a/.github/workflows/rust-benchmark.yml +++ b/.github/workflows/rust-benchmark.yml @@ -3,14 +3,18 @@ name: Rust Benchmark on: push: branches: [main, master] - paths: ['rust/**', 'spacetime-module/**', '.github/workflows/rust-benchmark.yml'] + paths: ['rust/**', '.github/workflows/rust-benchmark.yml'] pull_request: branches: [main, master] - paths: ['rust/**', 'spacetime-module/**', '.github/workflows/rust-benchmark.yml'] + paths: ['rust/**', '.github/workflows/rust-benchmark.yml'] env: CARGO_TERM_COLOR: always RUST_BACKTRACE: 1 + # Pinned nightly, kept in sync with rust/rust-toolchain.toml. The patched + # doublets crates rely on unstable APIs, so a rolling nightly breaks the build + # without any change in this repository. + toolchain: nightly-2026-04-14 defaults: run: @@ -28,10 +32,10 @@ jobs: steps: - uses: actions/checkout@v4 - - name: Setup Rust (nightly) + - name: Setup Rust (pinned nightly) uses: dtolnay/rust-toolchain@master with: - toolchain: nightly + toolchain: ${{ env.toolchain }} components: rustfmt, clippy targets: wasm32-unknown-unknown @@ -89,6 +93,27 @@ jobs: # Run tests sequentially to avoid parallel interference with shared SpacetimeDB state. run: cargo test -- --test-threads=1 + # Unit tests for the reporting pipeline (rust/out.py). They are cheap and must + # pass before a benchmark is run, otherwise a 40 minute benchmark could finish + # only to fail while publishing its results. + results-pipeline: + name: Results pipeline tests + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@v4 + + - name: Setup Python + uses: actions/setup-python@v5 + with: + python-version: '3.11' + + - name: Install Python dependencies + run: pip install matplotlib numpy + + - name: Run results pipeline tests + run: python3 -m unittest test_out -v + # Quick benchmark validation for pull requests. # Runs benchmarks with reduced scale to verify they work and produce results # in well under 10 minutes. Results are not committed but uploaded as artifacts. @@ -103,16 +128,21 @@ jobs: benchmark-pr: name: Benchmark (PR validation) runs-on: ubuntu-latest - needs: [test] + needs: [test, results-pipeline] if: github.event_name == 'pull_request' timeout-minutes: 20 steps: - uses: actions/checkout@v4 - - name: Setup Rust (nightly) + - name: Setup Rust (pinned nightly) uses: dtolnay/rust-toolchain@master with: - toolchain: nightly + toolchain: ${{ env.toolchain }} + # rust/rust-toolchain.toml requires these components. Installing them + # with the toolchain avoids rustup adding them lazily on the first + # cargo invocation, which fails on the runner image with + # "detected conflict: 'bin/cargo-fmt'". + components: rustfmt, clippy targets: wasm32-unknown-unknown - name: Setup Python @@ -173,6 +203,12 @@ jobs: SPACETIMEDB_URI: http://localhost:3000 SPACETIMEDB_DB: benchmark-links run: | + set -o pipefail + # Criterion compares against target/criterion///base and prints + # the failure to *stdout* when that directory was restored incompletely by + # the cache, which splices "Criterion.rs ERROR: ..." into the middle of the + # bencher records (see CI run 35028280108). Start from a clean state. + rm -rf target/criterion cargo bench --bench bench -- \ --output-format bencher \ --sample-size 10 \ @@ -181,15 +217,33 @@ jobs: --nresamples 1000 \ | tee out.txt - - name: Generate charts - run: python3 out.py + - name: Generate results table and charts + env: + BENCHMARK_LINK_COUNT: 10 + BACKGROUND_LINK_COUNT: 30 + run: python3 out.py out.txt --results results.md + + - name: Publish results to the job summary + run: | + { + echo "## Benchmark results (PR validation, reduced scale)" + echo + cat results.md + echo + echo "_Reduced scale: these numbers only prove the benchmark runs;" + echo "the published results come from the full run on \`main\`._" + } >> "$GITHUB_STEP_SUMMARY" - name: Upload PR benchmark artifacts + # Keep the raw output even when the results step fails, + # so a broken run can be diagnosed from the artifact. + if: always() uses: actions/upload-artifact@v4 with: name: benchmark-results-pr path: | rust/out.txt + rust/results.md rust/bench_rust.png rust/bench_rust_log_scale.png @@ -205,19 +259,27 @@ jobs: benchmark: name: Benchmark (full) runs-on: ubuntu-latest - needs: [test] + needs: [test, results-pipeline] if: github.event_name == 'push' && (github.ref == 'refs/heads/main' || github.ref == 'refs/heads/master') timeout-minutes: 180 + # Required to push the regenerated README table and docs/benchmarks/ charts back. + permissions: + contents: write steps: - uses: actions/checkout@v4 with: fetch-depth: 0 token: ${{ secrets.GITHUB_TOKEN }} - - name: Setup Rust (nightly) + - name: Setup Rust (pinned nightly) uses: dtolnay/rust-toolchain@master with: - toolchain: nightly + toolchain: ${{ env.toolchain }} + # rust/rust-toolchain.toml requires these components. Installing them + # with the toolchain avoids rustup adding them lazily on the first + # cargo invocation, which fails on the runner image with + # "detected conflict: 'bin/cargo-fmt'". + components: rustfmt, clippy targets: wasm32-unknown-unknown - name: Setup Python @@ -278,31 +340,66 @@ jobs: SPACETIMEDB_URI: http://localhost:3000 SPACETIMEDB_DB: benchmark-links run: | + set -o pipefail + # Criterion compares against target/criterion///base and prints + # the failure to *stdout* when that directory was restored incompletely by + # the cache, which splices "Criterion.rs ERROR: ..." into the middle of the + # bencher records (see CI run 35028280108). Start from a clean state. + rm -rf target/criterion cargo bench --bench bench -- \ --output-format bencher \ --sample-size 20 \ --nresamples 10000 \ | tee out.txt - - name: Generate charts - run: python3 out.py + # Regenerates results.md, both charts, copies the charts into docs/benchmarks/ and + # replaces the results section of README.md, so the numbers are readable + # in the repository without running the benchmark locally. + - name: Generate results table and charts + env: + BENCHMARK_LINK_COUNT: 1000 + BACKGROUND_LINK_COUNT: 3000 + run: | + python3 out.py out.txt \ + --results results.md \ + --readme ../README.md \ + --docs-dir ../docs/benchmarks + + - name: Publish results to the job summary + run: | + { + echo "## Benchmark results (full scale)" + echo + cat results.md + } >> "$GITHUB_STEP_SUMMARY" - name: Configure git run: | - git config user.name "github-actions[bot]" - git config user.email "github-actions[bot]@users.noreply.github.com" + git config user.email "linksplatform@gmail.com" + git config user.name "LinksPlatformBencher" - name: Commit benchmark results + working-directory: . run: | - git add -f out.txt bench_rust.png bench_rust_log_scale.png 2>/dev/null || true - git diff --staged --quiet || git commit -m "chore: update benchmark results [skip ci]" - git push + # The charts live in docs/benchmarks/ only; the copies in rust/ are build + # output and are kept as workflow artifacts instead of being committed twice. + git add docs/benchmarks README.md rust/results.md rust/out.txt + if git diff --staged --quiet; then + echo "No changes to commit" + else + git commit -m "Update benchmark results [skip ci]" + git push origin HEAD:${GITHUB_REF_NAME} + fi - name: Upload benchmark artifacts + # Keep the raw output even when the results step fails, + # so a broken run can be diagnosed from the artifact. + if: always() uses: actions/upload-artifact@v4 with: name: benchmark-results path: | rust/out.txt + rust/results.md rust/bench_rust.png rust/bench_rust_log_scale.png diff --git a/README.md b/README.md index aff2adf..887012c 100644 --- a/README.md +++ b/README.md @@ -40,9 +40,42 @@ Each benchmark iteration pre-populates the database with background links to sim ## Results -> _Benchmark results will be automatically generated and committed here by CI when changes are merged to main._ +The numbers below represent the amount of time (ns) a single benchmark iteration takes. - +- The first chart shows time in a pixel (linear) scale. Doublets bars are drawn with a + minimum visible width, otherwise they would not be visible next to SpacetimeDB. +- The second chart shows time in a logarithmic scale, to see the difference clearly, + because it is around 3-5 orders of magnitude. + +Charts and the table are recalculated by the +[Rust Benchmark workflow](.github/workflows/rust-benchmark.yml) on every push to `main` +and committed back to this repository, so the results are visible here without running +the benchmark locally. + +### Rust + +![Image of Rust benchmark (pixel scale)](https://github.com/linksplatform/Comparisons.SpacetimeDBVSDoublets/blob/main/docs/benchmarks/bench_rust.png?raw=true) +![Image of Rust benchmark (log scale)](https://github.com/linksplatform/Comparisons.SpacetimeDBVSDoublets/blob/main/docs/benchmarks/bench_rust_log_scale.png?raw=true) + +### Raw benchmark results (all numbers are in nanoseconds) + + +_No benchmark results have been published yet. They are generated by the first +[Rust Benchmark](.github/workflows/rust-benchmark.yml) run on `main`._ + + +Each Doublets cell is annotated with how many times faster (or slower) it is than +SpacetimeDB for the same operation. + +## Conclusion + +Doublets is an embedded store: an operation is a few pointer dereferences and tree +rotations in memory (or in a memory-mapped file), while every SpacetimeDB operation is a +reducer call over a WebSocket connection to a separate process, and every query is served +from the client-side subscription cache. The measured difference is dominated by that +architectural difference rather than by the data structures themselves. + +To get fresh numbers, please fork the repository and rerun the benchmark in GitHub Actions. ## Operation Complexity @@ -68,7 +101,7 @@ The algorithmic complexity is the same for volatile and non-volatile Doublets va ### Prerequisites -- Rust nightly (see `rust/rust-toolchain.toml`) +- Rust nightly, pinned in `rust/rust-toolchain.toml` (`rustup` installs it automatically) - SpacetimeDB CLI: `curl -sSf https://install.spacetimedb.com | sh` ### Start SpacetimeDB server and publish module @@ -78,8 +111,8 @@ The algorithmic complexity is the same for volatile and non-volatile Doublets va spacetime start & # Build and publish the links module -spacetime build --project-path spacetime-module -spacetime publish --project-path spacetime-module benchmark-links +spacetime build --project-path rust/spacetime-module +spacetime publish --project-path rust/spacetime-module --yes benchmark-links ``` ### Run benchmarks @@ -96,8 +129,13 @@ BENCHMARK_LINK_COUNT=10 BACKGROUND_LINK_COUNT=100 \ SPACETIMEDB_URI=http://localhost:3000 SPACETIMEDB_DB=benchmark-links \ cargo bench --bench bench -# Generate charts from results -python3 out.py +# Generate the results table and charts from out.txt +python3 out.py out.txt --results results.md + +# Regenerate everything the CI publishes: results.md, docs/benchmarks/ charts +# and the results section of README.md +python3 out.py out.txt --results results.md --readme ../README.md \ + --docs-dir ../docs/benchmarks ``` ### Run tests @@ -113,23 +151,32 @@ SPACETIMEDB_URI=http://localhost:3000 SPACETIMEDB_DB=benchmark-links cargo test cd rust cargo fmt --all cargo clippy --all-targets + +# Unit tests for the results reporting pipeline (no benchmark run required) +python3 -m unittest test_out -v ``` ## Project Structure ``` . -├── spacetime-module/ # SpacetimeDB WASM module (links table + reducers) -│ ├── Cargo.toml -│ └── src/ -│ └── lib.rs # Table definition and reducers using `spacetimedb` crate +├── docs/ +│ └── benchmarks/ # Benchmark charts published by CI and shown above +│ ├── bench_rust.png +│ └── bench_rust_log_scale.png ├── rust/ +│ ├── spacetime-module/ # SpacetimeDB WASM module (links table + reducers) +│ │ ├── Cargo.toml +│ │ └── src/ +│ │ └── lib.rs # Table definition and reducers using `spacetimedb` crate │ ├── Cargo.toml # Package manifest with pinned dependencies │ ├── doublets-patched/ # Local patches to doublets-rs for modern nightly compatibility │ │ └── PATCHES.md # Documents why patches are needed and what was changed │ ├── rust-toolchain.toml # Pinned Rust nightly toolchain │ ├── rustfmt.toml # Rust formatting config -│ ├── out.py # Chart generation script (matplotlib) +│ ├── out.py # Results table, charts and README update +│ ├── test_out.py # Unit tests for out.py +│ ├── results.md # Generated results table (committed by CI) │ ├── src/ │ │ ├── lib.rs # Links trait, constants (BENCHMARK_LINK_COUNT, BACKGROUND_LINK_COUNT) │ │ ├── module_bindings/ # spacetimedb-sdk client bindings for the links module @@ -145,7 +192,7 @@ cargo clippy --all-targets │ └── bench.rs # Criterion benchmark suite (7 operations x 5 backends) └── .github/ └── workflows/ - └── rust-benchmark.yml # CI: test on 3 OS + benchmark + chart generation + └── rust-benchmark.yml # CI: test on Linux/macOS, benchmark, publish results ``` ## License diff --git a/changelog.d/20260915_223000_benchmark_results_publication.md b/changelog.d/20260915_223000_benchmark_results_publication.md new file mode 100644 index 0000000..e1ba356 --- /dev/null +++ b/changelog.d/20260915_223000_benchmark_results_publication.md @@ -0,0 +1,21 @@ +--- +bump: patch +--- + +### Fixed +- Pinned the Rust nightly toolchain (`nightly-2026-04-14`), unbreaking the Rust Benchmark workflow that had been failing on every branch since April with `E0512` in `ethnum` and `E0277` in the patched `doublets` dev dependency +- Added `set -o pipefail` around `cargo bench ... | tee out.txt`, so a failing benchmark is no longer masked by `tee` +- The results parser now tolerates Criterion's own error messages, which it prints to stdout in the middle of a bencher record; a stale `target/criterion` baseline used to split every record over two lines and fail the run with `No benchmark data found in out.txt` +- The benchmark steps reset `target/criterion` before running, so a partially restored cache cannot pollute the benchmark output + +### Added +- Benchmark results are now published automatically: `rust/out.py` writes `rust/results.md`, copies both charts into `docs/benchmarks/` and replaces the results section of `README.md`, which CI commits back to `main` +- `rust/test_out.py` — unit tests for the results reporting pipeline, run by a dedicated `results-pipeline` CI job that gates the benchmark jobs +- Benchmark results are written to the GitHub Actions job summary for both the pull request and the full run + +### Changed +- `rust/out.py` now reports all five benchmarked backends (the two NonVolatile Doublets variants were previously missing) and annotates every Doublets result relative to the SpacetimeDB baseline +- `README.md` documents the results in the same style as the sibling Neo4j and PostgreSQL comparisons + +### Removed +- `rust/rust_out` — a 4.2 MB compiled binary committed by accident diff --git a/experiments/reproduce-criterion-error-interleaving.py b/experiments/reproduce-criterion-error-interleaving.py new file mode 100644 index 0000000..68ed258 --- /dev/null +++ b/experiments/reproduce-criterion-error-interleaving.py @@ -0,0 +1,56 @@ +#!/usr/bin/env python3 +"""Reproduces the `No benchmark data found in out.txt` CI failure (issue #14). + +Criterion prints its own error messages to *stdout* (see +criterion-0.3.6/src/macros_private.rs:36 `println!("Criterion.rs ERROR: {}", ...)`) +between `print!("test {} ... ")` and `println!("bench: ...")` +(criterion-0.3.6/src/report.rs:740 and :756). + +When a stale/partial `target/criterion///base` directory is present — +which happens in CI because the Rust cache restores `target/` — loading +`base/sample.json` fails and the error text is spliced into the middle of the +bencher line, so the measurement ends up split across two lines: + + test query_by_id/Doublets_Split_NonVolatile/10 ... Criterion.rs ERROR: error: Failed to access file "...": No such file or directory (os error 2) + bench: 37 ns/iter (+/- 0) + +This script feeds exactly that output (copied from CI run 35028280108) into +out.py's parser and reports how many measurements survive. +""" + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "rust")) + +import out # noqa: E402 + +POLLUTED = """\ +test create/SpacetimeDB/10 ... Criterion.rs ERROR: error: Failed to access file "/home/runner/work/rust/target/criterion/create_SpacetimeDB/10/base/sample.json": No such file or directory (os error 2) +bench: 24992765 ns/iter (+/- 1234567) + +test create/Doublets_United_Volatile/10 ... Criterion.rs ERROR: error: Failed to access file "/home/runner/work/rust/target/criterion/create_Doublets_United_Volatile/10/base/sample.json": No such file or directory (os error 2) +bench: 464 ns/iter (+/- 12) +""" + +CLEAN = """\ +test create/SpacetimeDB/10 ... bench: 24992765 ns/iter (+/- 1234567) + +test create/Doublets_United_Volatile/10 ... bench: 464 ns/iter (+/- 12) +""" + + +def main() -> int: + clean = out.parse_text(CLEAN) + polluted = out.parse_text(POLLUTED) + print(f"clean output -> {sum(len(v) for v in clean.values())} measurements: {clean}") + print(f"polluted output -> {sum(len(v) for v in polluted.values())} measurements: {polluted}") + if clean == polluted: + print("OK: the parser tolerates interleaved Criterion error messages") + return 0 + print("REPRODUCED: interleaved Criterion errors make the parser lose every measurement") + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/rust/out.py b/rust/out.py index 3dfccc3..5dfc796 100644 --- a/rust/out.py +++ b/rust/out.py @@ -1,218 +1,357 @@ #!/usr/bin/env python3 -""" -Benchmark result visualization for SpacetimeDB vs Doublets. +"""Benchmark result visualization and documentation for SpacetimeDB vs Doublets. + +Reads Criterion bencher-format output (produced by +`cargo bench --bench bench -- --output-format bencher`) and generates: -Reads Criterion bencher-format output from out.txt and generates: -- bench_rust.png: Linear scale comparison chart -- bench_rust_log_scale.png: Logarithmic scale comparison chart -- results.md: Markdown table with speedup ratios +- ``bench_rust.png`` — linear ("pixel") scale comparison chart +- ``bench_rust_log_scale.png``— logarithmic scale comparison chart +- ``results.md`` — Markdown results table with speedup ratios +- optionally updates the results section of ``README.md`` in place + +The chart and table style follows the sibling benchmarks +(Comparisons.Neo4jVSDoublets, Comparisons.PostgreSQLVSDoublets) so that all +LinksPlatform comparisons are documented the same way. Usage: - python3 out.py [out.txt] + python3 out.py [out.txt] [--readme ../README.md] [--docs-dir ../docs/benchmarks] """ +import argparse +import os import re +import shutil import sys -import os +from datetime import datetime, timezone try: import matplotlib - matplotlib.use('Agg') + + matplotlib.use("Agg") import matplotlib.pyplot as plt import numpy as np + HAS_MATPLOTLIB = True -except ImportError: +except ImportError: # pragma: no cover - exercised only without matplotlib print("Warning: matplotlib/numpy not installed, skipping chart generation") HAS_MATPLOTLIB = False -# Bencher output line format: -# test /// ... bench: ns/iter (+/- ) +# Bencher output format emitted by criterion: +# test // ... bench: ns/iter (+/- ) +# +# The two halves are printed by separate statements (`print!("test {} ... ")` and +# `println!("bench: ...")`, criterion-0.3.6/src/report.rs), and criterion logs its +# own errors to stdout in between (`println!("Criterion.rs ERROR: ...")`, +# criterion-0.3.6/src/macros_private.rs). A stale `target/criterion///base` +# directory is enough to splice such an error into the middle of the record and push +# `bench:` onto the next line, so the pattern skips anything that is not the start of +# the next `test ` record. BENCHER_PATTERN = re.compile( - r'test (\w+)/(\w+)/(\w+)/(\d+)\s+\.\.\.\s+bench:\s+([\d,]+)\s+ns/iter' + r"test\s+(\w+)/(\w+)/(\d+)\s+\.\.\.\s*" + r"(?:(?!\btest\s)[\s\S])*?" + r"bench:\s+([\d,]+)\s+ns/iter" ) +# Operations in the order they are reported, mapped to human readable labels. OPERATIONS = [ - 'create', - 'delete', - 'update', - 'query_all', - 'query_by_id', - 'query_by_source', - 'query_by_target', + ("create", "Create"), + ("update", "Update"), + ("delete", "Delete"), + ("query_all", "Query All"), + ("query_by_id", "Query by Id"), + ("query_by_source", "Query by Source"), + ("query_by_target", "Query by Target"), ] -OPERATION_LABELS = { - 'create': 'Create', - 'delete': 'Delete', - 'update': 'Update', - 'query_all': 'Query All', - 'query_by_id': 'Query by Id', - 'query_by_source': 'Query by Source', - 'query_by_target': 'Query by Target', -} - -VARIANTS = { - 'SpacetimeDB': 'SpacetimeDB 2.0', - 'Doublets_United_Volatile': 'Doublets (United/Volatile)', - 'Doublets_Split_Volatile': 'Doublets (Split/Volatile)', -} - -COLORS = { - 'SpacetimeDB': '#e74c3c', - 'Doublets_United_Volatile': '#2ecc71', - 'Doublets_Split_Volatile': '#3498db', -} - - -def parse_results(filename='out.txt'): - """Parse bencher-format output into a nested dict: operation -> variant -> ns_per_iter.""" - results = {op: {} for op in OPERATIONS} +# Benchmarked backends. The first entry is the baseline every other variant is +# compared against in the results table. +BASELINE = "SpacetimeDB" + +VARIANTS = [ + ("Doublets_United_Volatile", "Doublets United Volatile", "salmon"), + ("Doublets_United_NonVolatile", "Doublets United NonVolatile", "red"), + ("Doublets_Split_Volatile", "Doublets Split Volatile", "lightgreen"), + ("Doublets_Split_NonVolatile", "Doublets Split NonVolatile", "green"), + (BASELINE, "SpacetimeDB", "royalblue"), +] + +README_START_MARKER = "" +README_END_MARKER = "" + + +def parse_text(content): + """Parse bencher-format text into ``{operation: {variant: ns_per_iter}}``.""" + results = {op: {} for op, _ in OPERATIONS} + + for match in BENCHER_PATTERN.finditer(content): + operation, variant, _size, ns_str = match.groups() + if operation in results: + results[operation][variant] = int(ns_str.replace(",", "")) + + return results + +def parse_results(filename="out.txt"): + """Parse a bencher-format output file.""" if not os.path.exists(filename): print(f"Warning: {filename} not found") - return results - - with open(filename, 'r') as f: - content = f.read() - - for line in content.splitlines(): - m = BENCHER_PATTERN.search(line) - if m: - group, op, variant, size, ns_str = m.groups() - ns = int(ns_str.replace(',', '')) - if op in results: - results[op][variant] = ns - - # Also handle Criterion's default output format - # test group::benchmark/variant/size ... bench: X ns/iter (+/- Y) - CRITERION_PATTERN = re.compile( - r'test (\w+)::(\w+)/(\w+)/\d+\s+\.\.\.\s+bench:\s+([\d,]+)\s+ns/iter' + return {op: {} for op, _ in OPERATIONS} + + with open(filename, "r", encoding="utf-8") as handle: + return parse_text(handle.read()) + + +def report_input_excerpt(filename, lines=20): + """Describe the tail of ``filename`` so an unparsable run can be diagnosed.""" + if not os.path.exists(filename): + return f"{filename} does not exist" + + with open(filename, "r", encoding="utf-8") as handle: + content = handle.read() + + if not content.strip(): + return f"{filename} is empty" + + tail = content.splitlines()[-lines:] + return "\n".join([f"Last {len(tail)} line(s) of {filename}:", *tail]) + + +def has_any_results(results): + """Return ``True`` when at least one measurement was parsed.""" + return any(measurements for measurements in results.values()) + + +def format_speedup(value, baseline): + """Annotate ``value`` with how it compares to the ``baseline`` measurement.""" + if not value: + return "N/A" + if not baseline: + return f"{value}" + if value <= baseline: + return f"{value} ({baseline / value:.1f}x faster)" + return f"{value} ({value / baseline:.1f}x slower)" + + +def format_results_table(results): + """Render the Markdown results table (all numbers in nanoseconds).""" + labels = [label for _, label, _ in VARIANTS] + widths = [max(len(label), 13) for label in labels] + operation_width = max(len(label) for _, label in OPERATIONS) + + header = "| " + "Operation".ljust(operation_width) + " | " + header += " | ".join(label.ljust(width) for label, width in zip(labels, widths)) + header += " |" + separator = "|" + "-" * (operation_width + 2) + separator += "".join("|" + "-" * (width + 2) for width in widths) + "|" + + lines = [header, separator] + for op, op_label in OPERATIONS: + baseline = results[op].get(BASELINE, 0) + cells = [] + for key, _label, _color in VARIANTS: + value = results[op].get(key, 0) + if key == BASELINE: + cells.append(str(value) if value else "N/A") + else: + cells.append(format_speedup(value, baseline)) + row = "| " + op_label.ljust(operation_width) + " | " + row += " | ".join(cell.ljust(width) for cell, width in zip(cells, widths)) + row += " |" + lines.append(row) + + return "\n".join(lines) + + +def build_provenance(benchmark_links=None, background_links=None, generated_at=None): + """Describe how and when the committed results were produced.""" + benchmark_links = benchmark_links or os.environ.get("BENCHMARK_LINK_COUNT", "1000") + background_links = background_links or os.environ.get( + "BACKGROUND_LINK_COUNT", "3000" + ) + generated_at = generated_at or datetime.now(timezone.utc).strftime( + "%Y-%m-%d %H:%M UTC" ) - for line in content.splitlines(): - m = CRITERION_PATTERN.search(line) - if m: - _group, op, variant, ns_str = m.groups() - ns = int(ns_str.replace(',', '')) - if op in results: - results[op][variant] = ns - return results + source = "a local benchmark run" + repository = os.environ.get("GITHUB_REPOSITORY") + run_id = os.environ.get("GITHUB_RUN_ID") + if repository and run_id: + source = ( + f"[GitHub Actions run {run_id}]" + f"(https://github.com/{repository}/actions/runs/{run_id})" + ) + + return ( + f"_Generated {generated_at} by {source} — " + f"{benchmark_links} benchmarked links, " + f"{background_links} background links._" + ) -def generate_charts(results): - """Generate PNG comparison charts.""" - if not HAS_MATPLOTLIB: - return +def render_results_section(results, provenance=None): + """Render the README section: provenance line plus the results table.""" + provenance = provenance if provenance is not None else build_provenance() + return f"{provenance}\n\n{format_results_table(results)}" - variant_keys = list(VARIANTS.keys()) - op_labels = [OPERATION_LABELS.get(op, op) for op in OPERATIONS] - n_ops = len(OPERATIONS) - n_variants = len(variant_keys) - - # Collect data - data = {} - for variant in variant_keys: - data[variant] = [] - for op in OPERATIONS: - ns = results[op].get(variant, 0) - data[variant].append(ns) - - x = np.arange(n_ops) - width = 0.8 / n_variants - - def make_chart(ax, log_scale=False): - for i, variant in enumerate(variant_keys): - vals = [v if v > 0 else float('nan') for v in data[variant]] - offset = (i - n_variants / 2 + 0.5) * width - bars = ax.bar( - x + offset, vals, width, - label=VARIANTS[variant], - color=COLORS[variant], - alpha=0.85 - ) - - ax.set_xlabel('Operation') - ax.set_ylabel('Time (ns/iter)') - title = 'SpacetimeDB vs Doublets — Link CRUD Benchmark' - if log_scale: - title += ' (Log Scale)' - ax.set_yscale('log') - ax.set_title(title) - ax.set_xticks(x) - ax.set_xticklabels(op_labels, rotation=30, ha='right') - ax.legend() - ax.grid(axis='y', alpha=0.3) - plt.tight_layout() - - # Linear scale - fig, ax = plt.subplots(figsize=(12, 6)) - make_chart(ax, log_scale=False) - fig.savefig('bench_rust.png', dpi=150, bbox_inches='tight') - plt.close(fig) - print("Generated bench_rust.png") - - # Log scale - fig, ax = plt.subplots(figsize=(12, 6)) - make_chart(ax, log_scale=True) - fig.savefig('bench_rust_log_scale.png', dpi=150, bbox_inches='tight') - plt.close(fig) - print("Generated bench_rust_log_scale.png") - - -def generate_markdown_table(results): - """Generate a Markdown results table with speedup ratios.""" - baseline = 'SpacetimeDB' - doublets_variants = ['Doublets_United_Volatile', 'Doublets_Split_Volatile'] - - header = ( - '| Operation | SpacetimeDB (ns/iter) ' - '| Doublets United (ns/iter) | Doublets United Speedup ' - '| Doublets Split (ns/iter) | Doublets Split Speedup |' + +def update_readme(readme_path, section): + """Replace the marked results section of the README. + + Returns ``True`` when the file was modified. + """ + with open(readme_path, "r", encoding="utf-8") as handle: + readme = handle.read() + + if README_START_MARKER not in readme or README_END_MARKER not in readme: + raise ValueError( + f"{readme_path} does not contain the " + f"{README_START_MARKER} / {README_END_MARKER} markers" + ) + + pattern = re.compile( + re.escape(README_START_MARKER) + r".*?" + re.escape(README_END_MARKER), + re.DOTALL, ) - sep = '|---|---|---|---|---|---|' - - rows = [header, sep] - for op in OPERATIONS: - label = OPERATION_LABELS.get(op, op) - baseline_ns = results[op].get(baseline, 0) - baseline_str = f'{baseline_ns:,}' if baseline_ns else 'N/A' - - row = f'| {label} | {baseline_str} ' - for variant in doublets_variants: - ns = results[op].get(variant, 0) - ns_str = f'{ns:,}' if ns else 'N/A' - if baseline_ns > 0 and ns > 0: - speedup = baseline_ns / ns - speedup_str = f'{speedup:.0f}x' - else: - speedup_str = 'N/A' - row += f'| {ns_str} | {speedup_str} ' - row += '|' - rows.append(row) + replacement = f"{README_START_MARKER}\n{section}\n{README_END_MARKER}" + updated = pattern.sub(lambda _match: replacement, readme, count=1) + + if updated == readme: + return False + + with open(readme_path, "w", encoding="utf-8") as handle: + handle.write(updated) + return True + + +def _series(results, variant): + """Measurements of one variant across all operations, 0 when missing.""" + return [results[op].get(variant, 0) for op, _ in OPERATIONS] + + +def _ensure_min_visible(values, minimum): + """Keep non-zero bars at least ``minimum`` wide so they stay visible.""" + return [max(value, minimum) if value > 0 else 0 for value in values] + + +def _plot(results, filename, log_scale, output_dir): + positions = np.arange(len(OPERATIONS)) + width = 0.15 + figure, axes = plt.subplots(figsize=(12, 8)) + + series = {key: _series(results, key) for key, _, _ in VARIANTS} + + if log_scale: + plotted = series + else: + # On a linear scale Doublets bars are invisible next to SpacetimeDB, so + # give every non-zero measurement a minimum visible width (~0.5% of the + # maximum), matching the sibling benchmark charts. + all_values = [value for values in series.values() for value in values] + max_value = max(all_values) if all_values else 1 + minimum = max_value * 0.005 + plotted = { + key: _ensure_min_visible(values, minimum) for key, values in series.items() + } - table = '\n'.join(rows) - with open('results.md', 'w') as f: - f.write('# Benchmark Results\n\n') - f.write(f'> Background: {os.environ.get("BACKGROUND_LINK_COUNT", "3000")} links\n') - f.write(f'> Operations: {os.environ.get("BENCHMARK_LINK_COUNT", "1000")} links\n\n') - f.write(table) - f.write('\n') + offset_base = (len(VARIANTS) - 1) / 2 + for index, (key, label, color) in enumerate(VARIANTS): + offset = (index - offset_base) * width + axes.barh(positions + offset, plotted[key], width, label=label, color=color) - print("Generated results.md") - print('\n' + table) + axes.set_xlabel("Time (ns) – log scale" if log_scale else "Time (ns)") + axes.set_title("Benchmark Comparison: SpacetimeDB vs Doublets (Rust)") + axes.set_yticks(positions) + axes.set_yticklabels([label for _, label in OPERATIONS]) + if log_scale: + axes.set_xscale("log") + axes.legend() + figure.tight_layout() + path = os.path.join(output_dir, filename) if output_dir else filename + figure.savefig(path) + plt.close(figure) + print(f"Generated {path}") -def main(): - filename = sys.argv[1] if len(sys.argv) > 1 else 'out.txt' - results = parse_results(filename) - if all(not v for v in results.values()): - print(f"No benchmark data found in {filename}") +def generate_charts(results, output_dir=""): + """Generate the linear and logarithmic comparison charts.""" + if not HAS_MATPLOTLIB: return + _plot(results, "bench_rust.png", log_scale=False, output_dir=output_dir) + _plot(results, "bench_rust_log_scale.png", log_scale=True, output_dir=output_dir) + + +def copy_charts(docs_dir, output_dir=""): + """Copy generated charts into the documentation directory.""" + os.makedirs(docs_dir, exist_ok=True) + for chart in ("bench_rust.png", "bench_rust_log_scale.png"): + source = os.path.join(output_dir, chart) if output_dir else chart + if os.path.exists(source): + shutil.copyfile(source, os.path.join(docs_dir, chart)) + print(f"Copied {source} -> {os.path.join(docs_dir, chart)}") + + +def parse_args(argv): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "input", + nargs="?", + default="out.txt", + help="criterion bencher output file (default: out.txt)", + ) + parser.add_argument( + "--results", + default="results.md", + help="path of the generated Markdown table (default: results.md)", + ) + parser.add_argument( + "--readme", + default=None, + help="README to update in place between the benchmark result markers", + ) + parser.add_argument( + "--docs-dir", + default=None, + help="directory the generated charts are copied to (for example ../docs/benchmarks)", + ) + parser.add_argument( + "--output-dir", + default="", + help="directory the charts are written to (default: working directory)", + ) + return parser.parse_args(argv) + + +def main(argv=None): + args = parse_args(argv if argv is not None else sys.argv[1:]) + results = parse_results(args.input) + + if not has_any_results(results): + print(f"No benchmark data found in {args.input}") + print(report_input_excerpt(args.input)) + return 1 + + section = render_results_section(results) + table = format_results_table(results) + print(table) + + with open(args.results, "w", encoding="utf-8") as handle: + handle.write(section + "\n") + print(f"Generated {args.results}") + + generate_charts(results, args.output_dir) + + if args.docs_dir: + copy_charts(args.docs_dir, args.output_dir) + + if args.readme: + changed = update_readme(args.readme, section) + print( + f"{'Updated' if changed else 'No changes needed in'} {args.readme}", + ) - generate_charts(results) - generate_markdown_table(results) + return 0 -if __name__ == '__main__': - main() +if __name__ == "__main__": + sys.exit(main()) diff --git a/rust/rust-toolchain.toml b/rust/rust-toolchain.toml index 8e275b7..55c7ec7 100644 --- a/rust/rust-toolchain.toml +++ b/rust/rust-toolchain.toml @@ -1,3 +1,10 @@ +# Pinned nightly toolchain. +# +# The benchmark depends on `#![feature(allocator_api)]` and on the patched +# `doublets` crates, which use unstable `Try`/`Residual` APIs. Those APIs keep +# changing on rolling nightly, which repeatedly broke CI (see issue #14), so the +# toolchain is pinned to a known-good nightly and updated deliberately. [toolchain] -channel = "nightly" +channel = "nightly-2026-04-14" components = ["rustfmt", "clippy"] +targets = ["wasm32-unknown-unknown"] diff --git a/rust/rust_out b/rust/rust_out deleted file mode 100755 index 1ad1dd1..0000000 Binary files a/rust/rust_out and /dev/null differ diff --git a/rust/test_out.py b/rust/test_out.py new file mode 100644 index 0000000..cd202b2 --- /dev/null +++ b/rust/test_out.py @@ -0,0 +1,356 @@ +#!/usr/bin/env python3 +"""Tests for the benchmark reporting pipeline implemented in ``out.py``. + +These tests guard the automation that publishes benchmark results, so a broken +parser or README update is caught in CI instead of silently producing an empty +results table. + +Run with: + cd rust && python3 -m unittest test_out -v +""" + +import os +import tempfile +import unittest + +import out + +# Minimal but realistic criterion bencher output: every operation for every +# benchmarked backend, plus noise lines criterion prints around the results. +SAMPLE_OUT_TXT = """ +Gnuplot not found, using plotters backend +running 35 tests +test create/SpacetimeDB/1000 ... bench: 3,159,452,615 ns/iter (+/- 12,345) +test create/Doublets_United_Volatile/1000 ... bench: 84,823 ns/iter (+/- 1,000) +test create/Doublets_United_NonVolatile/1000 ... bench: 86,001 ns/iter (+/- 1,000) +test create/Doublets_Split_Volatile/1000 ... bench: 84,120 ns/iter (+/- 1,000) +test create/Doublets_Split_NonVolatile/1000 ... bench: 83,188 ns/iter (+/- 1,000) +test update/SpacetimeDB/1000 ... bench: 38,993,916 ns/iter (+/- 100) +test update/Doublets_United_Volatile/1000 ... bench: 1,484 ns/iter (+/- 10) +test update/Doublets_United_NonVolatile/1000 ... bench: 1,539 ns/iter (+/- 10) +test update/Doublets_Split_Volatile/1000 ... bench: 833 ns/iter (+/- 10) +test update/Doublets_Split_NonVolatile/1000 ... bench: 836 ns/iter (+/- 10) +test delete/SpacetimeDB/1000 ... bench: 1,989,558,566 ns/iter (+/- 100) +test delete/Doublets_United_Volatile/1000 ... bench: 143,527 ns/iter (+/- 100) +test delete/Doublets_United_NonVolatile/1000 ... bench: 153,180 ns/iter (+/- 100) +test delete/Doublets_Split_Volatile/1000 ... bench: 145,585 ns/iter (+/- 100) +test delete/Doublets_Split_NonVolatile/1000 ... bench: 143,649 ns/iter (+/- 100) +test query_all/SpacetimeDB/1000 ... bench: 943,914 ns/iter (+/- 100) +test query_all/Doublets_United_Volatile/1000 ... bench: 74 ns/iter (+/- 1) +test query_all/Doublets_United_NonVolatile/1000 ... bench: 77 ns/iter (+/- 1) +test query_all/Doublets_Split_Volatile/1000 ... bench: 78 ns/iter (+/- 1) +test query_all/Doublets_Split_NonVolatile/1000 ... bench: 77 ns/iter (+/- 1) +test query_by_id/SpacetimeDB/1000 ... bench: 8,801,871 ns/iter (+/- 100) +test query_by_id/Doublets_United_Volatile/1000 ... bench: 348 ns/iter (+/- 1) +test query_by_id/Doublets_United_NonVolatile/1000 ... bench: 348 ns/iter (+/- 1) +test query_by_id/Doublets_Split_Volatile/1000 ... bench: 348 ns/iter (+/- 1) +test query_by_id/Doublets_Split_NonVolatile/1000 ... bench: 348 ns/iter (+/- 1) +test query_by_source/SpacetimeDB/1000 ... bench: 8,729,692 ns/iter (+/- 100) +test query_by_source/Doublets_United_Volatile/1000 ... bench: 677 ns/iter (+/- 1) +test query_by_source/Doublets_United_NonVolatile/1000 ... bench: 493 ns/iter (+/- 1) +test query_by_source/Doublets_Split_Volatile/1000 ... bench: 422 ns/iter (+/- 1) +test query_by_source/Doublets_Split_NonVolatile/1000 ... bench: 408 ns/iter (+/- 1) +test query_by_target/SpacetimeDB/1000 ... bench: 8,742,366 ns/iter (+/- 100) +test query_by_target/Doublets_United_Volatile/1000 ... bench: 671 ns/iter (+/- 1) +test query_by_target/Doublets_United_NonVolatile/1000 ... bench: 487 ns/iter (+/- 1) +test query_by_target/Doublets_Split_Volatile/1000 ... bench: 411 ns/iter (+/- 1) +test query_by_target/Doublets_Split_NonVolatile/1000 ... bench: 411 ns/iter (+/- 1) +""" + +README_TEMPLATE = f"""# Title + +## Results + +{out.README_START_MARKER} +> _Benchmark results will be generated by CI._ +{out.README_END_MARKER} + +## Conclusion +""" + + +def write(path, content): + with open(path, "w", encoding="utf-8") as handle: + handle.write(content) + return path + + +class ParseResultsTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.path = write(os.path.join(self.tmp.name, "out.txt"), SAMPLE_OUT_TXT) + + def test_parses_every_operation_and_variant(self): + results = out.parse_results(self.path) + + self.assertEqual(len(results), len(out.OPERATIONS)) + for operation, _label in out.OPERATIONS: + measured = results[operation] + self.assertEqual( + sorted(measured), + sorted(key for key, _label, _color in out.VARIANTS), + f"missing variants for {operation}", + ) + + def test_parses_thousand_separators(self): + results = out.parse_results(self.path) + + self.assertEqual(results["create"]["SpacetimeDB"], 3_159_452_615) + self.assertEqual(results["create"]["Doublets_United_Volatile"], 84_823) + + def test_missing_file_yields_empty_results(self): + results = out.parse_results(os.path.join(self.tmp.name, "absent.txt")) + + self.assertFalse(out.has_any_results(results)) + + def test_unrelated_lines_are_ignored(self): + path = write( + os.path.join(self.tmp.name, "noise.txt"), + "Benchmarking create/SpacetimeDB/1000: Warming up for 3.0000 s\n", + ) + + self.assertFalse(out.has_any_results(out.parse_results(path))) + + def test_interleaved_criterion_errors_do_not_hide_measurements(self): + """Regression test for the `No benchmark data found` CI failure. + + Criterion prints its own errors to stdout between `test ... ` and + `bench: ...`, which splits the record over two lines (CI run 35028280108). + """ + path = write( + os.path.join(self.tmp.name, "interleaved.txt"), + 'test create/SpacetimeDB/10 ... Criterion.rs ERROR: error: Failed to ' + 'access file "target/criterion/create_SpacetimeDB/10/base/sample.json": ' + "No such file or directory (os error 2)\n" + "bench: 24992765 ns/iter (+/- 1790323177)\n" + "\n" + "test create/Doublets_United_Volatile/10 ... Criterion.rs ERROR: error: " + 'Failed to access file "sample.json": No such file or directory ' + "(os error 2)\n" + "bench: 464 ns/iter (+/- 12)\n", + ) + + results = out.parse_results(path) + + self.assertEqual(results["create"]["SpacetimeDB"], 24_992_765) + self.assertEqual(results["create"]["Doublets_United_Volatile"], 464) + + def test_measurement_is_not_borrowed_from_the_next_benchmark(self): + """An aborted benchmark must not take the following benchmark's number.""" + path = write( + os.path.join(self.tmp.name, "aborted.txt"), + "test create/SpacetimeDB/10 ... \n" + "test create/Doublets_United_Volatile/10 ... bench: 464 ns/iter (+/- 12)\n", + ) + + results = out.parse_results(path) + + self.assertNotIn("SpacetimeDB", results["create"]) + self.assertEqual(results["create"]["Doublets_United_Volatile"], 464) + + +class InputExcerptTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + + def test_missing_file(self): + path = os.path.join(self.tmp.name, "absent.txt") + + self.assertEqual(out.report_input_excerpt(path), f"{path} does not exist") + + def test_empty_file(self): + path = write(os.path.join(self.tmp.name, "empty.txt"), "\n\n") + + self.assertEqual(out.report_input_excerpt(path), f"{path} is empty") + + def test_reports_the_tail(self): + path = write( + os.path.join(self.tmp.name, "noise.txt"), + "".join(f"line {index}\n" for index in range(10)), + ) + + excerpt = out.report_input_excerpt(path, lines=3) + + self.assertIn("Last 3 line(s)", excerpt) + self.assertIn("line 9", excerpt) + self.assertNotIn("line 6", excerpt) + + +class SpeedupFormattingTests(unittest.TestCase): + def test_faster_than_baseline(self): + self.assertEqual(out.format_speedup(100, 1000), "100 (10.0x faster)") + + def test_slower_than_baseline(self): + self.assertEqual(out.format_speedup(1000, 100), "1000 (10.0x slower)") + + def test_missing_measurement(self): + self.assertEqual(out.format_speedup(0, 1000), "N/A") + + def test_missing_baseline_reports_raw_value(self): + self.assertEqual(out.format_speedup(100, 0), "100") + + +class ResultsTableTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + path = write(os.path.join(self.tmp.name, "out.txt"), SAMPLE_OUT_TXT) + self.results = out.parse_results(path) + self.table = out.format_results_table(self.results) + + def test_table_has_a_row_per_operation(self): + rows = [line for line in self.table.splitlines() if line.startswith("|")] + + # header + separator + one row per operation + self.assertEqual(len(rows), len(out.OPERATIONS) + 2) + + def test_header_lists_every_variant(self): + header = self.table.splitlines()[0] + + for _key, label, _color in out.VARIANTS: + self.assertIn(label, header) + + def test_doublets_cells_are_annotated_with_speedup(self): + create_row = next( + line for line in self.table.splitlines() if line.startswith("| Create") + ) + + # 3_159_452_615 / 84_823 = 37,247.6x + self.assertIn("84823 (37247.6x faster)", create_row) + self.assertIn("3159452615", create_row) + + def test_missing_measurements_render_as_not_available(self): + partial = {operation: {} for operation, _label in out.OPERATIONS} + partial["create"]["SpacetimeDB"] = 1000 + + table = out.format_results_table(partial) + create_row = next( + line for line in table.splitlines() if line.startswith("| Create") + ) + + self.assertIn("N/A", create_row) + + +class ProvenanceTests(unittest.TestCase): + def test_local_run_provenance(self): + provenance = out.build_provenance( + benchmark_links="1000", + background_links="3000", + generated_at="2026-01-01 00:00 UTC", + ) + + self.assertIn("2026-01-01 00:00 UTC", provenance) + self.assertIn("1000 benchmarked links", provenance) + self.assertIn("3000 background links", provenance) + + def test_github_actions_provenance_links_to_the_run(self): + os.environ["GITHUB_REPOSITORY"] = "linksplatform/Comparisons.SpacetimeDBVSDoublets" + os.environ["GITHUB_RUN_ID"] = "1234567890" + self.addCleanup(os.environ.pop, "GITHUB_REPOSITORY", None) + self.addCleanup(os.environ.pop, "GITHUB_RUN_ID", None) + + provenance = out.build_provenance(generated_at="2026-01-01 00:00 UTC") + + self.assertIn( + "https://github.com/linksplatform/Comparisons.SpacetimeDBVSDoublets" + "/actions/runs/1234567890", + provenance, + ) + + +class ReadmeUpdateTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.readme = write(os.path.join(self.tmp.name, "README.md"), README_TEMPLATE) + + def read(self): + with open(self.readme, "r", encoding="utf-8") as handle: + return handle.read() + + def test_replaces_content_between_markers(self): + self.assertTrue(out.update_readme(self.readme, "TABLE")) + + content = self.read() + self.assertIn(f"{out.README_START_MARKER}\nTABLE\n{out.README_END_MARKER}", content) + self.assertNotIn("will be generated by CI", content) + self.assertIn("## Conclusion", content) + + def test_update_is_idempotent(self): + out.update_readme(self.readme, "TABLE") + + self.assertFalse(out.update_readme(self.readme, "TABLE")) + + def test_special_regex_characters_survive_replacement(self): + section = r"| Create | 100 (1.5x faster) \g<0> $1 |" + + out.update_readme(self.readme, section) + + self.assertIn(section, self.read()) + + def test_missing_markers_raise(self): + path = write(os.path.join(self.tmp.name, "no-markers.md"), "# Title\n") + + with self.assertRaises(ValueError): + out.update_readme(path, "TABLE") + + +class MainTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.input = write(os.path.join(self.tmp.name, "out.txt"), SAMPLE_OUT_TXT) + self.readme = write(os.path.join(self.tmp.name, "README.md"), README_TEMPLATE) + self.results = os.path.join(self.tmp.name, "results.md") + self.docs = os.path.join(self.tmp.name, "docs", "benchmarks") + + def test_writes_results_and_updates_readme(self): + exit_code = out.main( + [ + self.input, + "--results", + self.results, + "--readme", + self.readme, + "--output-dir", + self.tmp.name, + "--docs-dir", + self.docs, + ] + ) + + self.assertEqual(exit_code, 0) + with open(self.results, "r", encoding="utf-8") as handle: + results_md = handle.read() + self.assertIn("| Create", results_md) + with open(self.readme, "r", encoding="utf-8") as handle: + self.assertIn("| Create", handle.read()) + + @unittest.skipUnless(out.HAS_MATPLOTLIB, "matplotlib is not installed") + def test_generates_and_copies_charts(self): + out.main( + [ + self.input, + "--results", + self.results, + "--output-dir", + self.tmp.name, + "--docs-dir", + self.docs, + ] + ) + + for chart in ("bench_rust.png", "bench_rust_log_scale.png"): + self.assertTrue(os.path.exists(os.path.join(self.tmp.name, chart)), chart) + self.assertTrue(os.path.exists(os.path.join(self.docs, chart)), chart) + + def test_empty_input_reports_failure(self): + empty = write(os.path.join(self.tmp.name, "empty.txt"), "") + + self.assertEqual(out.main([empty, "--results", self.results]), 1) + + +if __name__ == "__main__": + unittest.main()