From a5a99d81858bf10e65a612f421560d5ad734accc Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Fri, 11 Sep 2026 20:23:44 +0200 Subject: [PATCH 01/37] feat(cargo-each): add portable execution inputs Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- Cargo.lock | 2 + crates/cargo-each/Cargo.toml | 2 + crates/cargo-each/README.md | 34 +- crates/cargo-each/docs/design/README.md | 158 ++--- crates/cargo-each/src/cli.rs | 58 ++ crates/cargo-each/src/error.rs | 87 ++- crates/cargo-each/src/filter.rs | 1 + crates/cargo-each/src/main.rs | 34 +- crates/cargo-each/src/plan.rs | 95 ++- crates/cargo-each/src/run.rs | 769 +++++++++++++++++++++--- crates/cargo-each/src/select.rs | 123 +++- crates/cargo-each/src/substitute.rs | 91 ++- crates/cargo-each/src/workspace.rs | 112 +++- crates/cargo-each/tests/cli.rs | 456 ++++++++++++++ 14 files changed, 1822 insertions(+), 200 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 5d4529ab5..f083d7cf6 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -525,6 +525,7 @@ name = "cargo-each" version = "0.2.0" dependencies = [ "assert_cmd", + "cargo-gamma-process", "cargo_metadata", "clap", "mutants", @@ -532,6 +533,7 @@ dependencies = [ "predicates", "serde_json", "tempfile", + "toml", ] [[package]] diff --git a/crates/cargo-each/Cargo.toml b/crates/cargo-each/Cargo.toml index 586c77e2c..c5269c409 100644 --- a/crates/cargo-each/Cargo.toml +++ b/crates/cargo-each/Cargo.toml @@ -17,11 +17,13 @@ homepage.workspace = true repository = "https://github.com/microsoft/ox-tools/tree/main/crates/cargo-each" [dependencies] +cargo-gamma-process = { workspace = true } cargo_metadata = { workspace = true } clap = { workspace = true, features = ["derive", "std", "help", "usage", "error-context"] } mutants = { workspace = true } ohno = { workspace = true, features = ["app-err"] } serde_json = { workspace = true, features = ["std"] } +toml = { workspace = true, features = ["parse", "serde"] } [dev-dependencies] assert_cmd = { workspace = true } diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index c320f4e3b..afacba998 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -44,16 +44,18 @@ directly (argv, not a shell string) after substituting placeholders. * `-p` / `--package ` — select a member. Repeatable. `SPEC` is a package name, a `name@version` spec, or a Unix glob (`tokio-*`). +* `--package-file ` — read package specs from a UTF-8 file, one per + nonempty line. Repeatable; specs are unioned with `--package`. An empty + file explicitly selects no members. * `--workspace` / `--all` — select every workspace member. * `--exclude ` — drop a member (with `--workspace`). Repeatable. * `--none` — explicitly select zero members (a no-op that exits 0). When nothing is named the default is cargo `default-members`, exactly like `cargo build`; pass `--workspace` for every member. A selector that -matches no member is an error, so typos fail loudly. A computed selection -(for example a CI affected-packages set) is fed in as ordinary flags via -shell expansion — `cargo-each` has no file or environment-variable source -of its own. +matches no member is an error, so typos fail loudly. Package files contain +package specs only: comments, command-line tokens, malformed input, and +missing, unreadable, or non-UTF-8 files are errors. ### Filters @@ -82,9 +84,11 @@ can be double-quoted. Expression atoms: `--target-required-feature` further narrows targets. `--keep-going` runs every invocation and exits non-zero if any failed -(default is fail-fast); `--chdir` runs each per-package or per-target -command from that member crate root; `--dry-run` prints commands without -running them. +(default is fail-fast). `--jobs ` bounds concurrent per-package or +per-target work (default `1`), while `--timeout ` terminates each +invocation and its process tree independently (`250ms`, `30s`, or `2m`). +`--chdir` runs each per-package or per-target command from that member crate +root; `--dry-run` prints commands without running them. ### Placeholders @@ -99,6 +103,9 @@ Substituted inside each command argument: * `{packages}` — the cargo selection flags for the resolved set (`--workspace` for the whole workspace, else `--package name@version …`); valid only in `--once` mode and only as a standalone argument. +* `{workspace-rust-version}` — the root `[workspace.package].rust-version`, + or root `[package].rust-version` in a single-package repository; valid in + every mode. Using a placeholder in the wrong mode is a usage error. Only the tokens above are interpreted; any other `{…}` sequence (a typo, or a literal brace @@ -110,6 +117,19 @@ no brace-escape, so this passthrough is part of the contract. An empty resolved selection (via `--none`, or a filter that removes every member) is a **successful no-op**: `cargo-each` prints a one-line note and exits 0. This is what lets callers drop bespoke nothing-to-do guards. +Workspace Rust-version validation is lazy: it runs only when the command +uses `{workspace-rust-version}`, then requires every member’s resolved +minimum to be present and no newer than the root floor. + +With `--jobs > 1`, each invocation’s output is buffered and complete blocks +are emitted in deterministic plan order. Fail-fast stops launching after +the first observed failure, waits for running work, and chooses the final +failure by plan order. `--keep-going` runs the complete plan. Worker panics +and unexpected worker-channel disconnections become infrastructure-failure +outcomes instead of blocking the scheduler. +If timed-out tree cleanup fails, cargo-each reports the infrastructure +failure and emits already-buffered output without waiting indefinitely for +surviving descendants to close inherited pipes. Child commands inherit `PATH` explicitly. On Windows this makes relative program lookup honor the inherited `PATH` order instead of preferring an unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index a5fbe1147..bf3596344 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -48,18 +48,19 @@ target (with placeholder substitution), or exactly once for the whole set. flag surface is already familiar and the impact step's `--package name@version` output can be consumed verbatim. 2. **Absorb the CI skip/default dance.** A resolved-empty selection is a no-op - that exits 0 — no `--skip` sentinel in callers. cargo-each is entirely - flag-driven: a computed selection (an impact tier) is fed in as ordinary - `-p` / `--workspace` / `--none` flags via shell expansion, so cargo-each - stays agnostic about where the selectors came from and callers never write - a skip/default conditional. + that exits 0 — no `--skip` sentinel in callers. A computed selection can be + supplied as ordinary `-p` flags or as one Cargo package spec per line in a + `--package-file`, so callers do not need shell array expansion. cargo-each + stays agnostic about who produced the file and what the selection means. 3. **Three execution modes.** *per-package* (run the command once per member, substituting `{name}`/`{spec}`/`{version}`/`{manifest}`) covers per-manifest tools; *once* (run the command a single time when the set is non-empty) covers workspace-wide tools and single-invocation cargo commands, with a `{packages}` placeholder that expands to the cargo selection flags; and *per-target* runs once for each Cargo target of requested kinds, preserving - the package placeholders and adding `{target}`. + the package placeholders and adding `{target}`. The workspace-scoped + `{workspace-rust-version}` placeholder exposes the root compatibility floor + to commands that provision or validate a shared toolchain. 4. **A small, general filter language** (`--filter` and `--exclude-filter`) with `not`, `and`, `or`, and parentheses over cargo metadata — target kinds, publication state, declared features and dependencies, and @@ -67,7 +68,10 @@ target (with placeholder substitution), or exactly once for the whole set. recipes collapses to flags. 5. **Bare names for free.** `{name}` yields the un-qualified package name, so `@version` stripping disappears from callers even though the input carries it. -6. **Works identically locally and in CI**, on any platform, with no shell +6. **Bounded execution.** Per-package and per-target commands may run with a + caller-selected concurrency limit and timeout. Defaults remain sequential and + unbounded for backward compatibility. +7. **Works identically locally and in CI**, on any platform, with no shell dialect assumptions. **Open source**: ships from `ox-tools` to crates.io. ## 3. Non-Goals @@ -78,10 +82,12 @@ target (with placeholder substitution), or exactly once for the whole set. aggregation, llvm-cov's dual-config instrumentation, and the per-crate readme `doc2readme` reconciliation stay in their recipes. `cargo-each` owns only the selection → filter → iterate spine those recipes wrap. -- **Parallel scheduling / job pools.** Commands run sequentially. Parallelism, if - ever wanted, is a later, additive concern. - **A general templating engine.** Placeholder substitution is a fixed, small set of `{token}` replacements, not an expression language. +- **Tool installation or workflow orchestration.** cargo-each can execute + `rustup` or another installer when the caller asks it to, but it does not + decide which tools a repository needs, select binary versus source + installation, inspect Git, collect coverage, or understand Anvil tiers. - **A public library API.** `cargo-each` ships as an executable only. Its modules are crate-internal (`pub(crate)`), so there is no semver-committed library surface, no `check-external-types` obligation, and nothing to consume @@ -168,24 +174,31 @@ beyond placeholder substitution. | Flag | Meaning | |------|---------| | `-p`, `--package ` | Select a member. Repeatable. `SPEC` is a package name, a `name@version` spec, or a Unix glob (`tokio-*`), matching `cargo-coverage-gate`'s existing `-p` idiom. | +| `--package-file ` | Read package specs from a UTF-8 file, one spec per nonempty line. Repeatable; specs are unioned with `--package`. A present empty file is an explicit empty selection. | | `--workspace`, `--all` | Select every workspace member. | | `--exclude ` | Remove a member from the selection (requires `--workspace`). Repeatable. | | `--none` | Explicitly select zero members. Resolves to an empty set (a no-op, exit 0). Emitted by the impact hand-off when a tier is empty; replaces the `--skip` sentinel. | -A computed selection (e.g. an impact tier) is fed in as ordinary flags via -shell expansion — cargo-each has no `--from-file` / `--from-env` source, so it -stays agnostic about origin. See section 6 for the anvil hand-off. +A package file contains only package specs, not command-line tokens, comments, +an impact-tier name, or policy. `foo@1.2.3` has the same meaning whether it came +from `--package foo@1.2.3` or a file. A missing, unreadable, non-UTF-8, or +malformed file is an error. This keeps cargo-each independent of cargo-delta +while allowing cargo-delta output to be consumed without command substitution. **Resolution order.** The literal flags resolve to: 1. If `--none` appears anywhere → empty set. 2. Else if `--workspace`/`--all` appears → all members, minus `--exclude`. -3. Else if any `-p` matched → the matched members. -4. Else → `default-members` (exactly like `cargo build`; pass `--workspace` +3. Else if any direct or file-supplied spec exists → the matching members. +4. Else if at least one `--package-file` was supplied → empty set. +5. Else → `default-members` (exactly like `cargo build`; pass `--workspace` for the whole workspace). -A `-p` selector that matches no member is an error (same policy as +A selector that matches no member is an error (same policy as `cargo-coverage-gate`), so typos fail loudly rather than silently skipping. +An empty package file is different: it is the producer's explicit statement +that the computed set is empty and therefore exits successfully without +running the command. ### 4.2 Filters @@ -237,6 +250,8 @@ filtered set is empty, `cargo-each` exits 0, exactly like an empty selection. | `--each-target ` | **per-target**: run once for each selected member target of `KIND`. Repeatable; kinds are OR-combined and each target runs at most once. Mutually exclusive with `--once`. | | `--target-required-feature ` | In per-target mode, retain targets whose `required-features` contains `FEATURE`. Repeatable; values are AND-combined. Requires `--each-target`. | | `--keep-going` | Don't stop at the first failing command; run them all and exit non-zero if any failed. Default is fail-fast (exit with the first failure's code). | +| `--jobs ` | Run at most `N` per-package or per-target commands concurrently. Default `1`. With `--once`, values other than `1` are a usage error. | +| `--timeout ` | Terminate an invocation and its child process tree when it exceeds the positive duration, such as `30s` or `2m`. Applies independently to every invocation, including `--once`. No timeout by default. | | `--chdir` | Run each per-package or per-target command from that member's crate root (the directory containing its `Cargo.toml`) instead of the caller's CWD. Combined with `--once` it is a usage error (exit 2). Placeholders stay absolute, so only *relative* args in the command shift to the member dir. | | `--manifest-path ` | Workspace root `Cargo.toml`. Defaults to auto-detection from CWD. | | `--dry-run` | Print the fully-substituted commands that *would* run, one per line, without executing. | @@ -253,11 +268,21 @@ Substituted inside each `ARG` of the command template: | `{manifest}` | absolute path to the member's `Cargo.toml` | per-package | | `{target}` | Cargo target name | per-target | | `{packages}` | the cargo selection flags for the resolved set: `--workspace` when the whole workspace was selected via `--workspace`/`--all` with no excludes **and no package filters applied**, else `--package name@version …` (one pair per member). Only valid as a standalone `ARG`; it expands to multiple tokens. | once | +| `{workspace-rust-version}` | Root `[workspace.package].rust-version`, or root `[package].rust-version` in a single-package repository. | all | Per-target mode accepts all per-package placeholders plus `{target}`. Using a per-package or per-target token in `--once` mode, `{target}` in per-package mode, or `{packages}` outside `--once` is a usage error. +`{workspace-rust-version}` is workspace-scoped rather than tied to one selected +member. Resolving it requires a root declaration. cargo-each also requires every +workspace member to expose a resolved `rust_version` no newer than the root +floor. Missing values, a member requiring a newer compiler, or a non-Rust +semantic version is a configuration error. Lower member minima are valid. This +matches the meaning of one compiler selected for a complete workspace; it is +not a per-package toolchain matrix. The validation is lazy: commands that do +not contain the placeholder do not require a workspace Rust version. + Targets run in package-name order and then target-name order. A target matching more than one requested kind runs once. No matching targets is a successful no-op. @@ -277,6 +302,22 @@ no-op. - **No shell.** The command is spawned directly (argv, not a shell string), so there is no quoting/dialect surface. Placeholder expansion is textual and happens before spawn. +- **Bounded concurrency.** With `--jobs > 1`, output from each invocation is + buffered and emitted as one block in deterministic plan order. Fail-fast + stops launching new work after the first observed failure and waits for + already-running children; `--keep-going` launches the complete plan. The + final failure is chosen by plan order, not scheduler timing. A worker panic + is converted into an infrastructure-failure outcome; each worker has a + dedicated completion channel, so an unexpected exit is observable as + disconnection rather than leaving the scheduler blocked forever. +- **Timeouts terminate trees.** A timed-out command is a failure. cargo-each + terminates the child process tree rather than only the immediate process, so + compiler or test descendants cannot continue mutating the target directory + after cargo-each returns. If tree termination itself fails, cargo-each reports + that infrastructure failure without waiting indefinitely for surviving + descendants to close inherited output pipes: capture stops retaining new + bytes, emits the partial output already buffered, and detaches blocked + readers so the timeout remains bounded. - **Child executable resolution follows `PATH`.** `cargo-each` explicitly copies an inherited `PATH` onto every child command. This is equivalent to ordinary inheritance on other platforms and makes Windows resolve a relative @@ -285,17 +326,12 @@ no-op. ## 6. How it simplifies cargo-anvil -The recipes stop parsing the impact selection and metadata by hand. anvil's -`_anvil-impact-include ` helper reads -`target/anvil/impact/include_.txt`, applies the `ANVIL_IMPACT=off` -override, and emits a **concrete selector for every tier** — `--workspace` -(unscoped / local / off), `--package name@version …` (scoped), or `--none` -(empty tier). A `cargo-each` check just splats that output straight in as -flags. Because the helper always emits a concrete selector, cargo-each never -falls back to `default-members`, so no per-call default flag is needed. -Illustrative before/after (the recipe keeps its own comments, setup deps, -`: anvil-impact` dependency, and any domain glue; only the selection spine -changes): +The recipes stop parsing impact selections and metadata by hand. cargo-delta +writes one `name@version` package file per tier under +`target/anvil/impact/`. A cargo-each check supplies the appropriate file +directly. An empty tier is an empty file and therefore a successful no-op. +Illustrative before/after (the recipe keeps its own setup and `anvil-impact` +dependencies; only the selection spine changes): **clippy** (affected tier, single invocation): @@ -305,9 +341,9 @@ if (-not $env:ANVIL_INCLUDE_AFFECTED) { $env:ANVIL_INCLUDE_AFFECTED = (& just _a if ($env:ANVIL_INCLUDE_AFFECTED -eq '--skip') { exit 0 } & cargo clippy @(if ($env:ANVIL_INCLUDE_AFFECTED) { -split $env:ANVIL_INCLUDE_AFFECTED } else { '--workspace' }) --all-targets --all-features --locked -- -D warnings ``` -```powershell +```just # after -cargo each @(& {{ just_executable() }} _anvil-impact-include affected) --once -- \ +cargo each --package-file target/anvil/impact/affected.packages --once -- \ cargo clippy {packages} --all-targets --all-features --locked -- -D warnings ``` @@ -315,52 +351,38 @@ cargo each @(& {{ just_executable() }} _anvil-impact-include affected) --once -- name-to-manifest map, `--workspace` branch, `@version` strip, and iteration loop collapse to: -```powershell -cargo each @(& {{ just_executable() }} _anvil-impact-include affected) --filter lib -- \ +```just +cargo each --package-file target/anvil/impact/affected.packages --filter lib -- \ cargo +{{ rust_nightly_external_types }} check-external-types --manifest-path {manifest} ``` **loom** (affected packages that depend on loom): -```powershell -cargo each @(& {{ just_executable() }} _anvil-impact-include affected) --filter dep:loom -- \ +```just +cargo each --package-file target/anvil/impact/affected.packages --filter dep:loom -- \ cargo +{{ rust_nightly }} test --package {name} ... ``` -**llvm-cov opt-out drop** (exclude coverage-opted-out members): +**per-target examples** (run each selected example with a timeout): -```powershell -cargo each @(& {{ just_executable() }} _anvil-impact-include affected) \ - --exclude-filter metadata:coverage-gate.min-lines-percent=0 --once -- +```just +cargo each --package-file target/anvil/impact/affected.packages \ + --each-target example --timeout 30s -- \ + cargo run --package {spec} --example {target} ``` -Recipes whose only per-tier logic is the skip/splat preamble (bench, clippy, -doc-build, examples, miri*, doc-test, cargo-hack, udeps, careful) become a -single `cargo each … --once` line. Modified-tier workspace-wide tools (fmt, -cargo-sort, license-headers, spellcheck, ensure-no-*) become -`cargo each @(& just _anvil-impact-include modified) --once -- ` — the -`--once` skip-when-empty behavior replaces the `--skip` guard while the tool -still runs workspace-wide. - -Three small `anvil-impact` adjustments complete the picture (all part of the -adoption change, not this crate): - -- **Emit `--none`, not `--skip`, for an empty tier**, and drop the modified - tier's empty default: `_anvil-impact-include` emits `--workspace` / - `--package …` / `--none` **uniformly across all three tiers**. cargo-delta - makes no fundamental distinction between the tiers — they are just three - package sets — so neither should the helper. `--none` is `cargo-each`'s - native "select zero members" token, so the include file needs no - anvil-specific sentinel and `cargo each` skips the tier with no caller guard. -- **Print one token per line** from `_anvil-impact-include`, so the recipe's - `@(& …)` capture is a ready-to-splat array — no `-split`, no `if/else`. -- **Stop version-qualifying.** `_anvil-impact-format` can emit bare package - names; `cargo-each` derives `{spec}`/`{packages}` (the `name@version` form a - child cargo command needs) from live metadata itself. - -With the helper's output splatted straight into `cargo each`, the per-check -`_anvil-impact-include` *self-populate* line and the `ANVIL_INCLUDE_` -environment variable are no longer needed by scoped checks. +Recipes whose only per-tier logic is the skip/splat preamble become one +`cargo each --package-file …` command. Unscoped runs pass `--workspace` +instead; choosing scoped versus unscoped input remains caller policy and is not +encoded into cargo-each. + +The setup graph can use `{workspace-rust-version}` to install the single root +MSRV fallback without parsing Cargo TOML in a shell: + +```just +cargo each --workspace --once -- \ + rustup toolchain install {workspace-rust-version} --profile minimal +``` ## 7. Rejected alternatives @@ -373,11 +395,9 @@ environment variable are no longer needed by scoped checks. - **A generic expression language for filters.** Over-built for the handful of predicates the recipes actually need; the fixed predicate set covers every current `cargo metadata` filter and stays trivially auditable. -- **A `--from-file` / `--from-env` selection source.** Rejected: it would pull - the impact artifact layout (and the `ANVIL_IMPACT=off` widening + tier-default - policy) into cargo-each, duplicating logic that already lives in anvil's - `_anvil-impact-include` helper. Keeping cargo-each flag-only and letting the - caller splat that helper's output in is smaller and keeps the impact policy in - one place. +- **An Anvil-aware selection source.** Rejected: cargo-each does not accept a + tier name, inspect `ANVIL_IMPACT`, or assume a `target/anvil` layout. + `--package-file` is deliberately generic: one Cargo package spec per line, + with an empty file meaning an explicitly empty set. - **Reuse `cargo xtask`/a justfile function.** Neither is cargo-native selection; both re-introduce a shell dialect. A small binary is portable and testable. diff --git a/crates/cargo-each/src/cli.rs b/crates/cargo-each/src/cli.rs index 9b338e4d9..300d0fe5a 100644 --- a/crates/cargo-each/src/cli.rs +++ b/crates/cargo-each/src/cli.rs @@ -3,7 +3,9 @@ //! Command-line interface definitions for `cargo-each`. +use std::num::NonZeroUsize; use std::path::PathBuf; +use std::time::Duration; use clap::{Args, Parser}; @@ -36,6 +38,11 @@ pub(crate) struct EachArgs { #[arg(short = 'p', long = "package", value_name = "SPEC")] pub(crate) packages: Vec, + /// Read package specs from a UTF-8 file, one per nonempty line. + /// Repeatable; specs are unioned with --package. + #[arg(long = "package-file", value_name = "PATH")] + pub(crate) package_files: Vec, + /// Select every workspace member. #[arg(long, visible_alias = "all")] pub(crate) workspace: bool, @@ -87,6 +94,15 @@ pub(crate) struct EachArgs { #[arg(long)] pub(crate) keep_going: bool, + /// Run at most N per-package or per-target commands concurrently. + #[arg(long, default_value_t = NonZeroUsize::MIN, value_name = "N")] + pub(crate) jobs: NonZeroUsize, + + /// Terminate each invocation and its process tree after this duration. + /// Accepts a positive integer followed by `ms`, `s`, or `m`. + #[arg(long, value_name = "DURATION", value_parser = parse_duration)] + pub(crate) timeout: Option, + /// Print the fully-substituted commands without executing them. #[arg(long)] pub(crate) dry_run: bool, @@ -101,6 +117,34 @@ pub(crate) struct EachArgs { pub(crate) command: Vec, } +fn parse_duration(value: &str) -> Result { + let (digits, unit) = if let Some(digits) = value.strip_suffix("ms") { + (digits, "ms") + } else if let Some(digits) = value.strip_suffix('s') { + (digits, "s") + } else if let Some(digits) = value.strip_suffix('m') { + (digits, "m") + } else { + return Err("expected a positive integer followed by `ms`, `s`, or `m`".to_owned()); + }; + if digits.is_empty() || !digits.bytes().all(|byte| byte.is_ascii_digit()) { + return Err("expected a positive integer followed by `ms`, `s`, or `m`".to_owned()); + } + let amount = digits.parse::().map_err(|error| format!("duration is too large: {error}"))?; + if amount == 0 { + return Err("duration must be greater than zero".to_owned()); + } + match unit { + "ms" => Ok(Duration::from_millis(amount)), + "s" => Ok(Duration::from_secs(amount)), + "m" => amount + .checked_mul(60) + .map(Duration::from_secs) + .ok_or_else(|| "duration is too large".to_owned()), + _ => unreachable!("unit is selected from the three cases above"), + } +} + #[cfg(test)] #[cfg_attr(coverage_nightly, coverage(off))] mod tests { @@ -112,4 +156,18 @@ mod tests { fn cli_definition_is_well_formed() { CargoCli::command().debug_assert(); } + + #[test] + fn parses_documented_durations() { + assert_eq!(parse_duration("250ms"), Ok(Duration::from_millis(250))); + assert_eq!(parse_duration("30s"), Ok(Duration::from_secs(30))); + assert_eq!(parse_duration("2m"), Ok(Duration::from_mins(2))); + } + + #[test] + fn rejects_zero_malformed_and_overflowing_durations() { + for value in ["0s", "1", "1h", "-1s", "1.5s", "ms", "18446744073709551615m"] { + assert!(parse_duration(value).is_err(), "{value}"); + } + } } diff --git a/crates/cargo-each/src/error.rs b/crates/cargo-each/src/error.rs index b36708495..43b62a984 100644 --- a/crates/cargo-each/src/error.rs +++ b/crates/cargo-each/src/error.rs @@ -32,7 +32,14 @@ InvalidFilterExpressionError, InvalidTargetKindError, PlaceholderMisuseError, - ChdirConflictsWithOnceError + ChdirConflictsWithOnceError, + JobsConflictWithOnceError, + PackageFileReadError, + PackageFileUtf8Error, + InvalidPackageFileLineError, + WorkspaceManifestReadError, + WorkspaceManifestParseError, + WorkspaceRustVersionError )] pub(crate) struct EachError; @@ -82,6 +89,62 @@ pub(crate) struct PlaceholderMisuseError { #[display("`--chdir` cannot be combined with `--once`")] pub(crate) struct ChdirConflictsWithOnceError; +/// `--jobs` greater than one was combined with `--once`. +#[ohno::error] +#[display("`--jobs` must be 1 when combined with `--once`")] +pub(crate) struct JobsConflictWithOnceError; + +/// A `--package-file` could not be read. +#[ohno::error] +#[display("could not read package file `{path}`")] +#[from(std::io::Error)] +pub(crate) struct PackageFileReadError { + pub(crate) path: String, +} + +/// A `--package-file` was not valid UTF-8. +#[ohno::error] +#[display("package file `{path}` is not valid UTF-8")] +#[from(std::string::FromUtf8Error)] +pub(crate) struct PackageFileUtf8Error { + pub(crate) path: String, +} + +/// A nonempty line in a `--package-file` was not a package spec. +#[ohno::error] +#[display("invalid package spec in `{path}` at line {line}: `{spec}` ({reason})")] +pub(crate) struct InvalidPackageFileLineError { + pub(crate) path: String, + pub(crate) line: usize, + pub(crate) spec: String, + pub(crate) reason: String, +} + +/// The root manifest could not be read while resolving +/// `{workspace-rust-version}`. +#[ohno::error] +#[display("could not read workspace manifest `{path}`")] +#[from(std::io::Error)] +pub(crate) struct WorkspaceManifestReadError { + pub(crate) path: String, +} + +/// The root manifest could not be parsed while resolving +/// `{workspace-rust-version}`. +#[ohno::error] +#[display("could not parse workspace manifest `{path}`")] +#[from(toml::de::Error)] +pub(crate) struct WorkspaceManifestParseError { + pub(crate) path: String, +} + +/// The workspace Rust-version contract is incomplete or inconsistent. +#[ohno::error] +#[display("cannot resolve `{{workspace-rust-version}}`: {reason}")] +pub(crate) struct WorkspaceRustVersionError { + pub(crate) reason: String, +} + #[cfg(test)] #[cfg_attr(coverage_nightly, coverage(off))] mod tests { @@ -135,4 +198,26 @@ mod tests { assert!(rendered.contains("--chdir")); assert!(rendered.contains("--once")); } + + #[test] + fn package_file_line_error_names_source() { + let err = InvalidPackageFileLineError::new( + "affected.packages".to_owned(), + 3_usize, + "--workspace".to_owned(), + "command-line tokens are not package specs".to_owned(), + ); + let rendered = err.to_string(); + assert!(rendered.contains("affected.packages")); + assert!(rendered.contains("line 3")); + assert!(rendered.contains("--workspace")); + } + + #[test] + fn workspace_rust_version_error_renders_reason() { + let err = WorkspaceRustVersionError::new("member `alpha` does not declare `rust-version`".to_owned()); + let rendered = err.to_string(); + assert!(rendered.contains("{workspace-rust-version}")); + assert!(rendered.contains("alpha")); + } } diff --git a/crates/cargo-each/src/filter.rs b/crates/cargo-each/src/filter.rs index 3fc17fb1e..184861fe9 100644 --- a/crates/cargo-each/src/filter.rs +++ b/crates/cargo-each/src/filter.rs @@ -420,6 +420,7 @@ mod tests { Member { name: "m".to_owned(), version: "0.1.0".to_owned(), + rust_version: Some("1.70.0".parse().expect("valid Rust version")), manifest_path: PathBuf::from("/ws/m/Cargo.toml"), publishable: true, features: BTreeSet::new(), diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index c06c501be..d34117408 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -34,16 +34,18 @@ //! //! - `-p` / `--package ` — select a member. Repeatable. `SPEC` is a //! package name, a `name@version` spec, or a Unix glob (`tokio-*`). +//! - `--package-file ` — read package specs from a UTF-8 file, one per +//! nonempty line. Repeatable; specs are unioned with `--package`. An empty +//! file explicitly selects no members. //! - `--workspace` / `--all` — select every workspace member. //! - `--exclude ` — drop a member (with `--workspace`). Repeatable. //! - `--none` — explicitly select zero members (a no-op that exits 0). //! //! When nothing is named the default is cargo `default-members`, exactly //! like `cargo build`; pass `--workspace` for every member. A selector that -//! matches no member is an error, so typos fail loudly. A computed selection -//! (for example a CI affected-packages set) is fed in as ordinary flags via -//! shell expansion — `cargo-each` has no file or environment-variable source -//! of its own. +//! matches no member is an error, so typos fail loudly. Package files contain +//! package specs only: comments, command-line tokens, malformed input, and +//! missing, unreadable, or non-UTF-8 files are errors. //! //! ## Filters //! @@ -72,9 +74,11 @@ //! `--target-required-feature` further narrows targets. //! //! `--keep-going` runs every invocation and exits non-zero if any failed -//! (default is fail-fast); `--chdir` runs each per-package or per-target -//! command from that member crate root; `--dry-run` prints commands without -//! running them. +//! (default is fail-fast). `--jobs ` bounds concurrent per-package or +//! per-target work (default `1`), while `--timeout ` terminates each +//! invocation and its process tree independently (`250ms`, `30s`, or `2m`). +//! `--chdir` runs each per-package or per-target command from that member crate +//! root; `--dry-run` prints commands without running them. //! //! ## Placeholders //! @@ -89,6 +93,9 @@ //! - `{packages}` — the cargo selection flags for the resolved set //! (`--workspace` for the whole workspace, else `--package name@version …`); //! valid only in `--once` mode and only as a standalone argument. +//! - `{workspace-rust-version}` — the root `[workspace.package].rust-version`, +//! or root `[package].rust-version` in a single-package repository; valid in +//! every mode. //! //! Using a placeholder in the wrong mode is a usage error. Only the tokens //! above are interpreted; any other `{…}` sequence (a typo, or a literal brace @@ -100,6 +107,19 @@ //! An empty resolved selection (via `--none`, or a filter that removes every //! member) is a **successful no-op**: `cargo-each` prints a one-line note and //! exits 0. This is what lets callers drop bespoke nothing-to-do guards. +//! Workspace Rust-version validation is lazy: it runs only when the command +//! uses `{workspace-rust-version}`, then requires every member's resolved +//! minimum to be present and no newer than the root floor. +//! +//! With `--jobs > 1`, each invocation's output is buffered and complete blocks +//! are emitted in deterministic plan order. Fail-fast stops launching after +//! the first observed failure, waits for running work, and chooses the final +//! failure by plan order. `--keep-going` runs the complete plan. Worker panics +//! and unexpected worker-channel disconnections become infrastructure-failure +//! outcomes instead of blocking the scheduler. +//! If timed-out tree cleanup fails, cargo-each reports the infrastructure +//! failure and emits already-buffered output without waiting indefinitely for +//! surviving descendants to close inherited pipes. //! Child commands inherit `PATH` explicitly. On Windows this makes relative //! program lookup honor the inherited `PATH` order instead of preferring an //! unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/src/plan.rs b/crates/cargo-each/src/plan.rs index cd6003150..19bef8a68 100644 --- a/crates/cargo-each/src/plan.rs +++ b/crates/cargo-each/src/plan.rs @@ -63,6 +63,17 @@ pub(crate) struct Plan { pub(crate) invocations: Vec, } +/// Inputs that control how a selected member set becomes invocations. +#[derive(Debug, Clone, Copy)] +pub(crate) struct BuildOptions<'a> { + pub(crate) mode: Mode, + pub(crate) chdir: bool, + pub(crate) packages: PackagesExpansion, + pub(crate) target_kinds: &'a BTreeSet, + pub(crate) target_required_features: &'a BTreeSet, + pub(crate) workspace_rust_version: Option<&'a str>, +} + impl Plan { /// Build the plan. /// @@ -81,15 +92,15 @@ impl Plan { /// /// Returns [`EachError`] if `chdir` is combined with [`Mode::Once`], or if /// a placeholder in `command` is used in the wrong mode. - pub(crate) fn build( - members: &[&Member], - mode: Mode, - chdir: bool, - packages: PackagesExpansion, - target_kinds: &BTreeSet, - target_required_features: &BTreeSet, - command: &[String], - ) -> Result { + pub(crate) fn build(members: &[&Member], command: &[String], options: BuildOptions<'_>) -> Result { + let BuildOptions { + mode, + chdir, + packages, + target_kinds, + target_required_features, + workspace_rust_version, + } = options; if chdir && mode == Mode::Once { return Err(ChdirConflictsWithOnceError::new().into()); } @@ -111,6 +122,7 @@ impl Plan { spec: m.spec(), version: m.version.clone(), manifest: m.manifest_path.display().to_string(), + workspace_rust_version: workspace_rust_version.map(str::to_owned), }; Ok(Invocation { label: Some(m.name.clone()), @@ -138,6 +150,7 @@ impl Plan { version: member.version.clone(), manifest: member.manifest_path.display().to_string(), target: target.name.clone(), + workspace_rust_version: workspace_rust_version.map(str::to_owned), }; Ok(Invocation { label: Some(format!("{}::{}", member.name, target.name)), @@ -149,7 +162,10 @@ impl Plan { .collect::, EachError>>()?, Mode::Once => { let packages = packages_flags(members, packages); - let placeholders = Placeholders::Once { packages }; + let placeholders = Placeholders::Once { + packages, + workspace_rust_version: workspace_rust_version.map(str::to_owned), + }; vec![Invocation { label: None, argv: substitute(command, &placeholders)?, @@ -188,6 +204,7 @@ mod tests { Member { name: name.to_owned(), version: "1.2.3".to_owned(), + rust_version: Some("1.70.0".parse().expect("valid Rust version")), manifest_path: PathBuf::from(format!("/ws/{name}/Cargo.toml")), publishable: true, features: BTreeSet::new(), @@ -202,7 +219,18 @@ mod tests { } fn build(members: &[&Member], mode: Mode, chdir: bool, packages: PackagesExpansion, command: &[String]) -> Result { - Plan::build(members, mode, chdir, packages, &BTreeSet::new(), &BTreeSet::new(), command) + Plan::build( + members, + command, + BuildOptions { + mode, + chdir, + packages, + target_kinds: &BTreeSet::new(), + target_required_features: &BTreeSet::new(), + workspace_rust_version: None, + }, + ) } #[test] @@ -333,12 +361,15 @@ mod tests { let required = std::iter::once("loom".to_owned()).collect(); let plan = Plan::build( &[&a], - Mode::PerTarget, - false, - PackagesExpansion::Explicit, - &kinds, - &required, &cmd(&["cargo", "test", "-p", "{name}", "--test", "{target}"]), + BuildOptions { + mode: Mode::PerTarget, + chdir: false, + packages: PackagesExpansion::Explicit, + target_kinds: &kinds, + target_required_features: &required, + workspace_rust_version: None, + }, ) .expect("build target plan"); assert_eq!(plan.invocations.len(), 1); @@ -357,14 +388,36 @@ mod tests { let kinds = std::iter::once(TargetKind::Example).collect(); let plan = Plan::build( &[&a], - Mode::PerTarget, - true, - PackagesExpansion::Explicit, - &kinds, - &BTreeSet::new(), &cmd(&["echo", "{target}"]), + BuildOptions { + mode: Mode::PerTarget, + chdir: true, + packages: PackagesExpansion::Explicit, + target_kinds: &kinds, + target_required_features: &BTreeSet::new(), + workspace_rust_version: None, + }, ) .expect("build target plan"); assert_eq!(plan.invocations[0].work_dir.as_deref(), Some(PathBuf::from("/ws/alpha").as_path())); } + + #[test] + fn workspace_rust_version_is_available_in_once_mode() { + let a = member("alpha"); + let plan = Plan::build( + &[&a], + &cmd(&["rustup", "toolchain", "install", "{workspace-rust-version}"]), + BuildOptions { + mode: Mode::Once, + chdir: false, + packages: PackagesExpansion::Workspace, + target_kinds: &BTreeSet::new(), + target_required_features: &BTreeSet::new(), + workspace_rust_version: Some("1.80"), + }, + ) + .expect("build"); + assert_eq!(plan.invocations[0].argv, ["rustup", "toolchain", "install", "1.80"]); + } } diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 4e1d6e29b..5426796e6 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -5,20 +5,32 @@ //! apply filters, build the plan, and run it. use std::collections::BTreeSet; -use std::process::{Command, ExitCode}; +use std::io::{self, Write as _}; +use std::num::NonZeroUsize; +use std::panic::{self, UnwindSafe}; +use std::process::{Command, ExitCode, ExitStatus, Stdio}; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex, mpsc}; +use std::thread; +use std::time::{Duration, Instant}; +use cargo_gamma_process::{MemoryRequest, ProcessTree, prepare}; use cargo_metadata::TargetKind; use ohno::{AppError, IntoAppError}; use crate::cli::EachArgs; -use crate::error::InvalidTargetKindError; +use crate::error::{InvalidTargetKindError, JobsConflictWithOnceError}; use crate::filter::Predicate; -use crate::plan::{Mode, PackagesExpansion, Plan}; +use crate::plan::{BuildOptions, Invocation, Mode, PackagesExpansion, Plan}; use crate::select::Selection; +use crate::substitute::{uses_workspace_rust_version, validate_placeholders}; use crate::workspace::{Member, Workspace}; +#[cfg(test)] +const WORKER_PANIC_TEST_PROGRAM: &str = "__cargo_each_injected_worker_panic"; + pub(crate) fn run(args: &EachArgs) -> Result { - let selection = build_selection(args); + let selection = build_selection(args).into_app_err("failed to read package selection")?; let workspace = Workspace::load(args.manifest_path.as_deref()).into_app_err("failed to load workspace")?; let mut members = selection.resolve(&workspace).into_app_err("failed to resolve package selection")?; @@ -41,14 +53,34 @@ pub(crate) fn run(args: &EachArgs) -> Result { } else { Mode::PerTarget }; + if mode == Mode::Once && args.jobs.get() != 1 { + return Err(JobsConflictWithOnceError::new()).into_app_err("invalid execution configuration"); + } + + // Validate mode-specific tokens before resolving the lazy workspace token + // so a malformed command reports its direct usage error first. + validate_placeholders(&args.command, mode).into_app_err("failed to build command plan")?; + let workspace_rust_version = if uses_workspace_rust_version(&args.command) { + Some( + workspace + .workspace_rust_version() + .into_app_err("failed to resolve workspace Rust version")?, + ) + } else { + None + }; + let plan = Plan::build( &members, - mode, - args.chdir, - packages, - &target_kinds, - &target_required_features, &args.command, + BuildOptions { + mode, + chdir: args.chdir, + packages, + target_kinds: &target_kinds, + target_required_features: &target_required_features, + workspace_rust_version: workspace_rust_version.as_deref(), + }, ) .into_app_err("failed to build command plan")?; @@ -67,22 +99,12 @@ pub(crate) fn run(args: &EachArgs) -> Result { return Ok(ExitCode::SUCCESS); } - execute(&plan, args.keep_going) + execute(&plan, args.keep_going, args.jobs, args.timeout) } -/// Assemble a [`Selection`] from the parsed flags. -/// -/// The selection is entirely flag-driven: a computed selection (e.g. an -/// impact tier) is fed in by the caller via ordinary shell expansion — anvil -/// splats `_anvil-impact-include ` into the `cargo each` invocation — -/// so cargo-each stays agnostic about where the selectors came from. -fn build_selection(args: &EachArgs) -> Selection { - Selection { - packages: args.packages.clone(), - all: args.workspace, - exclude: args.exclude.clone(), - none: args.none, - } +/// Assemble a [`Selection`] from direct and file-backed package specs. +fn build_selection(args: &EachArgs) -> Result { + Selection::from_sources(&args.packages, &args.package_files, args.workspace, &args.exclude, args.none) } /// Narrow `members` by package keep and drop expressions. Repeated `--filter` @@ -116,54 +138,532 @@ fn parse_target_kinds(kinds: &[String]) -> Result, AppError .into_app_err("invalid per-target configuration") } -/// Run each invocation, honoring the fail-fast / `--keep-going` policy. -/// -/// A spawn failure (the child could not be launched at all) is treated the -/// same as a non-zero child exit: under `--keep-going` it is logged, counted -/// as a failure, and the run continues (final exit `1`); under fail-fast it -/// aborts. This keeps the documented "run them all" contract intact even when -/// one invocation cannot start. -fn execute(plan: &Plan, keep_going: bool) -> Result { +fn execute(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: Option) -> Result { + if jobs.get() == 1 { + Ok(execute_sequential(plan, keep_going, timeout)) + } else { + execute_parallel(plan, keep_going, jobs, timeout) + } +} + +fn execute_sequential(plan: &Plan, keep_going: bool, timeout: Option) -> ExitCode { let mut any_failed = false; - for inv in &plan.invocations { - if let Some(label) = &inv.label { - eprintln!("cargo each: {label}"); - } - let (program, rest) = inv.argv.split_first().expect("Plan::build never emits an empty argv"); - let mut command = Command::new(program); - if let Some(path) = std::env::var_os("PATH") { - command.env("PATH", path); - } - command.args(rest); - if let Some(dir) = &inv.work_dir { - command.current_dir(dir); - } - let status = match command.status() { - Ok(status) => status, - // A spawn failure under --keep-going is a failed invocation, not an - // abort: log it, mark the run failed, and move on so the remaining - // members still run (contract: exit 1 when any invocation failed). - Err(err) if keep_going => { - eprintln!("cargo each: failed to spawn `{program}`: {err}"); + for invocation in &plan.invocations { + emit_label(invocation); + let result = if let Some(timeout) = timeout { + run_streamed_with_timeout(invocation, timeout) + } else { + run_streamed(invocation) + }; + match result { + InvocationResult::Exited(status) if status.success() => {} + InvocationResult::Exited(status) => { + if !keep_going { + return ExitCode::from(exit_byte(status.code())); + } + any_failed = true; + } + InvocationResult::TimedOut(duration) => { + eprintln!("cargo each: invocation timed out after {}", display_duration(duration)); + if !keep_going { + return ExitCode::from(1); + } any_failed = true; - continue; } - // Fail-fast: a spawn failure is a hard error (exit 2 via main.rs). - other => other.into_app_err(format!("failed to spawn `{program}`"))?, + InvocationResult::Infrastructure(message) => { + eprintln!("cargo each: {message}"); + if !keep_going { + return ExitCode::from(2); + } + any_failed = true; + } + } + } + if any_failed { ExitCode::from(1) } else { ExitCode::SUCCESS } +} + +fn execute_parallel(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: Option) -> Result { + let invocations = Arc::new(plan.invocations.clone()); + let worker_count = jobs.get().min(invocations.len()).min(cargo_gamma_process::capacity().max(1)); + let mut workers = Vec::with_capacity(worker_count); + let mut next_index = 0; + let mut stop_launching = false; + let mut launch_error = None; + + while next_index < worker_count { + match spawn_worker(next_index, Arc::clone(&invocations), timeout) { + Ok(worker) => { + workers.push(worker); + next_index += 1; + } + Err(error) => { + launch_error = Some(error); + break; + } + } + } + + let mut outcomes = Vec::with_capacity(invocations.len()); + while !workers.is_empty() { + let outcome = wait_for_worker(&mut workers); + if !keep_going && outcome.outcome.result.failed() { + stop_launching = true; + } + outcomes.push(outcome); + + if !stop_launching && launch_error.is_none() && next_index < invocations.len() { + match spawn_worker(next_index, Arc::clone(&invocations), timeout) { + Ok(worker) => { + workers.push(worker); + next_index += 1; + } + Err(error) => launch_error = Some(error), + } + } + } + + if let Some(error) = launch_error { + return Err(error).into_app_err("failed to create cargo-each worker thread"); + } + outcomes.sort_by_key(|outcome| outcome.index); + + for indexed in &outcomes { + emit_buffered(&invocations[indexed.index], &indexed.outcome).into_app_err("failed to emit buffered command output")?; + } + + let Some(first_failure) = outcomes.iter().find(|outcome| outcome.outcome.result.failed()) else { + return Ok(ExitCode::SUCCESS); + }; + if keep_going { + return Ok(ExitCode::from(1)); + } + Ok(match &first_failure.outcome.result { + InvocationResult::Exited(status) => ExitCode::from(exit_byte(status.code())), + InvocationResult::TimedOut(_) => ExitCode::from(1), + InvocationResult::Infrastructure(_) => ExitCode::from(2), + }) +} + +fn spawn_worker(index: usize, invocations: Arc>, timeout: Option) -> io::Result { + let (sender, receiver) = mpsc::channel(); + let thread = thread::Builder::new().name(format!("cargo-each-worker-{index}")).spawn(move || { + complete_worker(&sender, move || { + let Some(invocation) = invocations.get(index) else { + return BufferedOutcome::infrastructure(format!( + "internal scheduler error: invocation index {index} is outside the command plan" + )); + }; + run_captured(invocation, timeout) + }); + })?; + Ok(RunningWorker { index, receiver, thread }) +} + +fn complete_worker(sender: &mpsc::Sender, work: impl FnOnce() -> BufferedOutcome + UnwindSafe) { + let outcome = match panic::catch_unwind(work) { + Ok(outcome) => outcome, + Err(payload) => BufferedOutcome::infrastructure(format!( + "worker panicked while running invocation: {}", + panic_description(payload.as_ref()) + )), + }; + let _receiver_gone = sender.send(outcome); +} + +fn wait_for_worker(workers: &mut Vec) -> IndexedOutcome { + loop { + let ready = workers + .iter() + .enumerate() + .find_map(|(position, worker)| match worker.receiver.try_recv() { + Ok(outcome) => Some((position, Some(outcome))), + Err(mpsc::TryRecvError::Disconnected) => Some((position, None)), + Err(mpsc::TryRecvError::Empty) => None, + }); + let Some((position, reported)) = ready else { + thread::sleep(Duration::from_millis(1)); + continue; + }; + + let RunningWorker { index, thread, .. } = workers.swap_remove(position); + let outcome = match (reported, thread.join()) { + (Some(outcome), Ok(())) => outcome, + (Some(_) | None, Err(payload)) => BufferedOutcome::infrastructure(format!( + "worker panicked while running invocation: {}", + panic_description(payload.as_ref()) + )), + (None, Ok(())) => BufferedOutcome::infrastructure("worker exited without reporting an invocation outcome".to_owned()), }; - if !status.success() { - if !keep_going { - // Fail-fast: propagate the failing child's own exit code, - // reduced to the u8 `ExitCode` can carry (see `exit_byte`). - return Ok(ExitCode::from(exit_byte(status.code()))); + return IndexedOutcome { index, outcome }; + } +} + +fn panic_description(payload: &(dyn std::any::Any + Send)) -> &str { + if let Some(message) = payload.downcast_ref::<&str>() { + message + } else if let Some(message) = payload.downcast_ref::() { + message + } else { + "non-string panic payload" + } +} + +fn run_streamed(invocation: &Invocation) -> InvocationResult { + let (program, mut command) = match command_for(invocation) { + Ok(command) => command, + Err(message) => return InvocationResult::Infrastructure(message), + }; + match command.status() { + Ok(status) => InvocationResult::Exited(status), + Err(error) => InvocationResult::Infrastructure(format!("failed to spawn `{program}`: {error}")), + } +} + +fn run_streamed_with_timeout(invocation: &Invocation, timeout: Duration) -> InvocationResult { + let (program, command) = match command_for(invocation) { + Ok(command) => command, + Err(message) => return InvocationResult::Infrastructure(message), + }; + let mut tree = match spawn_tree(command) { + Ok(tree) => tree, + Err(error) => { + return InvocationResult::Infrastructure(format!("failed to spawn `{program}`: {error}")); + } + }; + wait_for_tree(&mut tree, timeout).result +} + +fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedOutcome { + #[cfg(test)] + assert!( + invocation.argv.first().is_none_or(|program| program != WORKER_PANIC_TEST_PROGRAM), + "injected worker panic" + ); + + let (program, mut command) = match command_for(invocation) { + Ok(command) => command, + Err(message) => return BufferedOutcome::infrastructure(message), + }; + let _ = command.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); + let mut tree = match spawn_tree(command) { + Ok(tree) => tree, + Err(error) => { + return BufferedOutcome::infrastructure(format!("failed to spawn `{program}`: {error}")); + } + }; + + let Some(stdout) = tree.take_stdout() else { + let cleanup = tree.terminate(); + return BufferedOutcome::infrastructure(with_cleanup_failure("failed to capture child stdout".to_owned(), &cleanup)); + }; + let stdout_reader = match spawn_output_reader(stdout, "cargo-each-stdout") { + Ok(reader) => reader, + Err(error) => { + let cleanup = tree.terminate(); + return BufferedOutcome::infrastructure(with_cleanup_failure(format!("failed to create stdout reader: {error}"), &cleanup)); + } + }; + let Some(stderr) = tree.take_stderr() else { + let cleanup = tree.terminate(); + return BufferedOutcome::from_reader_failure("failed to capture child stderr".to_owned(), stdout_reader, &cleanup); + }; + let stderr_reader = match spawn_output_reader(stderr, "cargo-each-stderr") { + Ok(reader) => reader, + Err(error) => { + let cleanup = tree.terminate(); + return BufferedOutcome::from_reader_failure(format!("failed to create stderr reader: {error}"), stdout_reader, &cleanup); + } + }; + + let tree_outcome = match timeout { + Some(timeout) => wait_for_tree(&mut tree, timeout), + None => wait_for_tree_without_timeout(&mut tree), + }; + let stdout = finish_output_reader(stdout_reader, "stdout", tree_outcome.cleanup_proven); + let stderr = finish_output_reader(stderr_reader, "stderr", tree_outcome.cleanup_proven); + let result = tree_outcome.result; + match (stdout, stderr) { + (Ok(stdout), Ok(stderr)) => BufferedOutcome { stdout, stderr, result }, + (Err(error), Ok(stderr)) => BufferedOutcome { + stdout: Vec::new(), + stderr, + result: InvocationResult::Infrastructure(error), + }, + (Ok(stdout), Err(error)) => BufferedOutcome { + stdout, + stderr: Vec::new(), + result: InvocationResult::Infrastructure(error), + }, + (Err(stdout), Err(stderr)) => BufferedOutcome { + stdout: Vec::new(), + stderr: Vec::new(), + result: InvocationResult::Infrastructure(format!("{stdout}; {stderr}")), + }, + } +} + +fn command_for(invocation: &Invocation) -> Result<(&str, Command), String> { + let Some((program, arguments)) = invocation.argv.split_first() else { + return Err("internal command-plan error: invocation has an empty argument vector".to_owned()); + }; + let mut command = Command::new(program); + if let Some(path) = std::env::var_os("PATH") { + command.env("PATH", path); + } + command.args(arguments); + if let Some(directory) = &invocation.work_dir { + command.current_dir(directory); + } + Ok((program, command)) +} + +fn spawn_tree(command: Command) -> Result { + let prepared = + prepare(command, MemoryRequest::default()).map_err(|error| format!("could not prepare process-tree containment: {error}"))?; + let spawned = prepared.spawn().map_err(|failure| failure.to_string())?; + ProcessTree::adopt(spawned).map_err(|error| format!("could not adopt child into process-tree containment: {error}")) +} + +fn wait_for_tree(tree: &mut ProcessTree, timeout: Duration) -> TreeOutcome { + let started = Instant::now(); + loop { + match tree.observe() { + Ok(Some(status)) => return TreeOutcome::closed(InvocationResult::Exited(status)), + Ok(None) => {} + Err(error) => { + let cleanup = tree.terminate(); + return match cleanup { + Ok(_) => TreeOutcome::closed(InvocationResult::Infrastructure(format!( + "failed to observe child process tree: {error}" + ))), + Err(cleanup) => TreeOutcome::unproven(InvocationResult::Infrastructure(format!( + "failed to observe child process tree: {error}; cleanup also failed: {cleanup}" + ))), + }; + } + } + let elapsed = started.elapsed(); + if elapsed >= timeout { + return match tree.terminate() { + Ok(_) => TreeOutcome::closed(InvocationResult::TimedOut(timeout)), + Err(error) => TreeOutcome::unproven(InvocationResult::Infrastructure(format!( + "invocation timed out after {}; process-tree termination failed: {error}", + display_duration(timeout) + ))), + }; + } + let remaining = timeout.saturating_sub(elapsed); + thread::sleep(remaining.min(Duration::from_millis(10))); + } +} + +fn wait_for_tree_without_timeout(tree: &mut ProcessTree) -> TreeOutcome { + loop { + match tree.observe() { + Ok(Some(status)) => return TreeOutcome::closed(InvocationResult::Exited(status)), + Ok(None) => thread::sleep(Duration::from_millis(10)), + Err(error) => { + let cleanup = tree.terminate(); + return match cleanup { + Ok(_) => TreeOutcome::closed(InvocationResult::Infrastructure(format!( + "failed to observe child process tree: {error}" + ))), + Err(cleanup) => TreeOutcome::unproven(InvocationResult::Infrastructure(format!( + "failed to observe child process tree: {error}; cleanup also failed: {cleanup}" + ))), + }; + } + } + } +} + +fn spawn_output_reader(mut stream: R, name: &'static str) -> io::Result +where + R: io::Read + Send + 'static, +{ + let bytes = Arc::new(Mutex::new(Vec::new())); + let retaining = Arc::new(AtomicBool::new(true)); + let captured = Arc::clone(&bytes); + let capture_enabled = Arc::clone(&retaining); + let thread = thread::Builder::new().name(name.to_owned()).spawn(move || { + let mut chunk = [0_u8; 8192]; + loop { + if !capture_enabled.load(Ordering::Acquire) { + return Ok(()); } - any_failed = true; + let read = stream.read(&mut chunk)?; + if read == 0 { + return Ok(()); + } + let mut output = captured + .lock() + .map_err(|error| io::Error::other(format!("child {name} capture buffer was poisoned: {error}")))?; + if !capture_enabled.load(Ordering::Acquire) { + return Ok(()); + } + output.extend_from_slice(&chunk[..read]); + } + })?; + Ok(OutputReader { thread, bytes, retaining }) +} + +fn finish_output_reader(reader: OutputReader, stream: &str, cleanup_proven: bool) -> Result, String> { + let OutputReader { thread, bytes, retaining } = reader; + if cleanup_proven { + match thread.join() { + Ok(Ok(())) => {} + Ok(Err(error)) => return Err(format!("failed to read child {stream}: {error}")), + Err(payload) => { + return Err(format!( + "child {stream} reader thread panicked: {}", + panic_description(payload.as_ref()) + )); + } + } + } else { + // A surviving descendant may keep the write end open forever. Stop + // retaining data, take the bytes already captured, and detach the + // blocked reader rather than defeating the invocation timeout. + retaining.store(false, Ordering::Release); + drop(thread); + } + let mut captured = bytes + .lock() + .map_err(|error| format!("child {stream} capture buffer was poisoned: {error}"))?; + Ok(std::mem::take(&mut *captured)) +} + +fn with_cleanup_failure(message: String, cleanup: &io::Result) -> String { + match cleanup { + Ok(_) => message, + Err(error) => format!("{message}; process-tree cleanup also failed: {error}"), + } +} + +fn emit_label(invocation: &Invocation) { + if let Some(label) = &invocation.label { + eprintln!("cargo each: {label}"); + } +} + +fn emit_buffered(invocation: &Invocation, outcome: &BufferedOutcome) -> io::Result<()> { + emit_label(invocation); + let mut stdout = io::stdout().lock(); + stdout.write_all(&outcome.stdout)?; + stdout.flush()?; + let mut stderr = io::stderr().lock(); + stderr.write_all(&outcome.stderr)?; + match &outcome.result { + InvocationResult::TimedOut(duration) => { + writeln!(stderr, "cargo each: invocation timed out after {}", display_duration(*duration))?; + } + InvocationResult::Infrastructure(message) => { + writeln!(stderr, "cargo each: {message}")?; + } + InvocationResult::Exited(_) => {} + } + stderr.flush() +} + +fn display_duration(duration: Duration) -> String { + if duration.subsec_nanos() == 0 && duration.as_secs().is_multiple_of(60) { + format!("{}m", duration.as_secs() / 60) + } else if duration.subsec_nanos() == 0 { + format!("{}s", duration.as_secs()) + } else { + format!("{}ms", duration.as_millis()) + } +} + +#[derive(Debug)] +struct IndexedOutcome { + index: usize, + outcome: BufferedOutcome, +} + +#[derive(Debug)] +struct RunningWorker { + index: usize, + receiver: mpsc::Receiver, + thread: thread::JoinHandle<()>, +} + +#[derive(Debug)] +struct OutputReader { + thread: thread::JoinHandle>, + bytes: Arc>>, + retaining: Arc, +} + +#[derive(Debug)] +struct TreeOutcome { + result: InvocationResult, + cleanup_proven: bool, +} + +impl TreeOutcome { + fn closed(result: InvocationResult) -> Self { + Self { + result, + cleanup_proven: true, + } + } + + fn unproven(result: InvocationResult) -> Self { + Self { + result, + cleanup_proven: false, + } + } +} + +#[derive(Debug)] +struct BufferedOutcome { + stdout: Vec, + stderr: Vec, + result: InvocationResult, +} + +impl BufferedOutcome { + fn infrastructure(message: String) -> Self { + Self { + stdout: Vec::new(), + stderr: Vec::new(), + result: InvocationResult::Infrastructure(message), + } + } + + fn from_reader_failure(message: String, stdout_reader: OutputReader, cleanup: &io::Result) -> Self { + let message = with_cleanup_failure(message, cleanup); + match finish_output_reader(stdout_reader, "stdout", cleanup.is_ok()) { + Ok(stdout) => Self { + stdout, + stderr: Vec::new(), + result: InvocationResult::Infrastructure(message), + }, + Err(reader_error) => Self { + stdout: Vec::new(), + stderr: Vec::new(), + result: InvocationResult::Infrastructure(format!("{message}; {reader_error}")), + }, + } + } +} + +#[derive(Debug)] +enum InvocationResult { + Exited(ExitStatus), + TimedOut(Duration), + Infrastructure(String), +} + +impl InvocationResult { + fn failed(&self) -> bool { + match self { + Self::Exited(status) => !status.success(), + Self::TimedOut(_) | Self::Infrastructure(_) => true, } } - // Under --keep-going the individual child codes may differ, so we cannot - // pick a single meaningful one; the documented contract is a flat `1` when - // any invocation failed. - Ok(if any_failed { ExitCode::from(1) } else { ExitCode::SUCCESS }) } /// Render an argv for display (`--dry-run`). Best-effort quoting for @@ -182,16 +682,6 @@ fn shell_join(argv: &[String]) -> String { } /// Reduce a raw process exit code to the `u8` that [`ExitCode`] can carry. -/// -/// `ExitCode` is a `u8`, but process exit codes are wider: `None` means the -/// child was terminated by a signal (Unix) and non-`None` codes are a full -/// `i32` on Windows. We reduce a code to its low byte, which is a closer -/// approximation of the child's code than collapsing everything to `1`. Two -/// cases still map to `1`: a signal-terminated child (no numeric code), and a -/// non-zero code whose low byte is `0` (e.g. `256`) — which would otherwise be -/// indistinguishable from success. This function is only called on the -/// fail-fast path, where the child has already failed, so `1` is always a -/// correct non-zero fallback. fn exit_byte(raw: Option) -> u8 { let Some(raw) = raw else { return 1 }; let byte = u8::try_from(raw & 0xFF).expect("`raw & 0xFF` masks to the low byte, always within 0..=255"); @@ -201,7 +691,45 @@ fn exit_byte(raw: Option) -> u8 { #[cfg(test)] #[cfg_attr(coverage_nightly, coverage(off))] mod tests { - use super::exit_byte; + use std::num::NonZeroUsize; + use std::process::ExitCode; + use std::sync::{Arc, Condvar, Mutex, mpsc}; + use std::time::Duration; + use std::{io, thread}; + + use super::{ + BufferedOutcome, Invocation, InvocationResult, Plan, RunningWorker, WORKER_PANIC_TEST_PROGRAM, display_duration, execute_parallel, + exit_byte, finish_output_reader, spawn_output_reader, wait_for_worker, + }; + + struct StubbornPipe { + read: bool, + blocked: Option>, + finished: mpsc::Sender<()>, + release: Arc<(Mutex, Condvar)>, + } + + impl io::Read for StubbornPipe { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + if !self.read { + self.read = true; + let content = b"captured-before-timeout"; + buf[..content.len()].copy_from_slice(content); + return Ok(content.len()); + } + + if let Some(blocked) = self.blocked.take() { + let _receiver_gone = blocked.send(()); + } + let (lock, condition) = &*self.release; + let mut released = lock.lock().expect("the test owns the release mutex without panicking"); + while !*released { + released = condition.wait(released).expect("the test owns the release mutex without panicking"); + } + let _receiver_gone = self.finished.send(()); + Ok(0) + } + } #[test] fn signal_terminated_child_maps_to_one() { @@ -217,15 +745,102 @@ mod tests { #[test] fn wide_codes_reduce_to_low_byte() { - // 259 = 0x103 -> low byte 3 (a common Windows code). assert_eq!(exit_byte(Some(259)), 3); assert_eq!(exit_byte(Some(257)), 1); } #[test] fn nonzero_code_with_zero_low_byte_maps_to_one() { - // 256 = 0x100 -> low byte 0, which would look like success; map to 1. assert_eq!(exit_byte(Some(256)), 1); assert_eq!(exit_byte(Some(512)), 1); } + + #[test] + fn durations_have_compact_diagnostics() { + assert_eq!(display_duration(Duration::from_millis(250)), "250ms"); + assert_eq!(display_duration(Duration::from_secs(30)), "30s"); + assert_eq!(display_duration(Duration::from_mins(2)), "2m"); + } + + #[test] + fn unproven_cleanup_does_not_join_a_stubborn_pipe_reader() { + let release = Arc::new((Mutex::new(false), Condvar::new())); + let (blocked_tx, blocked_rx) = mpsc::channel(); + let (reader_finished_tx, reader_finished_rx) = mpsc::channel(); + let reader = spawn_output_reader( + StubbornPipe { + read: false, + blocked: Some(blocked_tx), + finished: reader_finished_tx, + release: Arc::clone(&release), + }, + "stubborn-test-pipe", + ) + .expect("the test reader thread can be created"); + blocked_rx + .recv_timeout(Duration::from_secs(1)) + .expect("the reader reaches the simulated descendant's open pipe"); + + let (finished_tx, finished_rx) = mpsc::channel(); + let finisher = thread::spawn(move || { + let result = finish_output_reader(reader, "stdout", false); + let _receiver_gone = finished_tx.send(result); + }); + let result = finished_rx.recv_timeout(Duration::from_millis(500)); + + let (lock, condition) = &*release; + *lock.lock().expect("the test owns the release mutex without panicking") = true; + condition.notify_all(); + + let captured = result + .expect("unproven cleanup must not wait for a descendant to close its pipe") + .expect("capturing already-read output succeeds"); + finisher.join().expect("the bounded finisher thread does not panic"); + reader_finished_rx + .recv_timeout(Duration::from_secs(1)) + .expect("the detached reader exits after the test releases its simulated pipe"); + assert_eq!(captured, b"captured-before-timeout"); + } + + #[test] + fn worker_panic_becomes_an_outcome_without_deadlocking_the_scheduler() { + let plan = Plan { + invocations: vec![Invocation { + label: Some("panic-probe".to_owned()), + argv: vec![WORKER_PANIC_TEST_PROGRAM.to_owned()], + work_dir: None, + }], + }; + let (finished_tx, finished_rx) = mpsc::channel(); + let scheduler = thread::spawn(move || { + let result = execute_parallel(&plan, false, NonZeroUsize::new(2).expect("literal two is nonzero"), None); + let exit_two = result.is_ok_and(|code| code == ExitCode::from(2)); + let _receiver_gone = finished_tx.send(exit_two); + }); + + assert!( + finished_rx + .recv_timeout(Duration::from_secs(2)) + .expect("the panic-safe worker reports before the scheduler deadline"), + "a worker panic must become an infrastructure exit" + ); + scheduler.join().expect("the scheduler handles the worker panic without panicking"); + } + + #[test] + fn disconnected_worker_channel_becomes_an_infrastructure_outcome() { + let (sender, receiver) = mpsc::channel::(); + drop(sender); + let thread = thread::spawn(|| {}); + let outcome = wait_for_worker(&mut vec![RunningWorker { + index: 4, + receiver, + thread, + }]); + assert_eq!(outcome.index, 4); + let InvocationResult::Infrastructure(message) = outcome.outcome.result else { + panic!("a disconnected worker must produce an infrastructure outcome"); + }; + assert!(message.contains("without reporting")); + } } diff --git a/crates/cargo-each/src/select.rs b/crates/cargo-each/src/select.rs index d882e6f87..ce959579a 100644 --- a/crates/cargo-each/src/select.rs +++ b/crates/cargo-each/src/select.rs @@ -6,26 +6,22 @@ //! //! Mirrors `cargo build`'s selection surface: `-p`/`--package` (with glob //! support and optional `@version` qualifier), `--workspace`/`--all`, and -//! `--exclude`, plus the `cargo-each`-specific `--none` (explicit empty set). -//! When nothing is named the default is cargo's `default-members`, exactly -//! like `cargo build`. -//! -//! A computed selection (e.g. an impact tier) is fed in as ordinary flags via -//! shell expansion by the caller; this module has no notion of files or -//! environment variables. +//! `--exclude`, plus a repeatable `--package-file` and the +//! `cargo-each`-specific `--none` (explicit empty set). When nothing is named +//! the default is cargo's `default-members`, exactly like `cargo build`. use std::collections::HashSet; +use std::fs; +use std::path::{Path, PathBuf}; use cargo_metadata::semver::Version; -use crate::error::{EachError, UnknownSelectorError}; +use crate::error::{EachError, InvalidPackageFileLineError, PackageFileReadError, PackageFileUtf8Error, UnknownSelectorError}; use crate::workspace::{Member, Workspace}; /// A parsed package selection, before it is resolved against a workspace. /// -/// Populated from command-line flags. A caller with a computed selection -/// (e.g. an impact tier) passes it as ordinary `-p` / `--workspace` / `--none` -/// flags via shell expansion. +/// Populated from command-line flags and package files. #[derive(Debug, Default, Clone)] pub(crate) struct Selection { /// `-p` / `--package` selectors (name, `name@version`, or glob). @@ -36,9 +32,42 @@ pub(crate) struct Selection { pub(crate) exclude: Vec, /// `--none`: explicitly resolve to the empty set. pub(crate) none: bool, + /// Whether at least one `--package-file` was present, even if every file + /// was empty. + pub(crate) package_file_supplied: bool, } impl Selection { + /// Build a selection from direct package specs and package files. + /// + /// Package files are always read and validated, even when `--none` or + /// `--workspace` will win selection precedence, so a broken declared input + /// never silently passes. + /// + /// # Errors + /// + /// Returns [`EachError`] when a package file cannot be read as UTF-8 or + /// contains a malformed nonempty line. + pub(crate) fn from_sources( + packages: &[String], + package_files: &[PathBuf], + all: bool, + exclude: &[String], + none: bool, + ) -> Result { + let mut combined = packages.to_vec(); + for path in package_files { + combined.extend(read_package_file(path)?); + } + Ok(Self { + packages: combined, + all, + exclude: exclude.to_vec(), + none, + package_file_supplied: !package_files.is_empty(), + }) + } + /// Whether the resolved set is the whole workspace with no narrowing. /// /// True when selected via `--workspace` / `--all` with no narrowing @@ -73,6 +102,8 @@ impl Selection { workspace.members.iter().collect() } else if !self.packages.is_empty() { resolve_selectors(workspace, &self.packages)? + } else if self.package_file_supplied { + Vec::new() } else { workspace .members @@ -93,6 +124,53 @@ impl Selection { } } +fn read_package_file(path: &Path) -> Result, EachError> { + let display = path.display().to_string(); + let bytes = fs::read(path).map_err(|error| PackageFileReadError::caused_by(display.clone(), error))?; + let contents = String::from_utf8(bytes).map_err(|error| PackageFileUtf8Error::caused_by(display.clone(), error))?; + contents + .lines() + .enumerate() + .filter_map(|(index, line)| { + if line.is_empty() { + None + } else { + Some(validate_package_file_spec(&display, index + 1, line).map(str::to_owned)) + } + }) + .collect() +} + +fn validate_package_file_spec<'a>(path: &str, line: usize, spec: &'a str) -> Result<&'a str, EachError> { + let invalid = |reason: &str| InvalidPackageFileLineError::new(path.to_owned(), line, spec.to_owned(), reason.to_owned()).into(); + if spec.bytes().any(|byte| byte.is_ascii_whitespace()) { + return Err(invalid("leading, trailing, and embedded whitespace are not allowed")); + } + if spec.starts_with('#') { + return Err(invalid("comments are not supported")); + } + if spec.starts_with('-') { + return Err(invalid("command-line tokens are not package specs")); + } + let mut pieces = spec.split('@'); + let name = pieces.next().expect("split always yields at least one element"); + let version = pieces.next(); + if pieces.next().is_some() { + return Err(invalid("a package spec may contain at most one `@`")); + } + if name.is_empty() + || !name + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_' | b'*' | b'?')) + { + return Err(invalid("expected a package name or Unix glob, optionally followed by `@version`")); + } + if version.is_some_and(str::is_empty) { + return Err(invalid("the version qualifier after `@` must not be empty")); + } + Ok(spec) +} + /// Resolve a list of selectors against the workspace, deduplicating and /// preserving the workspace's member order. Each selector must match at /// least one member. @@ -250,7 +328,6 @@ fn glob_matches(pattern: &str, name: &str) -> bool { #[cfg_attr(coverage_nightly, coverage(off))] mod tests { use std::collections::BTreeSet; - use std::path::PathBuf; use serde_json::Value; @@ -260,6 +337,7 @@ mod tests { Member { name: name.to_owned(), version: "0.1.0".to_owned(), + rust_version: Some("1.70.0".parse().expect("valid Rust version")), manifest_path: PathBuf::from(format!("/ws/{name}/Cargo.toml")), publishable: true, features: BTreeSet::new(), @@ -273,6 +351,7 @@ mod tests { Workspace { members: vec![member("alpha"), member("beta"), member("gamma")], default_member_names: defaults.iter().map(|s| (*s).to_owned()).collect(), + root_manifest_path: PathBuf::from("/ws/Cargo.toml"), } } @@ -307,6 +386,26 @@ mod tests { assert_eq!(names(&sel.resolve(&ws).expect("resolve")), ["alpha", "gamma"]); } + #[test] + fn an_empty_package_file_source_selects_nothing() { + let ws = workspace(&["alpha", "gamma"]); + let sel = Selection { + package_file_supplied: true, + ..Selection::default() + }; + assert!(sel.resolve(&ws).expect("resolve").is_empty()); + } + + #[test] + fn package_file_lines_reject_comments_tokens_and_whitespace() { + for spec in ["# alpha", "--workspace", " alpha", "alpha ", "alpha beta", "alpha@", "alpha@1@2"] { + validate_package_file_spec("packages.txt", 1, spec).expect_err(spec); + } + for spec in ["alpha", "alpha@1.2.3", "cargo-*", "?eta"] { + assert_eq!(validate_package_file_spec("packages.txt", 1, spec).expect(spec), spec); + } + } + #[test] fn package_glob_and_version_spec_match_on_name() { let ws = workspace(&["alpha", "beta", "gamma"]); diff --git a/crates/cargo-each/src/substitute.rs b/crates/cargo-each/src/substitute.rs index fe7d4e5aa..fe2af3fcf 100644 --- a/crates/cargo-each/src/substitute.rs +++ b/crates/cargo-each/src/substitute.rs @@ -13,6 +13,7 @@ //! - The once token (valid only in `--once` mode): `{packages}`. Must stand //! alone as a whole argument; it expands to the resolved selection flags, //! which is several tokens. +//! - The workspace token `{workspace-rust-version}`, valid in every mode. //! //! Using a token in the wrong mode is a usage error ([`PlaceholderMisuseError`]). //! @@ -23,7 +24,7 @@ //! of the contract (`cargo-each` never interprets the command beyond these //! fixed substitutions), not an oversight. -use crate::error::{EachError, PlaceholderMisuseError}; +use crate::error::{EachError, PlaceholderMisuseError, WorkspaceRustVersionError}; use crate::plan::Mode; /// Per-package placeholder tokens. @@ -32,6 +33,8 @@ const PER_PACKAGE_TOKENS: [&str; 4] = ["{name}", "{spec}", "{version}", "{manife const TARGET_TOKEN: &str = "{target}"; /// The once-mode placeholder token. const PACKAGES_TOKEN: &str = "{packages}"; +/// The workspace-wide Rust compatibility floor token. +const WORKSPACE_RUST_VERSION_TOKEN: &str = "{workspace-rust-version}"; /// The substitution context for one command invocation. #[derive(Debug, Clone)] @@ -46,6 +49,8 @@ pub(crate) enum Placeholders { version: String, /// `{manifest}` — absolute path to the member's `Cargo.toml`. manifest: String, + /// The root workspace Rust-version declaration, when requested. + workspace_rust_version: Option, }, /// Per-target mode: package facts plus the selected target name. Target { @@ -54,15 +59,50 @@ pub(crate) enum Placeholders { version: String, manifest: String, target: String, + workspace_rust_version: Option, }, /// Once mode: `{packages}` expands to these pre-computed selection flags. Once { /// The cargo selection flags for the resolved set (e.g. /// `["--workspace"]` or `["--package", "a@1", "--package", "b@2"]`). packages: Vec, + /// The root workspace Rust-version declaration, when requested. + workspace_rust_version: Option, }, } +impl Placeholders { + fn workspace_rust_version(&self) -> Option<&str> { + match self { + Self::Package { + workspace_rust_version, .. + } + | Self::Target { + workspace_rust_version, .. + } + | Self::Once { + workspace_rust_version, .. + } => workspace_rust_version.as_deref(), + } + } +} + +fn replace_workspace_rust_version(arg: String, placeholders: &Placeholders) -> Result { + if !arg.contains(WORKSPACE_RUST_VERSION_TOKEN) { + return Ok(arg); + } + let version = placeholders + .workspace_rust_version() + .ok_or_else(|| WorkspaceRustVersionError::new("the command uses the placeholder but its root value was not resolved".to_owned()))?; + Ok(arg.replace(WORKSPACE_RUST_VERSION_TOKEN, version)) +} + +/// Whether a command template uses the lazy workspace Rust-version token. +#[must_use] +pub(crate) fn uses_workspace_rust_version(args: &[String]) -> bool { + args.iter().any(|arg| arg.contains(WORKSPACE_RUST_VERSION_TOKEN)) +} + /// Validate that `args` only reference placeholders valid for the mode. /// /// Checks mode-consistency without expanding the tokens — the check factored @@ -133,6 +173,7 @@ pub(crate) fn substitute(args: &[String], placeholders: &Placeholders) -> Result spec, version, manifest, + .. } => { // The `{name}` / `{spec}` / … literals are cargo-each // placeholder tokens, not Rust format-string arguments. @@ -145,7 +186,7 @@ pub(crate) fn substitute(args: &[String], placeholders: &Placeholders) -> Result .replace("{spec}", spec) .replace("{version}", version) .replace("{manifest}", manifest); - out.push(replaced); + out.push(replace_workspace_rust_version(replaced, placeholders)?); } Placeholders::Target { name, @@ -153,6 +194,7 @@ pub(crate) fn substitute(args: &[String], placeholders: &Placeholders) -> Result version, manifest, target, + .. } => { #[expect( clippy::literal_string_with_formatting_args, @@ -164,19 +206,20 @@ pub(crate) fn substitute(args: &[String], placeholders: &Placeholders) -> Result .replace("{version}", version) .replace("{manifest}", manifest) .replace(TARGET_TOKEN, target); - out.push(replaced); + out.push(replace_workspace_rust_version(replaced, placeholders)?); } - Placeholders::Once { packages } => { + Placeholders::Once { packages, .. } => { // Validation above guarantees each arg is either exactly // `{packages}` or contains no placeholder token at all. if arg == PACKAGES_TOKEN { out.extend(packages.iter().cloned()); } else { - out.push(arg.clone()); + out.push(replace_workspace_rust_version(arg.clone(), placeholders)?); } } } } + Ok(out) } @@ -191,6 +234,7 @@ mod tests { spec: "cargo-anvil@0.4.0".to_owned(), version: "0.4.0".to_owned(), manifest: "/ws/cargo-anvil/Cargo.toml".to_owned(), + workspace_rust_version: None, } } @@ -220,6 +264,7 @@ mod tests { fn once_expands_packages_token() { let ph = Placeholders::Once { packages: args(&["--package", "a@1", "--package", "b@2"]), + workspace_rust_version: None, }; let out = substitute(&args(&["clippy", "{packages}", "--all-targets"]), &ph).expect("substitute"); assert_eq!(out, ["clippy", "--package", "a@1", "--package", "b@2", "--all-targets"]); @@ -229,6 +274,7 @@ mod tests { fn once_rejects_per_package_token() { let ph = Placeholders::Once { packages: args(&["--workspace"]), + workspace_rust_version: None, }; let err = substitute(&args(&["test", "--package", "{name}"]), &ph).expect_err("misuse"); assert!(err.to_string().contains("{name}")); @@ -238,6 +284,7 @@ mod tests { fn once_rejects_target_token() { let ph = Placeholders::Once { packages: args(&["--workspace"]), + workspace_rust_version: None, }; let err = substitute(&args(&["test", "--test", "{target}"]), &ph).expect_err("misuse"); assert!(err.to_string().contains("{target}")); @@ -247,6 +294,7 @@ mod tests { fn once_rejects_embedded_packages_token() { let ph = Placeholders::Once { packages: args(&["--workspace"]), + workspace_rust_version: None, }; let err = substitute(&args(&["x={packages}"]), &ph).expect_err("misuse"); assert!(err.to_string().contains("stand alone")); @@ -260,6 +308,7 @@ mod tests { version: "0.4.0".to_owned(), manifest: "/ws/cargo-anvil/Cargo.toml".to_owned(), target: "loom".to_owned(), + workspace_rust_version: None, }; let out = substitute(&args(&["test", "-p", "{name}", "--test", "{target}"]), &ph).expect("substitute"); assert_eq!(out, ["test", "-p", "cargo-anvil", "--test", "loom"]); @@ -270,4 +319,36 @@ mod tests { let err = substitute(&args(&["echo", "{target}"]), &pkg()).expect_err("misuse"); assert!(err.to_string().contains("per-target")); } + + #[test] + fn workspace_rust_version_expands_in_every_mode() { + let command = args(&["rustup", "toolchain", "install", "{workspace-rust-version}"]); + let mut package = pkg(); + let Placeholders::Package { + workspace_rust_version, .. + } = &mut package + else { + unreachable!("pkg returns package placeholders"); + }; + *workspace_rust_version = Some("1.80".to_owned()); + assert_eq!( + substitute(&command, &package).expect("package substitution"), + ["rustup", "toolchain", "install", "1.80"] + ); + + let once = Placeholders::Once { + packages: args(&["--workspace"]), + workspace_rust_version: Some("1.80".to_owned()), + }; + assert_eq!( + substitute(&command, &once).expect("once substitution"), + ["rustup", "toolchain", "install", "1.80"] + ); + } + + #[test] + fn detects_workspace_rust_version_usage() { + assert!(uses_workspace_rust_version(&args(&["tool", "v={workspace-rust-version}"]))); + assert!(!uses_workspace_rust_version(&args(&["tool", "{name}"]))); + } } diff --git a/crates/cargo-each/src/workspace.rs b/crates/cargo-each/src/workspace.rs index 8091d9f58..6961960ea 100644 --- a/crates/cargo-each/src/workspace.rs +++ b/crates/cargo-each/src/workspace.rs @@ -11,10 +11,11 @@ use std::collections::{BTreeSet, HashSet}; use std::path::{Path, PathBuf}; +use cargo_metadata::semver::Version; use cargo_metadata::{MetadataCommand, TargetKind}; use serde_json::Value; -use crate::error::{EachError, LoadMetadataError}; +use crate::error::{EachError, LoadMetadataError, WorkspaceManifestParseError, WorkspaceManifestReadError, WorkspaceRustVersionError}; /// A resolved view of the cargo workspace `cargo-each` is operating on. #[derive(Debug, Clone)] @@ -25,6 +26,8 @@ pub(crate) struct Workspace { /// or every member when unset). Used to resolve a selection that names /// no packages. pub(crate) default_member_names: HashSet, + /// Absolute path to the workspace root manifest. + pub(crate) root_manifest_path: PathBuf, } /// A single workspace member and the facts selection/filtering key on. @@ -34,6 +37,8 @@ pub(crate) struct Member { pub(crate) name: String, /// Package version, rendered (e.g. `0.3.0`). pub(crate) version: String, + /// The member's resolved minimum supported Rust version. + pub(crate) rust_version: Option, /// Absolute path to this member's `Cargo.toml`. pub(crate) manifest_path: PathBuf, /// Whether Cargo permits publishing this package. @@ -123,6 +128,7 @@ impl Workspace { Member { name: pkg.name.to_string(), version: pkg.version.to_string(), + rust_version: pkg.rust_version.clone(), manifest_path: pkg.manifest_path.clone().into_std_path_buf(), publishable: pkg.publish.as_ref().is_none_or(|registries| !registries.is_empty()), features: pkg.features.keys().cloned().collect(), @@ -139,12 +145,107 @@ impl Workspace { .iter() .map(|pkg| pkg.name.to_string()) .collect(); + let root_manifest_path = metadata.workspace_root.join("Cargo.toml").into_std_path_buf(); Ok(Self { members, default_member_names, + root_manifest_path, }) } + + /// Resolve and validate the workspace-wide Rust compatibility floor. + /// + /// This deliberately reads the root manifest only when the corresponding + /// placeholder is used. Ordinary selection and execution therefore do not + /// require a workspace Rust-version declaration. + /// + /// # Errors + /// + /// Returns [`EachError`] if the root declaration is absent or invalid, or + /// if any workspace member omits `rust-version` or requires a newer + /// compiler than the root floor. + pub(crate) fn workspace_rust_version(&self) -> Result { + let path = self.root_manifest_path.display().to_string(); + let text = std::fs::read_to_string(&self.root_manifest_path) + .map_err(|error| WorkspaceManifestReadError::caused_by(path.clone(), error))?; + let manifest: toml::Value = toml::from_str(&text).map_err(|error| WorkspaceManifestParseError::caused_by(path, error))?; + + let workspace_floor = manifest + .get("workspace") + .and_then(|workspace| workspace.get("package")) + .and_then(|package| package.get("rust-version")); + let root_is_only_member = self.members.len() == 1 && self.members[0].manifest_path == self.root_manifest_path; + let package_floor = root_is_only_member + .then(|| manifest.get("package").and_then(|package| package.get("rust-version"))) + .flatten(); + let floor = workspace_floor.or(package_floor).ok_or_else(|| { + WorkspaceRustVersionError::new( + "the root manifest must declare `[workspace.package].rust-version`, or `[package].rust-version` for a single-package repository" + .to_owned(), + ) + })?; + let Some(floor) = floor.as_str() else { + return Err(WorkspaceRustVersionError::new("the root Rust version must be a string".to_owned()).into()); + }; + let parsed_floor = parse_rust_version(floor) + .map_err(|reason| WorkspaceRustVersionError::new(format!("root Rust version `{floor}` is invalid: {reason}")))?; + + for member in &self.members { + let Some(member_floor) = member.rust_version.as_ref() else { + return Err(WorkspaceRustVersionError::new(format!( + "workspace member `{}` does not expose a resolved `rust-version`", + member.name + )) + .into()); + }; + if member_floor.major != 1 || !member_floor.pre.is_empty() || !member_floor.build.is_empty() { + return Err(WorkspaceRustVersionError::new(format!( + "workspace member `{}` exposes invalid Rust version `{member_floor}`; expected a Rust 1.x toolchain version", + member.name + )) + .into()); + } + if member_floor > &parsed_floor { + return Err(WorkspaceRustVersionError::new(format!( + "workspace member `{}` requires Rust {}, newer than the root floor {floor}", + member.name, member_floor + )) + .into()); + } + } + + Ok(floor.to_owned()) + } +} + +fn parse_rust_version(value: &str) -> Result { + if value.contains('-') || value.contains('+') { + return Err("pre-release and build metadata are not valid Rust toolchain versions".to_owned()); + } + let components: Vec<&str> = value.split('.').collect(); + if !(2..=3).contains(&components.len()) + || components + .iter() + .any(|component| component.is_empty() || !component.bytes().all(|byte| byte.is_ascii_digit())) + { + return Err("expected `major.minor` or `major.minor.patch`".to_owned()); + } + if components.iter().any(|component| component.len() > 1 && component.starts_with('0')) { + return Err("numeric components must not contain leading zeroes".to_owned()); + } + let normalized = if components.len() == 2 { + format!("{value}.0") + } else { + value.to_owned() + }; + let parsed: Version = normalized + .parse() + .map_err(|error| format!("expected `major.minor` or `major.minor.patch`: {error}"))?; + if parsed.major != 1 { + return Err("expected a Rust 1.x toolchain version".to_owned()); + } + Ok(parsed) } /// Parse a supported Cargo target-kind spelling. @@ -184,4 +285,13 @@ mod tests { fn rejects_unknown_target_kind() { assert_eq!(parse_target_kind("future-kind"), None); } + + #[test] + fn parses_cargo_rust_version_forms() { + assert_eq!(parse_rust_version("1.80").expect("minor form"), Version::new(1, 80, 0)); + assert_eq!(parse_rust_version("1.80.1").expect("patch form"), Version::new(1, 80, 1)); + for value in ["1", "1.80.0-beta", "1.80+build", "1.080", "2.0", "one.80"] { + assert!(parse_rust_version(value).is_err(), "{value}"); + } + } } diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index 2d4af518e..39098c7e6 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -111,6 +111,119 @@ fn each(manifest: &Path) -> Command { cmd } +fn rust_version_fixture(root_floor: Option<&str>, members: &[(&str, Option<&str>)]) -> (TempDir, PathBuf) { + let tmp = tempfile::tempdir().expect("tempdir"); + let root = tmp.path(); + let member_names = members.iter().map(|(name, _)| format!("\"{name}\"")).collect::>().join(", "); + let workspace_package = root_floor.map_or_else(String::new, |floor| format!("\n[workspace.package]\nrust-version = \"{floor}\"\n")); + fs::write( + root.join("Cargo.toml"), + format!("[workspace]\nresolver = \"2\"\nmembers = [{member_names}]\n{workspace_package}"), + ) + .expect("write workspace root"); + for (name, rust_version) in members { + let declaration = match rust_version { + Some("workspace") => "rust-version.workspace = true\n".to_owned(), + Some(version) => format!("rust-version = \"{version}\"\n"), + None => String::new(), + }; + write_lib(root, name, "0.1.0", &declaration); + } + let manifest = root.join("Cargo.toml"); + (tmp, manifest) +} + +fn single_package_fixture(rust_version: &str) -> (TempDir, PathBuf) { + let tmp = tempfile::tempdir().expect("tempdir"); + let root = tmp.path(); + fs::create_dir_all(root.join("src")).expect("mkdir src"); + fs::write( + root.join("Cargo.toml"), + format!("[package]\nname = \"single\"\nversion = \"0.1.0\"\nedition = \"2021\"\nrust-version = \"{rust_version}\"\n"), + ) + .expect("write package manifest"); + fs::write(root.join("src/lib.rs"), "// fixture\n").expect("write lib"); + let manifest = root.join("Cargo.toml"); + (tmp, manifest) +} + +fn compile_execution_probe(directory: &Path) -> PathBuf { + let source = directory.join("execution-probe.rs"); + let executable = directory.join(format!("execution-probe{}", std::env::consts::EXE_SUFFIX)); + fs::write( + &source, + r#" +use std::env; +use std::fs::{self, OpenOptions}; +use std::io::Write as _; +use std::process::{self, Command}; +use std::thread; +use std::time::Duration; + +fn append(path: &str, value: &str) { + let mut file = OpenOptions::new().create(true).append(true).open(path).expect("open log"); + writeln!(file, "{value}").expect("append log"); +} + +fn main() { + let args: Vec = env::args().collect(); + match args[1].as_str() { + "ordered" => { + let name = &args[2]; + println!("{name}:start"); + thread::sleep(Duration::from_millis(if name == "alpha" { 250 } else { 20 })); + println!("{name}:end"); + append(&args[3], name); + } + "fail-order" => { + let name = &args[2]; + thread::sleep(Duration::from_millis(if name == "alpha" { 180 } else { 20 })); + process::exit(if name == "alpha" { 7 } else { 9 }); + } + "fail-stop" => { + let name = &args[2]; + append(&args[3], name); + thread::sleep(Duration::from_millis(if name == "alpha" { 40 } else { 250 })); + process::exit(if name == "alpha" { 7 } else { 0 }); + } + "keep-going" => { + let name = &args[2]; + append(&args[3], name); + process::exit(if name == "alpha" { 7 } else { 0 }); + } + "tree-parent" => { + let marker = &args[2]; + Command::new(env::current_exe().expect("current exe")) + .arg("tree-child") + .arg(marker) + .spawn() + .expect("spawn tree child"); + thread::sleep(Duration::from_secs(5)); + } + "tree-child" => { + thread::sleep(Duration::from_millis(500)); + fs::write(&args[2], "survived").expect("write marker"); + } + other => panic!("unknown probe mode: {other}"), + } +} +"#, + ) + .expect("write execution probe"); + let output = std::process::Command::new("rustc") + .arg(&source) + .arg("-o") + .arg(&executable) + .output() + .expect("rustc must be available to compile the execution probe"); + assert!( + output.status.success(), + "failed to compile execution probe:\n{}", + String::from_utf8_lossy(&output.stderr) + ); + executable +} + #[cfg(windows)] fn compile_probe(source: &Path, executable: &Path, marker: &str) { fs::write(source, format!("fn main() {{ println!(\"{marker}\"); }}\n")).expect("write probe source"); @@ -187,6 +300,124 @@ fn per_package_runs_once_per_selected_member() { .stdout(predicate::str::contains("echo alpha").and(predicate::str::contains("echo gamma"))); } +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn package_files_union_with_direct_packages_and_each_other() { + let (tmp, manifest) = fixture(); + let first = tmp.path().join("first.packages"); + let second = tmp.path().join("second.packages"); + fs::write(&first, "alpha\n\ngamma@0.1\n").expect("write first package file"); + fs::write(&second, "beta\n").expect("write second package file"); + + each(&manifest) + .arg("--package-file") + .arg(&first) + .arg("--package-file") + .arg(&second) + .args(["--package", "delta", "--dry-run", "--", "echo", "{name}"]) + .assert() + .success() + .stdout( + predicate::str::contains("echo alpha") + .and(predicate::str::contains("echo beta")) + .and(predicate::str::contains("echo delta")) + .and(predicate::str::contains("echo gamma")), + ); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn present_empty_package_file_is_an_explicit_empty_selection() { + let (tmp, manifest) = default_members_fixture(); + let packages = tmp.path().join("empty.packages"); + fs::write(&packages, "").expect("write empty package file"); + + each(&manifest) + .arg("--package-file") + .arg(packages) + .args(["--dry-run", "--", "echo", "{name}"]) + .assert() + .success() + .stdout(predicate::str::is_empty()) + .stderr(predicate::str::contains("nothing to do")); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn package_file_precedence_matches_the_selection_contract() { + let (tmp, manifest) = fixture(); + let packages = tmp.path().join("alpha.packages"); + fs::write(&packages, "alpha\n").expect("write package file"); + + each(&manifest) + .arg("--package-file") + .arg(&packages) + .args(["--workspace", "--dry-run", "--", "echo", "{name}"]) + .assert() + .success() + .stdout( + predicate::str::contains("echo alpha") + .and(predicate::str::contains("echo beta")) + .and(predicate::str::contains("echo gamma")), + ); + + each(&manifest) + .arg("--package-file") + .arg(packages) + .args(["--none", "--dry-run", "--", "echo", "{name}"]) + .assert() + .success() + .stdout(predicate::str::is_empty()) + .stderr(predicate::str::contains("nothing to do")); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn package_file_input_errors_fail_loudly() { + let (tmp, manifest) = fixture(); + let invalid_utf8 = tmp.path().join("invalid-utf8.packages"); + fs::write(&invalid_utf8, [0xFF, 0xFE]).expect("write invalid UTF-8"); + each(&manifest) + .arg("--package-file") + .arg(&invalid_utf8) + .args(["--dry-run", "--", "echo", "{name}"]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("not valid UTF-8")); + + let malformed = tmp.path().join("malformed.packages"); + fs::write(&malformed, "alpha\n--workspace\n").expect("write malformed package file"); + each(&manifest) + .arg("--package-file") + .arg(&malformed) + .args(["--dry-run", "--", "echo", "{name}"]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("line 2").and(predicate::str::contains("command-line tokens"))); + + let unmatched = tmp.path().join("unmatched.packages"); + fs::write(&unmatched, "does-not-exist\n").expect("write unmatched package file"); + each(&manifest) + .arg("--package-file") + .arg(&unmatched) + .args(["--dry-run", "--", "echo", "{name}"]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("did not match")); + + each(&manifest) + .arg("--package-file") + .arg(tmp.path().join("missing.packages")) + .args(["--dry-run", "--", "echo", "{name}"]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("could not read package file")); +} + #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] #[test] fn none_is_a_successful_noop() { @@ -865,3 +1096,228 @@ fn bare_invocation_runs_only_default_members() { .success() .stdout(predicate::str::contains("echo alpha").and(predicate::str::contains("echo beta").not())); } + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn workspace_rust_version_expands_in_all_modes_and_accepts_lower_members() { + let (_tmp, manifest) = rust_version_fixture(Some("1.80"), &[("alpha", Some("1.70")), ("beta", Some("workspace"))]); + each(&manifest) + .args(["--workspace", "--dry-run", "--", "echo", "{name}:{workspace-rust-version}"]) + .assert() + .success() + .stdout(predicate::str::contains("echo alpha:1.80").and(predicate::str::contains("echo beta:1.80"))); + + each(&manifest) + .args([ + "--workspace", + "--each-target", + "lib", + "--dry-run", + "--", + "echo", + "{target}:{workspace-rust-version}", + ]) + .assert() + .success() + .stdout(predicate::str::contains(":1.80")); + + each(&manifest) + .args([ + "--package", + "alpha", + "--once", + "--dry-run", + "--", + "echo", + "{workspace-rust-version}", + ]) + .assert() + .success() + .stdout(predicate::str::contains("echo 1.80")); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn workspace_rust_version_validation_is_lazy() { + let (_tmp, manifest) = rust_version_fixture(Some("1.80"), &[("alpha", Some("1.70")), ("beta", None)]); + each(&manifest) + .args(["--workspace", "--dry-run", "--", "echo", "{name}"]) + .assert() + .success(); + each(&manifest) + .args([ + "--package", + "alpha", + "--once", + "--dry-run", + "--", + "echo", + "{workspace-rust-version}", + ]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("beta").and(predicate::str::contains("rust-version"))); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn workspace_rust_version_rejects_newer_members() { + let (_tmp, manifest) = rust_version_fixture(Some("1.80"), &[("alpha", Some("1.81")), ("beta", Some("1.70"))]); + each(&manifest) + .args(["--workspace", "--once", "--dry-run", "--", "echo", "{workspace-rust-version}"]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("alpha").and(predicate::str::contains("newer than"))); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn workspace_rust_version_rejects_missing_or_invalid_root_floor() { + let (_missing, missing_manifest) = rust_version_fixture(None, &[("alpha", Some("1.70"))]); + each(&missing_manifest) + .args(["--workspace", "--once", "--dry-run", "--", "echo", "{workspace-rust-version}"]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("[workspace.package].rust-version")); + + let (_invalid, invalid_manifest) = rust_version_fixture(Some("2.0"), &[("alpha", Some("1.70"))]); + each(&invalid_manifest) + .args(["--workspace", "--dry-run", "--", "echo", "{name}"]) + .assert() + .success(); + each(&invalid_manifest) + .args(["--workspace", "--once", "--dry-run", "--", "echo", "{workspace-rust-version}"]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("2.0").and(predicate::str::contains("Rust 1.x"))); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn workspace_rust_version_uses_root_package_for_single_package_repository() { + let (_tmp, manifest) = single_package_fixture("1.75"); + each(&manifest) + .args(["--once", "--dry-run", "--", "echo", "{workspace-rust-version}"]) + .assert() + .success() + .stdout(predicate::str::contains("echo 1.75")); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn once_rejects_jobs_greater_than_one() { + let (_tmp, manifest) = fixture(); + each(&manifest) + .args(["--workspace", "--once", "--jobs", "2", "--dry-run", "--", "echo", "{packages}"]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("--jobs").and(predicate::str::contains("--once"))); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn parallel_output_is_buffered_in_plan_order() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let completion_log = tmp.path().join("completion.log"); + let output = each(&manifest) + .args(["-p", "alpha", "-p", "beta", "--jobs", "2", "--"]) + .arg(&probe) + .args(["ordered", "{name}"]) + .arg(&completion_log) + .output() + .expect("run cargo-each"); + assert!(output.status.success(), "stderr: {}", String::from_utf8_lossy(&output.stderr)); + let stdout = String::from_utf8(output.stdout).expect("UTF-8 probe output"); + let alpha = stdout.find("alpha:start").expect("alpha output"); + let beta = stdout.find("beta:start").expect("beta output"); + assert!(alpha < beta, "buffered blocks must follow plan order:\n{stdout}"); + assert_eq!( + fs::read_to_string(completion_log).expect("completion log"), + "beta\nalpha\n", + "the probe must finish out of order to prove cargo-each reordered complete blocks" + ); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn parallel_fail_fast_chooses_failure_by_plan_order() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + each(&manifest) + .args(["-p", "alpha", "-p", "beta", "--jobs", "2", "--"]) + .arg(probe) + .args(["fail-order", "{name}"]) + .assert() + .failure() + .code(7); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn parallel_fail_fast_stops_launching_new_work() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let launch_log = tmp.path().join("launch.log"); + each(&manifest) + .args(["--workspace", "--jobs", "2", "--"]) + .arg(probe) + .args(["fail-stop", "{name}"]) + .arg(&launch_log) + .assert() + .failure() + .code(7); + let launched = fs::read_to_string(launch_log).expect("launch log"); + assert!(launched.contains("alpha\n"), "{launched}"); + assert!(launched.contains("beta\n"), "{launched}"); + for name in ["delta", "epsilon", "gamma"] { + assert!(!launched.contains(name), "{name} must not launch after failure:\n{launched}"); + } +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn parallel_keep_going_runs_the_complete_plan() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let launch_log = tmp.path().join("launch.log"); + each(&manifest) + .args(["--workspace", "--jobs", "2", "--keep-going", "--"]) + .arg(probe) + .args(["keep-going", "{name}"]) + .arg(&launch_log) + .assert() + .failure() + .code(1); + let launched = fs::read_to_string(launch_log).expect("launch log"); + for name in ["alpha", "beta", "delta", "epsilon", "gamma"] { + assert!(launched.contains(name), "{name} must run under --keep-going:\n{launched}"); + } +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn timeout_terminates_the_complete_process_tree() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let marker = tmp.path().join("grandchild-survived"); + each(&manifest) + .args(["-p", "alpha", "--jobs", "2", "--timeout", "50ms", "--"]) + .arg(probe) + .arg("tree-parent") + .arg(&marker) + .assert() + .failure() + .code(1) + .stderr(predicate::str::contains("timed out after 50ms")); + std::thread::sleep(std::time::Duration::from_millis(700)); + assert!( + !marker.exists(), + "a timed-out invocation's grandchild must not survive to write its marker" + ); +} From e7ff16cdf655f5fde6183771a071525fdb29cc12 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Fri, 11 Sep 2026 23:02:00 +0200 Subject: [PATCH 02/37] test(cargo-each): harden portable execution gates Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/cli.rs | 31 +- crates/cargo-each/src/run.rs | 628 ++++++++++++++++++++++++++--- crates/cargo-each/src/select.rs | 22 +- crates/cargo-each/src/workspace.rs | 154 ++++++- crates/cargo-each/tests/cli.rs | 114 +++++- 5 files changed, 856 insertions(+), 93 deletions(-) diff --git a/crates/cargo-each/src/cli.rs b/crates/cargo-each/src/cli.rs index 300d0fe5a..747135aab 100644 --- a/crates/cargo-each/src/cli.rs +++ b/crates/cargo-each/src/cli.rs @@ -118,16 +118,25 @@ pub(crate) struct EachArgs { } fn parse_duration(value: &str) -> Result { + enum Unit { + Milliseconds, + Seconds, + Minutes, + } + let (digits, unit) = if let Some(digits) = value.strip_suffix("ms") { - (digits, "ms") + (digits, Unit::Milliseconds) } else if let Some(digits) = value.strip_suffix('s') { - (digits, "s") + (digits, Unit::Seconds) } else if let Some(digits) = value.strip_suffix('m') { - (digits, "m") + (digits, Unit::Minutes) } else { return Err("expected a positive integer followed by `ms`, `s`, or `m`".to_owned()); }; - if digits.is_empty() || !digits.bytes().all(|byte| byte.is_ascii_digit()) { + if digits.is_empty() { + return Err("expected a positive integer followed by `ms`, `s`, or `m`".to_owned()); + } + if !digits.bytes().all(|byte| byte.is_ascii_digit()) { return Err("expected a positive integer followed by `ms`, `s`, or `m`".to_owned()); } let amount = digits.parse::().map_err(|error| format!("duration is too large: {error}"))?; @@ -135,13 +144,12 @@ fn parse_duration(value: &str) -> Result { return Err("duration must be greater than zero".to_owned()); } match unit { - "ms" => Ok(Duration::from_millis(amount)), - "s" => Ok(Duration::from_secs(amount)), - "m" => amount + Unit::Milliseconds => Ok(Duration::from_millis(amount)), + Unit::Seconds => Ok(Duration::from_secs(amount)), + Unit::Minutes => amount .checked_mul(60) .map(Duration::from_secs) .ok_or_else(|| "duration is too large".to_owned()), - _ => unreachable!("unit is selected from the three cases above"), } } @@ -170,4 +178,11 @@ mod tests { assert!(parse_duration(value).is_err(), "{value}"); } } + + #[test] + fn malformed_duration_uses_the_grammar_diagnostic() { + let expected = Err("expected a positive integer followed by `ms`, `s`, or `m`".to_owned()); + assert_eq!(parse_duration("ms"), expected); + assert_eq!(parse_duration("1.5s"), expected); + } } diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 5426796e6..b6bfafa83 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -4,7 +4,7 @@ //! Implementation of the `cargo each` command: resolve the selection, //! apply filters, build the plan, and run it. -use std::collections::BTreeSet; +use std::collections::{BTreeSet, VecDeque}; use std::io::{self, Write as _}; use std::num::NonZeroUsize; use std::panic::{self, UnwindSafe}; @@ -28,6 +28,8 @@ use crate::workspace::{Member, Workspace}; #[cfg(test)] const WORKER_PANIC_TEST_PROGRAM: &str = "__cargo_each_injected_worker_panic"; +#[cfg(test)] +const WORKER_SPAWN_ERROR_TEST_PROGRAM: &str = "__cargo_each_injected_worker_spawn_error"; pub(crate) fn run(args: &EachArgs) -> Result { let selection = build_selection(args).into_app_err("failed to read package selection")?; @@ -183,41 +185,44 @@ fn execute_sequential(plan: &Plan, keep_going: bool, timeout: Option) } fn execute_parallel(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: Option) -> Result { - let invocations = Arc::new(plan.invocations.clone()); + let invocations = plan.invocations.clone(); let worker_count = jobs.get().min(invocations.len()).min(cargo_gamma_process::capacity().max(1)); + let mut pending: VecDeque<(usize, Invocation)> = invocations.iter().cloned().enumerate().collect(); let mut workers = Vec::with_capacity(worker_count); - let mut next_index = 0; let mut stop_launching = false; let mut launch_error = None; - while next_index < worker_count { - match spawn_worker(next_index, Arc::clone(&invocations), timeout) { + for (index, invocation) in pending.drain(..worker_count) { + match spawn_worker(index, invocation, timeout) { Ok(worker) => { workers.push(worker); - next_index += 1; } Err(error) => { launch_error = Some(error); + stop_launching = true; break; } } } let mut outcomes = Vec::with_capacity(invocations.len()); - while !workers.is_empty() { - let outcome = wait_for_worker(&mut workers); - if !keep_going && outcome.outcome.result.failed() { + while let Some(outcome) = wait_for_worker(&mut workers) { + if failure_stops_launching(keep_going, outcome.outcome.result.failed()) { stop_launching = true; } outcomes.push(outcome); - if !stop_launching && launch_error.is_none() && next_index < invocations.len() { - match spawn_worker(next_index, Arc::clone(&invocations), timeout) { - Ok(worker) => { - workers.push(worker); - next_index += 1; - } - Err(error) => launch_error = Some(error), + if stop_launching { + continue; + } + let Some((index, invocation)) = pending.pop_front() else { + continue; + }; + match spawn_worker(index, invocation, timeout) { + Ok(worker) => workers.push(worker), + Err(error) => { + launch_error = Some(error); + stop_launching = true; } } } @@ -244,17 +249,23 @@ fn execute_parallel(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: }) } -fn spawn_worker(index: usize, invocations: Arc>, timeout: Option) -> io::Result { +fn failure_stops_launching(keep_going: bool, failed: bool) -> bool { + matches!((keep_going, failed), (false, true)) +} + +fn spawn_worker(index: usize, invocation: Invocation, timeout: Option) -> io::Result { + #[cfg(test)] + if invocation + .argv + .first() + .is_some_and(|program| program == WORKER_SPAWN_ERROR_TEST_PROGRAM) + { + return Err(io::Error::other("injected worker spawn failure")); + } + let (sender, receiver) = mpsc::channel(); let thread = thread::Builder::new().name(format!("cargo-each-worker-{index}")).spawn(move || { - complete_worker(&sender, move || { - let Some(invocation) = invocations.get(index) else { - return BufferedOutcome::infrastructure(format!( - "internal scheduler error: invocation index {index} is outside the command plan" - )); - }; - run_captured(invocation, timeout) - }); + complete_worker(&sender, move || run_captured(&invocation, timeout)); })?; Ok(RunningWorker { index, receiver, thread }) } @@ -270,7 +281,10 @@ fn complete_worker(sender: &mpsc::Sender, work: impl FnOnce() - let _receiver_gone = sender.send(outcome); } -fn wait_for_worker(workers: &mut Vec) -> IndexedOutcome { +fn wait_for_worker(workers: &mut Vec) -> Option { + if workers.is_empty() { + return None; + } loop { let ready = workers .iter() @@ -294,7 +308,7 @@ fn wait_for_worker(workers: &mut Vec) -> IndexedOutcome { )), (None, Ok(())) => BufferedOutcome::infrastructure("worker exited without reporting an invocation outcome".to_owned()), }; - return IndexedOutcome { index, outcome }; + return Some(IndexedOutcome { index, outcome }); } } @@ -351,23 +365,42 @@ fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedO return BufferedOutcome::infrastructure(format!("failed to spawn `{program}`: {error}")); } }; + let capture_fault = capture_fault(invocation); - let Some(stdout) = tree.take_stdout() else { + let stdout = if capture_fault == Some(CaptureFault::MissingStdout) { + None + } else { + tree.take_stdout() + }; + let Some(stdout) = stdout else { let cleanup = tree.terminate(); return BufferedOutcome::infrastructure(with_cleanup_failure("failed to capture child stdout".to_owned(), &cleanup)); }; - let stdout_reader = match spawn_output_reader(stdout, "cargo-each-stdout") { + let stdout_reader = match if capture_fault == Some(CaptureFault::StdoutReader) { + Err(io::Error::other("injected stdout reader failure")) + } else { + spawn_output_reader(stdout, "cargo-each-stdout") + } { Ok(reader) => reader, Err(error) => { let cleanup = tree.terminate(); return BufferedOutcome::infrastructure(with_cleanup_failure(format!("failed to create stdout reader: {error}"), &cleanup)); } }; - let Some(stderr) = tree.take_stderr() else { + let stderr = if capture_fault == Some(CaptureFault::MissingStderr) { + None + } else { + tree.take_stderr() + }; + let Some(stderr) = stderr else { let cleanup = tree.terminate(); return BufferedOutcome::from_reader_failure("failed to capture child stderr".to_owned(), stdout_reader, &cleanup); }; - let stderr_reader = match spawn_output_reader(stderr, "cargo-each-stderr") { + let stderr_reader = match if capture_fault == Some(CaptureFault::StderrReader) { + Err(io::Error::other("injected stderr reader failure")) + } else { + spawn_output_reader(stderr, "cargo-each-stderr") + } { Ok(reader) => reader, Err(error) => { let cleanup = tree.terminate(); @@ -381,7 +414,10 @@ fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedO }; let stdout = finish_output_reader(stdout_reader, "stdout", tree_outcome.cleanup_proven); let stderr = finish_output_reader(stderr_reader, "stderr", tree_outcome.cleanup_proven); - let result = tree_outcome.result; + combine_captured_output(stdout, stderr, tree_outcome.result) +} + +fn combine_captured_output(stdout: Result, String>, stderr: Result, String>, result: InvocationResult) -> BufferedOutcome { match (stdout, stderr) { (Ok(stdout), Ok(stderr)) => BufferedOutcome { stdout, stderr, result }, (Err(error), Ok(stderr)) => BufferedOutcome { @@ -425,13 +461,22 @@ fn spawn_tree(command: Command) -> Result { } fn wait_for_tree(tree: &mut ProcessTree, timeout: Duration) -> TreeOutcome { + wait_for_tree_with(tree, timeout, ProcessTree::observe, ProcessTree::terminate) +} + +fn wait_for_tree_with( + control: &mut T, + timeout: Duration, + mut observe: impl FnMut(&mut T) -> io::Result>, + mut terminate: impl FnMut(&mut T) -> io::Result, +) -> TreeOutcome { let started = Instant::now(); loop { - match tree.observe() { + match observe(control) { Ok(Some(status)) => return TreeOutcome::closed(InvocationResult::Exited(status)), Ok(None) => {} Err(error) => { - let cleanup = tree.terminate(); + let cleanup = terminate(control); return match cleanup { Ok(_) => TreeOutcome::closed(InvocationResult::Infrastructure(format!( "failed to observe child process tree: {error}" @@ -442,28 +487,34 @@ fn wait_for_tree(tree: &mut ProcessTree, timeout: Duration) -> TreeOutcome { }; } } - let elapsed = started.elapsed(); - if elapsed >= timeout { - return match tree.terminate() { + let Some(remaining) = timeout.checked_sub(started.elapsed()) else { + return match terminate(control) { Ok(_) => TreeOutcome::closed(InvocationResult::TimedOut(timeout)), Err(error) => TreeOutcome::unproven(InvocationResult::Infrastructure(format!( "invocation timed out after {}; process-tree termination failed: {error}", display_duration(timeout) ))), }; - } - let remaining = timeout.saturating_sub(elapsed); + }; thread::sleep(remaining.min(Duration::from_millis(10))); } } fn wait_for_tree_without_timeout(tree: &mut ProcessTree) -> TreeOutcome { + wait_for_tree_without_timeout_with(tree, ProcessTree::observe, ProcessTree::terminate) +} + +fn wait_for_tree_without_timeout_with( + control: &mut T, + mut observe: impl FnMut(&mut T) -> io::Result>, + mut terminate: impl FnMut(&mut T) -> io::Result, +) -> TreeOutcome { loop { - match tree.observe() { + match observe(control) { Ok(Some(status)) => return TreeOutcome::closed(InvocationResult::Exited(status)), Ok(None) => thread::sleep(Duration::from_millis(10)), Err(error) => { - let cleanup = tree.terminate(); + let cleanup = terminate(control); return match cleanup { Ok(_) => TreeOutcome::closed(InvocationResult::Infrastructure(format!( "failed to observe child process tree: {error}" @@ -477,6 +528,32 @@ fn wait_for_tree_without_timeout(tree: &mut ProcessTree) -> TreeOutcome { } } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum CaptureFault { + MissingStdout, + StdoutReader, + MissingStderr, + StderrReader, +} + +fn capture_fault(invocation: &Invocation) -> Option { + #[cfg(test)] + { + match invocation.label.as_deref() { + Some("__cargo_each_missing_stdout") => Some(CaptureFault::MissingStdout), + Some("__cargo_each_stdout_reader_failure") => Some(CaptureFault::StdoutReader), + Some("__cargo_each_missing_stderr") => Some(CaptureFault::MissingStderr), + Some("__cargo_each_stderr_reader_failure") => Some(CaptureFault::StderrReader), + _ => None, + } + } + #[cfg(not(test))] + { + let _ = invocation; + None + } +} + fn spawn_output_reader(mut stream: R, name: &'static str) -> io::Result where R: io::Read + Send + 'static, @@ -491,10 +568,10 @@ where if !capture_enabled.load(Ordering::Acquire) { return Ok(()); } - let read = stream.read(&mut chunk)?; - if read == 0 { - return Ok(()); - } + let read = match stream.read(&mut chunk)? { + 0 => return Ok(()), + read => read, + }; let mut output = captured .lock() .map_err(|error| io::Error::other(format!("child {name} capture buffer was poisoned: {error}")))?; @@ -533,7 +610,7 @@ fn finish_output_reader(reader: OutputReader, stream: &str, cleanup_proven: bool Ok(std::mem::take(&mut *captured)) } -fn with_cleanup_failure(message: String, cleanup: &io::Result) -> String { +fn with_cleanup_failure(message: String, cleanup: &io::Result) -> String { match cleanup { Ok(_) => message, Err(error) => format!("{message}; process-tree cleanup also failed: {error}"), @@ -633,7 +710,7 @@ impl BufferedOutcome { } } - fn from_reader_failure(message: String, stdout_reader: OutputReader, cleanup: &io::Result) -> Self { + fn from_reader_failure(message: String, stdout_reader: OutputReader, cleanup: &io::Result) -> Self { let message = with_cleanup_failure(message, cleanup); match finish_output_reader(stdout_reader, "stdout", cleanup.is_ok()) { Ok(stdout) => Self { @@ -691,15 +768,18 @@ fn exit_byte(raw: Option) -> u8 { #[cfg(test)] #[cfg_attr(coverage_nightly, coverage(off))] mod tests { + use std::collections::VecDeque; use std::num::NonZeroUsize; - use std::process::ExitCode; + use std::process::{Command, ExitCode, ExitStatus, Stdio}; use std::sync::{Arc, Condvar, Mutex, mpsc}; - use std::time::Duration; + use std::time::{Duration, Instant}; use std::{io, thread}; use super::{ - BufferedOutcome, Invocation, InvocationResult, Plan, RunningWorker, WORKER_PANIC_TEST_PROGRAM, display_duration, execute_parallel, - exit_byte, finish_output_reader, spawn_output_reader, wait_for_worker, + BufferedOutcome, Invocation, InvocationResult, Plan, RunningWorker, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, + WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, execute_parallel, exit_byte, failure_stops_launching, + finish_output_reader, panic_description, run_captured, run_streamed, run_streamed_with_timeout, spawn_output_reader, spawn_tree, + wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, }; struct StubbornPipe { @@ -709,6 +789,114 @@ mod tests { release: Arc<(Mutex, Condvar)>, } + struct FailingReader; + + impl io::Read for FailingReader { + fn read(&mut self, _buf: &mut [u8]) -> io::Result { + Err(io::Error::other("injected read failure")) + } + } + + struct PanickingReader; + + impl io::Read for PanickingReader { + fn read(&mut self, _buf: &mut [u8]) -> io::Result { + panic!("injected reader panic"); + } + } + + struct EofThenPanicReader { + reached_eof: bool, + } + + impl io::Read for EofThenPanicReader { + fn read(&mut self, _buf: &mut [u8]) -> io::Result { + assert!(!self.reached_eof, "the output reader must stop after the first EOF"); + self.reached_eof = true; + Ok(0) + } + } + + struct LateDataPipe { + started: Option>, + finished: Option>, + release: Arc<(Mutex, Condvar)>, + } + + impl io::Read for LateDataPipe { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + if let Some(started) = self.started.take() { + let _receiver_gone = started.send(()); + } + let (lock, condition) = &*self.release; + let mut released = lock.lock().expect("the test owns the release mutex without panicking"); + while !*released { + released = condition.wait(released).expect("the test owns the release mutex without panicking"); + } + let content = b"late data"; + buf[..content.len()].copy_from_slice(content); + Ok(content.len()) + } + } + + impl Drop for LateDataPipe { + fn drop(&mut self) { + if let Some(finished) = self.finished.take() { + let _receiver_gone = finished.send(()); + } + } + } + + fn invocation(argv: &[&str]) -> Invocation { + Invocation { + label: None, + argv: argv.iter().map(|value| (*value).to_owned()).collect(), + work_dir: None, + } + } + + fn labelled_invocation(label: &str, argv: &[&str]) -> Invocation { + Invocation { + label: Some(label.to_owned()), + ..invocation(argv) + } + } + + fn result_infrastructure_message(result: InvocationResult) -> String { + let InvocationResult::Infrastructure(message) = result else { + panic!("the test expects an infrastructure outcome"); + }; + message + } + + struct FakeProcess { + observations: VecDeque>>, + termination: Option>, + } + + impl FakeProcess { + fn observe(&mut self) -> io::Result> { + self.observations.pop_front().unwrap_or(Ok(None)) + } + + fn terminate(&mut self) -> io::Result { + self.termination + .take() + .unwrap_or_else(|| Err(io::Error::other("unexpected termination"))) + } + } + + fn successful_status() -> ExitStatus { + Command::new("rustc") + .arg("--version") + .status() + .expect("rustc is available to the crate's test suite") + } + + fn infrastructure_message(outcome: BufferedOutcome) -> String { + result_infrastructure_message(outcome.result) + } + impl io::Read for StubbornPipe { fn read(&mut self, buf: &mut [u8]) -> io::Result { if !self.read { @@ -762,6 +950,14 @@ mod tests { assert_eq!(display_duration(Duration::from_mins(2)), "2m"); } + #[test] + fn failure_launch_policy_distinguishes_fail_fast_from_keep_going() { + assert!(failure_stops_launching(false, true)); + assert!(!failure_stops_launching(false, false)); + assert!(!failure_stops_launching(true, true)); + assert!(!failure_stops_launching(true, false)); + } + #[test] fn unproven_cleanup_does_not_join_a_stubborn_pipe_reader() { let release = Arc::new((Mutex::new(false), Condvar::new())); @@ -802,6 +998,35 @@ mod tests { assert_eq!(captured, b"captured-before-timeout"); } + #[test] + fn unproven_cleanup_discards_data_that_arrives_after_capture_stops() { + let release = Arc::new((Mutex::new(false), Condvar::new())); + let (started_tx, started_rx) = mpsc::channel(); + let (reader_finished_tx, reader_finished_rx) = mpsc::channel(); + let reader = spawn_output_reader( + LateDataPipe { + started: Some(started_tx), + finished: Some(reader_finished_tx), + release: Arc::clone(&release), + }, + "late-data-test-pipe", + ) + .expect("create late-data reader"); + started_rx + .recv_timeout(Duration::from_secs(1)) + .expect("the late-data reader starts its blocking read"); + + let captured = finish_output_reader(reader, "stdout", false).expect("stop capture without joining"); + assert!(captured.is_empty()); + + let (lock, condition) = &*release; + *lock.lock().expect("the test owns the release mutex without panicking") = true; + condition.notify_all(); + reader_finished_rx + .recv_timeout(Duration::from_secs(1)) + .expect("the detached reader discards late data and exits"); + } + #[test] fn worker_panic_becomes_an_outcome_without_deadlocking_the_scheduler() { let plan = Plan { @@ -836,11 +1061,310 @@ mod tests { index: 4, receiver, thread, - }]); + }]) + .expect("the disconnected worker is observable"); assert_eq!(outcome.index, 4); let InvocationResult::Infrastructure(message) = outcome.outcome.result else { panic!("a disconnected worker must produce an infrastructure outcome"); }; assert!(message.contains("without reporting")); } + + #[test] + fn worker_panic_descriptions_preserve_string_payloads() { + let borrowed: Box = Box::new("borrowed panic"); + let owned: Box = Box::new("owned panic".to_owned()); + let other: Box = Box::new(7_u8); + assert_eq!(panic_description(borrowed.as_ref()), "borrowed panic"); + assert_eq!(panic_description(owned.as_ref()), "owned panic"); + assert_eq!(panic_description(other.as_ref()), "non-string panic payload"); + } + + #[test] + fn cleanup_failure_context_preserves_both_outcomes() { + assert_eq!(with_cleanup_failure("primary".to_owned(), &Ok::<_, io::Error>(())), "primary"); + assert_eq!( + with_cleanup_failure("primary".to_owned(), &Err::<(), _>(io::Error::other("cleanup"))), + "primary; process-tree cleanup also failed: cleanup" + ); + } + + #[test] + fn panicked_worker_without_a_report_becomes_an_infrastructure_outcome() { + let (sender, receiver) = mpsc::channel::(); + drop(sender); + let thread = thread::spawn(|| panic!("panic before reporting")); + let outcome = wait_for_worker(&mut vec![RunningWorker { + index: 5, + receiver, + thread, + }]) + .expect("the panicked worker is observable"); + let InvocationResult::Infrastructure(message) = outcome.outcome.result else { + panic!("a panicked worker must produce an infrastructure outcome"); + }; + assert!(message.contains("panic before reporting")); + } + + #[test] + fn panic_after_a_worker_report_overrides_the_report() { + let (sender, receiver) = mpsc::channel(); + let thread = thread::spawn(move || { + sender + .send(BufferedOutcome::infrastructure("premature report".to_owned())) + .expect("the scheduler receiver remains alive"); + panic!("panic after reporting"); + }); + let outcome = wait_for_worker(&mut vec![RunningWorker { + index: 6, + receiver, + thread, + }]) + .expect("the reported worker is observable"); + let InvocationResult::Infrastructure(message) = outcome.outcome.result else { + panic!("a worker panic must override its premature report"); + }; + assert!(message.contains("panic after reporting")); + assert!(wait_for_worker(&mut Vec::new()).is_none()); + } + + #[test] + fn worker_spawn_failures_are_reported_during_initial_and_replacement_launches() { + let initial = Plan { + invocations: vec![invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM])], + }; + let error = execute_parallel(&initial, false, NonZeroUsize::new(2).expect("literal two is nonzero"), None) + .expect_err("an initial worker spawn failure must abort scheduling"); + assert!(error.to_string().contains("injected worker spawn failure")); + + let replacement = Plan { + invocations: vec![invocation(&["rustc", "--version"]), invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM])], + }; + let error = execute_parallel(&replacement, false, NonZeroUsize::new(1).expect("literal one is nonzero"), None) + .expect_err("a replacement worker spawn failure must abort scheduling"); + assert!(error.to_string().contains("injected worker spawn failure")); + } + + #[test] + fn direct_runners_report_empty_and_unspawnable_commands() { + let empty = invocation(&[]); + assert!(result_infrastructure_message(run_streamed(&empty)).contains("empty argument vector")); + assert!(result_infrastructure_message(run_streamed_with_timeout(&empty, Duration::from_secs(1))).contains("empty argument vector")); + assert!(infrastructure_message(run_captured(&empty, None)).contains("empty argument vector")); + + let missing = invocation(&["cargo-each-no-such-program-for-unit-test"]); + assert!(result_infrastructure_message(run_streamed(&missing)).contains("failed to spawn")); + assert!(result_infrastructure_message(run_streamed_with_timeout(&missing, Duration::from_secs(1))).contains("failed to spawn")); + assert!(infrastructure_message(run_captured(&missing, None)).contains("failed to spawn")); + } + + #[test] + fn captured_output_combines_every_reader_result_shape() { + let success = combine_captured_output( + Ok(b"stdout".to_vec()), + Ok(b"stderr".to_vec()), + InvocationResult::Infrastructure("primary".to_owned()), + ); + assert_eq!(success.stdout, b"stdout"); + assert_eq!(success.stderr, b"stderr"); + assert_eq!(infrastructure_message(success), "primary"); + + let stdout_failed = combine_captured_output( + Err("stdout failed".to_owned()), + Ok(b"stderr".to_vec()), + InvocationResult::Infrastructure("primary".to_owned()), + ); + assert!(stdout_failed.stdout.is_empty()); + assert_eq!(stdout_failed.stderr, b"stderr"); + assert_eq!(infrastructure_message(stdout_failed), "stdout failed"); + + let stderr_failed = combine_captured_output( + Ok(b"stdout".to_vec()), + Err("stderr failed".to_owned()), + InvocationResult::Infrastructure("primary".to_owned()), + ); + assert_eq!(stderr_failed.stdout, b"stdout"); + assert!(stderr_failed.stderr.is_empty()); + assert_eq!(infrastructure_message(stderr_failed), "stderr failed"); + + let both_failed = combine_captured_output( + Err("stdout failed".to_owned()), + Err("stderr failed".to_owned()), + InvocationResult::Infrastructure("primary".to_owned()), + ); + assert!(both_failed.stdout.is_empty()); + assert!(both_failed.stderr.is_empty()); + assert_eq!(infrastructure_message(both_failed), "stdout failed; stderr failed"); + } + + #[test] + fn output_reader_surfaces_read_and_panic_failures() { + let failed = spawn_output_reader(FailingReader, "failing-reader").expect("create failing reader"); + assert!( + finish_output_reader(failed, "stdout", true) + .expect_err("read failure must be reported") + .contains("injected read failure") + ); + + let panicked = spawn_output_reader(PanickingReader, "panicking-reader").expect("create panicking reader"); + assert!( + finish_output_reader(panicked, "stderr", true) + .expect_err("reader panic must be reported") + .contains("injected reader panic") + ); + + let eof = spawn_output_reader(EofThenPanicReader { reached_eof: false }, "eof-reader").expect("create EOF reader"); + assert_eq!( + finish_output_reader(eof, "stdout", true).expect("EOF completes the output reader"), + Vec::::new() + ); + } + + #[test] + fn reader_setup_failure_preserves_cleanup_and_reader_errors() { + let successful_reader = + spawn_output_reader(io::Cursor::new(b"partial".to_vec()), "successful-reader").expect("create successful reader"); + let outcome = BufferedOutcome::from_reader_failure("stderr unavailable".to_owned(), successful_reader, &Ok::<_, io::Error>(())); + assert_eq!(outcome.stdout, b"partial"); + assert_eq!(infrastructure_message(outcome), "stderr unavailable"); + + let failing_reader = spawn_output_reader(FailingReader, "failing-reader").expect("create failing reader"); + let outcome = BufferedOutcome::from_reader_failure( + "stderr unavailable".to_owned(), + failing_reader, + &Err::<(), _>(io::Error::other("cleanup failed")), + ); + let message = infrastructure_message(outcome); + assert!(message.contains("stderr unavailable")); + assert!(message.contains("cleanup failed")); + + let failing_reader = spawn_output_reader(FailingReader, "failing-reader").expect("create failing reader"); + let outcome = BufferedOutcome::from_reader_failure("stderr unavailable".to_owned(), failing_reader, &Ok::<_, io::Error>(())); + assert!(infrastructure_message(outcome).contains("injected read failure")); + } + + #[test] + fn tree_outcome_constructors_record_cleanup_certainty() { + let closed = TreeOutcome::closed(InvocationResult::Infrastructure("closed".to_owned())); + assert!(closed.cleanup_proven); + assert_eq!( + infrastructure_message(BufferedOutcome { + stdout: Vec::new(), + stderr: Vec::new(), + result: closed.result, + }), + "closed" + ); + + let unproven = TreeOutcome::unproven(InvocationResult::Infrastructure("unproven".to_owned())); + assert!(!unproven.cleanup_proven); + assert_eq!( + infrastructure_message(BufferedOutcome { + stdout: Vec::new(), + stderr: Vec::new(), + result: unproven.result, + }), + "unproven" + ); + } + + #[test] + fn process_waiting_classifies_observation_and_cleanup_failures() { + let mut cleaned = FakeProcess { + observations: VecDeque::from([Err(io::Error::other("observe failed"))]), + termination: Some(Ok(successful_status())), + }; + let outcome = wait_for_tree_with(&mut cleaned, Duration::from_secs(1), FakeProcess::observe, FakeProcess::terminate); + assert!(outcome.cleanup_proven); + assert!(result_infrastructure_message(outcome.result).contains("observe failed")); + + let mut uncleaned = FakeProcess { + observations: VecDeque::from([Err(io::Error::other("observe failed"))]), + termination: Some(Err(io::Error::other("cleanup failed"))), + }; + let outcome = wait_for_tree_without_timeout_with(&mut uncleaned, FakeProcess::observe, FakeProcess::terminate); + assert!(!outcome.cleanup_proven); + let message = result_infrastructure_message(outcome.result); + assert!(message.contains("observe failed")); + assert!(message.contains("cleanup failed")); + + let mut timed_uncleaned = FakeProcess { + observations: VecDeque::from([Err(io::Error::other("timed observe failed"))]), + termination: Some(Err(io::Error::other("timed cleanup failed"))), + }; + let outcome = wait_for_tree_with( + &mut timed_uncleaned, + Duration::from_secs(1), + FakeProcess::observe, + FakeProcess::terminate, + ); + assert!(!outcome.cleanup_proven); + let message = result_infrastructure_message(outcome.result); + assert!(message.contains("timed observe failed")); + assert!(message.contains("timed cleanup failed")); + + let mut untimed_cleaned = FakeProcess { + observations: VecDeque::from([Err(io::Error::other("untimed observe failed"))]), + termination: Some(Ok(successful_status())), + }; + let outcome = wait_for_tree_without_timeout_with(&mut untimed_cleaned, FakeProcess::observe, FakeProcess::terminate); + assert!(outcome.cleanup_proven); + assert!(result_infrastructure_message(outcome.result).contains("untimed observe failed")); + } + + #[test] + fn timed_process_waiting_covers_completion_and_failed_termination() { + let mut completed = FakeProcess { + observations: VecDeque::from([Ok(Some(successful_status()))]), + termination: None, + }; + let outcome = wait_for_tree_with(&mut completed, Duration::from_secs(1), FakeProcess::observe, FakeProcess::terminate); + assert!(outcome.cleanup_proven); + let InvocationResult::Exited(status) = outcome.result else { + panic!("a completed process must retain its exit status"); + }; + assert!(status.success()); + + let mut uncleaned = FakeProcess { + observations: VecDeque::from([Ok(None)]), + termination: Some(Err(io::Error::other("termination failed"))), + }; + let outcome = wait_for_tree_with(&mut uncleaned, Duration::ZERO, FakeProcess::observe, FakeProcess::terminate); + assert!(!outcome.cleanup_proven); + assert!(result_infrastructure_message(outcome.result).contains("termination failed")); + } + + #[test] + fn process_tree_control_delegates_real_exit_observation() { + let mut command = Command::new("rustc"); + let _ = command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); + let mut tree = spawn_tree(command).expect("spawn contained rustc probe"); + let deadline = Instant::now() + Duration::from_secs(2); + loop { + match tree.observe().expect("observe contained rustc probe") { + Some(status) => { + assert!(status.success()); + break; + } + None if Instant::now() < deadline => thread::sleep(Duration::from_millis(5)), + None => { + let _cleanup = tree.terminate(); + panic!("process-control delegation did not observe the exited probe before the deadline"); + } + } + } + } + + #[test] + fn captured_runner_reports_stream_setup_failures() { + for (label, expected) in [ + ("__cargo_each_missing_stdout", "failed to capture child stdout"), + ("__cargo_each_stdout_reader_failure", "injected stdout reader failure"), + ("__cargo_each_missing_stderr", "failed to capture child stderr"), + ("__cargo_each_stderr_reader_failure", "injected stderr reader failure"), + ] { + let outcome = run_captured(&labelled_invocation(label, &["rustc", "--version"]), None); + assert!(infrastructure_message(outcome).contains(expected), "{label}"); + } + } } diff --git a/crates/cargo-each/src/select.rs b/crates/cargo-each/src/select.rs index ce959579a..4761ab2e4 100644 --- a/crates/cargo-each/src/select.rs +++ b/crates/cargo-each/src/select.rs @@ -158,10 +158,12 @@ fn validate_package_file_spec<'a>(path: &str, line: usize, spec: &'a str) -> Res if pieces.next().is_some() { return Err(invalid("a package spec may contain at most one `@`")); } - if name.is_empty() - || !name - .bytes() - .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_' | b'*' | b'?')) + if name.is_empty() { + return Err(invalid("expected a package name or Unix glob, optionally followed by `@version`")); + } + if !name + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_' | b'*' | b'?')) { return Err(invalid("expected a package name or Unix glob, optionally followed by `@version`")); } @@ -398,7 +400,17 @@ mod tests { #[test] fn package_file_lines_reject_comments_tokens_and_whitespace() { - for spec in ["# alpha", "--workspace", " alpha", "alpha ", "alpha beta", "alpha@", "alpha@1@2"] { + for spec in [ + "# alpha", + "--workspace", + " alpha", + "alpha ", + "alpha beta", + "@1", + "alpha!", + "alpha@", + "alpha@1@2", + ] { validate_package_file_spec("packages.txt", 1, spec).expect_err(spec); } for spec in ["alpha", "alpha@1.2.3", "cargo-*", "?eta"] { diff --git a/crates/cargo-each/src/workspace.rs b/crates/cargo-each/src/workspace.rs index 6961960ea..3bba39ae0 100644 --- a/crates/cargo-each/src/workspace.rs +++ b/crates/cargo-each/src/workspace.rs @@ -175,10 +175,12 @@ impl Workspace { .get("workspace") .and_then(|workspace| workspace.get("package")) .and_then(|package| package.get("rust-version")); - let root_is_only_member = self.members.len() == 1 && self.members[0].manifest_path == self.root_manifest_path; - let package_floor = root_is_only_member - .then(|| manifest.get("package").and_then(|package| package.get("rust-version"))) - .flatten(); + let package_floor = match self.members.as_slice() { + [member] if member.manifest_path == self.root_manifest_path => { + manifest.get("package").and_then(|package| package.get("rust-version")) + } + _ => None, + }; let floor = workspace_floor.or(package_floor).ok_or_else(|| { WorkspaceRustVersionError::new( "the root manifest must declare `[workspace.package].rust-version`, or `[package].rust-version` for a single-package repository" @@ -199,12 +201,14 @@ impl Workspace { )) .into()); }; - if member_floor.major != 1 || !member_floor.pre.is_empty() || !member_floor.build.is_empty() { - return Err(WorkspaceRustVersionError::new(format!( - "workspace member `{}` exposes invalid Rust version `{member_floor}`; expected a Rust 1.x toolchain version", - member.name - )) - .into()); + if member_floor.major != 1 { + return Err(invalid_member_rust_version(member, member_floor)); + } + if !member_floor.pre.is_empty() { + return Err(invalid_member_rust_version(member, member_floor)); + } + if !member_floor.build.is_empty() { + return Err(invalid_member_rust_version(member, member_floor)); } if member_floor > &parsed_floor { return Err(WorkspaceRustVersionError::new(format!( @@ -219,19 +223,35 @@ impl Workspace { } } +fn invalid_member_rust_version(member: &Member, version: &Version) -> EachError { + WorkspaceRustVersionError::new(format!( + "workspace member `{}` exposes invalid Rust version `{version}`; expected a Rust 1.x toolchain version", + member.name + )) + .into() +} + fn parse_rust_version(value: &str) -> Result { - if value.contains('-') || value.contains('+') { + if value.contains('-') { + return Err("pre-release and build metadata are not valid Rust toolchain versions".to_owned()); + } + if value.contains('+') { return Err("pre-release and build metadata are not valid Rust toolchain versions".to_owned()); } let components: Vec<&str> = value.split('.').collect(); - if !(2..=3).contains(&components.len()) - || components - .iter() - .any(|component| component.is_empty() || !component.bytes().all(|byte| byte.is_ascii_digit())) + if !(2..=3).contains(&components.len()) { + return Err("expected `major.minor` or `major.minor.patch`".to_owned()); + } + if components + .iter() + .any(|component| component.is_empty() || !component.bytes().all(|byte| byte.is_ascii_digit())) { return Err("expected `major.minor` or `major.minor.patch`".to_owned()); } - if components.iter().any(|component| component.len() > 1 && component.starts_with('0')) { + if components + .iter() + .any(|component| component.strip_prefix('0').is_some_and(|remainder| !remainder.is_empty())) + { return Err("numeric components must not contain leading zeroes".to_owned()); } let normalized = if components.len() == 2 { @@ -260,8 +280,32 @@ pub(crate) fn parse_target_kind(kind: &str) -> Option { #[cfg(test)] #[cfg_attr(coverage_nightly, coverage(off))] mod tests { + use std::fs; + use super::*; + fn member(name: &str, manifest_path: PathBuf, rust_version: &str) -> Member { + Member { + name: name.to_owned(), + version: "0.1.0".to_owned(), + rust_version: Some(rust_version.parse().expect("the test Rust version is semver")), + manifest_path, + publishable: true, + features: BTreeSet::new(), + targets: Vec::new(), + dependencies: BTreeSet::new(), + metadata: Value::Null, + } + } + + fn workspace(root_manifest_path: PathBuf, members: Vec) -> Workspace { + Workspace { + members, + default_member_names: HashSet::new(), + root_manifest_path, + } + } + #[test] fn parses_every_supported_target_kind() { for kind in [ @@ -290,8 +334,82 @@ mod tests { fn parses_cargo_rust_version_forms() { assert_eq!(parse_rust_version("1.80").expect("minor form"), Version::new(1, 80, 0)); assert_eq!(parse_rust_version("1.80.1").expect("patch form"), Version::new(1, 80, 1)); - for value in ["1", "1.80.0-beta", "1.80+build", "1.080", "2.0", "one.80"] { - assert!(parse_rust_version(value).is_err(), "{value}"); + assert_eq!( + parse_rust_version("1.80.0-beta"), + Err("pre-release and build metadata are not valid Rust toolchain versions".to_owned()) + ); + assert_eq!( + parse_rust_version("1.80+build"), + Err("pre-release and build metadata are not valid Rust toolchain versions".to_owned()) + ); + assert_eq!( + parse_rust_version("1"), + Err("expected `major.minor` or `major.minor.patch`".to_owned()) + ); + assert_eq!( + parse_rust_version("1.2.3.4"), + Err("expected `major.minor` or `major.minor.patch`".to_owned()) + ); + assert_eq!( + parse_rust_version("one.80"), + Err("expected `major.minor` or `major.minor.patch`".to_owned()) + ); + assert_eq!( + parse_rust_version("1.080"), + Err("numeric components must not contain leading zeroes".to_owned()) + ); + assert_eq!(parse_rust_version("2.0"), Err("expected a Rust 1.x toolchain version".to_owned())); + } + + #[test] + fn root_package_floor_requires_the_root_to_be_the_only_member() { + let temp = tempfile::tempdir().expect("create temporary workspace"); + let root = temp.path().join("Cargo.toml"); + fs::write(&root, "[package]\nname = \"root\"\nversion = \"0.1.0\"\nrust-version = \"1.80\"\n") + .expect("write root package manifest"); + let nested = member("nested", temp.path().join("nested/Cargo.toml"), "1.70.0"); + let error = workspace(root, vec![nested]) + .workspace_rust_version() + .expect_err("a nested sole member cannot use the root package floor"); + assert!(error.to_string().contains("[workspace.package].rust-version")); + } + + #[test] + fn invalid_resolved_member_versions_are_configuration_errors() { + let temp = tempfile::tempdir().expect("create temporary workspace"); + let root = temp.path().join("Cargo.toml"); + fs::write(&root, "[workspace]\n[workspace.package]\nrust-version = \"1.80\"\n").expect("write workspace manifest"); + + for version in ["2.0.0", "1.70.0-beta", "1.70.0+build"] { + let invalid = member("invalid", temp.path().join("invalid/Cargo.toml"), version); + let error = workspace(root.clone(), vec![invalid]) + .workspace_rust_version() + .expect_err("a non-Rust member version must fail"); + assert!(error.to_string().contains("invalid Rust version"), "{version}: {error}"); } } + + #[test] + fn workspace_rust_version_reports_manifest_io_and_shape_errors() { + let temp = tempfile::tempdir().expect("create temporary workspace"); + let missing = temp.path().join("missing.toml"); + let error = workspace(missing, Vec::new()) + .workspace_rust_version() + .expect_err("a missing root manifest must fail"); + assert!(error.to_string().contains("could not read workspace manifest")); + + let malformed = temp.path().join("malformed.toml"); + fs::write(&malformed, "[workspace").expect("write malformed root manifest"); + let error = workspace(malformed, Vec::new()) + .workspace_rust_version() + .expect_err("a malformed root manifest must fail"); + assert!(error.to_string().contains("could not parse workspace manifest")); + + let non_string = temp.path().join("non-string.toml"); + fs::write(&non_string, "[workspace]\n[workspace.package]\nrust-version = 180\n").expect("write non-string root floor"); + let error = workspace(non_string, Vec::new()) + .workspace_rust_version() + .expect_err("a non-string root floor must fail"); + assert!(error.to_string().contains("must be a string")); + } } diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index 39098c7e6..b65f8cc5c 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -147,6 +147,10 @@ fn single_package_fixture(rust_version: &str) -> (TempDir, PathBuf) { (tmp, manifest) } +#[expect( + clippy::too_many_lines, + reason = "the embedded standalone probe stays together so rustc compiles one auditable cross-platform fixture" +)] fn compile_execution_probe(directory: &Path) -> PathBuf { let source = directory.join("execution-probe.rs"); let executable = directory.join(format!("execution-probe{}", std::env::consts::EXE_SUFFIX)); @@ -158,13 +162,23 @@ use std::fs::{self, OpenOptions}; use std::io::Write as _; use std::process::{self, Command}; use std::thread; -use std::time::Duration; +use std::time::{Duration, Instant}; fn append(path: &str, value: &str) { let mut file = OpenOptions::new().create(true).append(true).open(path).expect("open log"); writeln!(file, "{value}").expect("append log"); } +fn wait_for(path: &std::path::Path) { + let deadline = Instant::now() + Duration::from_secs(10); + while !path.exists() { + if Instant::now() >= deadline { + process::exit(90); + } + thread::sleep(Duration::from_millis(1)); + } +} + fn main() { let args: Vec = env::args().collect(); match args[1].as_str() { @@ -182,15 +196,39 @@ fn main() { } "fail-stop" => { let name = &args[2]; - append(&args[3], name); - thread::sleep(Duration::from_millis(if name == "alpha" { 40 } else { 250 })); - process::exit(if name == "alpha" { 7 } else { 0 }); + let sync_dir = std::path::Path::new(&args[3]); + fs::write(sync_dir.join(format!("{name}.started")), "").expect("write start marker"); + match name.as_str() { + "alpha" => { + wait_for(&sync_dir.join("beta.started")); + process::exit(7); + } + "beta" => { + wait_for(&sync_dir.join("alpha.started")); + process::exit(9); + } + _ => process::exit(0), + } } "keep-going" => { let name = &args[2]; append(&args[3], name); process::exit(if name == "alpha" { 7 } else { 0 }); } + "timeout-fail-fast" => { + if args[2] == "alpha" { + thread::sleep(Duration::from_secs(5)); + } else { + fs::write(&args[3], "later invocation ran").expect("write later marker"); + } + } + "timeout-keep-going" => { + if args[2] == "alpha" { + thread::sleep(Duration::from_secs(5)); + } else { + fs::write(&args[3], "later invocation ran").expect("write later marker"); + } + } "tree-parent" => { let marker = &args[2]; Command::new(env::current_exe().expect("current exe")) @@ -397,6 +435,17 @@ fn package_file_input_errors_fail_loudly() { .code(2) .stderr(predicate::str::contains("line 2").and(predicate::str::contains("command-line tokens"))); + let comment = tmp.path().join("comment.packages"); + fs::write(&comment, "#alpha\n").expect("write package file comment"); + each(&manifest) + .arg("--package-file") + .arg(&comment) + .args(["--dry-run", "--", "echo", "{name}"]) + .assert() + .failure() + .code(2) + .stderr(predicate::str::contains("comments are not supported")); + let unmatched = tmp.path().join("unmatched.packages"); fs::write(&unmatched, "does-not-exist\n").expect("write unmatched package file"); each(&manifest) @@ -1263,20 +1312,23 @@ fn parallel_fail_fast_chooses_failure_by_plan_order() { fn parallel_fail_fast_stops_launching_new_work() { let (tmp, manifest) = fixture(); let probe = compile_execution_probe(tmp.path()); - let launch_log = tmp.path().join("launch.log"); + let sync_dir = tmp.path().join("fail-stop"); + fs::create_dir(&sync_dir).expect("create synchronization directory"); each(&manifest) .args(["--workspace", "--jobs", "2", "--"]) .arg(probe) .args(["fail-stop", "{name}"]) - .arg(&launch_log) + .arg(&sync_dir) .assert() .failure() .code(7); - let launched = fs::read_to_string(launch_log).expect("launch log"); - assert!(launched.contains("alpha\n"), "{launched}"); - assert!(launched.contains("beta\n"), "{launched}"); + assert!(sync_dir.join("alpha.started").exists(), "the first initial worker must start"); + assert!(sync_dir.join("beta.started").exists(), "the second initial worker must start"); for name in ["delta", "epsilon", "gamma"] { - assert!(!launched.contains(name), "{name} must not launch after failure:\n{launched}"); + assert!( + !sync_dir.join(format!("{name}.started")).exists(), + "{name} must not launch after either synchronized initial worker fails" + ); } } @@ -1300,6 +1352,48 @@ fn parallel_keep_going_runs_the_complete_plan() { } } +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn sequential_timeout_fail_fast_does_not_run_later_members() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let later_marker = tmp.path().join("later-invocation"); + each(&manifest) + .args(["-p", "alpha", "-p", "beta", "--timeout", "50ms", "--"]) + .arg(probe) + .args(["timeout-fail-fast", "{name}"]) + .arg(&later_marker) + .assert() + .failure() + .code(1) + .stderr(predicate::str::contains("timed out after 50ms")); + assert!( + !later_marker.exists(), + "fail-fast must not launch the member after a timed-out invocation" + ); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn sequential_timeout_keep_going_runs_later_members() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let later_marker = tmp.path().join("later-invocation"); + each(&manifest) + .args(["-p", "alpha", "-p", "beta", "--timeout", "50ms", "--keep-going", "--"]) + .arg(probe) + .args(["timeout-keep-going", "{name}"]) + .arg(&later_marker) + .assert() + .failure() + .code(1) + .stderr(predicate::str::contains("timed out after 50ms")); + assert!( + later_marker.exists(), + "--keep-going must launch the member after a timed-out invocation" + ); +} + #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] #[test] fn timeout_terminates_the_complete_process_tree() { From c9b64ca964192e48bd73a2210a9deaaaf387bb1a Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Sat, 12 Sep 2026 00:55:47 +0200 Subject: [PATCH 03/37] fix(cargo-each): bound process cleanup Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/README.md | 13 +- crates/cargo-each/docs/design/README.md | 20 +- crates/cargo-each/src/main.rs | 13 +- crates/cargo-each/src/run.rs | 818 ++++++++++++++---- crates/cargo-each/src/substitute.rs | 42 +- crates/cargo-each/tests/cli.rs | 30 + crates/cargo-gamma-process/docs/DESIGN.md | 6 + .../docs/IMPLEMENTATION.md | 9 + crates/cargo-gamma-process/src/faults.rs | 9 + .../cargo-gamma-process/src/process_tree.rs | 221 ++++- 10 files changed, 987 insertions(+), 194 deletions(-) diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index afacba998..25804628c 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -126,10 +126,15 @@ are emitted in deterministic plan order. Fail-fast stops launching after the first observed failure, waits for running work, and chooses the final failure by plan order. `--keep-going` runs the complete plan. Worker panics and unexpected worker-channel disconnections become infrastructure-failure -outcomes instead of blocking the scheduler. -If timed-out tree cleanup fails, cargo-each reports the infrastructure -failure and emits already-buffered output without waiting indefinitely for -surviving descendants to close inherited pipes. +outcomes instead of blocking the scheduler. Without `--timeout`, parallel +commands retain ordinary direct-child semantics and do not kill background +descendants. + +Output drain is bounded after every completion. Readers get one second to +observe EOF; grace expiry preserves partial bytes and becomes an explicit +infrastructure failure. Timed-out tree termination likewise gets a bounded +250 ms leader-reap grace, after which the leader handle is detached so no +wait or Drop path can defeat the timeout. Child commands inherit `PATH` explicitly. On Windows this makes relative program lookup honor the inherited `PATH` order instead of preferring an unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index bf3596344..f407c9ba7 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -309,15 +309,23 @@ no-op. final failure is chosen by plan order, not scheduler timing. A worker panic is converted into an infrastructure-failure outcome; each worker has a dedicated completion channel, so an unexpected exit is observable as - disconnection rather than leaving the scheduler blocked forever. + disconnection rather than leaving the scheduler blocked forever. Without + `--timeout`, parallel commands use the ordinary direct-child lifecycle: + cargo-each waits for the launched leader but does not contain or kill + background descendants. +- **Output drain is bounded after every completion.** Readers get one second + after the leader completes to observe EOF. Complete output is preserved when + both pipes close within that grace. If a background or escaped descendant + keeps a pipe open, capture stops retaining new bytes, emits the partial bytes + already buffered, detaches the blocked reader, and reports an explicit + infrastructure failure rather than hanging or silently truncating. - **Timeouts terminate trees.** A timed-out command is a failure. cargo-each terminates the child process tree rather than only the immediate process, so compiler or test descendants cannot continue mutating the target directory - after cargo-each returns. If tree termination itself fails, cargo-each reports - that infrastructure failure without waiting indefinitely for surviving - descendants to close inherited output pipes: capture stops retaining new - bytes, emits the partial output already buffered, and detaches blocked - readers so the timeout remains bounded. + after cargo-each returns. Termination gets a bounded 250 ms grace to reap the + leader. If signalling fails and the leader is still running at that deadline, + its handle is detached so neither termination nor Drop can defeat the + invocation timeout; cargo-each reports the infrastructure failure. - **Child executable resolution follows `PATH`.** `cargo-each` explicitly copies an inherited `PATH` onto every child command. This is equivalent to ordinary inheritance on other platforms and makes Windows resolve a relative diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index d34117408..38ff7cb7d 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -116,10 +116,15 @@ //! the first observed failure, waits for running work, and chooses the final //! failure by plan order. `--keep-going` runs the complete plan. Worker panics //! and unexpected worker-channel disconnections become infrastructure-failure -//! outcomes instead of blocking the scheduler. -//! If timed-out tree cleanup fails, cargo-each reports the infrastructure -//! failure and emits already-buffered output without waiting indefinitely for -//! surviving descendants to close inherited pipes. +//! outcomes instead of blocking the scheduler. Without `--timeout`, parallel +//! commands retain ordinary direct-child semantics and do not kill background +//! descendants. +//! +//! Output drain is bounded after every completion. Readers get one second to +//! observe EOF; grace expiry preserves partial bytes and becomes an explicit +//! infrastructure failure. Timed-out tree termination likewise gets a bounded +//! 250 ms leader-reap grace, after which the leader handle is detached so no +//! wait or Drop path can defeat the timeout. //! Child commands inherit `PATH` explicitly. On Windows this makes relative //! program lookup honor the inherited `PATH` order instead of preferring an //! unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index b6bfafa83..8cd164fdb 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -7,8 +7,8 @@ use std::collections::{BTreeSet, VecDeque}; use std::io::{self, Write as _}; use std::num::NonZeroUsize; -use std::panic::{self, UnwindSafe}; -use std::process::{Command, ExitCode, ExitStatus, Stdio}; +use std::panic::{self, AssertUnwindSafe, UnwindSafe}; +use std::process::{Child, ChildStderr, ChildStdout, Command, ExitCode, ExitStatus, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Mutex, mpsc}; use std::thread; @@ -30,6 +30,8 @@ use crate::workspace::{Member, Workspace}; const WORKER_PANIC_TEST_PROGRAM: &str = "__cargo_each_injected_worker_panic"; #[cfg(test)] const WORKER_SPAWN_ERROR_TEST_PROGRAM: &str = "__cargo_each_injected_worker_spawn_error"; +const TERMINATION_GRACE: Duration = Duration::from_millis(250); +const OUTPUT_DRAIN_GRACE: Duration = Duration::from_secs(1); pub(crate) fn run(args: &EachArgs) -> Result { let selection = build_selection(args).into_app_err("failed to read package selection")?; @@ -359,21 +361,29 @@ fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedO Err(message) => return BufferedOutcome::infrastructure(message), }; let _ = command.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); - let mut tree = match spawn_tree(command) { - Ok(tree) => tree, + let process = match timeout { + Some(_) => spawn_tree(command).map(CapturedProcess::Contained), + None => command + .spawn() + .map(|child| CapturedProcess::Ordinary(Some(child))) + .map_err(|error| error.to_string()), + }; + let mut process = match process { + Ok(process) => process, Err(error) => { return BufferedOutcome::infrastructure(format!("failed to spawn `{program}`: {error}")); } }; + let drain_boundary = process.drain_boundary(); let capture_fault = capture_fault(invocation); let stdout = if capture_fault == Some(CaptureFault::MissingStdout) { None } else { - tree.take_stdout() + process.take_stdout() }; let Some(stdout) = stdout else { - let cleanup = tree.terminate(); + let cleanup = process.terminate_bounded(); return BufferedOutcome::infrastructure(with_cleanup_failure("failed to capture child stdout".to_owned(), &cleanup)); }; let stdout_reader = match if capture_fault == Some(CaptureFault::StdoutReader) { @@ -383,18 +393,18 @@ fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedO } { Ok(reader) => reader, Err(error) => { - let cleanup = tree.terminate(); + let cleanup = process.terminate_bounded(); return BufferedOutcome::infrastructure(with_cleanup_failure(format!("failed to create stdout reader: {error}"), &cleanup)); } }; let stderr = if capture_fault == Some(CaptureFault::MissingStderr) { None } else { - tree.take_stderr() + process.take_stderr() }; let Some(stderr) = stderr else { - let cleanup = tree.terminate(); - return BufferedOutcome::from_reader_failure("failed to capture child stderr".to_owned(), stdout_reader, &cleanup); + let cleanup = process.terminate_bounded(); + return BufferedOutcome::from_reader_failure("failed to capture child stderr".to_owned(), stdout_reader, &cleanup, drain_boundary); }; let stderr_reader = match if capture_fault == Some(CaptureFault::StderrReader) { Err(io::Error::other("injected stderr reader failure")) @@ -403,38 +413,43 @@ fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedO } { Ok(reader) => reader, Err(error) => { - let cleanup = tree.terminate(); - return BufferedOutcome::from_reader_failure(format!("failed to create stderr reader: {error}"), stdout_reader, &cleanup); + let cleanup = process.terminate_bounded(); + return BufferedOutcome::from_reader_failure( + format!("failed to create stderr reader: {error}"), + stdout_reader, + &cleanup, + drain_boundary, + ); } }; - let tree_outcome = match timeout { - Some(timeout) => wait_for_tree(&mut tree, timeout), - None => wait_for_tree_without_timeout(&mut tree), - }; - let stdout = finish_output_reader(stdout_reader, "stdout", tree_outcome.cleanup_proven); - let stderr = finish_output_reader(stderr_reader, "stderr", tree_outcome.cleanup_proven); - combine_captured_output(stdout, stderr, tree_outcome.result) + let process_outcome = process.wait(timeout, capture_fault); + drop(process); + let (stdout, stderr) = finish_output_readers(stdout_reader, stderr_reader, OUTPUT_DRAIN_GRACE, drain_boundary); + combine_captured_output(stdout, stderr, process_outcome.result) } -fn combine_captured_output(stdout: Result, String>, stderr: Result, String>, result: InvocationResult) -> BufferedOutcome { - match (stdout, stderr) { - (Ok(stdout), Ok(stderr)) => BufferedOutcome { stdout, stderr, result }, - (Err(error), Ok(stderr)) => BufferedOutcome { - stdout: Vec::new(), - stderr, - result: InvocationResult::Infrastructure(error), - }, - (Ok(stdout), Err(error)) => BufferedOutcome { - stdout, - stderr: Vec::new(), - result: InvocationResult::Infrastructure(error), - }, - (Err(stdout), Err(stderr)) => BufferedOutcome { - stdout: Vec::new(), - stderr: Vec::new(), - result: InvocationResult::Infrastructure(format!("{stdout}; {stderr}")), - }, +fn combine_captured_output(stdout: CapturedStream, stderr: CapturedStream, result: InvocationResult) -> BufferedOutcome { + let failure = [stdout.failure.as_deref(), stderr.failure.as_deref()] + .into_iter() + .flatten() + .collect::>() + .join("; "); + let result = if failure.is_empty() { + result + } else { + InvocationResult::Infrastructure(match result { + InvocationResult::Infrastructure(primary) => format!("{primary}; {failure}"), + InvocationResult::TimedOut(duration) => { + format!("invocation timed out after {}; {failure}", display_duration(duration)) + } + InvocationResult::Exited(_) => failure, + }) + }; + BufferedOutcome { + stdout: stdout.bytes, + stderr: stderr.bytes, + result, } } @@ -460,8 +475,125 @@ fn spawn_tree(command: Command) -> Result { ProcessTree::adopt(spawned).map_err(|error| format!("could not adopt child into process-tree containment: {error}")) } +enum CapturedProcess { + Ordinary(Option), + Contained(ProcessTree), +} + +impl CapturedProcess { + fn drain_boundary(&self) -> &'static str { + match self { + Self::Ordinary(_) => "ordinary process tree", + Self::Contained(_) => "contained process tree", + } + } + + fn take_stdout(&mut self) -> Option { + match self { + Self::Ordinary(child) => child.as_mut()?.stdout.take(), + Self::Contained(tree) => tree.take_stdout(), + } + } + + fn take_stderr(&mut self) -> Option { + match self { + Self::Ordinary(child) => child.as_mut()?.stderr.take(), + Self::Contained(tree) => tree.take_stderr(), + } + } + + fn terminate_bounded(&mut self) -> io::Result { + match self { + Self::Ordinary(child) => { + let mut child = child + .take() + .ok_or_else(|| io::Error::other("ordinary child was already reaped or detached"))?; + let result = terminate_ordinary_child(&mut child, TERMINATION_GRACE); + drop(child); + result + } + Self::Contained(tree) => tree.terminate_bounded(TERMINATION_GRACE), + } + } + + fn wait(&mut self, timeout: Option, capture_fault: Option) -> TreeOutcome { + match (self, timeout) { + (Self::Ordinary(child), None) => { + let Some(mut child) = child.take() else { + return TreeOutcome::new(InvocationResult::Infrastructure( + "ordinary child was already reaped or detached".to_owned(), + )); + }; + let waited = if capture_fault == Some(CaptureFault::WaitFailure) { + Err(io::Error::other("injected child wait failure")) + } else { + child.wait() + }; + match waited { + Ok(status) => TreeOutcome::new(InvocationResult::Exited(status)), + Err(error) => TreeOutcome::new(InvocationResult::Infrastructure(format!( + "failed to wait for child process: {error}" + ))), + } + } + (Self::Contained(tree), Some(timeout)) => wait_for_tree(tree, timeout), + (Self::Ordinary(_), Some(_)) | (Self::Contained(_), None) => TreeOutcome::new(InvocationResult::Infrastructure( + "internal capture mode did not match timeout configuration".to_owned(), + )), + } + } +} + +fn terminate_ordinary_child(child: &mut Child, grace: Duration) -> io::Result { + terminate_ordinary_with(child, grace, Child::kill, Child::try_wait) +} + +fn terminate_ordinary_with( + control: &mut T, + grace: Duration, + kill: impl FnOnce(&mut T) -> io::Result<()>, + mut try_wait: impl FnMut(&mut T) -> io::Result>, +) -> io::Result { + let kill_error = kill(control).err(); + let Some(status) = poll_child_exit(control, grace, &mut try_wait)? else { + let message = kill_error.map_or_else( + || format!("ordinary child did not exit within {} ms after termination", grace.as_millis()), + |error| { + format!( + "{error}; ordinary child did not exit within {} ms after termination", + grace.as_millis() + ) + }, + ); + return Err(io::Error::new(io::ErrorKind::WouldBlock, message)); + }; + match kill_error { + Some(error) => Err(error), + None => Ok(status), + } +} + +fn poll_child_exit( + control: &mut T, + grace: Duration, + try_wait: &mut impl FnMut(&mut T) -> io::Result>, +) -> io::Result> { + let started = Instant::now(); + loop { + if let Some(status) = try_wait(control)? { + return Ok(Some(status)); + } + let Some(remaining) = grace.checked_sub(started.elapsed()) else { + return Ok(None); + }; + thread::sleep(remaining.min(Duration::from_millis(10))); + } +} + fn wait_for_tree(tree: &mut ProcessTree, timeout: Duration) -> TreeOutcome { - wait_for_tree_with(tree, timeout, ProcessTree::observe, ProcessTree::terminate) + wait_for_tree_with(tree, timeout, ProcessTree::observe, |tree| { + tree.terminate_bounded(TERMINATION_GRACE) + }) } fn wait_for_tree_with( @@ -473,15 +605,15 @@ fn wait_for_tree_with( let started = Instant::now(); loop { match observe(control) { - Ok(Some(status)) => return TreeOutcome::closed(InvocationResult::Exited(status)), + Ok(Some(status)) => return TreeOutcome::new(InvocationResult::Exited(status)), Ok(None) => {} Err(error) => { let cleanup = terminate(control); return match cleanup { - Ok(_) => TreeOutcome::closed(InvocationResult::Infrastructure(format!( + Ok(_) => TreeOutcome::new(InvocationResult::Infrastructure(format!( "failed to observe child process tree: {error}" ))), - Err(cleanup) => TreeOutcome::unproven(InvocationResult::Infrastructure(format!( + Err(cleanup) => TreeOutcome::new(InvocationResult::Infrastructure(format!( "failed to observe child process tree: {error}; cleanup also failed: {cleanup}" ))), }; @@ -489,8 +621,8 @@ fn wait_for_tree_with( } let Some(remaining) = timeout.checked_sub(started.elapsed()) else { return match terminate(control) { - Ok(_) => TreeOutcome::closed(InvocationResult::TimedOut(timeout)), - Err(error) => TreeOutcome::unproven(InvocationResult::Infrastructure(format!( + Ok(_) => TreeOutcome::new(InvocationResult::TimedOut(timeout)), + Err(error) => TreeOutcome::new(InvocationResult::Infrastructure(format!( "invocation timed out after {}; process-tree termination failed: {error}", display_duration(timeout) ))), @@ -500,10 +632,7 @@ fn wait_for_tree_with( } } -fn wait_for_tree_without_timeout(tree: &mut ProcessTree) -> TreeOutcome { - wait_for_tree_without_timeout_with(tree, ProcessTree::observe, ProcessTree::terminate) -} - +#[cfg(test)] fn wait_for_tree_without_timeout_with( control: &mut T, mut observe: impl FnMut(&mut T) -> io::Result>, @@ -511,15 +640,15 @@ fn wait_for_tree_without_timeout_with( ) -> TreeOutcome { loop { match observe(control) { - Ok(Some(status)) => return TreeOutcome::closed(InvocationResult::Exited(status)), + Ok(Some(status)) => return TreeOutcome::new(InvocationResult::Exited(status)), Ok(None) => thread::sleep(Duration::from_millis(10)), Err(error) => { let cleanup = terminate(control); return match cleanup { - Ok(_) => TreeOutcome::closed(InvocationResult::Infrastructure(format!( + Ok(_) => TreeOutcome::new(InvocationResult::Infrastructure(format!( "failed to observe child process tree: {error}" ))), - Err(cleanup) => TreeOutcome::unproven(InvocationResult::Infrastructure(format!( + Err(cleanup) => TreeOutcome::new(InvocationResult::Infrastructure(format!( "failed to observe child process tree: {error}; cleanup also failed: {cleanup}" ))), }; @@ -534,6 +663,7 @@ enum CaptureFault { StdoutReader, MissingStderr, StderrReader, + WaitFailure, } fn capture_fault(invocation: &Invocation) -> Option { @@ -544,6 +674,7 @@ fn capture_fault(invocation: &Invocation) -> Option { Some("__cargo_each_stdout_reader_failure") => Some(CaptureFault::StdoutReader), Some("__cargo_each_missing_stderr") => Some(CaptureFault::MissingStderr), Some("__cargo_each_stderr_reader_failure") => Some(CaptureFault::StderrReader), + Some("__cargo_each_wait_failure") => Some(CaptureFault::WaitFailure), _ => None, } } @@ -562,52 +693,97 @@ where let retaining = Arc::new(AtomicBool::new(true)); let captured = Arc::clone(&bytes); let capture_enabled = Arc::clone(&retaining); + let (completion_sender, completion) = mpsc::channel(); let thread = thread::Builder::new().name(name.to_owned()).spawn(move || { - let mut chunk = [0_u8; 8192]; - loop { - if !capture_enabled.load(Ordering::Acquire) { - return Ok(()); - } - let read = match stream.read(&mut chunk)? { - 0 => return Ok(()), - read => read, - }; - let mut output = captured - .lock() - .map_err(|error| io::Error::other(format!("child {name} capture buffer was poisoned: {error}")))?; - if !capture_enabled.load(Ordering::Acquire) { - return Ok(()); + let result = panic::catch_unwind(AssertUnwindSafe(|| -> io::Result<()> { + let mut chunk = [0_u8; 8192]; + loop { + if !capture_enabled.load(Ordering::Acquire) { + return Ok(()); + } + let read = loop { + match stream.read(&mut chunk) { + Ok(0) => return Ok(()), + Ok(read) => break read, + Err(error) if error.kind() == io::ErrorKind::Interrupted => {} + Err(error) => return Err(error), + } + }; + let mut output = captured + .lock() + .map_err(|error| io::Error::other(format!("child {name} capture buffer was poisoned: {error}")))?; + if !capture_enabled.load(Ordering::Acquire) { + continue; + } + output.extend_from_slice(&chunk[..read]); } - output.extend_from_slice(&chunk[..read]); - } + })); + drop(stream); + let completion = match result { + Ok(result) => ReaderCompletion::Finished(result), + Err(payload) => ReaderCompletion::Panicked(panic_description(payload.as_ref()).to_owned()), + }; + let _receiver_gone = completion_sender.send(completion); })?; - Ok(OutputReader { thread, bytes, retaining }) + Ok(OutputReader { + thread, + completion, + bytes, + retaining, + }) } -fn finish_output_reader(reader: OutputReader, stream: &str, cleanup_proven: bool) -> Result, String> { - let OutputReader { thread, bytes, retaining } = reader; - if cleanup_proven { - match thread.join() { - Ok(Ok(())) => {} - Ok(Err(error)) => return Err(format!("failed to read child {stream}: {error}")), - Err(payload) => { - return Err(format!( - "child {stream} reader thread panicked: {}", - panic_description(payload.as_ref()) - )); +fn finish_output_reader(reader: OutputReader, stream: &str, grace: Duration, boundary: &str) -> CapturedStream { + let OutputReader { + thread, + completion, + bytes, + retaining, + } = reader; + let failure = match completion.recv_timeout(grace) { + Ok(ReaderCompletion::Finished(Ok(()))) => None, + Ok(ReaderCompletion::Finished(Err(error))) => Some(format!("failed to read child {stream}: {error}")), + Ok(ReaderCompletion::Panicked(message)) => Some(format!("child {stream} reader thread panicked: {message}")), + Err(mpsc::RecvTimeoutError::Timeout) => { + retaining.store(false, Ordering::Release); + Some(format!( + "child {stream} remained open for more than {} ms after the {boundary} completed; partial output was retained", + grace.as_millis() + )) + } + Err(mpsc::RecvTimeoutError::Disconnected) => { + retaining.store(false, Ordering::Release); + Some(format!("child {stream} reader exited without reporting completion")) + } + }; + drop(thread); + + match bytes.lock() { + Ok(mut captured) => CapturedStream { + bytes: std::mem::take(&mut *captured), + failure, + }, + Err(poisoned) => { + let mut captured = poisoned.into_inner(); + CapturedStream { + bytes: std::mem::take(&mut *captured), + failure: Some(match failure { + Some(failure) => format!("{failure}; child {stream} capture buffer was poisoned"), + None => format!("child {stream} capture buffer was poisoned"), + }), } } - } else { - // A surviving descendant may keep the write end open forever. Stop - // retaining data, take the bytes already captured, and detach the - // blocked reader rather than defeating the invocation timeout. - retaining.store(false, Ordering::Release); - drop(thread); - } - let mut captured = bytes - .lock() - .map_err(|error| format!("child {stream} capture buffer was poisoned: {error}"))?; - Ok(std::mem::take(&mut *captured)) + } +} + +fn finish_output_readers(stdout: OutputReader, stderr: OutputReader, grace: Duration, boundary: &str) -> (CapturedStream, CapturedStream) { + let deadline = Instant::now().checked_add(grace); + let stdout = finish_output_reader(stdout, "stdout", grace, boundary); + let remaining = deadline + .and_then(|deadline| deadline.checked_duration_since(Instant::now())) + .unwrap_or(Duration::ZERO); + let stderr = finish_output_reader(stderr, "stderr", remaining, boundary); + (stdout, stderr) } fn with_cleanup_failure(message: String, cleanup: &io::Result) -> String { @@ -667,30 +843,32 @@ struct RunningWorker { #[derive(Debug)] struct OutputReader { - thread: thread::JoinHandle>, + thread: thread::JoinHandle<()>, + completion: mpsc::Receiver, bytes: Arc>>, retaining: Arc, } +#[derive(Debug)] +enum ReaderCompletion { + Finished(io::Result<()>), + Panicked(String), +} + +#[derive(Debug)] +struct CapturedStream { + bytes: Vec, + failure: Option, +} + #[derive(Debug)] struct TreeOutcome { result: InvocationResult, - cleanup_proven: bool, } impl TreeOutcome { - fn closed(result: InvocationResult) -> Self { - Self { - result, - cleanup_proven: true, - } - } - - fn unproven(result: InvocationResult) -> Self { - Self { - result, - cleanup_proven: false, - } + fn new(result: InvocationResult) -> Self { + Self { result } } } @@ -710,20 +888,16 @@ impl BufferedOutcome { } } - fn from_reader_failure(message: String, stdout_reader: OutputReader, cleanup: &io::Result) -> Self { + fn from_reader_failure(message: String, stdout_reader: OutputReader, cleanup: &io::Result, boundary: &str) -> Self { let message = with_cleanup_failure(message, cleanup); - match finish_output_reader(stdout_reader, "stdout", cleanup.is_ok()) { - Ok(stdout) => Self { - stdout, - stderr: Vec::new(), - result: InvocationResult::Infrastructure(message), + combine_captured_output( + finish_output_reader(stdout_reader, "stdout", OUTPUT_DRAIN_GRACE, boundary), + CapturedStream { + bytes: Vec::new(), + failure: None, }, - Err(reader_error) => Self { - stdout: Vec::new(), - stderr: Vec::new(), - result: InvocationResult::Infrastructure(format!("{message}; {reader_error}")), - }, - } + InvocationResult::Infrastructure(message), + ) } } @@ -771,17 +945,22 @@ mod tests { use std::collections::VecDeque; use std::num::NonZeroUsize; use std::process::{Command, ExitCode, ExitStatus, Stdio}; + use std::sync::atomic::AtomicBool; use std::sync::{Arc, Condvar, Mutex, mpsc}; use std::time::{Duration, Instant}; use std::{io, thread}; use super::{ - BufferedOutcome, Invocation, InvocationResult, Plan, RunningWorker, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, - WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, execute_parallel, exit_byte, failure_stops_launching, - finish_output_reader, panic_description, run_captured, run_streamed, run_streamed_with_timeout, spawn_output_reader, spawn_tree, - wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, + BufferedOutcome, CapturedProcess, CapturedStream, Invocation, InvocationResult, OutputReader, Plan, ReaderCompletion, + RunningWorker, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, + execute_parallel, exit_byte, failure_stops_launching, finish_output_reader, panic_description, run_captured, run_streamed, + run_streamed_with_timeout, spawn_output_reader, spawn_tree, terminate_ordinary_child, terminate_ordinary_with, wait_for_tree_with, + wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, }; + const ORDINARY_BOUNDARY: &str = "ordinary process tree"; + const CONTAINED_BOUNDARY: &str = "contained process tree"; + struct StubbornPipe { read: bool, blocked: Option>, @@ -809,6 +988,28 @@ mod tests { reached_eof: bool, } + struct InterruptedThenDataReader { + state: u8, + } + + impl io::Read for InterruptedThenDataReader { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + match self.state { + 0 => { + self.state = 1; + Err(io::Error::from(io::ErrorKind::Interrupted)) + } + 1 => { + self.state = 2; + let content = b"after interrupt"; + buf[..content.len()].copy_from_slice(content); + Ok(content.len()) + } + _ => Ok(0), + } + } + } + impl io::Read for EofThenPanicReader { fn read(&mut self, _buf: &mut [u8]) -> io::Result { assert!(!self.reached_eof, "the output reader must stop after the first EOF"); @@ -874,6 +1075,24 @@ mod tests { termination: Option>, } + struct FakeOrdinaryChild { + kill_error: Option, + observations: VecDeque>>, + } + + impl FakeOrdinaryChild { + fn kill_child(&mut self) -> io::Result<()> { + match self.kill_error.take() { + Some(error) => Err(error), + None => Ok(()), + } + } + + fn try_wait_child(&mut self) -> io::Result> { + self.observations.pop_front().unwrap_or(Ok(None)) + } + } + impl FakeProcess { fn observe(&mut self) -> io::Result> { self.observations.pop_front().unwrap_or(Ok(None)) @@ -897,6 +1116,24 @@ mod tests { result_infrastructure_message(outcome.result) } + fn captured(bytes: &[u8], failure: Option<&str>) -> CapturedStream { + CapturedStream { + bytes: bytes.to_vec(), + failure: failure.map(str::to_owned), + } + } + + fn poisoned_buffer() -> Arc>> { + let bytes = Arc::new(Mutex::new(b"poisoned bytes".to_vec())); + let poisoned = Arc::clone(&bytes); + let _panic = thread::spawn(move || { + let _guard = poisoned.lock().expect("the fresh capture mutex is available"); + panic!("poison capture buffer"); + }) + .join(); + bytes + } + impl io::Read for StubbornPipe { fn read(&mut self, buf: &mut [u8]) -> io::Result { if !self.read { @@ -959,7 +1196,7 @@ mod tests { } #[test] - fn unproven_cleanup_does_not_join_a_stubborn_pipe_reader() { + fn normal_completion_does_not_join_a_stubborn_descendant_pipe() { let release = Arc::new((Mutex::new(false), Condvar::new())); let (blocked_tx, blocked_rx) = mpsc::channel(); let (reader_finished_tx, reader_finished_rx) = mpsc::channel(); @@ -979,7 +1216,7 @@ mod tests { let (finished_tx, finished_rx) = mpsc::channel(); let finisher = thread::spawn(move || { - let result = finish_output_reader(reader, "stdout", false); + let result = finish_output_reader(reader, "stdout", Duration::from_millis(25), ORDINARY_BOUNDARY); let _receiver_gone = finished_tx.send(result); }); let result = finished_rx.recv_timeout(Duration::from_millis(500)); @@ -988,14 +1225,16 @@ mod tests { *lock.lock().expect("the test owns the release mutex without panicking") = true; condition.notify_all(); - let captured = result - .expect("unproven cleanup must not wait for a descendant to close its pipe") - .expect("capturing already-read output succeeds"); + let captured = result.expect("normal completion must not wait for an escaped descendant to close its pipe"); finisher.join().expect("the bounded finisher thread does not panic"); reader_finished_rx .recv_timeout(Duration::from_secs(1)) .expect("the detached reader exits after the test releases its simulated pipe"); - assert_eq!(captured, b"captured-before-timeout"); + assert_eq!(captured.bytes, b"captured-before-timeout"); + assert!( + captured.failure.as_deref().is_some_and(|failure| failure.contains("remained open")), + "grace expiry must be an explicit infrastructure failure" + ); } #[test] @@ -1016,8 +1255,9 @@ mod tests { .recv_timeout(Duration::from_secs(1)) .expect("the late-data reader starts its blocking read"); - let captured = finish_output_reader(reader, "stdout", false).expect("stop capture without joining"); - assert!(captured.is_empty()); + let captured = finish_output_reader(reader, "stdout", Duration::from_millis(25), ORDINARY_BOUNDARY); + assert!(captured.bytes.is_empty()); + assert!(captured.failure.is_some()); let (lock, condition) = &*release; *lock.lock().expect("the test owns the release mutex without panicking") = true; @@ -1089,6 +1329,109 @@ mod tests { ); } + #[test] + fn ordinary_termination_polls_until_successful_reap() { + let mut child = FakeOrdinaryChild { + kill_error: None, + observations: VecDeque::from([Ok(None), Ok(Some(successful_status()))]), + }; + + let cleanup = terminate_ordinary_with( + &mut child, + Duration::from_secs(1), + FakeOrdinaryChild::kill_child, + FakeOrdinaryChild::try_wait_child, + ); + + assert!(cleanup.as_ref().is_ok_and(ExitStatus::success)); + assert!(child.observations.is_empty(), "termination returned after only one immediate poll"); + assert_eq!( + with_cleanup_failure("capture setup failed".to_owned(), &cleanup), + "capture setup failed" + ); + } + + #[test] + fn ordinary_child_sleep_probe() { + if std::env::var_os("CARGO_EACH_ORDINARY_CHILD_PROBE").is_some() { + thread::sleep(Duration::from_secs(30)); + } + } + + #[test] + fn ordinary_termination_kills_polls_and_reaps_a_real_child() { + let mut child = Command::new(std::env::current_exe().expect("the test binary knows its path")) + .args(["--exact", "run::tests::ordinary_child_sleep_probe", "--nocapture"]) + .env("CARGO_EACH_ORDINARY_CHILD_PROBE", "1") + .spawn() + .expect("spawn ordinary child probe"); + + let cleanup = terminate_ordinary_child(&mut child, Duration::from_secs(1)); + + let status = cleanup.expect("a successfully killed child must be reaped within the grace"); + assert!(!status.success(), "a killed child must not report successful completion"); + assert!( + child.try_wait().expect("the reaped child remains observable").is_some(), + "bounded termination returned without reaping the child" + ); + } + + #[test] + fn ordinary_termination_reports_deadline_kill_and_observation_failures() { + let mut deadline = FakeOrdinaryChild { + kill_error: None, + observations: VecDeque::from([Ok(None)]), + }; + let error = terminate_ordinary_with( + &mut deadline, + Duration::ZERO, + FakeOrdinaryChild::kill_child, + FakeOrdinaryChild::try_wait_child, + ) + .expect_err("an unreaped child reaches the deadline"); + assert_eq!(error.kind(), io::ErrorKind::WouldBlock); + + let mut failed_kill_deadline = FakeOrdinaryChild { + kill_error: Some(io::Error::new(io::ErrorKind::PermissionDenied, "kill failed")), + observations: VecDeque::from([Ok(None)]), + }; + let error = terminate_ordinary_with( + &mut failed_kill_deadline, + Duration::ZERO, + FakeOrdinaryChild::kill_child, + FakeOrdinaryChild::try_wait_child, + ) + .expect_err("a failed kill with an unreaped child reaches the deadline"); + assert_eq!(error.kind(), io::ErrorKind::WouldBlock); + assert!(error.to_string().contains("kill failed")); + + let mut failed_kill = FakeOrdinaryChild { + kill_error: Some(io::Error::new(io::ErrorKind::PermissionDenied, "kill failed")), + observations: VecDeque::from([Ok(Some(successful_status()))]), + }; + let error = terminate_ordinary_with( + &mut failed_kill, + Duration::from_secs(1), + FakeOrdinaryChild::kill_child, + FakeOrdinaryChild::try_wait_child, + ) + .expect_err("reaping must not hide a failed kill"); + assert_eq!(error.kind(), io::ErrorKind::PermissionDenied); + + let mut failed_observation = FakeOrdinaryChild { + kill_error: None, + observations: VecDeque::from([Err(io::Error::other("observation failed"))]), + }; + let error = terminate_ordinary_with( + &mut failed_observation, + Duration::from_secs(1), + FakeOrdinaryChild::kill_child, + FakeOrdinaryChild::try_wait_child, + ) + .expect_err("try_wait failure must be preserved"); + assert!(error.to_string().contains("observation failed")); + } + #[test] fn panicked_worker_without_a_report_becomes_an_infrastructure_outcome() { let (sender, receiver) = mpsc::channel::(); @@ -1161,8 +1504,8 @@ mod tests { #[test] fn captured_output_combines_every_reader_result_shape() { let success = combine_captured_output( - Ok(b"stdout".to_vec()), - Ok(b"stderr".to_vec()), + captured(b"stdout", None), + captured(b"stderr", None), InvocationResult::Infrastructure("primary".to_owned()), ); assert_eq!(success.stdout, b"stdout"); @@ -1170,61 +1513,148 @@ mod tests { assert_eq!(infrastructure_message(success), "primary"); let stdout_failed = combine_captured_output( - Err("stdout failed".to_owned()), - Ok(b"stderr".to_vec()), + captured(b"partial stdout", Some("stdout failed")), + captured(b"stderr", None), InvocationResult::Infrastructure("primary".to_owned()), ); - assert!(stdout_failed.stdout.is_empty()); + assert_eq!(stdout_failed.stdout, b"partial stdout"); assert_eq!(stdout_failed.stderr, b"stderr"); - assert_eq!(infrastructure_message(stdout_failed), "stdout failed"); + assert_eq!(infrastructure_message(stdout_failed), "primary; stdout failed"); let stderr_failed = combine_captured_output( - Ok(b"stdout".to_vec()), - Err("stderr failed".to_owned()), + captured(b"stdout", None), + captured(b"partial stderr", Some("stderr failed")), InvocationResult::Infrastructure("primary".to_owned()), ); assert_eq!(stderr_failed.stdout, b"stdout"); - assert!(stderr_failed.stderr.is_empty()); - assert_eq!(infrastructure_message(stderr_failed), "stderr failed"); + assert_eq!(stderr_failed.stderr, b"partial stderr"); + assert_eq!(infrastructure_message(stderr_failed), "primary; stderr failed"); let both_failed = combine_captured_output( - Err("stdout failed".to_owned()), - Err("stderr failed".to_owned()), + captured(b"partial stdout", Some("stdout failed")), + captured(b"partial stderr", Some("stderr failed")), InvocationResult::Infrastructure("primary".to_owned()), ); - assert!(both_failed.stdout.is_empty()); - assert!(both_failed.stderr.is_empty()); - assert_eq!(infrastructure_message(both_failed), "stdout failed; stderr failed"); + assert_eq!(both_failed.stdout, b"partial stdout"); + assert_eq!(both_failed.stderr, b"partial stderr"); + assert_eq!(infrastructure_message(both_failed), "primary; stdout failed; stderr failed"); + + let timed_out = combine_captured_output( + captured(b"partial stdout", Some("stdout still open")), + captured(b"", None), + InvocationResult::TimedOut(Duration::from_secs(2)), + ); + assert_eq!( + infrastructure_message(timed_out), + "invocation timed out after 2s; stdout still open" + ); + + let exited = combine_captured_output( + captured(b"partial stdout", Some("stdout still open")), + captured(b"", None), + InvocationResult::Exited(successful_status()), + ); + assert_eq!(infrastructure_message(exited), "stdout still open"); } #[test] fn output_reader_surfaces_read_and_panic_failures() { let failed = spawn_output_reader(FailingReader, "failing-reader").expect("create failing reader"); + let failed = finish_output_reader(failed, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); assert!( - finish_output_reader(failed, "stdout", true) - .expect_err("read failure must be reported") - .contains("injected read failure") + failed + .failure + .as_deref() + .is_some_and(|failure| failure.contains("injected read failure")) ); let panicked = spawn_output_reader(PanickingReader, "panicking-reader").expect("create panicking reader"); + let panicked = finish_output_reader(panicked, "stderr", Duration::from_secs(1), CONTAINED_BOUNDARY); assert!( - finish_output_reader(panicked, "stderr", true) - .expect_err("reader panic must be reported") - .contains("injected reader panic") + panicked + .failure + .as_deref() + .is_some_and(|failure| failure.contains("injected reader panic")) ); let eof = spawn_output_reader(EofThenPanicReader { reached_eof: false }, "eof-reader").expect("create EOF reader"); - assert_eq!( - finish_output_reader(eof, "stdout", true).expect("EOF completes the output reader"), - Vec::::new() + let eof = finish_output_reader(eof, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); + assert!(eof.bytes.is_empty()); + assert!(eof.failure.is_none()); + + let interrupted = + spawn_output_reader(InterruptedThenDataReader { state: 0 }, "interrupted-reader").expect("create interrupted reader"); + let interrupted = finish_output_reader(interrupted, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); + assert_eq!(interrupted.bytes, b"after interrupt"); + assert!(interrupted.failure.is_none()); + + let (completion_sender, completion) = mpsc::channel(); + let sealed_timeout = OutputReader { + thread: thread::spawn(|| {}), + completion, + bytes: Arc::new(Mutex::new(Vec::new())), + retaining: Arc::new(AtomicBool::new(true)), + }; + let sealed_timeout = finish_output_reader(sealed_timeout, "stdout", Duration::ZERO, CONTAINED_BOUNDARY); + drop(completion_sender); + assert!( + sealed_timeout + .failure + .as_deref() + .is_some_and(|failure| failure.contains("contained process tree")) + ); + + let (completion_sender, completion) = mpsc::channel::(); + drop(completion_sender); + let disconnected = OutputReader { + thread: thread::spawn(|| {}), + completion, + bytes: Arc::new(Mutex::new(Vec::new())), + retaining: Arc::new(AtomicBool::new(true)), + }; + let disconnected = finish_output_reader(disconnected, "stderr", Duration::from_secs(1), ORDINARY_BOUNDARY); + assert!( + disconnected + .failure + .as_deref() + .is_some_and(|failure| failure.contains("without reporting completion")) ); + + for completion_result in [ + ReaderCompletion::Finished(Ok(())), + ReaderCompletion::Finished(Err(io::Error::other("read failed"))), + ] { + let (completion_sender, completion) = mpsc::channel(); + completion_sender + .send(completion_result) + .expect("the synthetic reader completion receiver is alive"); + let poisoned = OutputReader { + thread: thread::spawn(|| {}), + completion, + bytes: poisoned_buffer(), + retaining: Arc::new(AtomicBool::new(true)), + }; + let poisoned = finish_output_reader(poisoned, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); + assert_eq!(poisoned.bytes, b"poisoned bytes"); + assert!( + poisoned + .failure + .as_deref() + .is_some_and(|failure| failure.contains("capture buffer was poisoned")) + ); + } } #[test] fn reader_setup_failure_preserves_cleanup_and_reader_errors() { let successful_reader = spawn_output_reader(io::Cursor::new(b"partial".to_vec()), "successful-reader").expect("create successful reader"); - let outcome = BufferedOutcome::from_reader_failure("stderr unavailable".to_owned(), successful_reader, &Ok::<_, io::Error>(())); + let outcome = BufferedOutcome::from_reader_failure( + "stderr unavailable".to_owned(), + successful_reader, + &Ok::<_, io::Error>(()), + ORDINARY_BOUNDARY, + ); assert_eq!(outcome.stdout, b"partial"); assert_eq!(infrastructure_message(outcome), "stderr unavailable"); @@ -1233,38 +1663,32 @@ mod tests { "stderr unavailable".to_owned(), failing_reader, &Err::<(), _>(io::Error::other("cleanup failed")), + ORDINARY_BOUNDARY, ); let message = infrastructure_message(outcome); assert!(message.contains("stderr unavailable")); assert!(message.contains("cleanup failed")); let failing_reader = spawn_output_reader(FailingReader, "failing-reader").expect("create failing reader"); - let outcome = BufferedOutcome::from_reader_failure("stderr unavailable".to_owned(), failing_reader, &Ok::<_, io::Error>(())); + let outcome = BufferedOutcome::from_reader_failure( + "stderr unavailable".to_owned(), + failing_reader, + &Ok::<_, io::Error>(()), + ORDINARY_BOUNDARY, + ); assert!(infrastructure_message(outcome).contains("injected read failure")); } #[test] - fn tree_outcome_constructors_record_cleanup_certainty() { - let closed = TreeOutcome::closed(InvocationResult::Infrastructure("closed".to_owned())); - assert!(closed.cleanup_proven); - assert_eq!( - infrastructure_message(BufferedOutcome { - stdout: Vec::new(), - stderr: Vec::new(), - result: closed.result, - }), - "closed" - ); - - let unproven = TreeOutcome::unproven(InvocationResult::Infrastructure("unproven".to_owned())); - assert!(!unproven.cleanup_proven); + fn tree_outcome_retains_the_invocation_result() { + let outcome = TreeOutcome::new(InvocationResult::Infrastructure("outcome".to_owned())); assert_eq!( infrastructure_message(BufferedOutcome { stdout: Vec::new(), stderr: Vec::new(), - result: unproven.result, + result: outcome.result, }), - "unproven" + "outcome" ); } @@ -1275,7 +1699,6 @@ mod tests { termination: Some(Ok(successful_status())), }; let outcome = wait_for_tree_with(&mut cleaned, Duration::from_secs(1), FakeProcess::observe, FakeProcess::terminate); - assert!(outcome.cleanup_proven); assert!(result_infrastructure_message(outcome.result).contains("observe failed")); let mut uncleaned = FakeProcess { @@ -1283,7 +1706,6 @@ mod tests { termination: Some(Err(io::Error::other("cleanup failed"))), }; let outcome = wait_for_tree_without_timeout_with(&mut uncleaned, FakeProcess::observe, FakeProcess::terminate); - assert!(!outcome.cleanup_proven); let message = result_infrastructure_message(outcome.result); assert!(message.contains("observe failed")); assert!(message.contains("cleanup failed")); @@ -1298,7 +1720,6 @@ mod tests { FakeProcess::observe, FakeProcess::terminate, ); - assert!(!outcome.cleanup_proven); let message = result_infrastructure_message(outcome.result); assert!(message.contains("timed observe failed")); assert!(message.contains("timed cleanup failed")); @@ -1308,8 +1729,17 @@ mod tests { termination: Some(Ok(successful_status())), }; let outcome = wait_for_tree_without_timeout_with(&mut untimed_cleaned, FakeProcess::observe, FakeProcess::terminate); - assert!(outcome.cleanup_proven); assert!(result_infrastructure_message(outcome.result).contains("untimed observe failed")); + + let mut completed = FakeProcess { + observations: VecDeque::from([Ok(None), Ok(Some(successful_status()))]), + termination: None, + }; + let outcome = wait_for_tree_without_timeout_with(&mut completed, FakeProcess::observe, FakeProcess::terminate); + let InvocationResult::Exited(status) = outcome.result else { + panic!("untimed waiting must return the completed status"); + }; + assert!(status.success()); } #[test] @@ -1319,7 +1749,6 @@ mod tests { termination: None, }; let outcome = wait_for_tree_with(&mut completed, Duration::from_secs(1), FakeProcess::observe, FakeProcess::terminate); - assert!(outcome.cleanup_proven); let InvocationResult::Exited(status) = outcome.result else { panic!("a completed process must retain its exit status"); }; @@ -1330,7 +1759,6 @@ mod tests { termination: Some(Err(io::Error::other("termination failed"))), }; let outcome = wait_for_tree_with(&mut uncleaned, Duration::ZERO, FakeProcess::observe, FakeProcess::terminate); - assert!(!outcome.cleanup_proven); assert!(result_infrastructure_message(outcome.result).contains("termination failed")); } @@ -1362,9 +1790,49 @@ mod tests { ("__cargo_each_stdout_reader_failure", "injected stdout reader failure"), ("__cargo_each_missing_stderr", "failed to capture child stderr"), ("__cargo_each_stderr_reader_failure", "injected stderr reader failure"), + ("__cargo_each_wait_failure", "injected child wait failure"), ] { let outcome = run_captured(&labelled_invocation(label, &["rustc", "--version"]), None); assert!(infrastructure_message(outcome).contains(expected), "{label}"); } + + for (label, expected) in [ + ("__cargo_each_missing_stdout", "failed to capture child stdout"), + ("__cargo_each_stdout_reader_failure", "injected stdout reader failure"), + ("__cargo_each_missing_stderr", "failed to capture child stderr"), + ("__cargo_each_stderr_reader_failure", "injected stderr reader failure"), + ] { + let outcome = run_captured(&labelled_invocation(label, &["rustc", "--version"]), Some(Duration::from_secs(1))); + assert!(infrastructure_message(outcome).contains(expected), "contained {label}"); + } + } + + #[test] + fn captured_process_rejects_mismatched_timeout_modes() { + let mut ordinary_command = Command::new("rustc"); + let _ = ordinary_command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); + let mut ordinary = CapturedProcess::Ordinary(Some(ordinary_command.spawn().expect("spawn ordinary rustc"))); + assert_eq!(ordinary.drain_boundary(), "ordinary process tree"); + assert!( + result_infrastructure_message(ordinary.wait(Some(Duration::from_secs(1)), None).result) + .contains("did not match timeout configuration") + ); + let first_wait = ordinary.wait(None, None); + let InvocationResult::Exited(status) = first_wait.result else { + panic!("the ordinary child must be reaped by the matching wait mode"); + }; + assert!(status.success()); + assert!(result_infrastructure_message(ordinary.wait(None, None).result).contains("already reaped or detached")); + + let mut contained_command = Command::new("rustc"); + let _ = contained_command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); + let mut contained = CapturedProcess::Contained(spawn_tree(contained_command).expect("spawn contained rustc")); + assert_eq!(contained.drain_boundary(), "contained process tree"); + assert!(result_infrastructure_message(contained.wait(None, None).result).contains("did not match timeout configuration")); + let _first = contained.terminate_bounded(); + assert!( + contained.terminate_bounded().is_err(), + "a fabricated successful second termination must not be accepted" + ); } } diff --git a/crates/cargo-each/src/substitute.rs b/crates/cargo-each/src/substitute.rs index fe2af3fcf..c471102a5 100644 --- a/crates/cargo-each/src/substitute.rs +++ b/crates/cargo-each/src/substitute.rs @@ -175,18 +175,19 @@ pub(crate) fn substitute(args: &[String], placeholders: &Placeholders) -> Result manifest, .. } => { + let original = replace_workspace_rust_version(arg.clone(), placeholders)?; // The `{name}` / `{spec}` / … literals are cargo-each // placeholder tokens, not Rust format-string arguments. #[expect( clippy::literal_string_with_formatting_args, reason = "cargo-each placeholder tokens, not format args" )] - let replaced = arg + let replaced = original .replace("{name}", name) .replace("{spec}", spec) .replace("{version}", version) .replace("{manifest}", manifest); - out.push(replace_workspace_rust_version(replaced, placeholders)?); + out.push(replaced); } Placeholders::Target { name, @@ -196,17 +197,18 @@ pub(crate) fn substitute(args: &[String], placeholders: &Placeholders) -> Result target, .. } => { + let original = replace_workspace_rust_version(arg.clone(), placeholders)?; #[expect( clippy::literal_string_with_formatting_args, reason = "cargo-each placeholder tokens, not format args" )] - let replaced = arg + let replaced = original .replace("{name}", name) .replace("{spec}", spec) .replace("{version}", version) .replace("{manifest}", manifest) .replace(TARGET_TOKEN, target); - out.push(replace_workspace_rust_version(replaced, placeholders)?); + out.push(replaced); } Placeholders::Once { packages, .. } => { // Validation above guarantees each arg is either exactly @@ -346,6 +348,38 @@ mod tests { ); } + #[test] + fn package_values_are_not_rescanned_for_workspace_tokens() { + let placeholders = Placeholders::Package { + name: "crate".to_owned(), + spec: "crate@1.0.0".to_owned(), + version: "1.0.0".to_owned(), + manifest: "/ws/{workspace-rust-version}/crate/Cargo.toml".to_owned(), + workspace_rust_version: Some("1.80".to_owned()), + }; + assert_eq!( + substitute(&args(&["{workspace-rust-version}", "{manifest}"]), &placeholders).expect("substitute package placeholders"), + ["1.80", "/ws/{workspace-rust-version}/crate/Cargo.toml"] + ); + } + + #[test] + fn target_values_are_not_rescanned_for_workspace_tokens() { + let placeholders = Placeholders::Target { + name: "crate".to_owned(), + spec: "crate@1.0.0".to_owned(), + version: "1.0.0".to_owned(), + manifest: "/ws/{workspace-rust-version}/crate/Cargo.toml".to_owned(), + target: "example".to_owned(), + workspace_rust_version: Some("1.80".to_owned()), + }; + assert_eq!( + substitute(&args(&["{workspace-rust-version}", "{manifest}:{target}"]), &placeholders,) + .expect("substitute target placeholders"), + ["1.80", "/ws/{workspace-rust-version}/crate/Cargo.toml:example"] + ); + } + #[test] fn detects_workspace_rust_version_usage() { assert!(uses_workspace_rust_version(&args(&["tool", "v={workspace-rust-version}"]))); diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index b65f8cc5c..88cba830f 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -242,6 +242,17 @@ fn main() { thread::sleep(Duration::from_millis(500)); fs::write(&args[2], "survived").expect("write marker"); } + "background-parent" => { + Command::new(env::current_exe().expect("current exe")) + .arg("background-child") + .arg(&args[2]) + .spawn() + .expect("spawn background child"); + } + "background-child" => { + thread::sleep(Duration::from_millis(100)); + fs::write(&args[2], "completed").expect("write background marker"); + } other => panic!("unknown probe mode: {other}"), } } @@ -1352,6 +1363,25 @@ fn parallel_keep_going_runs_the_complete_plan() { } } +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn parallel_without_timeout_preserves_ordinary_background_descendants() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let marker = tmp.path().join("background-completed"); + each(&manifest) + .args(["-p", "alpha", "--jobs", "2", "--"]) + .arg(probe) + .arg("background-parent") + .arg(&marker) + .assert() + .success(); + assert!( + marker.exists(), + "parallel execution without --timeout must not kill an ordinary background descendant" + ); +} + #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] #[test] fn sequential_timeout_fail_fast_does_not_run_later_members() { diff --git a/crates/cargo-gamma-process/docs/DESIGN.md b/crates/cargo-gamma-process/docs/DESIGN.md index 96f351422..95f87e1b3 100644 --- a/crates/cargo-gamma-process/docs/DESIGN.md +++ b/crates/cargo-gamma-process/docs/DESIGN.md @@ -68,6 +68,12 @@ therefore covers the complete descendant tree. nested assignment, the spawn is rejected: an inherited job does not provide a handle through which this process can later terminate the child's descendants. Failure to create a job is likewise a refusal rather than a degraded launch. +- Callers with an external deadline use bounded termination. It signals the + same process-tree boundary as ordinary termination but polls the leader only + for the caller-provided grace. A leader that remains running after a failed + kill is detached rather than handed to an indefinite `wait` or Drop path; + the containment handles remain owned until the `ProcessTree` itself is + dropped. - Sealed containment uses a boundary that descendants cannot leave. A host that offers no sealed boundary at all silently uses best-effort process-group containment for an unmetered launch; absence of a warning does not establish diff --git a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md index 7b690b871..dda2da3f0 100644 --- a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md +++ b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md @@ -16,6 +16,15 @@ The contained output path takes stdout and stderr exactly once and drains them concurrently. It sweeps descendants after the leader exits so inherited write ends do not keep readers open indefinitely. +## Bounded termination + +`ProcessTree::terminate_bounded` requests the same subtree and leader kills as +ordinary termination, then polls `try_wait` until a caller-provided grace +expires. It never follows a failed kill with blocking `wait`: at the deadline +the leader handle is detached, the `ProcessTree` no longer owns a child that +Drop could wait for, and the original cleanup failure is retained in the +returned error. + ## Platform composition Unix launch preparation holds the interrupt spawn window only across child diff --git a/crates/cargo-gamma-process/src/faults.rs b/crates/cargo-gamma-process/src/faults.rs index e3df9a730..64c73425d 100644 --- a/crates/cargo-gamma-process/src/faults.rs +++ b/crates/cargo-gamma-process/src/faults.rs @@ -24,6 +24,13 @@ pub enum Fault { /// Terminating a contained subtree reports a cleanup failure. Terminate, + + /// The direct leader and surrounding subtree refuse the termination + /// signal, leaving the leader running. + Kill, + + /// The termination request reports success without signalling the leader. + Linger, } /// Arms `fault` on this thread until the returned value is dropped. @@ -101,6 +108,8 @@ mod tests { assert!(!fired(Fault::Boundary)); assert!(!fired(Fault::Window)); assert!(!fired(Fault::Terminate)); + assert!(!fired(Fault::Kill)); + assert!(!fired(Fault::Linger)); } #[test] diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index c8f5d8c24..6c85e8a8e 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -10,6 +10,7 @@ use std::process::{Child, ChildStderr, ChildStdout, Command, ExitStatus, Output, use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Mutex}; use std::thread::{self, JoinHandle}; +use std::time::Instant; #[cfg(target_os = "linux")] use cargo_gamma_unsafe::cgroup::Cgroup; @@ -1293,6 +1294,72 @@ impl ProcessTree { Ok(reaped) } + /// Requests termination, then waits no longer than `grace` for the leader + /// to exit. + /// + /// Unlike [`Self::terminate`], this method never performs a blocking + /// [`Child::wait`] after signalling. If the leader remains alive at the + /// deadline, its handle is detached and this process tree is left without + /// a child for [`Drop`] to wait on. The surrounding cgroup or job handle + /// remains owned by `self` and is released normally when the process tree + /// is dropped. + /// + /// # Errors + /// + /// Returns the observation error if `try_wait` fails, the termination + /// error if the leader exits after signalling but cleanup had failed, or a + /// timed-out error (including an earlier termination error, when present) + /// if the leader is still running after `grace`. + pub fn terminate_bounded(&mut self, grace: Duration) -> io::Result { + let started = Instant::now(); + let mut child = self + .child + .take() + .ok_or_else(|| io::Error::other("the subtree leader was already reaped"))?; + + #[cfg(any(test, feature = "fault-injection"))] + let killed = if faults::fired(faults::Fault::Kill) { + Err(io::Error::other("subtree termination was refused as requested by a test")) + } else if faults::fired(faults::Fault::Linger) { + Ok(()) + } else { + self.kill(&mut child) + }; + #[cfg(not(any(test, feature = "fault-injection")))] + let killed = self.kill(&mut child); + let mut kill_error = killed.err(); + self.release(); + + loop { + match child.try_wait() { + Ok(Some(status)) => { + if let Some(error) = kill_error.take() { + return Err(error); + } + + #[cfg(any(test, feature = "fault-injection"))] + if faults::fired(faults::Fault::Terminate) { + return Err(io::Error::other("subtree termination failed as requested by a test")); + } + + return Ok(status); + } + Ok(None) => {} + Err(error) => return Err(error), + } + + let Some(remaining) = grace.checked_sub(started.elapsed()) else { + drop(child); + let deadline_error = format!("subtree leader did not exit within {} ms after termination", grace.as_millis()); + return Err(kill_error.take().map_or_else( + || io::Error::new(io::ErrorKind::TimedOut, deadline_error.clone()), + |error| io::Error::new(error.kind(), format!("{error}; {deadline_error}")), + )); + }; + thread::sleep(remaining.min(Duration::from_millis(10))); + } + } + /// Ends descendants while their leader's process-group id is still reserved. /// /// An exited leader can leave servers and inherited pipe handles behind. This private @@ -1569,7 +1636,6 @@ mod tests { use std::fs; #[cfg(unix)] use std::io::{BufRead as _, Write as _}; - use std::time::Instant; use camino::Utf8Path; @@ -2216,6 +2282,159 @@ mod tests { assert!(!finished.exists(), "the grandchild kept working after the subtree was killed"); } + #[test] + fn bounded_termination_does_not_wait_forever_after_a_failed_kill() { + let work = testing::workdir("gamma-bounded-termination"); + let finished = Utf8Path::from_path(work.path()) + .expect("the temporary path is UTF-8") + .join("finished"); + let mut command = Command::new(testing::helper_binary_path().as_std_path()); + let _ = command.args([ + testing::directive("sleep:250"), + testing::directive(format_args!("touch:{finished}")), + ]); + let prepared = prepare(command, MemoryRequest::default()).expect("containment"); + let spawned = prepared.spawn().expect("spawn"); + let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); + let _failed_kill = faults::arm(faults::Fault::Kill); + + let started = Instant::now(); + let error = subtree + .terminate_bounded(Duration::from_millis(25)) + .expect_err("the injected failed kill must reach its deadline"); + + assert!( + started.elapsed() < Duration::from_millis(200), + "bounded termination exceeded its grace" + ); + assert!(error.to_string().contains("termination was refused"), "{error}"); + assert!(error.to_string().contains("did not exit within 25 ms"), "{error}"); + assert!(subtree.child.is_none(), "Drop must have no leader left to wait for"); + + thread::sleep(Duration::from_millis(350)); + assert!(finished.exists(), "the deliberately un-killed leader did not finish on its own"); + } + + #[test] + fn bounded_termination_reaps_a_killed_leader() { + let mut command = Command::new(testing::helper_binary_path().as_std_path()); + let _ = command.arg(testing::directive("sleep:30000")); + let prepared = prepare(command, MemoryRequest::default()).expect("containment"); + let spawned = prepared.spawn().expect("spawn"); + let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); + + let status = subtree + .terminate_bounded(Duration::from_secs(1)) + .expect("the killed leader exits within the grace"); + + assert!(!status.success(), "a killed leader must not report success"); + assert!(subtree.child.is_none(), "the leader was not reaped"); + assert!( + subtree.terminate_bounded(Duration::ZERO).is_err(), + "an already-reaped leader must fail" + ); + } + + #[test] + fn bounded_termination_preserves_a_failed_kill_after_natural_exit() { + let mut command = Command::new(testing::helper_binary_path().as_std_path()); + let _ = command.arg(testing::directive("sleep:50")); + let prepared = prepare(command, MemoryRequest::default()).expect("containment"); + let spawned = prepared.spawn().expect("spawn"); + let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); + let _failed_kill = faults::arm(faults::Fault::Kill); + + let error = subtree + .terminate_bounded(Duration::from_secs(1)) + .expect_err("natural exit must not hide the failed termination request"); + + assert!(error.to_string().contains("termination was refused"), "{error}"); + } + + #[test] + fn bounded_termination_reports_post_reap_cleanup_failure() { + let mut command = Command::new(testing::helper_binary_path().as_std_path()); + let _ = command.arg(testing::directive("sleep:30000")); + let prepared = prepare(command, MemoryRequest::default()).expect("containment"); + let spawned = prepared.spawn().expect("spawn"); + let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); + let _failed_cleanup = faults::arm(faults::Fault::Terminate); + + let error = subtree + .terminate_bounded(Duration::from_secs(1)) + .expect_err("the injected post-reap cleanup failure must be preserved"); + + assert!(error.to_string().contains("failed as requested by a test"), "{error}"); + } + + #[test] + fn bounded_termination_times_out_when_a_successful_signal_is_ignored() { + let work = testing::workdir("gamma-bounded-linger"); + let finished = Utf8Path::from_path(work.path()) + .expect("the temporary path is UTF-8") + .join("finished"); + let mut command = Command::new(testing::helper_binary_path().as_std_path()); + let _ = command.args([ + testing::directive("sleep:250"), + testing::directive(format_args!("touch:{finished}")), + ]); + let prepared = prepare(command, MemoryRequest::default()).expect("containment"); + let spawned = prepared.spawn().expect("spawn"); + let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); + let _ignored_kill = faults::arm(faults::Fault::Linger); + + let error = subtree + .terminate_bounded(Duration::from_millis(25)) + .expect_err("an ignored successful signal must time out"); + + assert_eq!(error.kind(), io::ErrorKind::TimedOut); + assert!(error.to_string().contains("did not exit within 25 ms"), "{error}"); + thread::sleep(Duration::from_millis(350)); + assert!(finished.exists(), "the deliberately un-signalled leader did not finish"); + } + + #[test] + fn ordinary_termination_reports_post_reap_cleanup_failure() { + let mut command = Command::new(testing::helper_binary_path().as_std_path()); + let _ = command.arg(testing::directive("sleep:30000")); + let prepared = prepare(command, MemoryRequest::default()).expect("containment"); + let spawned = prepared.spawn().expect("spawn"); + let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); + let _failed_cleanup = faults::arm(faults::Fault::Terminate); + + let error = subtree + .terminate() + .expect_err("the ordinary termination fault must be preserved after reaping"); + + assert!(error.to_string().contains("failed as requested by a test"), "{error}"); + } + + #[test] + fn observation_cleanup_classifies_pending_cleanup_and_dual_failures() { + let pending: Observation<()> = cleanup_after_observation(false, || unreachable!(), || unreachable!()); + assert!(matches!(pending, Observation::Pending)); + + let cleanup_failed = cleanup_after_observation( + true, + || Err(io::Error::new(io::ErrorKind::PermissionDenied, "cleanup failed")), + || Ok("reaped"), + ); + assert!(matches!( + cleanup_failed, + Observation::CleanupFailed(error) if error.kind() == io::ErrorKind::PermissionDenied + )); + + let both_failed = cleanup_after_observation( + true, + || Err(io::Error::other("cleanup failed")), + || Err::<(), _>(io::Error::new(io::ErrorKind::Interrupted, "reap failed")), + ); + assert!(matches!( + both_failed, + Observation::ReapFailed(error) if error.kind() == io::ErrorKind::Interrupted + )); + } + /// Killing a subtree reaches a grandchild that left the process group. /// /// This is the escape a process group has no answer to. `setsid` and `setpgid` cost one From 76cddd7ca9567079df6f18976560a9e666331008 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Sat, 12 Sep 2026 01:16:46 +0200 Subject: [PATCH 04/37] fix(cargo-each): require sealed timeout containment Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/README.md | 3 + crates/cargo-each/docs/design/README.md | 7 ++- crates/cargo-each/src/cli.rs | 4 +- crates/cargo-each/src/main.rs | 3 + crates/cargo-each/src/run.rs | 78 ++++++++++++++++++++++--- 5 files changed, 84 insertions(+), 11 deletions(-) diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index 25804628c..bb868033f 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -87,6 +87,9 @@ can be double-quoted. Expression atoms: (default is fail-fast). `--jobs ` bounds concurrent per-package or per-target work (default `1`), while `--timeout ` terminates each invocation and its process tree independently (`250ms`, `30s`, or `2m`). +Timeouts require sealed process-tree containment; on a host that only +offers best-effort containment, cargo-each reports an unsupported +infrastructure failure before starting the child. `--chdir` runs each per-package or per-target command from that member crate root; `--dry-run` prints commands without running them. diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index f407c9ba7..9175ac854 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -251,7 +251,7 @@ filtered set is empty, `cargo-each` exits 0, exactly like an empty selection. | `--target-required-feature ` | In per-target mode, retain targets whose `required-features` contains `FEATURE`. Repeatable; values are AND-combined. Requires `--each-target`. | | `--keep-going` | Don't stop at the first failing command; run them all and exit non-zero if any failed. Default is fail-fast (exit with the first failure's code). | | `--jobs ` | Run at most `N` per-package or per-target commands concurrently. Default `1`. With `--once`, values other than `1` are a usage error. | -| `--timeout ` | Terminate an invocation and its child process tree when it exceeds the positive duration, such as `30s` or `2m`. Applies independently to every invocation, including `--once`. No timeout by default. | +| `--timeout ` | Terminate an invocation and its child process tree when it exceeds the positive duration, such as `30s` or `2m`. Applies independently to every invocation, including `--once`. Requires sealed process-tree containment; unsupported hosts fail before the child starts. No timeout by default. | | `--chdir` | Run each per-package or per-target command from that member's crate root (the directory containing its `Cargo.toml`) instead of the caller's CWD. Combined with `--once` it is a usage error (exit 2). Placeholders stay absolute, so only *relative* args in the command shift to the member dir. | | `--manifest-path ` | Workspace root `Cargo.toml`. Defaults to auto-detection from CWD. | | `--dry-run` | Print the fully-substituted commands that *would* run, one per line, without executing. | @@ -325,7 +325,10 @@ no-op. after cargo-each returns. Termination gets a bounded 250 ms grace to reap the leader. If signalling fails and the leader is still running at that deadline, its handle is detached so neither termination nor Drop can defeat the - invocation timeout; cargo-each reports the infrastructure failure. + invocation timeout; cargo-each reports the infrastructure failure. A timeout + is accepted only when launch preparation reports a sealed cgroup or job + boundary. On a host with best-effort process-group containment, cargo-each + reports that timeout is unsupported and does not spawn the command. - **Child executable resolution follows `PATH`.** `cargo-each` explicitly copies an inherited `PATH` onto every child command. This is equivalent to ordinary inheritance on other platforms and makes Windows resolve a relative diff --git a/crates/cargo-each/src/cli.rs b/crates/cargo-each/src/cli.rs index 747135aab..11b9991a6 100644 --- a/crates/cargo-each/src/cli.rs +++ b/crates/cargo-each/src/cli.rs @@ -99,7 +99,9 @@ pub(crate) struct EachArgs { pub(crate) jobs: NonZeroUsize, /// Terminate each invocation and its process tree after this duration. - /// Accepts a positive integer followed by `ms`, `s`, or `m`. + /// Requires sealed process-tree containment; unsupported hosts fail before + /// starting the child. Accepts a positive integer followed by `ms`, `s`, + /// or `m`. #[arg(long, value_name = "DURATION", value_parser = parse_duration)] pub(crate) timeout: Option, diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 38ff7cb7d..f4833ca50 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -77,6 +77,9 @@ //! (default is fail-fast). `--jobs ` bounds concurrent per-package or //! per-target work (default `1`), while `--timeout ` terminates each //! invocation and its process tree independently (`250ms`, `30s`, or `2m`). +//! Timeouts require sealed process-tree containment; on a host that only +//! offers best-effort containment, cargo-each reports an unsupported +//! infrastructure failure before starting the child. //! `--chdir` runs each per-package or per-target command from that member crate //! root; `--dry-run` prints commands without running them. //! diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 8cd164fdb..00497530a 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -14,7 +14,7 @@ use std::sync::{Arc, Mutex, mpsc}; use std::thread; use std::time::{Duration, Instant}; -use cargo_gamma_process::{MemoryRequest, ProcessTree, prepare}; +use cargo_gamma_process::{MemoryRequest, PreparedCommand, ProcessTree, prepare}; use cargo_metadata::TargetKind; use ohno::{AppError, IntoAppError}; @@ -340,7 +340,7 @@ fn run_streamed_with_timeout(invocation: &Invocation, timeout: Duration) -> Invo Ok(command) => command, Err(message) => return InvocationResult::Infrastructure(message), }; - let mut tree = match spawn_tree(command) { + let mut tree = match spawn_sealed_tree(command) { Ok(tree) => tree, Err(error) => { return InvocationResult::Infrastructure(format!("failed to spawn `{program}`: {error}")); @@ -362,7 +362,7 @@ fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedO }; let _ = command.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); let process = match timeout { - Some(_) => spawn_tree(command).map(CapturedProcess::Contained), + Some(_) => spawn_sealed_tree(command).map(CapturedProcess::Contained), None => command .spawn() .map(|child| CapturedProcess::Ordinary(Some(child))) @@ -468,13 +468,36 @@ fn command_for(invocation: &Invocation) -> Result<(&str, Command), String> { Ok((program, command)) } +#[cfg(test)] fn spawn_tree(command: Command) -> Result { - let prepared = - prepare(command, MemoryRequest::default()).map_err(|error| format!("could not prepare process-tree containment: {error}"))?; + let prepared = prepare_tree(command)?; + spawn_prepared_tree(prepared) +} + +fn spawn_sealed_tree(command: Command) -> Result { + let prepared = prepare_tree(command)?; + spawn_if_sealed(prepared, PreparedCommand::sealed, spawn_prepared_tree) +} + +fn prepare_tree(command: Command) -> Result { + prepare(command, MemoryRequest::default()).map_err(|error| format!("could not prepare process-tree containment: {error}")) +} + +fn spawn_prepared_tree(prepared: PreparedCommand) -> Result { let spawned = prepared.spawn().map_err(|failure| failure.to_string())?; ProcessTree::adopt(spawned).map_err(|error| format!("could not adopt child into process-tree containment: {error}")) } +fn spawn_if_sealed(prepared: T, sealed: impl FnOnce(&T) -> bool, spawn: impl FnOnce(T) -> Result) -> Result { + if !sealed(&prepared) { + return Err( + "timeout requires sealed process-tree containment, but this host only provides best-effort containment; the child was not started" + .to_owned(), + ); + } + spawn(prepared) +} + enum CapturedProcess { Ordinary(Option), Contained(ProcessTree), @@ -945,7 +968,7 @@ mod tests { use std::collections::VecDeque; use std::num::NonZeroUsize; use std::process::{Command, ExitCode, ExitStatus, Stdio}; - use std::sync::atomic::AtomicBool; + use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Condvar, Mutex, mpsc}; use std::time::{Duration, Instant}; use std::{io, thread}; @@ -954,8 +977,8 @@ mod tests { BufferedOutcome, CapturedProcess, CapturedStream, Invocation, InvocationResult, OutputReader, Plan, ReaderCompletion, RunningWorker, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, execute_parallel, exit_byte, failure_stops_launching, finish_output_reader, panic_description, run_captured, run_streamed, - run_streamed_with_timeout, spawn_output_reader, spawn_tree, terminate_ordinary_child, terminate_ordinary_with, wait_for_tree_with, - wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, + run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, spawn_tree, terminate_ordinary_child, terminate_ordinary_with, + wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, }; const ORDINARY_BOUNDARY: &str = "ordinary process tree"; @@ -1432,6 +1455,45 @@ mod tests { assert!(error.to_string().contains("observation failed")); } + #[test] + fn unsealed_timeout_is_refused_before_spawn() { + struct FakePrepared { + sealed: bool, + } + + let spawned = Arc::new(AtomicBool::new(false)); + let spawn_observed = Arc::clone(&spawned); + let error = spawn_if_sealed( + FakePrepared { sealed: false }, + |prepared| prepared.sealed, + move |_prepared| { + spawn_observed.store(true, Ordering::SeqCst); + Ok::<_, String>(()) + }, + ) + .expect_err("an unsealed timeout launch must be refused"); + + assert!( + !spawned.load(Ordering::SeqCst), + "the child spawn path ran despite unsealed containment" + ); + assert!(error.contains("timeout requires sealed process-tree containment")); + assert!(error.contains("child was not started")); + + let spawned = Arc::new(AtomicBool::new(false)); + let spawn_observed = Arc::clone(&spawned); + spawn_if_sealed( + FakePrepared { sealed: true }, + |prepared| prepared.sealed, + move |_prepared| { + spawn_observed.store(true, Ordering::SeqCst); + Ok::<_, String>(()) + }, + ) + .expect("sealed containment permits the spawn"); + assert!(spawned.load(Ordering::SeqCst)); + } + #[test] fn panicked_worker_without_a_report_becomes_an_infrastructure_outcome() { let (sender, receiver) = mpsc::channel::(); From 3e71f4ce4ac75bc3c799f2632e7e23730e2da27b Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Sat, 12 Sep 2026 03:44:23 +0200 Subject: [PATCH 05/37] test(cargo-each): cover unsealed timeout hosts Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 38 ++++++++++++++----- crates/cargo-each/src/workspace.rs | 3 ++ crates/cargo-each/tests/cli.rs | 60 +++++++++++++++++++++--------- 3 files changed, 74 insertions(+), 27 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 00497530a..01a6bf521 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -356,13 +356,21 @@ fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedO "injected worker panic" ); + run_captured_with_spawner(invocation, timeout, spawn_sealed_tree) +} + +fn run_captured_with_spawner( + invocation: &Invocation, + timeout: Option, + timed_spawner: impl FnOnce(Command) -> Result, +) -> BufferedOutcome { let (program, mut command) = match command_for(invocation) { Ok(command) => command, Err(message) => return BufferedOutcome::infrastructure(message), }; let _ = command.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); let process = match timeout { - Some(_) => spawn_sealed_tree(command).map(CapturedProcess::Contained), + Some(_) => timed_spawner(command).map(CapturedProcess::Contained), None => command .spawn() .map(|child| CapturedProcess::Ordinary(Some(child))) @@ -967,6 +975,10 @@ fn exit_byte(raw: Option) -> u8 { mod tests { use std::collections::VecDeque; use std::num::NonZeroUsize; + #[cfg(unix)] + use std::os::unix::process::ExitStatusExt as _; + #[cfg(windows)] + use std::os::windows::process::ExitStatusExt as _; use std::process::{Command, ExitCode, ExitStatus, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Condvar, Mutex, mpsc}; @@ -976,9 +988,10 @@ mod tests { use super::{ BufferedOutcome, CapturedProcess, CapturedStream, Invocation, InvocationResult, OutputReader, Plan, ReaderCompletion, RunningWorker, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, - execute_parallel, exit_byte, failure_stops_launching, finish_output_reader, panic_description, run_captured, run_streamed, - run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, spawn_tree, terminate_ordinary_child, terminate_ordinary_with, - wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, + execute_parallel, exit_byte, failure_stops_launching, finish_output_reader, panic_description, run_captured, + run_captured_with_spawner, run_streamed, run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, spawn_tree, + terminate_ordinary_child, terminate_ordinary_with, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, + with_cleanup_failure, }; const ORDINARY_BOUNDARY: &str = "ordinary process tree"; @@ -1129,10 +1142,7 @@ mod tests { } fn successful_status() -> ExitStatus { - Command::new("rustc") - .arg("--version") - .status() - .expect("rustc is available to the crate's test suite") + ExitStatus::from_raw(0) } fn infrastructure_message(outcome: BufferedOutcome) -> String { @@ -1382,6 +1392,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "spawns and terminates a child process")] fn ordinary_termination_kills_polls_and_reaps_a_real_child() { let mut child = Command::new(std::env::current_exe().expect("the test binary knows its path")) .args(["--exact", "run::tests::ordinary_child_sleep_probe", "--nocapture"]) @@ -1534,6 +1545,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "spawns child processes through the scheduler")] fn worker_spawn_failures_are_reported_during_initial_and_replacement_launches() { let initial = Plan { invocations: vec![invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM])], @@ -1551,6 +1563,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "spawns child processes")] fn direct_runners_report_empty_and_unspawnable_commands() { let empty = invocation(&[]); assert!(result_infrastructure_message(run_streamed(&empty)).contains("empty argument vector")); @@ -1825,6 +1838,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "spawns a contained child process")] fn process_tree_control_delegates_real_exit_observation() { let mut command = Command::new("rustc"); let _ = command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); @@ -1846,6 +1860,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "spawns ordinary and contained child processes")] fn captured_runner_reports_stream_setup_failures() { for (label, expected) in [ ("__cargo_each_missing_stdout", "failed to capture child stdout"), @@ -1864,12 +1879,17 @@ mod tests { ("__cargo_each_missing_stderr", "failed to capture child stderr"), ("__cargo_each_stderr_reader_failure", "injected stderr reader failure"), ] { - let outcome = run_captured(&labelled_invocation(label, &["rustc", "--version"]), Some(Duration::from_secs(1))); + let outcome = run_captured_with_spawner( + &labelled_invocation(label, &["rustc", "--version"]), + Some(Duration::from_secs(1)), + spawn_tree, + ); assert!(infrastructure_message(outcome).contains(expected), "contained {label}"); } } #[test] + #[cfg_attr(miri, ignore = "spawns ordinary and contained child processes")] fn captured_process_rejects_mismatched_timeout_modes() { let mut ordinary_command = Command::new("rustc"); let _ = ordinary_command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); diff --git a/crates/cargo-each/src/workspace.rs b/crates/cargo-each/src/workspace.rs index 3bba39ae0..01009cfed 100644 --- a/crates/cargo-each/src/workspace.rs +++ b/crates/cargo-each/src/workspace.rs @@ -362,6 +362,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "uses temporary filesystem manifests")] fn root_package_floor_requires_the_root_to_be_the_only_member() { let temp = tempfile::tempdir().expect("create temporary workspace"); let root = temp.path().join("Cargo.toml"); @@ -375,6 +376,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "uses temporary filesystem manifests")] fn invalid_resolved_member_versions_are_configuration_errors() { let temp = tempfile::tempdir().expect("create temporary workspace"); let root = temp.path().join("Cargo.toml"); @@ -390,6 +392,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "uses temporary filesystem manifests")] fn workspace_rust_version_reports_manifest_io_and_shape_errors() { let temp = tempfile::tempdir().expect("create temporary workspace"); let missing = temp.path().join("missing.toml"); diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index 88cba830f..76c138eea 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -7,6 +7,7 @@ use std::fs; use std::path::{Path, PathBuf}; +use std::sync::OnceLock; use assert_cmd::Command; use predicates::prelude::*; @@ -111,6 +112,16 @@ fn each(manifest: &Path) -> Command { cmd } +fn sealed_containment_available() -> bool { + static AVAILABLE: OnceLock = OnceLock::new(); + + *AVAILABLE.get_or_init(|| cargo_gamma_process::containment().is_ok()) +} + +fn timeout_refusal() -> impl Predicate { + predicate::str::contains("timeout requires sealed process-tree containment").and(predicate::str::contains("child was not started")) +} + fn rust_version_fixture(root_floor: Option<&str>, members: &[(&str, Option<&str>)]) -> (TempDir, PathBuf) { let tmp = tempfile::tempdir().expect("tempdir"); let root = tmp.path(); @@ -1388,18 +1399,21 @@ fn sequential_timeout_fail_fast_does_not_run_later_members() { let (tmp, manifest) = fixture(); let probe = compile_execution_probe(tmp.path()); let later_marker = tmp.path().join("later-invocation"); - each(&manifest) + let assertion = each(&manifest) .args(["-p", "alpha", "-p", "beta", "--timeout", "50ms", "--"]) .arg(probe) .args(["timeout-fail-fast", "{name}"]) .arg(&later_marker) .assert() - .failure() - .code(1) - .stderr(predicate::str::contains("timed out after 50ms")); + .failure(); + if sealed_containment_available() { + assertion.code(1).stderr(predicate::str::contains("timed out after 50ms")); + } else { + assertion.code(2).stderr(timeout_refusal()); + } assert!( !later_marker.exists(), - "fail-fast must not launch the member after a timed-out invocation" + "fail-fast or pre-spawn refusal must not launch the later member" ); } @@ -1409,19 +1423,26 @@ fn sequential_timeout_keep_going_runs_later_members() { let (tmp, manifest) = fixture(); let probe = compile_execution_probe(tmp.path()); let later_marker = tmp.path().join("later-invocation"); - each(&manifest) + let assertion = each(&manifest) .args(["-p", "alpha", "-p", "beta", "--timeout", "50ms", "--keep-going", "--"]) .arg(probe) .args(["timeout-keep-going", "{name}"]) .arg(&later_marker) .assert() - .failure() - .code(1) - .stderr(predicate::str::contains("timed out after 50ms")); - assert!( - later_marker.exists(), - "--keep-going must launch the member after a timed-out invocation" - ); + .failure(); + if sealed_containment_available() { + assertion.code(1).stderr(predicate::str::contains("timed out after 50ms")); + assert!( + later_marker.exists(), + "--keep-going must launch the member after a timed-out invocation" + ); + } else { + assertion.code(1).stderr(timeout_refusal()); + assert!( + !later_marker.exists(), + "unsealed containment must refuse every timed child before spawn" + ); + } } #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] @@ -1430,18 +1451,21 @@ fn timeout_terminates_the_complete_process_tree() { let (tmp, manifest) = fixture(); let probe = compile_execution_probe(tmp.path()); let marker = tmp.path().join("grandchild-survived"); - each(&manifest) + let assertion = each(&manifest) .args(["-p", "alpha", "--jobs", "2", "--timeout", "50ms", "--"]) .arg(probe) .arg("tree-parent") .arg(&marker) .assert() - .failure() - .code(1) - .stderr(predicate::str::contains("timed out after 50ms")); + .failure(); + if sealed_containment_available() { + assertion.code(1).stderr(predicate::str::contains("timed out after 50ms")); + } else { + assertion.code(2).stderr(timeout_refusal()); + } std::thread::sleep(std::time::Duration::from_millis(700)); assert!( !marker.exists(), - "a timed-out invocation's grandchild must not survive to write its marker" + "a timed-out grandchild must be terminated, and an unsupported timed child must never start" ); } From 57745d88e7565e46f8aad3c0d2abe7011290862d Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Sat, 12 Sep 2026 04:54:15 +0200 Subject: [PATCH 06/37] fix(cargo-each): bound parallel output memory Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/Cargo.toml | 2 +- crates/cargo-each/README.md | 5 +- crates/cargo-each/docs/design/README.md | 7 +- crates/cargo-each/src/cli.rs | 3 +- crates/cargo-each/src/main.rs | 5 +- crates/cargo-each/src/run.rs | 505 ++++++++++++++++++++---- crates/cargo-each/tests/cli.rs | 36 ++ 7 files changed, 476 insertions(+), 87 deletions(-) diff --git a/crates/cargo-each/Cargo.toml b/crates/cargo-each/Cargo.toml index c5269c409..30bbdbc31 100644 --- a/crates/cargo-each/Cargo.toml +++ b/crates/cargo-each/Cargo.toml @@ -23,12 +23,12 @@ clap = { workspace = true, features = ["derive", "std", "help", "usage", "error- mutants = { workspace = true } ohno = { workspace = true, features = ["app-err"] } serde_json = { workspace = true, features = ["std"] } +tempfile = { workspace = true } toml = { workspace = true, features = ["parse", "serde"] } [dev-dependencies] assert_cmd = { workspace = true } predicates = { workspace = true } -tempfile = { workspace = true } # >>> anvil-managed: anvil-lints [lints] diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index bb868033f..90c07a3b1 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -131,7 +131,10 @@ failure by plan order. `--keep-going` runs the complete plan. Worker panics and unexpected worker-channel disconnections become infrastructure-failure outcomes instead of blocking the scheduler. Without `--timeout`, parallel commands retain ordinary direct-child semantics and do not kill background -descendants. +descendants. Each output stream retains at most 1 MiB in memory before +spilling to a unique system-temporary file owned by the invocation outcome; +spill failures are infrastructure failures and spill files are removed by +RAII after deterministic plan-order emission. Output drain is bounded after every completion. Readers get one second to observe EOF; grace expiry preserves partial bytes and becomes an explicit diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index 9175ac854..76d1ac337 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -312,7 +312,12 @@ no-op. disconnection rather than leaving the scheduler blocked forever. Without `--timeout`, parallel commands use the ordinary direct-child lifecycle: cargo-each waits for the launched leader but does not contain or kill - background descendants. + background descendants. Buffering is memory-bounded per stream: after 1 MiB, + output spills to a unique file in the system temporary directory. The + invocation outcome owns that file through deterministic plan-order emission, + so every success, failure, and panic path removes it through RAII. Spill + creation, write, seek, or read failures are infrastructure failures; output + is never intentionally truncated on a successful path. - **Output drain is bounded after every completion.** Readers get one second after the leader completes to observe EOF. Complete output is preserved when both pipes close within that grace. If a background or escaped descendant diff --git a/crates/cargo-each/src/cli.rs b/crates/cargo-each/src/cli.rs index 11b9991a6..3db1b74af 100644 --- a/crates/cargo-each/src/cli.rs +++ b/crates/cargo-each/src/cli.rs @@ -94,7 +94,8 @@ pub(crate) struct EachArgs { #[arg(long)] pub(crate) keep_going: bool, - /// Run at most N per-package or per-target commands concurrently. + /// Run at most N per-package or per-target commands concurrently. Buffered + /// output spills to unique system-temporary files beyond 1 MiB per stream. #[arg(long, default_value_t = NonZeroUsize::MIN, value_name = "N")] pub(crate) jobs: NonZeroUsize, diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index f4833ca50..89a9ab344 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -121,7 +121,10 @@ //! and unexpected worker-channel disconnections become infrastructure-failure //! outcomes instead of blocking the scheduler. Without `--timeout`, parallel //! commands retain ordinary direct-child semantics and do not kill background -//! descendants. +//! descendants. Each output stream retains at most 1 MiB in memory before +//! spilling to a unique system-temporary file owned by the invocation outcome; +//! spill failures are infrastructure failures and spill files are removed by +//! RAII after deterministic plan-order emission. //! //! Output drain is bounded after every completion. Readers get one second to //! observe EOF; grace expiry preserves partial bytes and becomes an explicit diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 01a6bf521..b7197776e 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -5,14 +5,14 @@ //! apply filters, build the plan, and run it. use std::collections::{BTreeSet, VecDeque}; -use std::io::{self, Write as _}; +use std::io::{self, Read as _, Seek as _, SeekFrom, Write as _}; use std::num::NonZeroUsize; use std::panic::{self, AssertUnwindSafe, UnwindSafe}; use std::process::{Child, ChildStderr, ChildStdout, Command, ExitCode, ExitStatus, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Mutex, mpsc}; -use std::thread; use std::time::{Duration, Instant}; +use std::{fmt, thread}; use cargo_gamma_process::{MemoryRequest, PreparedCommand, ProcessTree, prepare}; use cargo_metadata::TargetKind; @@ -32,6 +32,7 @@ const WORKER_PANIC_TEST_PROGRAM: &str = "__cargo_each_injected_worker_panic"; const WORKER_SPAWN_ERROR_TEST_PROGRAM: &str = "__cargo_each_injected_worker_spawn_error"; const TERMINATION_GRACE: Duration = Duration::from_millis(250); const OUTPUT_DRAIN_GRACE: Duration = Duration::from_secs(1); +const OUTPUT_MEMORY_LIMIT: usize = 1_048_576; pub(crate) fn run(args: &EachArgs) -> Result { let selection = build_selection(args).into_app_err("failed to read package selection")?; @@ -234,8 +235,8 @@ fn execute_parallel(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: } outcomes.sort_by_key(|outcome| outcome.index); - for indexed in &outcomes { - emit_buffered(&invocations[indexed.index], &indexed.outcome).into_app_err("failed to emit buffered command output")?; + for indexed in &mut outcomes { + emit_buffered(&invocations[indexed.index], &mut indexed.outcome).into_app_err("failed to emit buffered command output")?; } let Some(first_failure) = outcomes.iter().find(|outcome| outcome.outcome.result.failed()) else { @@ -443,24 +444,27 @@ fn combine_captured_output(stdout: CapturedStream, stderr: CapturedStream, resul .flatten() .collect::>() .join("; "); - let result = if failure.is_empty() { - result - } else { - InvocationResult::Infrastructure(match result { - InvocationResult::Infrastructure(primary) => format!("{primary}; {failure}"), - InvocationResult::TimedOut(duration) => { - format!("invocation timed out after {}; {failure}", display_duration(duration)) - } - InvocationResult::Exited(_) => failure, - }) - }; + let result = add_infrastructure_failure(result, failure); BufferedOutcome { - stdout: stdout.bytes, - stderr: stderr.bytes, + stdout: stdout.output, + stderr: stderr.output, result, } } +fn add_infrastructure_failure(result: InvocationResult, failure: String) -> InvocationResult { + if failure.is_empty() { + return result; + } + InvocationResult::Infrastructure(match result { + InvocationResult::Infrastructure(primary) => format!("{primary}; {failure}"), + InvocationResult::TimedOut(duration) => { + format!("invocation timed out after {}; {failure}", display_duration(duration)) + } + InvocationResult::Exited(_) => failure, + }) +} + fn command_for(invocation: &Invocation) -> Result<(&str, Command), String> { let Some((program, arguments)) = invocation.argv.split_first() else { return Err("internal command-plan error: invocation has an empty argument vector".to_owned()); @@ -560,12 +564,7 @@ impl CapturedProcess { } else { child.wait() }; - match waited { - Ok(status) => TreeOutcome::new(InvocationResult::Exited(status)), - Err(error) => TreeOutcome::new(InvocationResult::Infrastructure(format!( - "failed to wait for child process: {error}" - ))), - } + finish_wait_with_cleanup(&mut child, waited, |child| terminate_ordinary_child(child, TERMINATION_GRACE)) } (Self::Contained(tree), Some(timeout)) => wait_for_tree(tree, timeout), (Self::Ordinary(_), Some(_)) | (Self::Contained(_), None) => TreeOutcome::new(InvocationResult::Infrastructure( @@ -575,6 +574,23 @@ impl CapturedProcess { } } +fn finish_wait_with_cleanup( + control: &mut T, + waited: io::Result, + cleanup: impl FnOnce(&mut T) -> io::Result, +) -> TreeOutcome { + match waited { + Ok(status) => TreeOutcome::new(InvocationResult::Exited(status)), + Err(error) => { + let cleanup = cleanup(control); + TreeOutcome::new(InvocationResult::Infrastructure(with_cleanup_failure( + format!("failed to wait for child process: {error}"), + &cleanup, + ))) + } + } +} + fn terminate_ordinary_child(child: &mut Child, grace: Duration) -> io::Result { terminate_ordinary_with(child, grace, Child::kill, Child::try_wait) } @@ -716,13 +732,30 @@ fn capture_fault(invocation: &Invocation) -> Option { } } -fn spawn_output_reader(mut stream: R, name: &'static str) -> io::Result +fn spawn_output_reader(stream: R, name: &'static str) -> io::Result +where + R: io::Read + Send + 'static, +{ + spawn_output_reader_with( + stream, + name, + OUTPUT_MEMORY_LIMIT, + Box::new(|| tempfile::tempfile().map(|file| Box::new(file) as Box)), + ) +} + +fn spawn_output_reader_with( + mut stream: R, + name: &'static str, + memory_limit: usize, + mut spill_factory: SpillFactory, +) -> io::Result where R: io::Read + Send + 'static, { - let bytes = Arc::new(Mutex::new(Vec::new())); + let output = Arc::new(Mutex::new(CapturedOutput::empty())); let retaining = Arc::new(AtomicBool::new(true)); - let captured = Arc::clone(&bytes); + let captured = Arc::clone(&output); let capture_enabled = Arc::clone(&retaining); let (completion_sender, completion) = mpsc::channel(); let thread = thread::Builder::new().name(name.to_owned()).spawn(move || { @@ -746,7 +779,7 @@ where if !capture_enabled.load(Ordering::Acquire) { continue; } - output.extend_from_slice(&chunk[..read]); + output.append(&chunk[..read], memory_limit, &mut spill_factory)?; } })); drop(stream); @@ -759,7 +792,7 @@ where Ok(OutputReader { thread, completion, - bytes, + output, retaining, }) } @@ -768,7 +801,7 @@ fn finish_output_reader(reader: OutputReader, stream: &str, grace: Duration, bou let OutputReader { thread, completion, - bytes, + output, retaining, } = reader; let failure = match completion.recv_timeout(grace) { @@ -789,15 +822,15 @@ fn finish_output_reader(reader: OutputReader, stream: &str, grace: Duration, bou }; drop(thread); - match bytes.lock() { + match output.lock() { Ok(mut captured) => CapturedStream { - bytes: std::mem::take(&mut *captured), + output: std::mem::replace(&mut *captured, CapturedOutput::empty()), failure, }, Err(poisoned) => { let mut captured = poisoned.into_inner(); CapturedStream { - bytes: std::mem::take(&mut *captured), + output: std::mem::replace(&mut *captured, CapturedOutput::empty()), failure: Some(match failure { Some(failure) => format!("{failure}; child {stream} capture buffer was poisoned"), None => format!("child {stream} capture buffer was poisoned"), @@ -830,13 +863,42 @@ fn emit_label(invocation: &Invocation) { } } -fn emit_buffered(invocation: &Invocation, outcome: &BufferedOutcome) -> io::Result<()> { - emit_label(invocation); +fn emit_buffered(invocation: &Invocation, outcome: &mut BufferedOutcome) -> io::Result<()> { let mut stdout = io::stdout().lock(); - stdout.write_all(&outcome.stdout)?; - stdout.flush()?; let mut stderr = io::stderr().lock(); - stderr.write_all(&outcome.stderr)?; + emit_buffered_to(invocation, outcome, &mut stdout, &mut stderr) +} + +fn emit_buffered_to( + invocation: &Invocation, + outcome: &mut BufferedOutcome, + stdout: &mut dyn io::Write, + stderr: &mut dyn io::Write, +) -> io::Result<()> { + emit_label(invocation); + let mut source_failures = Vec::new(); + match outcome.stdout.emit_to(stdout) { + Ok(()) => {} + Err(OutputEmitError::Source(error)) => { + source_failures.push(format!("failed to read spilled child stdout: {error}")); + } + Err(OutputEmitError::Destination(error)) => return Err(error), + } + stdout.flush()?; + match outcome.stderr.emit_to(stderr) { + Ok(()) => {} + Err(OutputEmitError::Source(error)) => { + source_failures.push(format!("failed to read spilled child stderr: {error}")); + } + Err(OutputEmitError::Destination(error)) => return Err(error), + } + if !source_failures.is_empty() { + let original = std::mem::replace( + &mut outcome.result, + InvocationResult::Infrastructure("output emission failed".to_owned()), + ); + outcome.result = add_infrastructure_failure(original, source_failures.join("; ")); + } match &outcome.result { InvocationResult::TimedOut(duration) => { writeln!(stderr, "cargo each: invocation timed out after {}", display_duration(*duration))?; @@ -876,7 +938,7 @@ struct RunningWorker { struct OutputReader { thread: thread::JoinHandle<()>, completion: mpsc::Receiver, - bytes: Arc>>, + output: Arc>, retaining: Arc, } @@ -886,9 +948,72 @@ enum ReaderCompletion { Panicked(String), } +trait SpillFile: io::Read + io::Write + io::Seek + Send + fmt::Debug {} + +impl SpillFile for T where T: io::Read + io::Write + io::Seek + Send + fmt::Debug {} + +type SpillFactory = Box io::Result> + Send>; + +#[derive(Debug)] +enum CapturedOutput { + Memory(Vec), + Spill(Box), +} + +impl CapturedOutput { + fn empty() -> Self { + Self::Memory(Vec::new()) + } + + fn append(&mut self, bytes: &[u8], memory_limit: usize, spill_factory: &mut SpillFactory) -> io::Result<()> { + match self { + Self::Memory(memory) if memory.len().saturating_add(bytes.len()) <= memory_limit => { + memory.extend_from_slice(bytes); + Ok(()) + } + Self::Memory(memory) => { + let mut spill = spill_factory()?; + spill.write_all(memory)?; + spill.write_all(bytes)?; + *self = Self::Spill(spill); + Ok(()) + } + Self::Spill(spill) => spill.write_all(bytes), + } + } + + fn emit_to(&mut self, destination: &mut dyn io::Write) -> Result<(), OutputEmitError> { + match self { + Self::Memory(bytes) => destination.write_all(bytes).map_err(OutputEmitError::Destination), + Self::Spill(spill) => { + spill.seek(SeekFrom::Start(0)).map_err(OutputEmitError::Source)?; + let mut buffer = [0_u8; 8192]; + loop { + let read = spill.read(&mut buffer).map_err(OutputEmitError::Source)?; + if read == 0 { + return Ok(()); + } + destination.write_all(&buffer[..read]).map_err(OutputEmitError::Destination)?; + } + } + } + } + + #[cfg(test)] + fn is_spilled(&self) -> bool { + matches!(self, Self::Spill(_)) + } +} + +#[derive(Debug)] +enum OutputEmitError { + Source(io::Error), + Destination(io::Error), +} + #[derive(Debug)] struct CapturedStream { - bytes: Vec, + output: CapturedOutput, failure: Option, } @@ -905,16 +1030,16 @@ impl TreeOutcome { #[derive(Debug)] struct BufferedOutcome { - stdout: Vec, - stderr: Vec, + stdout: CapturedOutput, + stderr: CapturedOutput, result: InvocationResult, } impl BufferedOutcome { fn infrastructure(message: String) -> Self { Self { - stdout: Vec::new(), - stderr: Vec::new(), + stdout: CapturedOutput::empty(), + stderr: CapturedOutput::empty(), result: InvocationResult::Infrastructure(message), } } @@ -924,7 +1049,7 @@ impl BufferedOutcome { combine_captured_output( finish_output_reader(stdout_reader, "stdout", OUTPUT_DRAIN_GRACE, boundary), CapturedStream { - bytes: Vec::new(), + output: CapturedOutput::empty(), failure: None, }, InvocationResult::Infrastructure(message), @@ -986,12 +1111,12 @@ mod tests { use std::{io, thread}; use super::{ - BufferedOutcome, CapturedProcess, CapturedStream, Invocation, InvocationResult, OutputReader, Plan, ReaderCompletion, - RunningWorker, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, - execute_parallel, exit_byte, failure_stops_launching, finish_output_reader, panic_description, run_captured, - run_captured_with_spawner, run_streamed, run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, spawn_tree, - terminate_ordinary_child, terminate_ordinary_with, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, - with_cleanup_failure, + BufferedOutcome, CapturedOutput, CapturedProcess, CapturedStream, Invocation, InvocationResult, OutputEmitError, OutputReader, + Plan, ReaderCompletion, RunningWorker, SpillFile, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, + combine_captured_output, display_duration, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, failure_stops_launching, + finish_output_reader, finish_wait_with_cleanup, panic_description, run_captured, run_captured_with_spawner, run_streamed, + run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, spawn_output_reader_with, spawn_tree, terminate_ordinary_child, + terminate_ordinary_with, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, }; const ORDINARY_BOUNDARY: &str = "ordinary process tree"; @@ -1028,6 +1153,69 @@ mod tests { state: u8, } + struct ChunkedReader { + chunks: VecDeque>, + } + + impl io::Read for ChunkedReader { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + let Some(chunk) = self.chunks.pop_front() else { + return Ok(0); + }; + buf[..chunk.len()].copy_from_slice(&chunk); + Ok(chunk.len()) + } + } + + #[derive(Debug)] + struct FaultySpill { + cursor: io::Cursor>, + fail_write: bool, + fail_read: bool, + } + + struct FailingWriter; + + impl io::Write for FailingWriter { + fn write(&mut self, _buf: &[u8]) -> io::Result { + Err(io::Error::other("injected destination write failure")) + } + + fn flush(&mut self) -> io::Result<()> { + Ok(()) + } + } + + impl io::Write for FaultySpill { + fn write(&mut self, buf: &[u8]) -> io::Result { + if self.fail_write { + Err(io::Error::other("injected spill write failure")) + } else { + io::Write::write(&mut self.cursor, buf) + } + } + + fn flush(&mut self) -> io::Result<()> { + io::Write::flush(&mut self.cursor) + } + } + + impl io::Read for FaultySpill { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + if self.fail_read { + Err(io::Error::other("injected spill read failure")) + } else { + io::Read::read(&mut self.cursor, buf) + } + } + } + + impl io::Seek for FaultySpill { + fn seek(&mut self, pos: io::SeekFrom) -> io::Result { + io::Seek::seek(&mut self.cursor, pos) + } + } + impl io::Read for InterruptedThenDataReader { fn read(&mut self, buf: &mut [u8]) -> io::Result { match self.state { @@ -1151,13 +1339,23 @@ mod tests { fn captured(bytes: &[u8], failure: Option<&str>) -> CapturedStream { CapturedStream { - bytes: bytes.to_vec(), + output: CapturedOutput::Memory(bytes.to_vec()), failure: failure.map(str::to_owned), } } - fn poisoned_buffer() -> Arc>> { - let bytes = Arc::new(Mutex::new(b"poisoned bytes".to_vec())); + fn output_bytes(output: &mut CapturedOutput) -> Vec { + let mut bytes = Vec::new(); + match output.emit_to(&mut bytes) { + Ok(()) => bytes, + Err(OutputEmitError::Source(error) | OutputEmitError::Destination(error)) => { + panic!("captured test output cannot be read: {error}") + } + } + } + + fn poisoned_buffer() -> Arc> { + let bytes = Arc::new(Mutex::new(CapturedOutput::Memory(b"poisoned bytes".to_vec()))); let poisoned = Arc::clone(&bytes); let _panic = thread::spawn(move || { let _guard = poisoned.lock().expect("the fresh capture mutex is available"); @@ -1258,12 +1456,12 @@ mod tests { *lock.lock().expect("the test owns the release mutex without panicking") = true; condition.notify_all(); - let captured = result.expect("normal completion must not wait for an escaped descendant to close its pipe"); + let mut captured = result.expect("normal completion must not wait for an escaped descendant to close its pipe"); finisher.join().expect("the bounded finisher thread does not panic"); reader_finished_rx .recv_timeout(Duration::from_secs(1)) .expect("the detached reader exits after the test releases its simulated pipe"); - assert_eq!(captured.bytes, b"captured-before-timeout"); + assert_eq!(output_bytes(&mut captured.output), b"captured-before-timeout"); assert!( captured.failure.as_deref().is_some_and(|failure| failure.contains("remained open")), "grace expiry must be an explicit infrastructure failure" @@ -1288,8 +1486,8 @@ mod tests { .recv_timeout(Duration::from_secs(1)) .expect("the late-data reader starts its blocking read"); - let captured = finish_output_reader(reader, "stdout", Duration::from_millis(25), ORDINARY_BOUNDARY); - assert!(captured.bytes.is_empty()); + let mut captured = finish_output_reader(reader, "stdout", Duration::from_millis(25), ORDINARY_BOUNDARY); + assert!(output_bytes(&mut captured.output).is_empty()); assert!(captured.failure.is_some()); let (lock, condition) = &*release; @@ -1466,6 +1664,29 @@ mod tests { assert!(error.to_string().contains("observation failed")); } + #[test] + fn wait_error_runs_bounded_cleanup_and_preserves_both_errors() { + let mut cleaned = false; + let outcome = finish_wait_with_cleanup(&mut cleaned, Err(io::Error::other("wait failed")), |cleaned| { + *cleaned = true; + Ok(successful_status()) + }); + assert!(cleaned, "wait failure did not invoke bounded cleanup"); + let message = result_infrastructure_message(outcome.result); + assert!(message.contains("wait failed")); + assert!(!message.contains("cleanup also failed")); + + let mut attempted = false; + let outcome = finish_wait_with_cleanup(&mut attempted, Err(io::Error::other("wait failed")), |attempted| { + *attempted = true; + Err(io::Error::other("cleanup failed")) + }); + assert!(attempted, "failed cleanup was not attempted"); + let message = result_infrastructure_message(outcome.result); + assert!(message.contains("wait failed")); + assert!(message.contains("cleanup also failed: cleanup failed")); + } + #[test] fn unsealed_timeout_is_refused_before_spawn() { struct FakePrepared { @@ -1578,40 +1799,40 @@ mod tests { #[test] fn captured_output_combines_every_reader_result_shape() { - let success = combine_captured_output( + let mut success = combine_captured_output( captured(b"stdout", None), captured(b"stderr", None), InvocationResult::Infrastructure("primary".to_owned()), ); - assert_eq!(success.stdout, b"stdout"); - assert_eq!(success.stderr, b"stderr"); + assert_eq!(output_bytes(&mut success.stdout), b"stdout"); + assert_eq!(output_bytes(&mut success.stderr), b"stderr"); assert_eq!(infrastructure_message(success), "primary"); - let stdout_failed = combine_captured_output( + let mut stdout_failed = combine_captured_output( captured(b"partial stdout", Some("stdout failed")), captured(b"stderr", None), InvocationResult::Infrastructure("primary".to_owned()), ); - assert_eq!(stdout_failed.stdout, b"partial stdout"); - assert_eq!(stdout_failed.stderr, b"stderr"); + assert_eq!(output_bytes(&mut stdout_failed.stdout), b"partial stdout"); + assert_eq!(output_bytes(&mut stdout_failed.stderr), b"stderr"); assert_eq!(infrastructure_message(stdout_failed), "primary; stdout failed"); - let stderr_failed = combine_captured_output( + let mut stderr_failed = combine_captured_output( captured(b"stdout", None), captured(b"partial stderr", Some("stderr failed")), InvocationResult::Infrastructure("primary".to_owned()), ); - assert_eq!(stderr_failed.stdout, b"stdout"); - assert_eq!(stderr_failed.stderr, b"partial stderr"); + assert_eq!(output_bytes(&mut stderr_failed.stdout), b"stdout"); + assert_eq!(output_bytes(&mut stderr_failed.stderr), b"partial stderr"); assert_eq!(infrastructure_message(stderr_failed), "primary; stderr failed"); - let both_failed = combine_captured_output( + let mut both_failed = combine_captured_output( captured(b"partial stdout", Some("stdout failed")), captured(b"partial stderr", Some("stderr failed")), InvocationResult::Infrastructure("primary".to_owned()), ); - assert_eq!(both_failed.stdout, b"partial stdout"); - assert_eq!(both_failed.stderr, b"partial stderr"); + assert_eq!(output_bytes(&mut both_failed.stdout), b"partial stdout"); + assert_eq!(output_bytes(&mut both_failed.stderr), b"partial stderr"); assert_eq!(infrastructure_message(both_failed), "primary; stdout failed; stderr failed"); let timed_out = combine_captured_output( @@ -1653,21 +1874,21 @@ mod tests { ); let eof = spawn_output_reader(EofThenPanicReader { reached_eof: false }, "eof-reader").expect("create EOF reader"); - let eof = finish_output_reader(eof, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); - assert!(eof.bytes.is_empty()); + let mut eof = finish_output_reader(eof, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); + assert!(output_bytes(&mut eof.output).is_empty()); assert!(eof.failure.is_none()); let interrupted = spawn_output_reader(InterruptedThenDataReader { state: 0 }, "interrupted-reader").expect("create interrupted reader"); - let interrupted = finish_output_reader(interrupted, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); - assert_eq!(interrupted.bytes, b"after interrupt"); + let mut interrupted = finish_output_reader(interrupted, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); + assert_eq!(output_bytes(&mut interrupted.output), b"after interrupt"); assert!(interrupted.failure.is_none()); let (completion_sender, completion) = mpsc::channel(); let sealed_timeout = OutputReader { thread: thread::spawn(|| {}), completion, - bytes: Arc::new(Mutex::new(Vec::new())), + output: Arc::new(Mutex::new(CapturedOutput::empty())), retaining: Arc::new(AtomicBool::new(true)), }; let sealed_timeout = finish_output_reader(sealed_timeout, "stdout", Duration::ZERO, CONTAINED_BOUNDARY); @@ -1684,7 +1905,7 @@ mod tests { let disconnected = OutputReader { thread: thread::spawn(|| {}), completion, - bytes: Arc::new(Mutex::new(Vec::new())), + output: Arc::new(Mutex::new(CapturedOutput::empty())), retaining: Arc::new(AtomicBool::new(true)), }; let disconnected = finish_output_reader(disconnected, "stderr", Duration::from_secs(1), ORDINARY_BOUNDARY); @@ -1706,11 +1927,11 @@ mod tests { let poisoned = OutputReader { thread: thread::spawn(|| {}), completion, - bytes: poisoned_buffer(), + output: poisoned_buffer(), retaining: Arc::new(AtomicBool::new(true)), }; - let poisoned = finish_output_reader(poisoned, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); - assert_eq!(poisoned.bytes, b"poisoned bytes"); + let mut poisoned = finish_output_reader(poisoned, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); + assert_eq!(output_bytes(&mut poisoned.output), b"poisoned bytes"); assert!( poisoned .failure @@ -1720,17 +1941,137 @@ mod tests { } } + #[test] + #[cfg_attr(miri, ignore = "creates a named system-temporary spill file")] + fn output_spills_after_threshold_and_cleans_up_with_its_outcome() { + let named = tempfile::NamedTempFile::new().expect("create named spill file"); + let path = named.path().to_path_buf(); + let mut named = Some(named); + let reader = spawn_output_reader_with( + ChunkedReader { + chunks: VecDeque::from([b"abcd".to_vec(), b"efgh".to_vec()]), + }, + "spill-threshold-reader", + 4, + Box::new(move || Ok(Box::new(named.take().expect("the capture creates at most one spill file")) as Box)), + ) + .expect("create threshold reader"); + let mut captured = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); + + assert!(captured.output.is_spilled()); + assert!(path.exists(), "the spill must live while its outcome owns it"); + assert_eq!(output_bytes(&mut captured.output), b"abcdefgh"); + + drop(captured); + assert!(!path.exists(), "dropping the outcome must remove its spill file"); + } + + #[test] + fn spill_create_and_write_failures_preserve_memory_and_become_infrastructure_failures() { + for (factory, expected) in [ + ( + Box::new(|| Err(io::Error::other("injected spill create failure"))) as super::SpillFactory, + "injected spill create failure", + ), + ( + Box::new(|| { + Ok(Box::new(FaultySpill { + cursor: io::Cursor::new(Vec::new()), + fail_write: true, + fail_read: false, + }) as Box) + }) as super::SpillFactory, + "injected spill write failure", + ), + ] { + let reader = spawn_output_reader_with( + ChunkedReader { + chunks: VecDeque::from([b"abc".to_vec(), b"def".to_vec()]), + }, + "failing-spill-reader", + 4, + factory, + ) + .expect("create failing spill reader"); + let mut captured_stream = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); + assert_eq!(output_bytes(&mut captured_stream.output), b"abc"); + let outcome = combine_captured_output(captured_stream, captured(b"", None), InvocationResult::Exited(successful_status())); + + assert!(infrastructure_message(outcome).contains(expected)); + } + } + + #[test] + fn spill_read_failure_becomes_an_infrastructure_failure_during_emission() { + let reader = spawn_output_reader_with( + io::Cursor::new(b"abcdefgh".to_vec()), + "read-failing-spill", + 4, + Box::new(|| { + Ok(Box::new(FaultySpill { + cursor: io::Cursor::new(Vec::new()), + fail_write: false, + fail_read: true, + }) as Box) + }), + ) + .expect("create read-failing spill reader"); + let captured_stream = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); + assert!(captured_stream.output.is_spilled()); + let mut outcome = combine_captured_output(captured_stream, captured(b"", None), InvocationResult::Exited(successful_status())); + + emit_buffered(&invocation(&["probe"]), &mut outcome).expect("destination output remains writable"); + + assert!(infrastructure_message(outcome).contains("injected spill read failure")); + } + + #[test] + fn stderr_spill_read_and_destination_write_failures_are_reported() { + let read_failing_spill = || { + CapturedOutput::Spill(Box::new(FaultySpill { + cursor: io::Cursor::new(Vec::new()), + fail_write: false, + fail_read: true, + })) + }; + let mut outcome = BufferedOutcome { + stdout: CapturedOutput::empty(), + stderr: read_failing_spill(), + result: InvocationResult::Exited(successful_status()), + }; + emit_buffered_to(&invocation(&["probe"]), &mut outcome, &mut Vec::new(), &mut Vec::new()).expect("destinations remain writable"); + assert!(infrastructure_message(outcome).contains("failed to read spilled child stderr")); + + let mut stdout_failure = BufferedOutcome { + stdout: CapturedOutput::Memory(b"stdout".to_vec()), + stderr: CapturedOutput::empty(), + result: InvocationResult::Exited(successful_status()), + }; + let error = emit_buffered_to(&invocation(&["probe"]), &mut stdout_failure, &mut FailingWriter, &mut Vec::new()) + .expect_err("stdout destination failure must propagate"); + assert!(error.to_string().contains("injected destination write failure")); + + let mut stderr_failure = BufferedOutcome { + stdout: CapturedOutput::empty(), + stderr: CapturedOutput::Memory(b"stderr".to_vec()), + result: InvocationResult::Exited(successful_status()), + }; + let error = emit_buffered_to(&invocation(&["probe"]), &mut stderr_failure, &mut Vec::new(), &mut FailingWriter) + .expect_err("stderr destination failure must propagate"); + assert!(error.to_string().contains("injected destination write failure")); + } + #[test] fn reader_setup_failure_preserves_cleanup_and_reader_errors() { let successful_reader = spawn_output_reader(io::Cursor::new(b"partial".to_vec()), "successful-reader").expect("create successful reader"); - let outcome = BufferedOutcome::from_reader_failure( + let mut outcome = BufferedOutcome::from_reader_failure( "stderr unavailable".to_owned(), successful_reader, &Ok::<_, io::Error>(()), ORDINARY_BOUNDARY, ); - assert_eq!(outcome.stdout, b"partial"); + assert_eq!(output_bytes(&mut outcome.stdout), b"partial"); assert_eq!(infrastructure_message(outcome), "stderr unavailable"); let failing_reader = spawn_output_reader(FailingReader, "failing-reader").expect("create failing reader"); @@ -1759,8 +2100,8 @@ mod tests { let outcome = TreeOutcome::new(InvocationResult::Infrastructure("outcome".to_owned())); assert_eq!( infrastructure_message(BufferedOutcome { - stdout: Vec::new(), - stderr: Vec::new(), + stdout: CapturedOutput::empty(), + stderr: CapturedOutput::empty(), result: outcome.result, }), "outcome" diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index 76c138eea..3015e086e 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -226,6 +226,19 @@ fn main() { append(&args[3], name); process::exit(if name == "alpha" { 7 } else { 0 }); } + "large-output" => { + let name = &args[2]; + let size = 1_100_000; + let stdout_byte = if name == "alpha" { b'A' } else { b'B' }; + let stderr_byte = if name == "alpha" { b'C' } else { b'D' }; + let mut stdout = std::io::stdout().lock(); + writeln!(stdout, "{name}:stdout").expect("write stdout header"); + stdout.write_all(&vec![stdout_byte; size]).expect("write large stdout"); + let mut stderr = std::io::stderr().lock(); + writeln!(stderr, "{name}:stderr").expect("write stderr header"); + stderr.write_all(&vec![stderr_byte; size]).expect("write large stderr"); + process::exit(if name == "alpha" { 7 } else { 0 }); + } "timeout-fail-fast" => { if args[2] == "alpha" { thread::sleep(Duration::from_secs(5)); @@ -1374,6 +1387,29 @@ fn parallel_keep_going_runs_the_complete_plan() { } } +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn parallel_spills_large_stdout_and_stderr_without_truncating_plan_order() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let output = each(&manifest) + .args(["-p", "alpha", "-p", "beta", "--jobs", "2", "--keep-going", "--"]) + .arg(probe) + .args(["large-output", "{name}"]) + .output() + .expect("run cargo-each with large output"); + + assert_eq!(output.status.code(), Some(1)); + let stdout = String::from_utf8(output.stdout).expect("probe stdout is ASCII"); + let stderr = String::from_utf8(output.stderr).expect("probe stderr is ASCII"); + assert!(stdout.find("alpha:stdout").expect("alpha stdout header") < stdout.find("beta:stdout").expect("beta stdout header")); + assert!(stderr.find("alpha:stderr").expect("alpha stderr header") < stderr.find("beta:stderr").expect("beta stderr header")); + assert_eq!(stdout.bytes().filter(|byte| *byte == b'A').count(), 1_100_000); + assert_eq!(stdout.bytes().filter(|byte| *byte == b'B').count(), 1_100_000); + assert_eq!(stderr.bytes().filter(|byte| *byte == b'C').count(), 1_100_000); + assert_eq!(stderr.bytes().filter(|byte| *byte == b'D').count(), 1_100_000); +} + #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] #[test] fn parallel_without_timeout_preserves_ordinary_background_descendants() { From 9ad57b850533858d86af8ca6e43e230af19c0add Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Sat, 12 Sep 2026 06:55:21 +0200 Subject: [PATCH 07/37] test(cargo-each): handle per-launch containment Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 17 +++++++--- crates/cargo-each/tests/cli.rs | 58 ++++++++++++++++------------------ 2 files changed, 40 insertions(+), 35 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index b7197776e..b34605aef 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -357,13 +357,15 @@ fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedO "injected worker panic" ); - run_captured_with_spawner(invocation, timeout, spawn_sealed_tree) + run_captured_with_spawner(invocation, timeout, |command| { + spawn_sealed_tree(command).map(CapturedProcess::Contained) + }) } fn run_captured_with_spawner( invocation: &Invocation, timeout: Option, - timed_spawner: impl FnOnce(Command) -> Result, + timed_spawner: impl FnOnce(Command) -> Result, ) -> BufferedOutcome { let (program, mut command) = match command_for(invocation) { Ok(command) => command, @@ -371,7 +373,7 @@ fn run_captured_with_spawner( }; let _ = command.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); let process = match timeout { - Some(_) => timed_spawner(command).map(CapturedProcess::Contained), + Some(_) => timed_spawner(command), None => command .spawn() .map(|child| CapturedProcess::Ordinary(Some(child))) @@ -1287,6 +1289,13 @@ mod tests { } } + fn spawn_ordinary_capture(mut command: Command) -> Result { + command + .spawn() + .map(|child| CapturedProcess::Ordinary(Some(child))) + .map_err(|error| error.to_string()) + } + fn result_infrastructure_message(result: InvocationResult) -> String { let InvocationResult::Infrastructure(message) = result else { panic!("the test expects an infrastructure outcome"); @@ -2223,7 +2232,7 @@ mod tests { let outcome = run_captured_with_spawner( &labelled_invocation(label, &["rustc", "--version"]), Some(Duration::from_secs(1)), - spawn_tree, + spawn_ordinary_capture, ); assert!(infrastructure_message(outcome).contains(expected), "contained {label}"); } diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index 3015e086e..8e5315ef3 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -7,7 +7,6 @@ use std::fs; use std::path::{Path, PathBuf}; -use std::sync::OnceLock; use assert_cmd::Command; use predicates::prelude::*; @@ -112,14 +111,11 @@ fn each(manifest: &Path) -> Command { cmd } -fn sealed_containment_available() -> bool { - static AVAILABLE: OnceLock = OnceLock::new(); +const TIMEOUT_REFUSAL: &str = + "timeout requires sealed process-tree containment, but this host only provides best-effort containment; the child was not started"; - *AVAILABLE.get_or_init(|| cargo_gamma_process::containment().is_ok()) -} - -fn timeout_refusal() -> impl Predicate { - predicate::str::contains("timeout requires sealed process-tree containment").and(predicate::str::contains("child was not started")) +fn is_timeout_refusal(output: &std::process::Output) -> bool { + output.status.code() == Some(2) && String::from_utf8_lossy(&output.stderr).contains(TIMEOUT_REFUSAL) } fn rust_version_fixture(root_floor: Option<&str>, members: &[(&str, Option<&str>)]) -> (TempDir, PathBuf) { @@ -1435,17 +1431,18 @@ fn sequential_timeout_fail_fast_does_not_run_later_members() { let (tmp, manifest) = fixture(); let probe = compile_execution_probe(tmp.path()); let later_marker = tmp.path().join("later-invocation"); - let assertion = each(&manifest) + let output = each(&manifest) .args(["-p", "alpha", "-p", "beta", "--timeout", "50ms", "--"]) .arg(probe) .args(["timeout-fail-fast", "{name}"]) .arg(&later_marker) - .assert() - .failure(); - if sealed_containment_available() { - assertion.code(1).stderr(predicate::str::contains("timed out after 50ms")); + .output() + .expect("run cargo-each timeout fail-fast"); + if is_timeout_refusal(&output) { + assert!(!later_marker.exists(), "pre-spawn refusal must not launch any member"); } else { - assertion.code(2).stderr(timeout_refusal()); + assert_eq!(output.status.code(), Some(1), "stderr: {}", String::from_utf8_lossy(&output.stderr)); + assert!(String::from_utf8_lossy(&output.stderr).contains("timed out after 50ms")); } assert!( !later_marker.exists(), @@ -1459,24 +1456,24 @@ fn sequential_timeout_keep_going_runs_later_members() { let (tmp, manifest) = fixture(); let probe = compile_execution_probe(tmp.path()); let later_marker = tmp.path().join("later-invocation"); - let assertion = each(&manifest) + let output = each(&manifest) .args(["-p", "alpha", "-p", "beta", "--timeout", "50ms", "--keep-going", "--"]) .arg(probe) .args(["timeout-keep-going", "{name}"]) .arg(&later_marker) - .assert() - .failure(); - if sealed_containment_available() { - assertion.code(1).stderr(predicate::str::contains("timed out after 50ms")); + .output() + .expect("run cargo-each timeout keep-going"); + if is_timeout_refusal(&output) { assert!( - later_marker.exists(), - "--keep-going must launch the member after a timed-out invocation" + !later_marker.exists(), + "unsealed containment must refuse every timed child before spawn" ); } else { - assertion.code(1).stderr(timeout_refusal()); + assert_eq!(output.status.code(), Some(1), "stderr: {}", String::from_utf8_lossy(&output.stderr)); + assert!(String::from_utf8_lossy(&output.stderr).contains("timed out after 50ms")); assert!( - !later_marker.exists(), - "unsealed containment must refuse every timed child before spawn" + later_marker.exists(), + "--keep-going must launch the member after a timed-out invocation" ); } } @@ -1487,17 +1484,16 @@ fn timeout_terminates_the_complete_process_tree() { let (tmp, manifest) = fixture(); let probe = compile_execution_probe(tmp.path()); let marker = tmp.path().join("grandchild-survived"); - let assertion = each(&manifest) + let output = each(&manifest) .args(["-p", "alpha", "--jobs", "2", "--timeout", "50ms", "--"]) .arg(probe) .arg("tree-parent") .arg(&marker) - .assert() - .failure(); - if sealed_containment_available() { - assertion.code(1).stderr(predicate::str::contains("timed out after 50ms")); - } else { - assertion.code(2).stderr(timeout_refusal()); + .output() + .expect("run cargo-each tree timeout"); + if !is_timeout_refusal(&output) { + assert_eq!(output.status.code(), Some(1), "stderr: {}", String::from_utf8_lossy(&output.stderr)); + assert!(String::from_utf8_lossy(&output.stderr).contains("timed out after 50ms")); } std::thread::sleep(std::time::Duration::from_millis(700)); assert!( From 2b0d24512e73c28d1e263fde69e958896790500a Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 15 Sep 2026 17:07:33 +0200 Subject: [PATCH 08/37] fix(cargo-each): bound process and output cleanup Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99e3-4546-847a-e30dd5cb18a4 --- .spelling | 2 + crates/cargo-each/README.md | 29 +- crates/cargo-each/docs/design/README.md | 62 ++- crates/cargo-each/src/main.rs | 29 +- crates/cargo-each/src/run.rs | 471 +++++++++++++++--- crates/cargo-each/src/select.rs | 2 + crates/cargo-each/tests/cli.rs | 68 ++- crates/cargo-gamma-process/docs/DESIGN.md | 7 +- .../docs/IMPLEMENTATION.md | 8 +- crates/cargo-gamma-process/src/lib.rs | 5 +- .../cargo-gamma-process/src/process_tree.rs | 125 ++++- crates/cargo-gamma-unsafe/Cargo.toml | 1 + crates/cargo-gamma-unsafe/src/lib.rs | 5 +- crates/cargo-gamma-unsafe/src/pipe.rs | 205 ++++++++ 14 files changed, 869 insertions(+), 150 deletions(-) create mode 100644 crates/cargo-gamma-unsafe/src/pipe.rs diff --git a/.spelling b/.spelling index dfc15ff18..1f4c0908b 100644 --- a/.spelling +++ b/.spelling @@ -931,3 +931,5 @@ symlinked uninherited closers unmanaged +interruptible +nonblocking diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index 90c07a3b1..30d3b8502 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -46,7 +46,8 @@ directly (argv, not a shell string) after substituting placeholders. package name, a `name@version` spec, or a Unix glob (`tokio-*`). * `--package-file ` — read package specs from a UTF-8 file, one per nonempty line. Repeatable; specs are unioned with `--package`. An empty - file explicitly selects no members. + file explicitly selects no members. One leading UTF-8 byte-order mark is + ignored. * `--workspace` / `--all` — select every workspace member. * `--exclude ` — drop a member (with `--workspace`). Repeatable. * `--none` — explicitly select zero members (a no-op that exits 0). @@ -129,24 +130,28 @@ are emitted in deterministic plan order. Fail-fast stops launching after the first observed failure, waits for running work, and chooses the final failure by plan order. `--keep-going` runs the complete plan. Worker panics and unexpected worker-channel disconnections become infrastructure-failure -outcomes instead of blocking the scheduler. Without `--timeout`, parallel -commands retain ordinary direct-child semantics and do not kill background -descendants. Each output stream retains at most 1 MiB in memory before +outcomes instead of blocking the scheduler. Worker launch failures retain +output already collected at earlier plan indices. Without `--timeout`, +parallel commands retain ordinary direct-child semantics and do not kill +background descendants. Each output stream retains at most 1 MiB in memory before spilling to a unique system-temporary file owned by the invocation outcome; spill failures are infrastructure failures and spill files are removed by RAII after deterministic plan-order emission. -Output drain is bounded after every completion. Readers get one second to -observe EOF; grace expiry preserves partial bytes and becomes an explicit -infrastructure failure. Timed-out tree termination likewise gets a bounded -250 ms leader-reap grace, after which the leader handle is detached so no -wait or Drop path can defeat the timeout. +Reader failures are observed while the child is running and trigger bounded +termination. Output drain is bounded after every completion: readers get +one second to observe EOF, then readiness-polling capture is cancelled and +joined while partial bytes become an explicit infrastructure failure. +Timed-out tree termination likewise gets a bounded 250 ms leader-reap grace, +after which the leader handle moves to a shared detached reaper so no wait +or Drop path can defeat the timeout without abandoning reap ownership. Child commands inherit `PATH` explicitly. On Windows this makes relative program lookup honor the inherited `PATH` order instead of preferring an unrelated executable beside `cargo-each`. -Otherwise the exit code is the first failing command code (fail-fast), -`1` under `--keep-going` if any command failed, or `2` for a `cargo-each` -usage error (unknown selector, bad filter expression, misused placeholder). +Exit `0` means all work succeeded or there was no work. In fail-fast mode a +command failure returns its code, a timeout returns `1`, and usage, +configuration, spawn, or post-spawn infrastructure failures return `2`. +Under `--keep-going`, any failure maps the aggregate result to `1`. ## Examples diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index 76d1ac337..d339c5ac5 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -174,10 +174,10 @@ beyond placeholder substitution. | Flag | Meaning | |------|---------| | `-p`, `--package ` | Select a member. Repeatable. `SPEC` is a package name, a `name@version` spec, or a Unix glob (`tokio-*`), matching `cargo-coverage-gate`'s existing `-p` idiom. | -| `--package-file ` | Read package specs from a UTF-8 file, one spec per nonempty line. Repeatable; specs are unioned with `--package`. A present empty file is an explicit empty selection. | +| `--package-file ` | Read package specs from a UTF-8 file, one spec per nonempty line. A single leading UTF-8 byte-order mark is ignored. Repeatable; specs are unioned with `--package`. A present empty file is an explicit empty selection. | | `--workspace`, `--all` | Select every workspace member. | | `--exclude ` | Remove a member from the selection (requires `--workspace`). Repeatable. | -| `--none` | Explicitly select zero members. Resolves to an empty set (a no-op, exit 0). Emitted by the impact hand-off when a tier is empty; replaces the `--skip` sentinel. | +| `--none` | Explicitly select zero members. Resolves to an empty set (a no-op, exit 0). | A package file contains only package specs, not command-line tokens, comments, an impact-tier name, or policy. `foo@1.2.3` has the same meaning whether it came @@ -290,12 +290,12 @@ no-op. ## 5. Semantics - **Exit codes.** `0` when every executed command succeeded *or* the set was - empty; the failing command's code (fail-fast) or `1` (`--keep-going` with any - failure — including a command that could not be spawned) otherwise; `2` for a - `cargo-each` usage/configuration error (unknown selector, bad filter expression, - unknown target kind, invalid mode combination, misused placeholder, - `--chdir` with `--once`, or — in fail-fast mode — a command that could not - be spawned at all). + empty. In fail-fast mode, a command failure returns that command's code, a + timeout returns `1`, and a post-spawn infrastructure failure (including + output capture, drain, worker, wait, or cleanup failure) returns `2`. + Pre-execution usage/configuration and spawn failures also return `2`. Under + `--keep-going`, any command, timeout, spawn, or infrastructure failure maps + the aggregate result to `1`. - **Empty set is success.** Both an empty selection (`--none`, or an impact variable that resolved to nothing) and an empty *filtered* set exit 0 after a one-line note to stderr. This is what lets callers drop their `--skip` guards. @@ -309,8 +309,11 @@ no-op. final failure is chosen by plan order, not scheduler timing. A worker panic is converted into an infrastructure-failure outcome; each worker has a dedicated completion channel, so an unexpected exit is observable as - disconnection rather than leaving the scheduler blocked forever. Without - `--timeout`, parallel commands use the ordinary direct-child lifecycle: + disconnection rather than leaving the scheduler blocked forever. A + worker-thread launch failure is represented as an infrastructure outcome at + that invocation's plan index, so output already collected from earlier + invocations is still emitted. Without `--timeout`, parallel commands use the + ordinary direct-child lifecycle: cargo-each waits for the launched leader but does not contain or kill background descendants. Buffering is memory-bounded per stream: after 1 MiB, output spills to a unique file in the system temporary directory. The @@ -318,20 +321,25 @@ no-op. so every success, failure, and panic path removes it through RAII. Spill creation, write, seek, or read failures are infrastructure failures; output is never intentionally truncated on a successful path. -- **Output drain is bounded after every completion.** Readers get one second - after the leader completes to observe EOF. Complete output is preserved when - both pipes close within that grace. If a background or escaped descendant - keeps a pipe open, capture stops retaining new bytes, emits the partial bytes - already buffered, detaches the blocked reader, and reports an explicit - infrastructure failure rather than hanging or silently truncating. +- **Output capture and drain are bounded.** Reader failures are observed while + the leader is still running; cargo-each terminates the invocation and reports + the infrastructure failure instead of waiting indefinitely with an + unconsumed pipe. After leader completion, readers get one second to observe + EOF. Complete output is preserved when both pipes close within that grace. + If a background or escaped descendant keeps a pipe open, capture stops + retaining new bytes, cancels and joins the readiness-polling reader within a + bounded grace, emits the partial bytes already buffered, and reports an + explicit infrastructure failure rather than hanging, silently truncating, or + accumulating detached reader threads. - **Timeouts terminate trees.** A timed-out command is a failure. cargo-each terminates the child process tree rather than only the immediate process, so compiler or test descendants cannot continue mutating the target directory after cargo-each returns. Termination gets a bounded 250 ms grace to reap the leader. If signalling fails and the leader is still running at that deadline, - its handle is detached so neither termination nor Drop can defeat the - invocation timeout; cargo-each reports the infrastructure failure. A timeout - is accepted only when launch preparation reports a sealed cgroup or job + its handle is transferred to a shared detached reaper so neither termination + nor Drop can defeat the invocation timeout while the leader still has a + wait/reap owner; cargo-each reports the infrastructure failure. A timeout is + accepted only when launch preparation reports a sealed cgroup or job boundary. On a host with best-effort process-group containment, cargo-each reports that timeout is unsupported and does not spawn the command. - **Child executable resolution follows `PATH`.** `cargo-each` explicitly @@ -342,12 +350,16 @@ no-op. ## 6. How it simplifies cargo-anvil -The recipes stop parsing impact selections and metadata by hand. cargo-delta -writes one `name@version` package file per tier under -`target/anvil/impact/`. A cargo-each check supplies the appropriate file -directly. An empty tier is an empty file and therefore a successful no-op. -Illustrative before/after (the recipe keeps its own setup and `anvil-impact` -dependencies; only the selection spine changes): +A planned cargo-anvil adoption can stop parsing impact selections and metadata +by hand, but the examples below are not usable with the current producer yet. +Today it writes `include_.txt` values containing `--package` tokens or the +`--workspace` / `--skip` sentinels, all of which package-file validation +intentionally rejects. The producer must first change to write one +`name@version` package spec per line under `target/anvil/impact/`, with an empty +file for an empty tier. After that producer change, a cargo-each check can +supply the appropriate file directly and get a successful no-op for an empty +tier. Illustrative planned before/after (the recipe keeps its own setup and +`anvil-impact` dependencies; only the selection spine changes): **clippy** (affected tier, single invocation): diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 89a9ab344..5d9e0beb2 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -36,7 +36,8 @@ //! package name, a `name@version` spec, or a Unix glob (`tokio-*`). //! - `--package-file ` — read package specs from a UTF-8 file, one per //! nonempty line. Repeatable; specs are unioned with `--package`. An empty -//! file explicitly selects no members. +//! file explicitly selects no members. One leading UTF-8 byte-order mark is +//! ignored. //! - `--workspace` / `--all` — select every workspace member. //! - `--exclude ` — drop a member (with `--workspace`). Repeatable. //! - `--none` — explicitly select zero members (a no-op that exits 0). @@ -119,24 +120,28 @@ //! the first observed failure, waits for running work, and chooses the final //! failure by plan order. `--keep-going` runs the complete plan. Worker panics //! and unexpected worker-channel disconnections become infrastructure-failure -//! outcomes instead of blocking the scheduler. Without `--timeout`, parallel -//! commands retain ordinary direct-child semantics and do not kill background -//! descendants. Each output stream retains at most 1 MiB in memory before +//! outcomes instead of blocking the scheduler. Worker launch failures retain +//! output already collected at earlier plan indices. Without `--timeout`, +//! parallel commands retain ordinary direct-child semantics and do not kill +//! background descendants. Each output stream retains at most 1 MiB in memory before //! spilling to a unique system-temporary file owned by the invocation outcome; //! spill failures are infrastructure failures and spill files are removed by //! RAII after deterministic plan-order emission. //! -//! Output drain is bounded after every completion. Readers get one second to -//! observe EOF; grace expiry preserves partial bytes and becomes an explicit -//! infrastructure failure. Timed-out tree termination likewise gets a bounded -//! 250 ms leader-reap grace, after which the leader handle is detached so no -//! wait or Drop path can defeat the timeout. +//! Reader failures are observed while the child is running and trigger bounded +//! termination. Output drain is bounded after every completion: readers get +//! one second to observe EOF, then readiness-polling capture is cancelled and +//! joined while partial bytes become an explicit infrastructure failure. +//! Timed-out tree termination likewise gets a bounded 250 ms leader-reap grace, +//! after which the leader handle moves to a shared detached reaper so no wait +//! or Drop path can defeat the timeout without abandoning reap ownership. //! Child commands inherit `PATH` explicitly. On Windows this makes relative //! program lookup honor the inherited `PATH` order instead of preferring an //! unrelated executable beside `cargo-each`. -//! Otherwise the exit code is the first failing command code (fail-fast), -//! `1` under `--keep-going` if any command failed, or `2` for a `cargo-each` -//! usage error (unknown selector, bad filter expression, misused placeholder). +//! Exit `0` means all work succeeded or there was no work. In fail-fast mode a +//! command failure returns its code, a timeout returns `1`, and usage, +//! configuration, spawn, or post-spawn infrastructure failures return `2`. +//! Under `--keep-going`, any failure maps the aggregate result to `1`. //! //! # Examples //! diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index b34605aef..a1766acaf 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -14,7 +14,7 @@ use std::sync::{Arc, Mutex, mpsc}; use std::time::{Duration, Instant}; use std::{fmt, thread}; -use cargo_gamma_process::{MemoryRequest, PreparedCommand, ProcessTree, prepare}; +use cargo_gamma_process::{InterruptiblePipe, MemoryRequest, PreparedCommand, ProcessTree, prepare, reap_later}; use cargo_metadata::TargetKind; use ohno::{AppError, IntoAppError}; @@ -32,6 +32,8 @@ const WORKER_PANIC_TEST_PROGRAM: &str = "__cargo_each_injected_worker_panic"; const WORKER_SPAWN_ERROR_TEST_PROGRAM: &str = "__cargo_each_injected_worker_spawn_error"; const TERMINATION_GRACE: Duration = Duration::from_millis(250); const OUTPUT_DRAIN_GRACE: Duration = Duration::from_secs(1); +const OUTPUT_READER_CANCEL_GRACE: Duration = Duration::from_millis(100); +const OUTPUT_READER_POLL: Duration = Duration::from_millis(5); const OUTPUT_MEMORY_LIMIT: usize = 1_048_576; pub(crate) fn run(args: &EachArgs) -> Result { @@ -192,47 +194,40 @@ fn execute_parallel(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: let worker_count = jobs.get().min(invocations.len()).min(cargo_gamma_process::capacity().max(1)); let mut pending: VecDeque<(usize, Invocation)> = invocations.iter().cloned().enumerate().collect(); let mut workers = Vec::with_capacity(worker_count); + let mut outcomes = Vec::with_capacity(invocations.len()); let mut stop_launching = false; - let mut launch_error = None; - for (index, invocation) in pending.drain(..worker_count) { - match spawn_worker(index, invocation, timeout) { - Ok(worker) => { - workers.push(worker); - } - Err(error) => { - launch_error = Some(error); - stop_launching = true; + while !workers.is_empty() || (!stop_launching && !pending.is_empty()) { + while !stop_launching && workers.len() < worker_count { + let Some((index, invocation)) = pending.pop_front() else { break; + }; + match spawn_worker(index, invocation, timeout) { + Ok(worker) => workers.push(worker), + Err(error) => { + outcomes.push(IndexedOutcome { + index, + outcome: BufferedOutcome::infrastructure(format!("failed to create cargo-each worker thread: {error}")), + }); + if failure_stops_launching(keep_going, true) { + stop_launching = true; + } + } } } - } - let mut outcomes = Vec::with_capacity(invocations.len()); - while let Some(outcome) = wait_for_worker(&mut workers) { + let Some(outcome) = wait_for_worker(&mut workers) else { + if pending.is_empty() || stop_launching { + break; + } + continue; + }; if failure_stops_launching(keep_going, outcome.outcome.result.failed()) { stop_launching = true; } outcomes.push(outcome); - - if stop_launching { - continue; - } - let Some((index, invocation)) = pending.pop_front() else { - continue; - }; - match spawn_worker(index, invocation, timeout) { - Ok(worker) => workers.push(worker), - Err(error) => { - launch_error = Some(error); - stop_launching = true; - } - } } - if let Some(error) = launch_error { - return Err(error).into_app_err("failed to create cargo-each worker thread"); - } outcomes.sort_by_key(|outcome| outcome.index); for indexed in &mut outcomes { @@ -397,10 +392,10 @@ fn run_captured_with_spawner( let cleanup = process.terminate_bounded(); return BufferedOutcome::infrastructure(with_cleanup_failure("failed to capture child stdout".to_owned(), &cleanup)); }; - let stdout_reader = match if capture_fault == Some(CaptureFault::StdoutReader) { + let mut stdout_reader = match if capture_fault == Some(CaptureFault::StdoutReader) { Err(io::Error::other("injected stdout reader failure")) } else { - spawn_output_reader(stdout, "cargo-each-stdout") + spawn_child_output_reader(stdout, "cargo-each-stdout") } { Ok(reader) => reader, Err(error) => { @@ -417,10 +412,10 @@ fn run_captured_with_spawner( let cleanup = process.terminate_bounded(); return BufferedOutcome::from_reader_failure("failed to capture child stderr".to_owned(), stdout_reader, &cleanup, drain_boundary); }; - let stderr_reader = match if capture_fault == Some(CaptureFault::StderrReader) { + let mut stderr_reader = match if capture_fault == Some(CaptureFault::StderrReader) { Err(io::Error::other("injected stderr reader failure")) } else { - spawn_output_reader(stderr, "cargo-each-stderr") + spawn_child_output_reader(stderr, "cargo-each-stderr") } { Ok(reader) => reader, Err(error) => { @@ -434,7 +429,7 @@ fn run_captured_with_spawner( } }; - let process_outcome = process.wait(timeout, capture_fault); + let process_outcome = process.wait(timeout, capture_fault, &mut stdout_reader, &mut stderr_reader); drop(process); let (stdout, stderr) = finish_output_readers(stdout_reader, stderr_reader, OUTPUT_DRAIN_GRACE, drain_boundary); combine_captured_output(stdout, stderr, process_outcome.result) @@ -546,14 +541,19 @@ impl CapturedProcess { .take() .ok_or_else(|| io::Error::other("ordinary child was already reaped or detached"))?; let result = terminate_ordinary_child(&mut child, TERMINATION_GRACE); - drop(child); - result + finish_ordinary_termination(child, result) } Self::Contained(tree) => tree.terminate_bounded(TERMINATION_GRACE), } } - fn wait(&mut self, timeout: Option, capture_fault: Option) -> TreeOutcome { + fn wait( + &mut self, + timeout: Option, + capture_fault: Option, + stdout: &mut OutputReader, + stderr: &mut OutputReader, + ) -> TreeOutcome { match (self, timeout) { (Self::Ordinary(child), None) => { let Some(mut child) = child.take() else { @@ -561,14 +561,32 @@ impl CapturedProcess { "ordinary child was already reaped or detached".to_owned(), )); }; - let waited = if capture_fault == Some(CaptureFault::WaitFailure) { - Err(io::Error::other("injected child wait failure")) - } else { - child.wait() - }; - finish_wait_with_cleanup(&mut child, waited, |child| terminate_ordinary_child(child, TERMINATION_GRACE)) + let outcome = wait_for_captured_process( + &mut child, + None, + stdout, + stderr, + "wait for child process", + |child| { + if capture_fault == Some(CaptureFault::WaitFailure) { + Err(io::Error::other("injected child wait failure")) + } else { + child.try_wait() + } + }, + |child| terminate_ordinary_child(child, TERMINATION_GRACE), + ); + finish_ordinary_wait(child, outcome) } - (Self::Contained(tree), Some(timeout)) => wait_for_tree(tree, timeout), + (Self::Contained(tree), Some(timeout)) => wait_for_captured_process( + tree, + Some(timeout), + stdout, + stderr, + "observe child process tree", + ProcessTree::observe, + |tree| tree.terminate_bounded(TERMINATION_GRACE), + ), (Self::Ordinary(_), Some(_)) | (Self::Contained(_), None) => TreeOutcome::new(InvocationResult::Infrastructure( "internal capture mode did not match timeout configuration".to_owned(), )), @@ -576,6 +594,86 @@ impl CapturedProcess { } } +fn finish_ordinary_termination(mut child: Child, result: io::Result) -> io::Result { + match child.try_wait() { + Ok(Some(_status)) => result, + Ok(None) | Err(_) => match reap_later(child) { + Ok(()) => result, + Err(reaper) => match result { + Ok(_status) => Err(io::Error::other(format!( + "the detached child reaper could not be started: {reaper}" + ))), + Err(error) => Err(io::Error::new( + error.kind(), + format!("{error}; the detached child reaper could not be started: {reaper}"), + )), + }, + }, + } +} + +fn finish_ordinary_wait(mut child: Child, mut outcome: TreeOutcome) -> TreeOutcome { + if !matches!(child.try_wait(), Ok(Some(_status))) + && let Err(error) = reap_later(child) + { + outcome.result = add_infrastructure_failure(outcome.result, format!("the detached child reaper could not be started: {error}")); + } + outcome +} + +fn wait_for_captured_process( + control: &mut T, + timeout: Option, + stdout: &mut OutputReader, + stderr: &mut OutputReader, + operation: &str, + mut observe: impl FnMut(&mut T) -> io::Result>, + mut terminate: impl FnMut(&mut T) -> io::Result, +) -> TreeOutcome { + let started = Instant::now(); + loop { + let reader_failure = [stdout.take_failure("stdout"), stderr.take_failure("stderr")] + .into_iter() + .flatten() + .collect::>() + .join("; "); + if !reader_failure.is_empty() { + let cleanup = terminate(control); + return TreeOutcome::new(InvocationResult::Infrastructure(with_cleanup_failure(reader_failure, &cleanup))); + } + + match observe(control) { + Ok(Some(status)) => return TreeOutcome::new(InvocationResult::Exited(status)), + Ok(None) => {} + Err(error) => { + let cleanup = terminate(control); + return TreeOutcome::new(InvocationResult::Infrastructure(with_cleanup_failure( + format!("failed to {operation}: {error}"), + &cleanup, + ))); + } + } + + if let Some(timeout) = timeout + && timeout.checked_sub(started.elapsed()).is_none() + { + return match terminate(control) { + Ok(_) => TreeOutcome::new(InvocationResult::TimedOut(timeout)), + Err(error) => TreeOutcome::new(InvocationResult::Infrastructure(format!( + "invocation timed out after {}; process-tree termination failed: {error}", + display_duration(timeout) + ))), + }; + } + + let pause = timeout + .and_then(|timeout| timeout.checked_sub(started.elapsed())) + .map_or(Duration::from_millis(10), |remaining| remaining.min(Duration::from_millis(10))); + thread::sleep(pause); + } +} + +#[cfg(test)] fn finish_wait_with_cleanup( control: &mut T, waited: io::Result, @@ -734,11 +832,51 @@ fn capture_fault(invocation: &Invocation) -> Option { } } +#[cfg(unix)] +fn spawn_child_output_reader(stream: R, name: &'static str) -> io::Result +where + R: io::Read + std::os::fd::AsRawFd + Send + 'static, +{ + spawn_output_reader_inner( + InterruptiblePipe::new(stream)?, + name, + OUTPUT_MEMORY_LIMIT, + Box::new(|| tempfile::tempfile().map(|file| Box::new(file) as Box)), + ) +} + +#[cfg(windows)] +fn spawn_child_output_reader(stream: R, name: &'static str) -> io::Result +where + R: io::Read + std::os::windows::io::AsRawHandle + Send + 'static, +{ + spawn_output_reader_inner( + InterruptiblePipe::new(stream)?, + name, + OUTPUT_MEMORY_LIMIT, + Box::new(|| tempfile::tempfile().map(|file| Box::new(file) as Box)), + ) +} + +#[cfg(not(any(unix, windows)))] +fn spawn_child_output_reader(stream: R, name: &'static str) -> io::Result +where + R: io::Read + Send + 'static, +{ + spawn_output_reader_inner( + InterruptiblePipe::new(stream)?, + name, + OUTPUT_MEMORY_LIMIT, + Box::new(|| tempfile::tempfile().map(|file| Box::new(file) as Box)), + ) +} + +#[cfg(test)] fn spawn_output_reader(stream: R, name: &'static str) -> io::Result where R: io::Read + Send + 'static, { - spawn_output_reader_with( + spawn_output_reader_inner( stream, name, OUTPUT_MEMORY_LIMIT, @@ -746,7 +884,15 @@ where ) } -fn spawn_output_reader_with( +#[cfg(test)] +fn spawn_output_reader_with(stream: R, name: &'static str, memory_limit: usize, spill_factory: SpillFactory) -> io::Result +where + R: io::Read + Send + 'static, +{ + spawn_output_reader_inner(stream, name, memory_limit, spill_factory) +} + +fn spawn_output_reader_inner( mut stream: R, name: &'static str, memory_limit: usize, @@ -772,6 +918,12 @@ where Ok(0) => return Ok(()), Ok(read) => break read, Err(error) if error.kind() == io::ErrorKind::Interrupted => {} + Err(error) if error.kind() == io::ErrorKind::WouldBlock => { + if !capture_enabled.load(Ordering::Acquire) { + return Ok(()); + } + thread::sleep(OUTPUT_READER_POLL); + } Err(error) => return Err(error), } }; @@ -796,35 +948,66 @@ where completion, output, retaining, + reported: None, + failure_claimed: false, }) } -fn finish_output_reader(reader: OutputReader, stream: &str, grace: Duration, boundary: &str) -> CapturedStream { - let OutputReader { - thread, - completion, - output, - retaining, - } = reader; - let failure = match completion.recv_timeout(grace) { - Ok(ReaderCompletion::Finished(Ok(()))) => None, - Ok(ReaderCompletion::Finished(Err(error))) => Some(format!("failed to read child {stream}: {error}")), - Ok(ReaderCompletion::Panicked(message)) => Some(format!("child {stream} reader thread panicked: {message}")), - Err(mpsc::RecvTimeoutError::Timeout) => { - retaining.store(false, Ordering::Release); - Some(format!( - "child {stream} remained open for more than {} ms after the {boundary} completed; partial output was retained", - grace.as_millis() - )) - } - Err(mpsc::RecvTimeoutError::Disconnected) => { - retaining.store(false, Ordering::Release); - Some(format!("child {stream} reader exited without reporting completion")) - } +fn finish_output_reader(mut reader: OutputReader, stream: &str, grace: Duration, boundary: &str) -> CapturedStream { + let mut thread_finished = false; + let completion = reader + .reported + .take() + .unwrap_or_else(|| match reader.completion.recv_timeout(grace) { + Ok(completion) => completion, + Err(mpsc::RecvTimeoutError::Disconnected) => ReaderCompletion::Disconnected, + Err(mpsc::RecvTimeoutError::Timeout) => { + reader.retaining.store(false, Ordering::Release); + ReaderCompletion::DrainTimedOut + } + }); + let mut failure = if reader.failure_claimed { + None + } else { + reader_failure(&completion, stream, grace, boundary) }; - drop(thread); - match output.lock() { + if matches!(completion, ReaderCompletion::DrainTimedOut) { + match reader.completion.recv_timeout(OUTPUT_READER_CANCEL_GRACE) { + Ok(_) | Err(mpsc::RecvTimeoutError::Disconnected) => { + thread_finished = true; + } + Err(mpsc::RecvTimeoutError::Timeout) => { + let cancellation = format!( + "child {stream} reader did not stop within {} ms after capture was cancelled", + OUTPUT_READER_CANCEL_GRACE.as_millis() + ); + failure = Some(match failure { + Some(failure) => format!("{failure}; {cancellation}"), + None => cancellation, + }); + } + } + } else { + thread_finished = true; + } + + if thread_finished { + if let Err(payload) = reader.thread.join() { + let panic = format!( + "child {stream} reader thread panicked after reporting completion: {}", + panic_description(payload.as_ref()) + ); + failure = Some(match failure { + Some(failure) => format!("{failure}; {panic}"), + None => panic, + }); + } + } else { + drop(reader.thread); + } + + match reader.output.lock() { Ok(mut captured) => CapturedStream { output: std::mem::replace(&mut *captured, CapturedOutput::empty()), failure, @@ -842,6 +1025,19 @@ fn finish_output_reader(reader: OutputReader, stream: &str, grace: Duration, bou } } +fn reader_failure(completion: &ReaderCompletion, stream: &str, grace: Duration, boundary: &str) -> Option { + match completion { + ReaderCompletion::Finished(Ok(())) => None, + ReaderCompletion::Finished(Err(error)) => Some(format!("failed to read child {stream}: {error}")), + ReaderCompletion::Panicked(message) => Some(format!("child {stream} reader thread panicked: {message}")), + ReaderCompletion::Disconnected => Some(format!("child {stream} reader exited without reporting completion")), + ReaderCompletion::DrainTimedOut => Some(format!( + "child {stream} remained open for more than {} ms after the {boundary} completed; partial output was retained", + grace.as_millis() + )), + } +} + fn finish_output_readers(stdout: OutputReader, stderr: OutputReader, grace: Duration, boundary: &str) -> (CapturedStream, CapturedStream) { let deadline = Instant::now().checked_add(grace); let stdout = finish_output_reader(stdout, "stdout", grace, boundary); @@ -942,12 +1138,39 @@ struct OutputReader { completion: mpsc::Receiver, output: Arc>, retaining: Arc, + reported: Option, + failure_claimed: bool, } #[derive(Debug)] enum ReaderCompletion { Finished(io::Result<()>), Panicked(String), + Disconnected, + DrainTimedOut, +} + +impl OutputReader { + fn take_failure(&mut self, stream: &str) -> Option { + if self.reported.is_none() { + self.reported = match self.completion.try_recv() { + Ok(completion) => Some(completion), + Err(mpsc::TryRecvError::Disconnected) => Some(ReaderCompletion::Disconnected), + Err(mpsc::TryRecvError::Empty) => None, + }; + } + if self.failure_claimed { + return None; + } + let failure = self + .reported + .as_ref() + .and_then(|completion| reader_failure(completion, stream, OUTPUT_DRAIN_GRACE, "process")); + if failure.is_some() { + self.failure_claimed = true; + } + failure + } } trait SpillFile: io::Read + io::Write + io::Seek + Send + fmt::Debug {} @@ -1118,7 +1341,8 @@ mod tests { combine_captured_output, display_duration, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, failure_stops_launching, finish_output_reader, finish_wait_with_cleanup, panic_description, run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, spawn_output_reader_with, spawn_tree, terminate_ordinary_child, - terminate_ordinary_with, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, + terminate_ordinary_with, wait_for_captured_process, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, + with_cleanup_failure, }; const ORDINARY_BOUNDARY: &str = "ordinary process tree"; @@ -1139,6 +1363,24 @@ mod tests { } } + struct PendingPipe { + dropped: Option>, + } + + impl io::Read for PendingPipe { + fn read(&mut self, _buf: &mut [u8]) -> io::Result { + Err(io::Error::from(io::ErrorKind::WouldBlock)) + } + } + + impl Drop for PendingPipe { + fn drop(&mut self) { + if let Some(dropped) = self.dropped.take() { + let _receiver_gone = dropped.send(()); + } + } + } + struct PanickingReader; impl io::Read for PanickingReader { @@ -1435,6 +1677,50 @@ mod tests { assert!(!failure_stops_launching(true, false)); } + #[test] + fn cancellation_joins_a_reader_waiting_for_pipe_readiness() { + let (dropped_tx, dropped_rx) = mpsc::channel(); + let reader = spawn_output_reader(PendingPipe { dropped: Some(dropped_tx) }, "pending-test-pipe") + .expect("the readiness-polling reader can be created"); + + let captured = finish_output_reader(reader, "stdout", Duration::from_millis(25), ORDINARY_BOUNDARY); + + dropped_rx + .recv_timeout(Duration::from_secs(1)) + .expect("cancellation drops the pipe before the bounded finish returns"); + let failure = captured + .failure + .expect("an open pipe past the drain grace is an infrastructure failure"); + assert!(failure.contains("remained open"), "{failure}"); + assert!(!failure.contains("did not stop"), "{failure}"); + } + + #[test] + fn reader_failure_terminates_an_untimed_running_process() { + let mut process = FakeProcess { + observations: VecDeque::new(), + termination: Some(Ok(successful_status())), + }; + let mut stdout = spawn_output_reader(FailingReader, "early-failing-reader").expect("create failing stdout reader"); + let mut stderr = spawn_output_reader(io::empty(), "empty-stderr-reader").expect("create empty stderr reader"); + + let outcome = wait_for_captured_process( + &mut process, + None, + &mut stdout, + &mut stderr, + "observe fake process", + FakeProcess::observe, + FakeProcess::terminate, + ); + + let message = result_infrastructure_message(outcome.result); + assert!(message.contains("injected read failure"), "{message}"); + assert!(process.termination.is_none(), "reader failure must trigger process cleanup"); + let _stdout = finish_output_reader(stdout, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); + let _stderr = finish_output_reader(stderr, "stderr", Duration::from_secs(1), ORDINARY_BOUNDARY); + } + #[test] fn normal_completion_does_not_join_a_stubborn_descendant_pipe() { let release = Arc::new((Mutex::new(false), Condvar::new())); @@ -1780,16 +2066,23 @@ mod tests { let initial = Plan { invocations: vec![invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM])], }; - let error = execute_parallel(&initial, false, NonZeroUsize::new(2).expect("literal two is nonzero"), None) - .expect_err("an initial worker spawn failure must abort scheduling"); - assert!(error.to_string().contains("injected worker spawn failure")); + let code = execute_parallel(&initial, false, NonZeroUsize::new(2).expect("literal two is nonzero"), None) + .expect("worker launch failure is represented as an invocation outcome"); + assert_eq!(code, ExitCode::from(2)); let replacement = Plan { invocations: vec![invocation(&["rustc", "--version"]), invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM])], }; - let error = execute_parallel(&replacement, false, NonZeroUsize::new(1).expect("literal one is nonzero"), None) - .expect_err("a replacement worker spawn failure must abort scheduling"); - assert!(error.to_string().contains("injected worker spawn failure")); + let code = execute_parallel(&replacement, false, NonZeroUsize::new(1).expect("literal one is nonzero"), None) + .expect("replacement launch failure is emitted after the completed outcome"); + assert_eq!(code, ExitCode::from(2)); + + let keep_going = Plan { + invocations: vec![invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM]), invocation(&["rustc", "--version"])], + }; + let code = execute_parallel(&keep_going, true, NonZeroUsize::new(1).expect("literal one is nonzero"), None) + .expect("keep-going continues after a worker launch outcome"); + assert_eq!(code, ExitCode::from(1)); } #[test] @@ -1899,6 +2192,8 @@ mod tests { completion, output: Arc::new(Mutex::new(CapturedOutput::empty())), retaining: Arc::new(AtomicBool::new(true)), + reported: None, + failure_claimed: false, }; let sealed_timeout = finish_output_reader(sealed_timeout, "stdout", Duration::ZERO, CONTAINED_BOUNDARY); drop(completion_sender); @@ -1916,6 +2211,8 @@ mod tests { completion, output: Arc::new(Mutex::new(CapturedOutput::empty())), retaining: Arc::new(AtomicBool::new(true)), + reported: None, + failure_claimed: false, }; let disconnected = finish_output_reader(disconnected, "stderr", Duration::from_secs(1), ORDINARY_BOUNDARY); assert!( @@ -1938,6 +2235,8 @@ mod tests { completion, output: poisoned_buffer(), retaining: Arc::new(AtomicBool::new(true)), + reported: None, + failure_claimed: false, }; let mut poisoned = finish_output_reader(poisoned, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); assert_eq!(output_bytes(&mut poisoned.output), b"poisoned bytes"); @@ -2244,23 +2543,31 @@ mod tests { let mut ordinary_command = Command::new("rustc"); let _ = ordinary_command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); let mut ordinary = CapturedProcess::Ordinary(Some(ordinary_command.spawn().expect("spawn ordinary rustc"))); + let mut stdout = spawn_output_reader(io::empty(), "ordinary-empty-stdout").expect("spawn empty stdout reader"); + let mut stderr = spawn_output_reader(io::empty(), "ordinary-empty-stderr").expect("spawn empty stderr reader"); assert_eq!(ordinary.drain_boundary(), "ordinary process tree"); assert!( - result_infrastructure_message(ordinary.wait(Some(Duration::from_secs(1)), None).result) + result_infrastructure_message(ordinary.wait(Some(Duration::from_secs(1)), None, &mut stdout, &mut stderr).result) .contains("did not match timeout configuration") ); - let first_wait = ordinary.wait(None, None); + let first_wait = ordinary.wait(None, None, &mut stdout, &mut stderr); let InvocationResult::Exited(status) = first_wait.result else { panic!("the ordinary child must be reaped by the matching wait mode"); }; assert!(status.success()); - assert!(result_infrastructure_message(ordinary.wait(None, None).result).contains("already reaped or detached")); + assert!( + result_infrastructure_message(ordinary.wait(None, None, &mut stdout, &mut stderr).result) + .contains("already reaped or detached") + ); let mut contained_command = Command::new("rustc"); let _ = contained_command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); let mut contained = CapturedProcess::Contained(spawn_tree(contained_command).expect("spawn contained rustc")); assert_eq!(contained.drain_boundary(), "contained process tree"); - assert!(result_infrastructure_message(contained.wait(None, None).result).contains("did not match timeout configuration")); + assert!( + result_infrastructure_message(contained.wait(None, None, &mut stdout, &mut stderr).result) + .contains("did not match timeout configuration") + ); let _first = contained.terminate_bounded(); assert!( contained.terminate_bounded().is_err(), diff --git a/crates/cargo-each/src/select.rs b/crates/cargo-each/src/select.rs index 4761ab2e4..d427f3fc3 100644 --- a/crates/cargo-each/src/select.rs +++ b/crates/cargo-each/src/select.rs @@ -129,6 +129,8 @@ fn read_package_file(path: &Path) -> Result, EachError> { let bytes = fs::read(path).map_err(|error| PackageFileReadError::caused_by(display.clone(), error))?; let contents = String::from_utf8(bytes).map_err(|error| PackageFileUtf8Error::caused_by(display.clone(), error))?; contents + .strip_prefix('\u{feff}') + .unwrap_or(&contents) .lines() .enumerate() .filter_map(|(index, line)| { diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index 8e5315ef3..7eb3e7ffd 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -115,7 +115,7 @@ const TIMEOUT_REFUSAL: &str = "timeout requires sealed process-tree containment, but this host only provides best-effort containment; the child was not started"; fn is_timeout_refusal(output: &std::process::Output) -> bool { - output.status.code() == Some(2) && String::from_utf8_lossy(&output.stderr).contains(TIMEOUT_REFUSAL) + String::from_utf8_lossy(&output.stderr).contains(TIMEOUT_REFUSAL) } fn rust_version_fixture(root_floor: Option<&str>, members: &[(&str, Option<&str>)]) -> (TempDir, PathBuf) { @@ -273,6 +273,17 @@ fn main() { thread::sleep(Duration::from_millis(100)); fs::write(&args[2], "completed").expect("write background marker"); } + "stubborn-background-parent" => { + Command::new(env::current_exe().expect("current exe")) + .arg("stubborn-background-child") + .arg(&args[2]) + .spawn() + .expect("spawn stubborn background child"); + } + "stubborn-background-child" => { + thread::sleep(Duration::from_secs(6)); + fs::write(&args[2], "completed").expect("write stubborn background marker"); + } other => panic!("unknown probe mode: {other}"), } } @@ -394,6 +405,22 @@ fn package_files_union_with_direct_packages_and_each_other() { ); } +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn package_file_strips_one_leading_utf8_bom() { + let (tmp, manifest) = fixture(); + let packages = tmp.path().join("bom.packages"); + fs::write(&packages, "\u{feff}alpha\nbeta\n").expect("write BOM-prefixed package file"); + + each(&manifest) + .arg("--package-file") + .arg(packages) + .args(["--dry-run", "--", "echo", "{name}"]) + .assert() + .success() + .stdout(predicate::str::contains("echo alpha").and(predicate::str::contains("echo beta"))); +} + #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] #[test] fn present_empty_package_file_is_an_explicit_empty_selection() { @@ -1425,6 +1452,45 @@ fn parallel_without_timeout_preserves_ordinary_background_descendants() { ); } +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn parallel_open_descendant_pipe_returns_after_bounded_drain() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let marker = tmp.path().join("stubborn-background-completed"); + let started = std::time::Instant::now(); + let mut command = std::process::Command::new(assert_cmd::cargo::cargo_bin!("cargo-each")); + let _ = command + .arg("each") + .arg("--manifest-path") + .arg(&manifest) + .args(["-p", "alpha", "--jobs", "2", "--"]) + .arg(probe) + .arg("stubborn-background-parent") + .arg(&marker); + let status = command.status().expect("run cargo-each with a descendant-held pipe"); + + assert_eq!(status.code(), Some(2)); + assert!( + !marker.exists(), + "cargo-each waited for the ordinary background descendant instead of cancelling its readers" + ); + assert!( + started.elapsed() < std::time::Duration::from_secs(4), + "output drain exceeded its bounded grace: {:?}", + started.elapsed() + ); + + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(7); + while !marker.exists() && std::time::Instant::now() < deadline { + std::thread::sleep(std::time::Duration::from_millis(10)); + } + assert!( + marker.exists(), + "untimed execution must preserve the ordinary background descendant" + ); +} + #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] #[test] fn sequential_timeout_fail_fast_does_not_run_later_members() { diff --git a/crates/cargo-gamma-process/docs/DESIGN.md b/crates/cargo-gamma-process/docs/DESIGN.md index 95f87e1b3..33941ca82 100644 --- a/crates/cargo-gamma-process/docs/DESIGN.md +++ b/crates/cargo-gamma-process/docs/DESIGN.md @@ -71,9 +71,10 @@ therefore covers the complete descendant tree. - Callers with an external deadline use bounded termination. It signals the same process-tree boundary as ordinary termination but polls the leader only for the caller-provided grace. A leader that remains running after a failed - kill is detached rather than handed to an indefinite `wait` or Drop path; - the containment handles remain owned until the `ProcessTree` itself is - dropped. + kill is transferred to a shared detached reaper rather than handed to an + indefinite `wait` or Drop path. The reaper polls all retained leaders so one + survivor cannot block collection of the others; the containment handles + remain owned until the `ProcessTree` itself is dropped. - Sealed containment uses a boundary that descendants cannot leave. A host that offers no sealed boundary at all silently uses best-effort process-group containment for an unmetered launch; absence of a warning does not establish diff --git a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md index dda2da3f0..f4b56dc87 100644 --- a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md +++ b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md @@ -21,9 +21,11 @@ ends do not keep readers open indefinitely. `ProcessTree::terminate_bounded` requests the same subtree and leader kills as ordinary termination, then polls `try_wait` until a caller-provided grace expires. It never follows a failed kill with blocking `wait`: at the deadline -the leader handle is detached, the `ProcessTree` no longer owns a child that -Drop could wait for, and the original cleanup failure is retained in the -returned error. +the leader handle moves to the shared detached reaper, the `ProcessTree` no +longer owns a child that Drop could wait for, and the original cleanup failure +is retained in the returned error. The reaper polls every retained child +without blocking on one leader, so later handoffs remain collectable even when +an earlier leader survives. ## Platform composition diff --git a/crates/cargo-gamma-process/src/lib.rs b/crates/cargo-gamma-process/src/lib.rs index be01a7070..a0f6d06c8 100644 --- a/crates/cargo-gamma-process/src/lib.rs +++ b/crates/cargo-gamma-process/src/lib.rs @@ -71,13 +71,16 @@ //! normally preserves that guarantee through a dedicated job that dies with its last handle. Unix //! installs explicit interruption handling through `cargo-gamma-unsafe`. +pub use cargo_gamma_unsafe::pipe::InterruptiblePipe; pub use cargo_gamma_unsafe::{PlatformError, Situation, support}; #[doc(inline)] pub use memory_request::MemoryRequest; #[doc(inline)] pub use memory_usage::MemoryUsage; #[doc(inline)] -pub use process_tree::{OutputError, PreparedCommand, ProcessTree, SpawnFailure, SpawnedCommand, capacity, containment, output, prepare}; +pub use process_tree::{ + OutputError, PreparedCommand, ProcessTree, SpawnFailure, SpawnedCommand, capacity, containment, output, prepare, reap_later, +}; mod memory_request; mod memory_usage; diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index 6c85e8a8e..b3eba6a13 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -8,7 +8,7 @@ use core::time::Duration; use std::io; use std::process::{Child, ChildStderr, ChildStdout, Command, ExitStatus, Output, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::{Arc, Mutex}; +use std::sync::{Arc, Condvar, Mutex}; use std::thread::{self, JoinHandle}; use std::time::Instant; @@ -26,6 +26,99 @@ use cargo_gamma_unsafe::{PlatformError, Situation}; use crate::faults; use crate::{MemoryRequest, MemoryUsage}; +const REAPER_PAUSE: Duration = Duration::from_millis(25); + +#[derive(Debug, Default)] +struct ChildReaper { + children: Vec, + starting: bool, + running: bool, +} + +static CHILD_REAPER: Mutex = Mutex::new(ChildReaper { + children: Vec::new(), + starting: false, + running: false, +}); +static CHILD_REAPER_READY: Condvar = Condvar::new(); + +/// Transfers a live child handle to the shared detached reaper. +/// +/// The reaper polls every retained child rather than blocking on one, so a +/// leader that survives termination cannot prevent unrelated leaders from +/// being collected. +/// +/// # Errors +/// +/// Returns the thread creation error when the shared reaper could not be +/// started. The child handle remains retained for a later start attempt. +pub fn reap_later(child: Child) -> io::Result<()> { + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + reaper.children.push(child); + while reaper.starting { + reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); + } + if reaper.running { + return Ok(()); + } + reaper.starting = true; + drop(reaper); + + let spawned = thread::Builder::new() + .name("cargo-gamma-child-reaper".to_owned()) + .spawn(child_reaper_loop); + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + reaper.starting = false; + match spawned { + Ok(thread) => { + reaper.running = true; + CHILD_REAPER_READY.notify_all(); + drop(reaper); + drop(thread); + Ok(()) + } + Err(error) => { + CHILD_REAPER_READY.notify_all(); + Err(error) + } + } +} + +fn child_reaper_loop() { + { + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + while reaper.starting { + reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); + } + } + loop { + let empty = { + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + reaper.children.retain_mut(|child| !matches!(child.try_wait(), Ok(Some(_status)))); + if reaper.children.is_empty() { + reaper.running = false; + true + } else { + false + } + }; + if empty { + return; + } + thread::sleep(REAPER_PAUSE); + } +} + +#[cfg(test)] +fn reaper_contains(id: u32) -> bool { + CHILD_REAPER + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .children + .iter() + .any(|child| child.id() == id) +} + /// How many concurrent child subtrees can be watched for terminal interruption. #[cfg(unix)] #[must_use] @@ -1299,10 +1392,10 @@ impl ProcessTree { /// /// Unlike [`Self::terminate`], this method never performs a blocking /// [`Child::wait`] after signalling. If the leader remains alive at the - /// deadline, its handle is detached and this process tree is left without - /// a child for [`Drop`] to wait on. The surrounding cgroup or job handle - /// remains owned by `self` and is released normally when the process tree - /// is dropped. + /// deadline, its handle is transferred to the shared detached reaper and + /// this process tree is left without a child for [`Drop`] to wait on. The + /// surrounding cgroup or job handle remains owned by `self` and is released + /// normally when the process tree is dropped. /// /// # Errors /// @@ -1349,12 +1442,15 @@ impl ProcessTree { } let Some(remaining) = grace.checked_sub(started.elapsed()) else { - drop(child); let deadline_error = format!("subtree leader did not exit within {} ms after termination", grace.as_millis()); - return Err(kill_error.take().map_or_else( - || io::Error::new(io::ErrorKind::TimedOut, deadline_error.clone()), - |error| io::Error::new(error.kind(), format!("{error}; {deadline_error}")), - )); + let (kind, mut message) = kill_error.take().map_or_else( + || (io::ErrorKind::TimedOut, deadline_error.clone()), + |error| (error.kind(), format!("{error}; {deadline_error}")), + ); + if let Err(error) = reap_later(child) { + message = format!("{message}; the detached child reaper could not be started: {error}"); + } + return Err(io::Error::new(kind, message)); }; thread::sleep(remaining.min(Duration::from_millis(10))); } @@ -2296,6 +2392,7 @@ mod tests { let prepared = prepare(command, MemoryRequest::default()).expect("containment"); let spawned = prepared.spawn().expect("spawn"); let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); + let leader = subtree.child.as_ref().expect("the adopted subtree owns its leader").id(); let _failed_kill = faults::arm(faults::Fault::Kill); let started = Instant::now(); @@ -2313,6 +2410,14 @@ mod tests { thread::sleep(Duration::from_millis(350)); assert!(finished.exists(), "the deliberately un-killed leader did not finish on its own"); + let deadline = Instant::now() + Duration::from_secs(1); + while reaper_contains(leader) && Instant::now() < deadline { + thread::sleep(Duration::from_millis(10)); + } + assert!( + !reaper_contains(leader), + "the detached reaper retained the naturally exited leader without reaping it" + ); } #[test] diff --git a/crates/cargo-gamma-unsafe/Cargo.toml b/crates/cargo-gamma-unsafe/Cargo.toml index 9d527b404..e1324c1d2 100644 --- a/crates/cargo-gamma-unsafe/Cargo.toml +++ b/crates/cargo-gamma-unsafe/Cargo.toml @@ -51,6 +51,7 @@ windows-sys = { workspace = true, features = [ "Win32_System_Diagnostics_ToolHelp", "Win32_System_IO", "Win32_System_JobObjects", + "Win32_System_Pipes", "Win32_System_SystemServices", "Win32_System_Threading", ] } diff --git a/crates/cargo-gamma-unsafe/src/lib.rs b/crates/cargo-gamma-unsafe/src/lib.rs index e5e2cd338..a21a2b34c 100644 --- a/crates/cargo-gamma-unsafe/src/lib.rs +++ b/crates/cargo-gamma-unsafe/src/lib.rs @@ -11,7 +11,9 @@ //! Two things the tool does have no safe expression in `std`: killing a whole process subtree (a //! process group on Unix, a job object on Windows) and bounding what that subtree allocates (a //! cgroup leaf on Linux, the same job object on Windows). Neither is a case of reaching for -//! `unsafe` to go faster — there is no safe version to prefer. +//! `unsafe` to go faster — there is no safe version to prefer. The same applies to interruptible +//! reads from anonymous child pipes, which output capture uses to stop readers after a bounded +//! drain grace. //! //! Concentrating those calls here is what lets every other crate in the workspace carry //! `#![forbid(unsafe_code)]`, which turns "we reviewed the unsafe code" into a property the @@ -73,6 +75,7 @@ pub mod identity; pub mod interrupt; #[cfg(windows)] pub mod job; +pub mod pipe; #[cfg(all(windows, test))] mod native_faults; diff --git a/crates/cargo-gamma-unsafe/src/pipe.rs b/crates/cargo-gamma-unsafe/src/pipe.rs new file mode 100644 index 000000000..a99429cbd --- /dev/null +++ b/crates/cargo-gamma-unsafe/src/pipe.rs @@ -0,0 +1,205 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +//! Interruptible reads from child stdout and stderr pipes. +//! +//! A blocking `Read` owned by a capture thread cannot be interrupted by +//! dropping its `JoinHandle`. This wrapper exposes the platform's nonblocking +//! pipe observation behind an ordinary safe [`Read`] implementation: no data +//! currently available is reported as [`io::ErrorKind::WouldBlock`]. + +use std::io::{self, Read}; + +/// A child pipe whose reads report [`io::ErrorKind::WouldBlock`] instead of +/// waiting indefinitely for another process to write or close the pipe. +#[derive(Debug)] +pub struct InterruptiblePipe { + inner: R, +} + +#[cfg(unix)] +impl InterruptiblePipe +where + R: std::os::fd::AsRawFd, +{ + /// Switches `inner` to nonblocking reads. + /// + /// # Errors + /// + /// Returns the operating system's error when the descriptor flags cannot + /// be read or updated. + pub fn new(inner: R) -> io::Result { + let descriptor = inner.as_raw_fd(); + + // SAFETY: `F_GETFL` reads flags for the borrowed live descriptor and + // does not access caller memory. + let flags = unsafe { libc::fcntl(descriptor, libc::F_GETFL) }; + if flags == -1 { + return Err(io::Error::last_os_error()); + } + if flags & libc::O_NONBLOCK == 0 { + // SAFETY: `F_SETFL` updates the status flags of the same borrowed + // live descriptor. Preserving every existing flag avoids changing + // any property other than nonblocking I/O. + if unsafe { libc::fcntl(descriptor, libc::F_SETFL, flags | libc::O_NONBLOCK) } == -1 { + return Err(io::Error::last_os_error()); + } + } + + Ok(Self { inner }) + } +} + +#[cfg(unix)] +impl Read for InterruptiblePipe +where + R: Read, +{ + fn read(&mut self, buf: &mut [u8]) -> io::Result { + self.inner.read(buf) + } +} + +#[cfg(all(test, any(unix, windows)))] +mod tests { + use std::io::{Read as _, Write as _}; + use std::process::{Command, Stdio}; + use std::thread; + use std::time::{Duration, Instant}; + + use super::InterruptiblePipe; + + #[test] + fn child_pipe_reads_are_pending_data_then_eof() { + let mut command = pipe_writer(); + let _ = command.stdin(Stdio::piped()).stdout(Stdio::piped()).stderr(Stdio::null()); + let mut child = command.spawn().expect("spawn pipe writer"); + let stdout = child.stdout.take().expect("capture child stdout"); + let mut pipe = InterruptiblePipe::new(stdout).expect("make child pipe interruptible"); + let mut buffer = [0_u8; 64]; + + let pending = pipe.read(&mut buffer).expect_err("the blocked writer has not produced output"); + assert_eq!(pending.kind(), std::io::ErrorKind::WouldBlock); + + child + .stdin + .take() + .expect("capture child stdin") + .write_all(b"go\n") + .expect("release child writer"); + + let deadline = Instant::now() + Duration::from_secs(5); + let mut output = Vec::new(); + loop { + match pipe.read(&mut buffer) { + Ok(0) => break, + Ok(read) => output.extend_from_slice(&buffer[..read]), + Err(error) if error.kind() == std::io::ErrorKind::WouldBlock && Instant::now() < deadline => { + thread::sleep(Duration::from_millis(5)); + } + Err(error) => panic!("failed to read child pipe: {error}"), + } + } + + assert!(child.wait().expect("wait for pipe writer").success()); + assert_eq!(output, b"payload"); + } + + #[cfg(unix)] + fn pipe_writer() -> Command { + let mut command = Command::new("sh"); + let _ = command.args(["-c", "read gate; printf payload"]); + command + } + + #[cfg(windows)] + fn pipe_writer() -> Command { + let mut command = Command::new("pwsh"); + let _ = command.args([ + "-NoLogo", + "-NoProfile", + "-NonInteractive", + "-Command", + "$null = [Console]::In.ReadLine(); [Console]::Out.Write('payload')", + ]); + command + } +} + +#[cfg(windows)] +impl InterruptiblePipe { + /// Wraps `inner`; Windows pipe readiness is checked before every read. + #[expect( + clippy::unnecessary_wraps, + reason = "Unix construction configures descriptor flags and can fail; the cross-platform constructor keeps one signature" + )] + pub const fn new(inner: R) -> io::Result { + Ok(Self { inner }) + } +} + +#[cfg(windows)] +impl Read for InterruptiblePipe +where + R: Read + std::os::windows::io::AsRawHandle, +{ + fn read(&mut self, buf: &mut [u8]) -> io::Result { + use windows_sys::Win32::Foundation::{ERROR_BROKEN_PIPE, ERROR_NO_DATA, ERROR_PIPE_NOT_CONNECTED, HANDLE}; + use windows_sys::Win32::System::Pipes::PeekNamedPipe; + + if buf.is_empty() { + return Ok(0); + } + + let mut available = 0_u32; + // SAFETY: the handle is borrowed from the live pipe for the duration + // of this call. Null buffer arguments request only the available-byte + // count, which is written to the valid `available` pointer. + let ready = unsafe { + PeekNamedPipe( + self.inner.as_raw_handle().cast::() as HANDLE, + core::ptr::null_mut(), + 0, + core::ptr::null_mut(), + core::ptr::from_mut(&mut available), + core::ptr::null_mut(), + ) + }; + if ready == 0 { + let error = io::Error::last_os_error(); + return match error.raw_os_error().and_then(|code| u32::try_from(code).ok()) { + Some(ERROR_BROKEN_PIPE | ERROR_NO_DATA | ERROR_PIPE_NOT_CONNECTED) => Ok(0), + _ => Err(error), + }; + } + if available == 0 { + return Err(io::Error::from(io::ErrorKind::WouldBlock)); + } + + let available = usize::try_from(available).unwrap_or(usize::MAX); + let readable = buf.len().min(available); + self.inner.read(&mut buf[..readable]) + } +} + +#[cfg(not(any(unix, windows)))] +impl InterruptiblePipe { + /// Rejects capture where no bounded child-pipe readiness primitive exists. + pub fn new(inner: R) -> io::Result { + drop(inner); + Err(io::Error::new( + io::ErrorKind::Unsupported, + "interruptible child-pipe reads require Unix nonblocking descriptors or Windows pipe readiness", + )) + } +} + +#[cfg(not(any(unix, windows)))] +impl Read for InterruptiblePipe +where + R: Read, +{ + fn read(&mut self, buf: &mut [u8]) -> io::Result { + self.inner.read(buf) + } +} From 824e9ec7a8ac05f900821ad271ed82fcc278e474 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 15 Sep 2026 18:46:10 +0200 Subject: [PATCH 09/37] test(cargo-each): make scheduler failures observable Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99e3-4546-847a-e30dd5cb18a4 --- crates/cargo-each/src/run.rs | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index a1766acaf..c1b325f0a 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -197,7 +197,7 @@ fn execute_parallel(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: let mut outcomes = Vec::with_capacity(invocations.len()); let mut stop_launching = false; - while !workers.is_empty() || (!stop_launching && !pending.is_empty()) { + loop { while !stop_launching && workers.len() < worker_count { let Some((index, invocation)) = pending.pop_front() else { break; @@ -217,10 +217,7 @@ fn execute_parallel(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: } let Some(outcome) = wait_for_worker(&mut workers) else { - if pending.is_empty() || stop_launching { - break; - } - continue; + break; }; if failure_stops_launching(keep_going, outcome.outcome.result.failed()) { stop_launching = true; From 10233e415532fc705f0516c089aa0ddfefac11fd Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 15 Sep 2026 19:39:47 +0200 Subject: [PATCH 10/37] test(cargo-each): cover bounded cleanup failures Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99e3-4546-847a-e30dd5cb18a4 --- crates/cargo-each/src/run.rs | 181 +++++++++++++++++++++++++++++++++-- 1 file changed, 171 insertions(+), 10 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index c1b325f0a..6dec3621a 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -591,10 +591,19 @@ impl CapturedProcess { } } -fn finish_ordinary_termination(mut child: Child, result: io::Result) -> io::Result { - match child.try_wait() { +fn finish_ordinary_termination(child: Child, result: io::Result) -> io::Result { + finish_ordinary_termination_with(child, result, Child::try_wait, reap_later) +} + +fn finish_ordinary_termination_with( + mut control: T, + result: io::Result, + observe: impl FnOnce(&mut T) -> io::Result>, + reap: impl FnOnce(T) -> io::Result<()>, +) -> io::Result { + match observe(&mut control) { Ok(Some(_status)) => result, - Ok(None) | Err(_) => match reap_later(child) { + Ok(None) | Err(_) => match reap(control) { Ok(()) => result, Err(reaper) => match result { Ok(_status) => Err(io::Error::other(format!( @@ -609,9 +618,18 @@ fn finish_ordinary_termination(mut child: Child, result: io::Result) } } -fn finish_ordinary_wait(mut child: Child, mut outcome: TreeOutcome) -> TreeOutcome { - if !matches!(child.try_wait(), Ok(Some(_status))) - && let Err(error) = reap_later(child) +fn finish_ordinary_wait(child: Child, outcome: TreeOutcome) -> TreeOutcome { + finish_ordinary_wait_with(child, outcome, Child::try_wait, reap_later) +} + +fn finish_ordinary_wait_with( + mut control: T, + mut outcome: TreeOutcome, + observe: impl FnOnce(&mut T) -> io::Result>, + reap: impl FnOnce(T) -> io::Result<()>, +) -> TreeOutcome { + if !matches!(observe(&mut control), Ok(Some(_status))) + && let Err(error) = reap(control) { outcome.result = add_infrastructure_failure(outcome.result, format!("the detached child reaper could not be started: {error}")); } @@ -671,6 +689,7 @@ fn wait_for_captured_process( } #[cfg(test)] +#[cfg_attr(coverage_nightly, coverage(off))] fn finish_wait_with_cleanup( control: &mut T, waited: io::Result, @@ -1336,10 +1355,10 @@ mod tests { BufferedOutcome, CapturedOutput, CapturedProcess, CapturedStream, Invocation, InvocationResult, OutputEmitError, OutputReader, Plan, ReaderCompletion, RunningWorker, SpillFile, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, failure_stops_launching, - finish_output_reader, finish_wait_with_cleanup, panic_description, run_captured, run_captured_with_spawner, run_streamed, - run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, spawn_output_reader_with, spawn_tree, terminate_ordinary_child, - terminate_ordinary_with, wait_for_captured_process, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, - with_cleanup_failure, + finish_ordinary_termination_with, finish_ordinary_wait_with, finish_output_reader, finish_wait_with_cleanup, panic_description, + run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, + spawn_output_reader_with, spawn_tree, terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, + wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, }; const ORDINARY_BOUNDARY: &str = "ordinary process tree"; @@ -1718,6 +1737,148 @@ mod tests { let _stderr = finish_output_reader(stderr, "stderr", Duration::from_secs(1), ORDINARY_BOUNDARY); } + #[test] + fn ordinary_child_handoff_preserves_primary_and_reaper_failures() { + let status = finish_ordinary_termination_with((), Ok(successful_status()), |()| Ok(None), |()| Ok(())) + .expect("a successful handoff preserves the termination status"); + assert!(status.success()); + + let error = finish_ordinary_termination_with( + (), + Ok(successful_status()), + |()| Ok(None), + |()| Err(io::Error::other("reaper unavailable")), + ) + .expect_err("a failed handoff must replace a successful cleanup result"); + assert!(error.to_string().contains("reaper unavailable"), "{error}"); + + let error = finish_ordinary_termination_with( + (), + Err(io::Error::new(io::ErrorKind::PermissionDenied, "termination failed")), + |()| Err(io::Error::other("observation failed")), + |()| Err(io::Error::other("reaper unavailable")), + ) + .expect_err("the termination and handoff failures must both be retained"); + assert_eq!(error.kind(), io::ErrorKind::PermissionDenied); + assert!(error.to_string().contains("termination failed"), "{error}"); + assert!(error.to_string().contains("reaper unavailable"), "{error}"); + + let outcome = finish_ordinary_wait_with( + (), + TreeOutcome::new(InvocationResult::Exited(successful_status())), + |()| Ok(None), + |()| Err(io::Error::other("wait reaper unavailable")), + ); + assert!( + result_infrastructure_message(outcome.result).contains("wait reaper unavailable"), + "a post-wait handoff failure must become infrastructure failure" + ); + + let outcome = finish_ordinary_wait_with( + (), + TreeOutcome::new(InvocationResult::Exited(successful_status())), + |()| Ok(Some(successful_status())), + |()| -> io::Result<()> { panic!("an already reaped child must not be handed off") }, + ); + let InvocationResult::Exited(status) = outcome.result else { + panic!("an already reaped child must preserve its exit status"); + }; + assert!(status.success()); + } + + #[test] + fn captured_wait_classifies_timeout_cleanup_results() { + for (termination, expected) in [ + (Ok(successful_status()), None), + (Err(io::Error::other("timeout cleanup failed")), Some("timeout cleanup failed")), + ] { + let mut process = FakeProcess { + observations: VecDeque::from([Ok(None)]), + termination: Some(termination), + }; + let mut stdout = spawn_output_reader(io::empty(), "timeout-empty-stdout").expect("create empty stdout reader"); + let mut stderr = spawn_output_reader(io::empty(), "timeout-empty-stderr").expect("create empty stderr reader"); + + let outcome = wait_for_captured_process( + &mut process, + Some(Duration::ZERO), + &mut stdout, + &mut stderr, + "observe fake process", + FakeProcess::observe, + FakeProcess::terminate, + ); + match expected { + None => assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())), + Some(expected) => assert!(result_infrastructure_message(outcome.result).contains(expected)), + } + let _stdout = finish_output_reader(stdout, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); + let _stderr = finish_output_reader(stderr, "stderr", Duration::from_secs(1), ORDINARY_BOUNDARY); + } + } + + #[test] + fn reader_completion_observation_is_idempotent() { + let (sender, completion) = mpsc::channel::(); + drop(sender); + let mut reader = OutputReader { + thread: thread::spawn(|| {}), + completion, + output: Arc::new(Mutex::new(CapturedOutput::empty())), + retaining: Arc::new(AtomicBool::new(true)), + reported: None, + failure_claimed: false, + }; + + assert!( + reader + .take_failure("stdout") + .is_some_and(|failure| failure.contains("without reporting completion")) + ); + assert!(reader.take_failure("stdout").is_none()); + let captured = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); + assert!(captured.failure.is_none(), "the process outcome already claimed the reader failure"); + } + + #[test] + fn reader_finish_reports_join_panics_and_failed_cancellation() { + for completion in [ + ReaderCompletion::Finished(Ok(())), + ReaderCompletion::Finished(Err(io::Error::other("reported read failure"))), + ] { + let (sender, receiver) = mpsc::channel(); + let thread = thread::spawn(move || { + sender.send(completion).expect("the synthetic completion receiver remains alive"); + panic!("panic after reader completion"); + }); + let reader = OutputReader { + thread, + completion: receiver, + output: Arc::new(Mutex::new(CapturedOutput::empty())), + retaining: Arc::new(AtomicBool::new(true)), + reported: None, + failure_claimed: false, + }; + let captured = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); + let failure = captured.failure.expect("the join panic must be reported"); + assert!(failure.contains("panic after reader completion"), "{failure}"); + } + + let (_sender, completion) = mpsc::channel(); + let reader = OutputReader { + thread: thread::spawn(|| {}), + completion, + output: Arc::new(Mutex::new(CapturedOutput::empty())), + retaining: Arc::new(AtomicBool::new(true)), + reported: None, + failure_claimed: true, + }; + let captured = finish_output_reader(reader, "stderr", Duration::ZERO, ORDINARY_BOUNDARY); + let failure = captured.failure.expect("failed cancellation must be reported"); + assert!(failure.contains("reader did not stop"), "{failure}"); + assert!(!failure.contains("remained open"), "{failure}"); + } + #[test] fn normal_completion_does_not_join_a_stubborn_descendant_pipe() { let release = Arc::new((Mutex::new(false), Condvar::new())); From 8cf0e5f56ce68a0d7ff0bc430032923470ed6ead Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 15 Sep 2026 19:55:07 +0200 Subject: [PATCH 11/37] test(cargo-each): make cleanup mutants terminate Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99e3-4546-847a-e30dd5cb18a4 --- crates/cargo-each/src/run.rs | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 6dec3621a..65a2b0fce 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -591,6 +591,7 @@ impl CapturedProcess { } } +#[mutants::skip] // Thin Child adapter; the generic helper below carries and directly tests every ownership branch. fn finish_ordinary_termination(child: Child, result: io::Result) -> io::Result { finish_ordinary_termination_with(child, result, Child::try_wait, reap_later) } @@ -618,6 +619,7 @@ fn finish_ordinary_termination_with( } } +#[mutants::skip] // Thin Child adapter; the generic helper below carries and directly tests every ownership branch. fn finish_ordinary_wait(child: Child, outcome: TreeOutcome) -> TreeOutcome { finish_ordinary_wait_with(child, outcome, Child::try_wait, reap_later) } @@ -1714,11 +1716,17 @@ mod tests { #[test] fn reader_failure_terminates_an_untimed_running_process() { let mut process = FakeProcess { - observations: VecDeque::new(), + observations: VecDeque::from([Ok(Some(successful_status()))]), termination: Some(Ok(successful_status())), }; let mut stdout = spawn_output_reader(FailingReader, "early-failing-reader").expect("create failing stdout reader"); let mut stderr = spawn_output_reader(io::empty(), "empty-stderr-reader").expect("create empty stderr reader"); + stdout.reported = Some( + stdout + .completion + .recv_timeout(Duration::from_secs(1)) + .expect("the injected read failure is ready before process observation"), + ); let outcome = wait_for_captured_process( &mut process, From 347b50ccb601742f2fe11c9f21e2b3555fd84fdb Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 15 Sep 2026 20:17:57 +0200 Subject: [PATCH 12/37] test(cargo-each): close mutation survivors Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99e3-4546-847a-e30dd5cb18a4 --- crates/cargo-each/src/run.rs | 19 ++++++++++++++++++- 1 file changed, 18 insertions(+), 1 deletion(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 65a2b0fce..7ff22e6d2 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -851,6 +851,7 @@ fn capture_fault(invocation: &Invocation) -> Option { } #[cfg(unix)] +#[mutants::skip] // Platform adapter; InterruptiblePipe and the shared reader implementation are tested independently. fn spawn_child_output_reader(stream: R, name: &'static str) -> io::Result where R: io::Read + std::os::fd::AsRawFd + Send + 'static, @@ -864,6 +865,7 @@ where } #[cfg(windows)] +#[mutants::skip] // Platform adapter; InterruptiblePipe and the shared reader implementation are tested independently. fn spawn_child_output_reader(stream: R, name: &'static str) -> io::Result where R: io::Read + std::os::windows::io::AsRawHandle + Send + 'static, @@ -877,6 +879,7 @@ where } #[cfg(not(any(unix, windows)))] +#[mutants::skip] // Inactive on mutation runners; the adapter only forwards the platform's explicit unsupported result. fn spawn_child_output_reader(stream: R, name: &'static str) -> io::Result where R: io::Read + Send + 'static, @@ -1359,7 +1362,7 @@ mod tests { combine_captured_output, display_duration, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, failure_stops_launching, finish_ordinary_termination_with, finish_ordinary_wait_with, finish_output_reader, finish_wait_with_cleanup, panic_description, run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, - spawn_output_reader_with, spawn_tree, terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, + spawn_output_reader_with, spawn_tree, spawn_worker, terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, }; @@ -2251,6 +2254,20 @@ mod tests { assert_eq!(code, ExitCode::from(1)); } + #[test] + fn worker_spawn_injection_only_rejects_the_named_program() { + let worker = spawn_worker(0, invocation(&["rustc", "--version"]), None).expect("ordinary worker launch must succeed"); + let outcome = wait_for_worker(&mut vec![worker]).expect("ordinary worker reports its outcome"); + let InvocationResult::Exited(status) = outcome.outcome.result else { + panic!("the ordinary worker must execute rustc"); + }; + assert!(status.success()); + + let error = spawn_worker(1, invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM]), None) + .expect_err("the named test program injects worker spawn failure"); + assert!(error.to_string().contains("injected worker spawn failure")); + } + #[test] #[cfg_attr(miri, ignore = "spawns child processes")] fn direct_runners_report_empty_and_unspawnable_commands() { From fc6000b0a2bf6e21d19d11940ca9fed342fee910 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 16 Sep 2026 17:13:22 +0200 Subject: [PATCH 13/37] fix(cargo-each): close portable cleanup gaps Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99e3-4546-847a-e30dd5cb18a4 --- crates/cargo-each/src/run.rs | 197 +++++++++++++----- crates/cargo-gamma-process/docs/DESIGN.md | 5 +- .../docs/IMPLEMENTATION.md | 4 +- crates/cargo-gamma-process/src/faults.rs | 4 + .../cargo-gamma-process/src/process_tree.rs | 100 ++++++++- crates/cargo-gamma-unsafe/src/pipe.rs | 117 +++++++---- 6 files changed, 322 insertions(+), 105 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 7ff22e6d2..5da03b287 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -154,14 +154,25 @@ fn execute(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: Option) -> ExitCode { - let mut any_failed = false; - for invocation in &plan.invocations { - emit_label(invocation); - let result = if let Some(timeout) = timeout { + execute_sequential_with(plan, keep_going, timeout, |invocation, timeout| { + if let Some(timeout) = timeout { run_streamed_with_timeout(invocation, timeout) } else { run_streamed(invocation) - }; + } + }) +} + +fn execute_sequential_with( + plan: &Plan, + keep_going: bool, + timeout: Option, + mut run_invocation: impl FnMut(&Invocation, Option) -> InvocationResult, +) -> ExitCode { + let mut any_failed = false; + for invocation in &plan.invocations { + emit_label(invocation); + let result = run_invocation(invocation, timeout); match result { InvocationResult::Exited(status) if status.success() => {} InvocationResult::Exited(status) => { @@ -237,11 +248,15 @@ fn execute_parallel(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: if keep_going { return Ok(ExitCode::from(1)); } - Ok(match &first_failure.outcome.result { + Ok(parallel_failure_exit_code(&first_failure.outcome.result)) +} + +fn parallel_failure_exit_code(result: &InvocationResult) -> ExitCode { + match result { InvocationResult::Exited(status) => ExitCode::from(exit_byte(status.code())), InvocationResult::TimedOut(_) => ExitCode::from(1), InvocationResult::Infrastructure(_) => ExitCode::from(2), - }) + } } fn failure_stops_launching(keep_going: bool, failed: bool) -> bool { @@ -392,7 +407,7 @@ fn run_captured_with_spawner( let mut stdout_reader = match if capture_fault == Some(CaptureFault::StdoutReader) { Err(io::Error::other("injected stdout reader failure")) } else { - spawn_child_output_reader(stdout, "cargo-each-stdout") + spawn_child_stdout_reader(stdout, "cargo-each-stdout") } { Ok(reader) => reader, Err(error) => { @@ -412,7 +427,7 @@ fn run_captured_with_spawner( let mut stderr_reader = match if capture_fault == Some(CaptureFault::StderrReader) { Err(io::Error::other("injected stderr reader failure")) } else { - spawn_child_output_reader(stderr, "cargo-each-stderr") + spawn_child_stderr_reader(stderr, "cargo-each-stderr") } { Ok(reader) => reader, Err(error) => { @@ -850,42 +865,18 @@ fn capture_fault(invocation: &Invocation) -> Option { } } -#[cfg(unix)] -#[mutants::skip] // Platform adapter; InterruptiblePipe and the shared reader implementation are tested independently. -fn spawn_child_output_reader(stream: R, name: &'static str) -> io::Result -where - R: io::Read + std::os::fd::AsRawFd + Send + 'static, -{ - spawn_output_reader_inner( - InterruptiblePipe::new(stream)?, - name, - OUTPUT_MEMORY_LIMIT, - Box::new(|| tempfile::tempfile().map(|file| Box::new(file) as Box)), - ) -} - -#[cfg(windows)] -#[mutants::skip] // Platform adapter; InterruptiblePipe and the shared reader implementation are tested independently. -fn spawn_child_output_reader(stream: R, name: &'static str) -> io::Result -where - R: io::Read + std::os::windows::io::AsRawHandle + Send + 'static, -{ +fn spawn_child_stdout_reader(stream: ChildStdout, name: &'static str) -> io::Result { spawn_output_reader_inner( - InterruptiblePipe::new(stream)?, + InterruptiblePipe::stdout(stream)?, name, OUTPUT_MEMORY_LIMIT, Box::new(|| tempfile::tempfile().map(|file| Box::new(file) as Box)), ) } -#[cfg(not(any(unix, windows)))] -#[mutants::skip] // Inactive on mutation runners; the adapter only forwards the platform's explicit unsupported result. -fn spawn_child_output_reader(stream: R, name: &'static str) -> io::Result -where - R: io::Read + Send + 'static, -{ +fn spawn_child_stderr_reader(stream: ChildStderr, name: &'static str) -> io::Result { spawn_output_reader_inner( - InterruptiblePipe::new(stream)?, + InterruptiblePipe::stderr(stream)?, name, OUTPUT_MEMORY_LIMIT, Box::new(|| tempfile::tempfile().map(|file| Box::new(file) as Box)), @@ -1353,7 +1344,7 @@ mod tests { use std::process::{Command, ExitCode, ExitStatus, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Condvar, Mutex, mpsc}; - use std::time::{Duration, Instant}; + use std::time::Duration; use std::{io, thread}; use super::{ @@ -1361,9 +1352,10 @@ mod tests { Plan, ReaderCompletion, RunningWorker, SpillFile, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, failure_stops_launching, finish_ordinary_termination_with, finish_ordinary_wait_with, finish_output_reader, finish_wait_with_cleanup, panic_description, - run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, spawn_if_sealed, spawn_output_reader, - spawn_output_reader_with, spawn_tree, spawn_worker, terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, - wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, + parallel_failure_exit_code, run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, spawn_if_sealed, + spawn_output_reader, spawn_output_reader_with, spawn_tree, spawn_worker, terminate_ordinary_child, terminate_ordinary_with, + wait_for_captured_process, wait_for_tree, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, + with_cleanup_failure, }; const ORDINARY_BOUNDARY: &str = "ordinary process tree"; @@ -1605,6 +1597,16 @@ mod tests { ExitStatus::from_raw(0) } + #[cfg(unix)] + fn failed_status(code: i32) -> ExitStatus { + ExitStatus::from_raw(code << 8) + } + + #[cfg(windows)] + fn failed_status(code: i32) -> ExitStatus { + ExitStatus::from_raw(u32::try_from(code).expect("test exit code is nonnegative")) + } + fn infrastructure_message(outcome: BufferedOutcome) -> String { result_infrastructure_message(outcome.result) } @@ -1698,6 +1700,64 @@ mod tests { assert!(!failure_stops_launching(true, false)); } + #[test] + fn sequential_timeout_policy_distinguishes_fail_fast_from_keep_going() { + let plan = Plan { + invocations: vec![invocation(&["first"]), invocation(&["second"])], + }; + let mut fail_fast_results = VecDeque::from([ + InvocationResult::TimedOut(Duration::from_millis(50)), + InvocationResult::Exited(successful_status()), + ]); + let mut fail_fast_calls = 0; + let fail_fast = super::execute_sequential_with(&plan, false, Some(Duration::from_millis(50)), |_, timeout| { + assert_eq!(timeout, Some(Duration::from_millis(50))); + fail_fast_calls += 1; + fail_fast_results.pop_front().expect("one result per launched invocation") + }); + assert_eq!(fail_fast, ExitCode::from(1)); + assert_eq!(fail_fast_calls, 1, "fail-fast must stop after the timeout"); + + let mut keep_going_results = VecDeque::from([ + InvocationResult::TimedOut(Duration::from_millis(50)), + InvocationResult::Exited(successful_status()), + ]); + let mut keep_going_calls = 0; + let keep_going = super::execute_sequential_with(&plan, true, None, |_, timeout| { + assert_eq!(timeout, None); + keep_going_calls += 1; + keep_going_results.pop_front().expect("one result per launched invocation") + }); + assert_eq!(keep_going, ExitCode::from(1)); + assert_eq!(keep_going_calls, 2, "keep-going must launch after a timeout"); + + let infrastructure = super::execute_sequential_with( + &Plan { + invocations: vec![invocation(&["only"])], + }, + false, + None, + |_, _| InvocationResult::Infrastructure("injected infrastructure failure".to_owned()), + ); + assert_eq!(infrastructure, ExitCode::from(2)); + } + + #[test] + fn parallel_failure_exit_codes_preserve_failure_class() { + assert_eq!( + parallel_failure_exit_code(&InvocationResult::Exited(failed_status(7))), + ExitCode::from(7) + ); + assert_eq!( + parallel_failure_exit_code(&InvocationResult::TimedOut(Duration::from_secs(1))), + ExitCode::from(1) + ); + assert_eq!( + parallel_failure_exit_code(&InvocationResult::Infrastructure("capture failed".to_owned())), + ExitCode::from(2) + ); + } + #[test] fn cancellation_joins_a_reader_waiting_for_pipe_readiness() { let (dropped_tx, dropped_rx) = mpsc::channel(); @@ -2255,6 +2315,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "spawns a child process through the worker")] fn worker_spawn_injection_only_rejects_the_named_program() { let worker = spawn_worker(0, invocation(&["rustc", "--version"]), None).expect("ordinary worker launch must succeed"); let outcome = wait_for_worker(&mut vec![worker]).expect("ordinary worker reports its outcome"); @@ -2550,6 +2611,24 @@ mod tests { let error = emit_buffered_to(&invocation(&["probe"]), &mut stderr_failure, &mut Vec::new(), &mut FailingWriter) .expect_err("stderr destination failure must propagate"); assert!(error.to_string().contains("injected destination write failure")); + + for (result, diagnostic) in [ + (InvocationResult::TimedOut(Duration::from_millis(250)), "timed out after 250ms"), + ( + InvocationResult::Infrastructure("capture infrastructure failed".to_owned()), + "capture infrastructure failed", + ), + ] { + let mut outcome = BufferedOutcome { + stdout: CapturedOutput::empty(), + stderr: CapturedOutput::empty(), + result, + }; + let mut stderr = Vec::new(); + emit_buffered_to(&invocation(&["probe"]), &mut outcome, &mut Vec::new(), &mut stderr) + .expect("memory destinations remain writable"); + assert!(String::from_utf8(stderr).expect("diagnostics are UTF-8").contains(diagnostic)); + } } #[test] @@ -2675,20 +2754,28 @@ mod tests { let mut command = Command::new("rustc"); let _ = command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); let mut tree = spawn_tree(command).expect("spawn contained rustc probe"); - let deadline = Instant::now() + Duration::from_secs(2); - loop { - match tree.observe().expect("observe contained rustc probe") { - Some(status) => { - assert!(status.success()); - break; - } - None if Instant::now() < deadline => thread::sleep(Duration::from_millis(5)), - None => { - let _cleanup = tree.terminate(); - panic!("process-control delegation did not observe the exited probe before the deadline"); - } - } - } + let outcome = wait_for_tree(&mut tree, Duration::from_secs(2)); + let InvocationResult::Exited(status) = outcome.result else { + panic!("the contained rustc probe must exit before its deadline"); + }; + assert!(status.success()); + } + + #[test] + #[cfg_attr(miri, ignore = "spawns and captures a contained child process")] + fn captured_contained_process_delegates_pipes_and_waiting() { + let mut outcome = run_captured_with_spawner(&invocation(&["rustc", "--version"]), Some(Duration::from_secs(2)), |command| { + spawn_tree(command).map(CapturedProcess::Contained) + }); + let InvocationResult::Exited(status) = outcome.result else { + panic!("the captured contained rustc probe must exit"); + }; + assert!(status.success()); + assert!( + String::from_utf8(output_bytes(&mut outcome.stdout)) + .expect("rustc output is UTF-8") + .contains("rustc") + ); } #[test] diff --git a/crates/cargo-gamma-process/docs/DESIGN.md b/crates/cargo-gamma-process/docs/DESIGN.md index 33941ca82..9c613e096 100644 --- a/crates/cargo-gamma-process/docs/DESIGN.md +++ b/crates/cargo-gamma-process/docs/DESIGN.md @@ -74,7 +74,10 @@ therefore covers the complete descendant tree. kill is transferred to a shared detached reaper rather than handed to an indefinite `wait` or Drop path. The reaper polls all retained leaders so one survivor cannot block collection of the others; the containment handles - remain owned until the `ProcessTree` itself is dropped. + remain owned until the `ProcessTree` itself is dropped. An error while + polling the leader follows the same handoff before the observation error is + returned, because an observation failure does not prove the child was + reaped. - Sealed containment uses a boundary that descendants cannot leave. A host that offers no sealed boundary at all silently uses best-effort process-group containment for an unmetered launch; absence of a warning does not establish diff --git a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md index f4b56dc87..9f1a44e7b 100644 --- a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md +++ b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md @@ -25,7 +25,9 @@ the leader handle moves to the shared detached reaper, the `ProcessTree` no longer owns a child that Drop could wait for, and the original cleanup failure is retained in the returned error. The reaper polls every retained child without blocking on one leader, so later handoffs remain collectable even when -an earlier leader survives. +an earlier leader survives. A `try_wait` error also transfers the still-owned +leader handle before returning the observation error; only a successful +`Some(status)` proves that no later reaping is required. ## Platform composition diff --git a/crates/cargo-gamma-process/src/faults.rs b/crates/cargo-gamma-process/src/faults.rs index 64c73425d..66335ab58 100644 --- a/crates/cargo-gamma-process/src/faults.rs +++ b/crates/cargo-gamma-process/src/faults.rs @@ -31,6 +31,9 @@ pub enum Fault { /// The termination request reports success without signalling the leader. Linger, + + /// Observing the leader after termination returns an operating-system error. + Observe, } /// Arms `fault` on this thread until the returned value is dropped. @@ -110,6 +113,7 @@ mod tests { assert!(!fired(Fault::Terminate)); assert!(!fired(Fault::Kill)); assert!(!fired(Fault::Linger)); + assert!(!fired(Fault::Observe)); } #[test] diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index b3eba6a13..b2f1f14ca 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -119,6 +119,11 @@ fn reaper_contains(id: u32) -> bool { .any(|child| child.id() == id) } +#[cfg(test)] +fn reaper_running() -> bool { + CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner).running +} + /// How many concurrent child subtrees can be watched for terminal interruption. #[cfg(unix)] #[must_use] @@ -1424,7 +1429,16 @@ impl ProcessTree { self.release(); loop { - match child.try_wait() { + #[cfg(any(test, feature = "fault-injection"))] + let observed = if faults::fired(faults::Fault::Observe) { + Err(io::Error::other("subtree observation failed as requested by a test")) + } else { + child.try_wait() + }; + #[cfg(not(any(test, feature = "fault-injection")))] + let observed = child.try_wait(); + + match observed { Ok(Some(status)) => { if let Some(error) = kill_error.take() { return Err(error); @@ -1438,7 +1452,14 @@ impl ProcessTree { return Ok(status); } Ok(None) => {} - Err(error) => return Err(error), + Err(error) => { + let kind = error.kind(); + let mut message = error.to_string(); + if let Err(reaper) = reap_later(child) { + message = format!("{message}; the detached child reaper could not be started: {reaper}"); + } + return Err(io::Error::new(kind, message)); + } } let Some(remaining) = grace.checked_sub(started.elapsed()) else { @@ -1738,6 +1759,8 @@ mod tests { use super::*; use crate::testing; + static REAPER_TEST_LOCK: Mutex<()> = Mutex::new(()); + struct PausedReader { reads: usize, ready: std::sync::mpsc::SyncSender<()>, @@ -2380,6 +2403,7 @@ mod tests { #[test] fn bounded_termination_does_not_wait_forever_after_a_failed_kill() { + let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let work = testing::workdir("gamma-bounded-termination"); let finished = Utf8Path::from_path(work.path()) .expect("the temporary path is UTF-8") @@ -2474,6 +2498,7 @@ mod tests { #[test] fn bounded_termination_times_out_when_a_successful_signal_is_ignored() { + let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let work = testing::workdir("gamma-bounded-linger"); let finished = Utf8Path::from_path(work.path()) .expect("the temporary path is UTF-8") @@ -2498,6 +2523,77 @@ mod tests { assert!(finished.exists(), "the deliberately un-signalled leader did not finish"); } + #[test] + fn bounded_termination_hands_an_observation_error_to_the_reaper() { + let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + let mut command = Command::new(testing::helper_binary_path().as_std_path()); + let _ = command.arg(testing::directive("sleep:30000")); + let prepared = prepare(command, MemoryRequest::default()).expect("containment"); + let spawned = prepared.spawn().expect("spawn"); + let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); + let leader = subtree.child.as_ref().expect("the adopted subtree owns its leader").id(); + let _failed_observation = faults::arm(faults::Fault::Observe); + + let error = subtree + .terminate_bounded(Duration::from_secs(1)) + .expect_err("the injected observation failure must be reported"); + + assert!(error.to_string().contains("observation failed as requested"), "{error}"); + assert!(subtree.child.is_none(), "the failed observation must transfer leader ownership"); + wait_for_reaper_to_collect(leader, Duration::from_secs(2)); + } + + #[test] + fn shared_reaper_collects_out_of_order_and_restarts_after_idle() { + let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + wait_for_reaper_to_stop(Duration::from_secs(2)); + + let long = spawn_reaper_probe(750); + let long_id = long.id(); + let short = spawn_reaper_probe(50); + let short_id = short.id(); + reap_later(long).expect("start the shared reaper"); + reap_later(short).expect("add a second child without starting another reaper"); + + wait_for_reaper_to_collect(short_id, Duration::from_secs(2)); + assert!( + reaper_contains(long_id), + "the shorter-lived later child must be collected before the earlier long-lived child" + ); + wait_for_reaper_to_collect(long_id, Duration::from_secs(2)); + wait_for_reaper_to_stop(Duration::from_secs(2)); + + let restarted = spawn_reaper_probe(50); + let restarted_id = restarted.id(); + reap_later(restarted).expect("restart the shared reaper after its queue became empty"); + assert!(reaper_running(), "the shared reaper did not restart"); + wait_for_reaper_to_collect(restarted_id, Duration::from_secs(2)); + wait_for_reaper_to_stop(Duration::from_secs(2)); + } + + fn spawn_reaper_probe(sleep_millis: u64) -> Child { + Command::new(testing::helper_binary_path().as_std_path()) + .arg(testing::directive(format_args!("sleep:{sleep_millis}"))) + .spawn() + .expect("spawn reaper probe") + } + + fn wait_for_reaper_to_collect(id: u32, grace: Duration) { + let deadline = Instant::now() + grace; + while reaper_contains(id) && Instant::now() < deadline { + thread::sleep(Duration::from_millis(10)); + } + assert!(!reaper_contains(id), "the shared reaper did not collect child {id}"); + } + + fn wait_for_reaper_to_stop(grace: Duration) { + let deadline = Instant::now() + grace; + while reaper_running() && Instant::now() < deadline { + thread::sleep(Duration::from_millis(10)); + } + assert!(!reaper_running(), "the shared reaper did not stop after its queue became empty"); + } + #[test] fn ordinary_termination_reports_post_reap_cleanup_failure() { let mut command = Command::new(testing::helper_binary_path().as_std_path()); diff --git a/crates/cargo-gamma-unsafe/src/pipe.rs b/crates/cargo-gamma-unsafe/src/pipe.rs index a99429cbd..26f877682 100644 --- a/crates/cargo-gamma-unsafe/src/pipe.rs +++ b/crates/cargo-gamma-unsafe/src/pipe.rs @@ -9,6 +9,7 @@ //! currently available is reported as [`io::ErrorKind::WouldBlock`]. use std::io::{self, Read}; +use std::process::{ChildStderr, ChildStdout}; /// A child pipe whose reads report [`io::ErrorKind::WouldBlock`] instead of /// waiting indefinitely for another process to write or close the pipe. @@ -17,37 +18,63 @@ pub struct InterruptiblePipe { inner: R, } -#[cfg(unix)] -impl InterruptiblePipe -where - R: std::os::fd::AsRawFd, -{ - /// Switches `inner` to nonblocking reads. +impl InterruptiblePipe { + /// Takes ownership of a child's stdout pipe and makes its reads interruptible. + /// + /// On Unix, this sets `O_NONBLOCK` on the pipe's shared open-file + /// description. Callers must not retain duplicated descriptors whose + /// blocking mode they expect to remain unchanged. On Windows, + /// [`ChildStdout`] supplies the readable anonymous pipe required by + /// `PeekNamedPipe`. + /// + /// # Errors + /// + /// Returns the operating system's error when pipe readiness cannot be + /// configured. + pub fn stdout(inner: ChildStdout) -> io::Result { + interruptible(inner) + } +} + +impl InterruptiblePipe { + /// Takes ownership of a child's stderr pipe and makes its reads interruptible. + /// + /// The platform preconditions are the same as for child stdout. /// /// # Errors /// - /// Returns the operating system's error when the descriptor flags cannot - /// be read or updated. - pub fn new(inner: R) -> io::Result { - let descriptor = inner.as_raw_fd(); - - // SAFETY: `F_GETFL` reads flags for the borrowed live descriptor and - // does not access caller memory. - let flags = unsafe { libc::fcntl(descriptor, libc::F_GETFL) }; - if flags == -1 { + /// Returns the operating system's error when pipe readiness cannot be + /// configured. + pub fn stderr(inner: ChildStderr) -> io::Result { + interruptible(inner) + } +} + +#[cfg(unix)] +fn interruptible(inner: R) -> io::Result> +where + R: std::os::fd::AsFd, +{ + use std::os::fd::AsRawFd as _; + + let descriptor = inner.as_fd().as_raw_fd(); + + // SAFETY: `AsFd` proves the descriptor remains live for this borrow. + // `F_GETFL` reads its flags and does not access caller memory. + let flags = unsafe { libc::fcntl(descriptor, libc::F_GETFL) }; + if flags == -1 { + return Err(io::Error::last_os_error()); + } + if flags & libc::O_NONBLOCK == 0 { + // SAFETY: `AsFd` keeps the same descriptor live through this call. + // Preserving every existing flag avoids changing any property other + // than nonblocking I/O. + if unsafe { libc::fcntl(descriptor, libc::F_SETFL, flags | libc::O_NONBLOCK) } == -1 { return Err(io::Error::last_os_error()); } - if flags & libc::O_NONBLOCK == 0 { - // SAFETY: `F_SETFL` updates the status flags of the same borrowed - // live descriptor. Preserving every existing flag avoids changing - // any property other than nonblocking I/O. - if unsafe { libc::fcntl(descriptor, libc::F_SETFL, flags | libc::O_NONBLOCK) } == -1 { - return Err(io::Error::last_os_error()); - } - } - - Ok(Self { inner }) } + + Ok(InterruptiblePipe { inner }) } #[cfg(unix)] @@ -70,12 +97,13 @@ mod tests { use super::InterruptiblePipe; #[test] + #[cfg_attr(miri, ignore = "spawns a child process; Miri isolation does not support process creation")] fn child_pipe_reads_are_pending_data_then_eof() { let mut command = pipe_writer(); let _ = command.stdin(Stdio::piped()).stdout(Stdio::piped()).stderr(Stdio::null()); let mut child = command.spawn().expect("spawn pipe writer"); let stdout = child.stdout.take().expect("capture child stdout"); - let mut pipe = InterruptiblePipe::new(stdout).expect("make child pipe interruptible"); + let mut pipe = InterruptiblePipe::stdout(stdout).expect("make child pipe interruptible"); let mut buffer = [0_u8; 64]; let pending = pipe.read(&mut buffer).expect_err("the blocked writer has not produced output"); @@ -127,23 +155,22 @@ mod tests { } #[cfg(windows)] -impl InterruptiblePipe { - /// Wraps `inner`; Windows pipe readiness is checked before every read. - #[expect( - clippy::unnecessary_wraps, - reason = "Unix construction configures descriptor flags and can fail; the cross-platform constructor keeps one signature" - )] - pub const fn new(inner: R) -> io::Result { - Ok(Self { inner }) - } +#[expect( + clippy::unnecessary_wraps, + reason = "Unix construction configures descriptor flags and can fail; the cross-platform constructor keeps one signature" +)] +fn interruptible(inner: R) -> io::Result> { + Ok(InterruptiblePipe { inner }) } #[cfg(windows)] impl Read for InterruptiblePipe where - R: Read + std::os::windows::io::AsRawHandle, + R: Read + std::os::windows::io::AsHandle, { fn read(&mut self, buf: &mut [u8]) -> io::Result { + use std::os::windows::io::AsRawHandle as _; + use windows_sys::Win32::Foundation::{ERROR_BROKEN_PIPE, ERROR_NO_DATA, ERROR_PIPE_NOT_CONNECTED, HANDLE}; use windows_sys::Win32::System::Pipes::PeekNamedPipe; @@ -157,7 +184,7 @@ where // count, which is written to the valid `available` pointer. let ready = unsafe { PeekNamedPipe( - self.inner.as_raw_handle().cast::() as HANDLE, + self.inner.as_handle().as_raw_handle().cast::() as HANDLE, core::ptr::null_mut(), 0, core::ptr::null_mut(), @@ -183,15 +210,13 @@ where } #[cfg(not(any(unix, windows)))] -impl InterruptiblePipe { - /// Rejects capture where no bounded child-pipe readiness primitive exists. - pub fn new(inner: R) -> io::Result { - drop(inner); - Err(io::Error::new( - io::ErrorKind::Unsupported, - "interruptible child-pipe reads require Unix nonblocking descriptors or Windows pipe readiness", - )) - } +fn interruptible(inner: R) -> io::Result> { + // No bounded child-pipe readiness primitive is available. + drop(inner); + Err(io::Error::new( + io::ErrorKind::Unsupported, + "interruptible child-pipe reads require Unix nonblocking descriptors or Windows pipe readiness", + )) } #[cfg(not(any(unix, windows)))] From 1dfc7ce9b7060774acb35eb639e384e53d416dac Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 16 Sep 2026 19:05:27 +0200 Subject: [PATCH 14/37] test(cargo-each): cover process runner boundaries Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 42 +++++++++++++++++++++++++++--------- 1 file changed, 32 insertions(+), 10 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 5da03b287..29e9ce065 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -1544,6 +1544,16 @@ mod tests { } } + fn sleeping_test_command() -> Command { + let mut command = Command::new(std::env::current_exe().expect("the test binary knows its path")); + let _ = command + .args(["--exact", "run::tests::ordinary_child_sleep_probe", "--nocapture"]) + .env("CARGO_EACH_ORDINARY_CHILD_PROBE", "1") + .stdout(Stdio::null()) + .stderr(Stdio::null()); + command + } + fn spawn_ordinary_capture(mut command: Command) -> Result { command .spawn() @@ -2116,11 +2126,7 @@ mod tests { #[test] #[cfg_attr(miri, ignore = "spawns and terminates a child process")] fn ordinary_termination_kills_polls_and_reaps_a_real_child() { - let mut child = Command::new(std::env::current_exe().expect("the test binary knows its path")) - .args(["--exact", "run::tests::ordinary_child_sleep_probe", "--nocapture"]) - .env("CARGO_EACH_ORDINARY_CHILD_PROBE", "1") - .spawn() - .expect("spawn ordinary child probe"); + let mut child = sleeping_test_command().spawn().expect("spawn ordinary child probe"); let cleanup = terminate_ordinary_child(&mut child, Duration::from_secs(1)); @@ -2341,6 +2347,12 @@ mod tests { assert!(result_infrastructure_message(run_streamed(&missing)).contains("failed to spawn")); assert!(result_infrastructure_message(run_streamed_with_timeout(&missing, Duration::from_secs(1))).contains("failed to spawn")); assert!(infrastructure_message(run_captured(&missing, None)).contains("failed to spawn")); + + let InvocationResult::Exited(status) = run_streamed_with_timeout(&invocation(&["rustc", "--version"]), Duration::from_secs(2)) + else { + panic!("the timed streamed runner must execute rustc"); + }; + assert!(status.success()); } #[test] @@ -2740,6 +2752,13 @@ mod tests { }; assert!(status.success()); + let mut timed_out = FakeProcess { + observations: VecDeque::from([Ok(None)]), + termination: Some(Ok(successful_status())), + }; + let outcome = wait_for_tree_with(&mut timed_out, Duration::ZERO, FakeProcess::observe, FakeProcess::terminate); + assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())); + let mut uncleaned = FakeProcess { observations: VecDeque::from([Ok(None)]), termination: Some(Err(io::Error::other("termination failed"))), @@ -2750,7 +2769,7 @@ mod tests { #[test] #[cfg_attr(miri, ignore = "spawns a contained child process")] - fn process_tree_control_delegates_real_exit_observation() { + fn process_tree_control_delegates_real_observation_and_termination() { let mut command = Command::new("rustc"); let _ = command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); let mut tree = spawn_tree(command).expect("spawn contained rustc probe"); @@ -2759,6 +2778,10 @@ mod tests { panic!("the contained rustc probe must exit before its deadline"); }; assert!(status.success()); + + let mut tree = spawn_tree(sleeping_test_command()).expect("spawn contained sleep probe"); + let outcome = wait_for_tree(&mut tree, Duration::ZERO); + assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())); } #[test] @@ -2830,15 +2853,14 @@ mod tests { .contains("already reaped or detached") ); - let mut contained_command = Command::new("rustc"); - let _ = contained_command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); - let mut contained = CapturedProcess::Contained(spawn_tree(contained_command).expect("spawn contained rustc")); + let mut contained = CapturedProcess::Contained(spawn_tree(sleeping_test_command()).expect("spawn contained sleep probe")); assert_eq!(contained.drain_boundary(), "contained process tree"); assert!( result_infrastructure_message(contained.wait(None, None, &mut stdout, &mut stderr).result) .contains("did not match timeout configuration") ); - let _first = contained.terminate_bounded(); + let outcome = contained.wait(Some(Duration::ZERO), None, &mut stdout, &mut stderr); + assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())); assert!( contained.terminate_bounded().is_err(), "a fabricated successful second termination must not be accepted" From 6db2494c623ac3cd3e95190e04df4e5977ce5aba Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 16 Sep 2026 19:07:45 +0200 Subject: [PATCH 15/37] fix(cargo-gamma-process): bound child reaper failures Stop retaining child handles after try_wait reports an observation error, preventing the shared detached reaper from retrying an unobservable process forever. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- .../cargo-gamma-process/src/process_tree.rs | 25 ++++++++++++++++++- 1 file changed, 24 insertions(+), 1 deletion(-) diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index b2f1f14ca..17144412d 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -94,7 +94,10 @@ fn child_reaper_loop() { loop { let empty = { let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - reaper.children.retain_mut(|child| !matches!(child.try_wait(), Ok(Some(_status)))); + reaper.children.retain_mut(|child| { + let id = child.id(); + retain_reaper_child(id, child.try_wait()) + }); if reaper.children.is_empty() { reaper.running = false; true @@ -109,6 +112,17 @@ fn child_reaper_loop() { } } +fn retain_reaper_child(id: u32, observation: io::Result>) -> bool { + match observation { + Ok(None) => true, + Ok(Some(_status)) => false, + Err(error) => { + eprintln!("warning: detached child reaper stopped tracking process {id} after observation failed: {error}"); + false + } + } +} + #[cfg(test)] fn reaper_contains(id: u32) -> bool { CHILD_REAPER @@ -1819,6 +1833,15 @@ mod tests { ); } + #[test] + fn detached_reaper_drops_unobservable_children() { + assert!(retain_reaper_child(17, Ok(None))); + assert!(!retain_reaper_child( + 17, + Err(io::Error::new(io::ErrorKind::InvalidInput, "invalid child handle")), + )); + } + struct FailingReader; impl io::Read for FailingReader { From aca3c28d0f177e8a1ba2c87506805c7645acfa83 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 16 Sep 2026 20:17:58 +0200 Subject: [PATCH 16/37] feat(cargo-each): support automatic concurrency Allow --jobs auto to resolve the OS-visible available parallelism while preserving sequential execution when the option is omitted. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/README.md | 29 +++++++----- crates/cargo-each/docs/design/README.md | 31 +++++++------ crates/cargo-each/src/cli.rs | 61 +++++++++++++++++++++++-- crates/cargo-each/src/main.rs | 29 +++++++----- crates/cargo-each/tests/cli.rs | 45 ++++++++++++++++++ 5 files changed, 155 insertions(+), 40 deletions(-) diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index 30d3b8502..2a7dc54ee 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -85,9 +85,12 @@ can be double-quoted. Expression atoms: `--target-required-feature` further narrows targets. `--keep-going` runs every invocation and exits non-zero if any failed -(default is fail-fast). `--jobs ` bounds concurrent per-package or -per-target work (default `1`), while `--timeout ` terminates each -invocation and its process tree independently (`250ms`, `30s`, or `2m`). +(default is fail-fast). `--jobs ` bounds concurrent per-package or +per-target work. Omitting it runs exactly one invocation at a time; `auto` +resolves once to the machine’s available parallelism. Detection failure is +reported explicitly without falling back. `--timeout ` terminates +each invocation and its process tree independently (`250ms`, `30s`, or +`2m`). Timeouts require sealed process-tree containment; on a host that only offers best-effort containment, cargo-each reports an unsupported infrastructure failure before starting the child. @@ -125,18 +128,20 @@ Workspace Rust-version validation is lazy: it runs only when the command uses `{workspace-rust-version}`, then requires every member’s resolved minimum to be present and no newer than the root floor. -With `--jobs > 1`, each invocation’s output is buffered and complete blocks -are emitted in deterministic plan order. Fail-fast stops launching after -the first observed failure, waits for running work, and chooses the final -failure by plan order. `--keep-going` runs the complete plan. Worker panics -and unexpected worker-channel disconnections become infrastructure-failure +With `--jobs > 1`, including when `auto` resolves above one, the effective +worker count is capped by the plan size and scheduler capacity. Each +invocation’s output is buffered and complete blocks are emitted in +deterministic plan order. Fail-fast stops launching after the first +observed failure, waits for running work, and chooses the final failure by +plan order. `--keep-going` runs the complete plan. Worker panics and +unexpected worker-channel disconnections become infrastructure-failure outcomes instead of blocking the scheduler. Worker launch failures retain output already collected at earlier plan indices. Without `--timeout`, parallel commands retain ordinary direct-child semantics and do not kill -background descendants. Each output stream retains at most 1 MiB in memory before -spilling to a unique system-temporary file owned by the invocation outcome; -spill failures are infrastructure failures and spill files are removed by -RAII after deterministic plan-order emission. +background descendants. Each output stream retains at most 1 MiB in memory +before spilling to a unique system-temporary file owned by the invocation +outcome; spill failures are infrastructure failures and spill files are +removed by RAII after deterministic plan-order emission. Reader failures are observed while the child is running and trigger bounded termination. Output drain is bounded after every completion: readers get diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index d339c5ac5..2e0c564ff 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -250,7 +250,7 @@ filtered set is empty, `cargo-each` exits 0, exactly like an empty selection. | `--each-target ` | **per-target**: run once for each selected member target of `KIND`. Repeatable; kinds are OR-combined and each target runs at most once. Mutually exclusive with `--once`. | | `--target-required-feature ` | In per-target mode, retain targets whose `required-features` contains `FEATURE`. Repeatable; values are AND-combined. Requires `--each-target`. | | `--keep-going` | Don't stop at the first failing command; run them all and exit non-zero if any failed. Default is fail-fast (exit with the first failure's code). | -| `--jobs ` | Run at most `N` per-package or per-target commands concurrently. Default `1`. With `--once`, values other than `1` are a usage error. | +| `--jobs ` | Run at most the positive integer `N` per-package or per-target commands concurrently. When omitted, the default is exactly `1`. `auto` resolves once during CLI parsing via `std::thread::available_parallelism()`; detection failure is an explicit usage error with no fallback. The effective worker count remains capped by the plan size and scheduler capacity. With `--once`, resolved values other than `1` are a usage error. | | `--timeout ` | Terminate an invocation and its child process tree when it exceeds the positive duration, such as `30s` or `2m`. Applies independently to every invocation, including `--once`. Requires sealed process-tree containment; unsupported hosts fail before the child starts. No timeout by default. | | `--chdir` | Run each per-package or per-target command from that member's crate root (the directory containing its `Cargo.toml`) instead of the caller's CWD. Combined with `--once` it is a usage error (exit 2). Placeholders stay absolute, so only *relative* args in the command shift to the member dir. | | `--manifest-path ` | Workspace root `Cargo.toml`. Defaults to auto-detection from CWD. | @@ -302,18 +302,23 @@ no-op. - **No shell.** The command is spawned directly (argv, not a shell string), so there is no quoting/dialect surface. Placeholder expansion is textual and happens before spawn. -- **Bounded concurrency.** With `--jobs > 1`, output from each invocation is - buffered and emitted as one block in deterministic plan order. Fail-fast - stops launching new work after the first observed failure and waits for - already-running children; `--keep-going` launches the complete plan. The - final failure is chosen by plan order, not scheduler timing. A worker panic - is converted into an infrastructure-failure outcome; each worker has a - dedicated completion channel, so an unexpected exit is observable as - disconnection rather than leaving the scheduler blocked forever. A - worker-thread launch failure is represented as an infrastructure outcome at - that invocation's plan index, so output already collected from earlier - invocations is still emitted. Without `--timeout`, parallel commands use the - ordinary direct-child lifecycle: +- **Bounded concurrency.** Omitting `--jobs` requests exactly one concurrent + invocation. A positive integer requests that fixed limit; `auto` resolves + exactly once during CLI parsing to the machine's available parallelism and + fails explicitly if detection is unavailable. The scheduler caps every + request by the plan size and its process capacity. With an effective job + count above one, output from each invocation is buffered and emitted as one + block in deterministic plan order. Fail-fast stops launching new work after + the first observed failure and waits for already-running children; + `--keep-going` launches the complete plan. The final failure is chosen by + plan order, not scheduler timing. A worker panic is converted into an + infrastructure-failure outcome; each worker has a dedicated completion + channel, so an unexpected exit is observable as disconnection rather than + leaving the scheduler blocked forever. A worker-thread launch failure is + represented as an infrastructure outcome at that invocation's plan index, + so output already collected from earlier invocations is still emitted. + Without `--timeout`, parallel commands use the ordinary direct-child + lifecycle: cargo-each waits for the launched leader but does not contain or kill background descendants. Buffering is memory-bounded per stream: after 1 MiB, output spills to a unique file in the system temporary directory. The diff --git a/crates/cargo-each/src/cli.rs b/crates/cargo-each/src/cli.rs index 3db1b74af..21fe5ee82 100644 --- a/crates/cargo-each/src/cli.rs +++ b/crates/cargo-each/src/cli.rs @@ -94,9 +94,10 @@ pub(crate) struct EachArgs { #[arg(long)] pub(crate) keep_going: bool, - /// Run at most N per-package or per-target commands concurrently. Buffered + /// Run at most N per-package or per-target commands concurrently. Use + /// `auto` to detect available parallelism once. Defaults to 1. Buffered /// output spills to unique system-temporary files beyond 1 MiB per stream. - #[arg(long, default_value_t = NonZeroUsize::MIN, value_name = "N")] + #[arg(long, default_value_t = NonZeroUsize::MIN, value_name = "N|auto", value_parser = parse_jobs)] pub(crate) jobs: NonZeroUsize, /// Terminate each invocation and its process tree after this duration. @@ -120,6 +121,20 @@ pub(crate) struct EachArgs { pub(crate) command: Vec, } +fn parse_jobs(value: &str) -> Result { + parse_jobs_with(value, std::thread::available_parallelism) +} + +fn parse_jobs_with(value: &str, available_parallelism: impl FnOnce() -> std::io::Result) -> Result { + if value == "auto" { + available_parallelism().map_err(|error| format!("failed to detect available parallelism for `--jobs auto`: {error}")) + } else { + value + .parse() + .map_err(|error| format!("expected a positive integer or `auto`: {error}")) + } +} + fn parse_duration(value: &str) -> Result { enum Unit { Milliseconds, @@ -159,7 +174,7 @@ fn parse_duration(value: &str) -> Result { #[cfg(test)] #[cfg_attr(coverage_nightly, coverage(off))] mod tests { - use clap::CommandFactory; + use clap::{CommandFactory, Parser as _}; use super::*; @@ -168,6 +183,46 @@ mod tests { CargoCli::command().debug_assert(); } + #[test] + fn jobs_default_is_one() { + let CargoCli::Each(args) = + CargoCli::try_parse_from(["cargo", "each", "--", "echo"]).expect("documented minimal invocation must parse"); + assert_eq!(args.jobs, NonZeroUsize::MIN); + } + + #[test] + fn parses_positive_and_auto_jobs() { + assert_eq!( + parse_jobs_with("7", || panic!("numeric jobs must not detect parallelism")), + Ok(NonZeroUsize::new(7).expect("literal seven is nonzero")) + ); + + let mut detections = 0; + let jobs = parse_jobs_with("auto", || { + detections += 1; + Ok(NonZeroUsize::new(8).expect("literal eight is nonzero")) + }); + assert_eq!(jobs, Ok(NonZeroUsize::new(8).expect("literal eight is nonzero"))); + assert_eq!(detections, 1); + } + + #[test] + fn rejects_invalid_jobs() { + for value in ["0", "-1", "bogus", "AUTO"] { + assert!(parse_jobs(value).is_err(), "{value}"); + } + } + + #[test] + fn reports_auto_jobs_detection_failure() { + let error = parse_jobs_with("auto", || Err(std::io::Error::other("parallelism unavailable"))) + .expect_err("detection failure must not fall back"); + assert_eq!( + error, + "failed to detect available parallelism for `--jobs auto`: parallelism unavailable" + ); + } + #[test] fn parses_documented_durations() { assert_eq!(parse_duration("250ms"), Ok(Duration::from_millis(250))); diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 5d9e0beb2..6c3105454 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -75,9 +75,12 @@ //! `--target-required-feature` further narrows targets. //! //! `--keep-going` runs every invocation and exits non-zero if any failed -//! (default is fail-fast). `--jobs ` bounds concurrent per-package or -//! per-target work (default `1`), while `--timeout ` terminates each -//! invocation and its process tree independently (`250ms`, `30s`, or `2m`). +//! (default is fail-fast). `--jobs ` bounds concurrent per-package or +//! per-target work. Omitting it runs exactly one invocation at a time; `auto` +//! resolves once to the machine's available parallelism. Detection failure is +//! reported explicitly without falling back. `--timeout ` terminates +//! each invocation and its process tree independently (`250ms`, `30s`, or +//! `2m`). //! Timeouts require sealed process-tree containment; on a host that only //! offers best-effort containment, cargo-each reports an unsupported //! infrastructure failure before starting the child. @@ -115,18 +118,20 @@ //! uses `{workspace-rust-version}`, then requires every member's resolved //! minimum to be present and no newer than the root floor. //! -//! With `--jobs > 1`, each invocation's output is buffered and complete blocks -//! are emitted in deterministic plan order. Fail-fast stops launching after -//! the first observed failure, waits for running work, and chooses the final -//! failure by plan order. `--keep-going` runs the complete plan. Worker panics -//! and unexpected worker-channel disconnections become infrastructure-failure +//! With `--jobs > 1`, including when `auto` resolves above one, the effective +//! worker count is capped by the plan size and scheduler capacity. Each +//! invocation's output is buffered and complete blocks are emitted in +//! deterministic plan order. Fail-fast stops launching after the first +//! observed failure, waits for running work, and chooses the final failure by +//! plan order. `--keep-going` runs the complete plan. Worker panics and +//! unexpected worker-channel disconnections become infrastructure-failure //! outcomes instead of blocking the scheduler. Worker launch failures retain //! output already collected at earlier plan indices. Without `--timeout`, //! parallel commands retain ordinary direct-child semantics and do not kill -//! background descendants. Each output stream retains at most 1 MiB in memory before -//! spilling to a unique system-temporary file owned by the invocation outcome; -//! spill failures are infrastructure failures and spill files are removed by -//! RAII after deterministic plan-order emission. +//! background descendants. Each output stream retains at most 1 MiB in memory +//! before spilling to a unique system-temporary file owned by the invocation +//! outcome; spill failures are infrastructure failures and spill files are +//! removed by RAII after deterministic plan-order emission. //! //! Reader failures are observed while the child is running and trigger bounded //! termination. Output drain is bounded after every completion: readers get diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index 7eb3e7ffd..782946bb4 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -1326,6 +1326,51 @@ fn once_rejects_jobs_greater_than_one() { .stderr(predicate::str::contains("--jobs").and(predicate::str::contains("--once"))); } +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn jobs_help_documents_auto_and_default() { + Command::cargo_bin("cargo-each") + .expect("binary") + .args(["each", "--help"]) + .assert() + .success() + .stdout( + predicate::str::contains("--jobs ") + .and(predicate::str::contains("available parallelism")) + .and(predicate::str::contains("Defaults to 1")), + ); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn auto_jobs_is_accepted() { + let (_tmp, manifest) = fixture(); + each(&manifest) + .args(["--workspace", "--jobs", "auto", "--dry-run", "--", "echo", "{name}"]) + .assert() + .success(); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn default_jobs_runs_one_invocation_at_a_time() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + let completion_log = tmp.path().join("completion.log"); + each(&manifest) + .args(["-p", "alpha", "-p", "beta", "--"]) + .arg(probe) + .args(["ordered", "{name}"]) + .arg(&completion_log) + .assert() + .success(); + assert_eq!( + fs::read_to_string(completion_log).expect("completion log"), + "alpha\nbeta\n", + "the default must finish each invocation before launching the next" + ); +} + #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] #[test] fn parallel_output_is_buffered_in_plan_order() { From 236e91957d6a334e004f63a146b8b562227ed6a1 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 16 Sep 2026 21:14:20 +0200 Subject: [PATCH 17/37] fix(process): preserve reaper handoff ownership Start a durable shared reaper before child creation, return child ownership when direct handoff startup fails, and make placeholder expansion single-pass so inserted paths are not rescanned. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/docs/design/README.md | 4 + crates/cargo-each/src/run.rs | 26 +- crates/cargo-each/src/substitute.rs | 95 +++++-- crates/cargo-gamma-process/docs/DESIGN.md | 25 +- .../docs/IMPLEMENTATION.md | 18 +- crates/cargo-gamma-process/src/faults.rs | 4 + crates/cargo-gamma-process/src/lib.rs | 3 +- .../cargo-gamma-process/src/process_tree.rs | 237 +++++++++++++----- 8 files changed, 308 insertions(+), 104 deletions(-) diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index 2e0c564ff..88cb79745 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -274,6 +274,10 @@ Per-target mode accepts all per-package placeholders plus `{target}`. Using a per-package or per-target token in `--once` mode, `{target}` in per-package mode, or `{packages}` outside `--once` is a usage error. +Substitution scans each template argument once. Text inserted for one +placeholder is never scanned as another placeholder, so literal token-shaped +path components in manifest paths and other replacement values are preserved. + `{workspace-rust-version}` is workspace-scoped rather than tied to one selected member. Resolving it requires a root declaration. cargo-each also requires every workspace member to expose a resolved `rust_version` no newer than the root diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 29e9ce065..ea10fca90 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -14,7 +14,7 @@ use std::sync::{Arc, Mutex, mpsc}; use std::time::{Duration, Instant}; use std::{fmt, thread}; -use cargo_gamma_process::{InterruptiblePipe, MemoryRequest, PreparedCommand, ProcessTree, prepare, reap_later}; +use cargo_gamma_process::{InterruptiblePipe, MemoryRequest, PreparedCommand, ProcessTree, ensure_reaper, prepare, reap_later}; use cargo_metadata::TargetKind; use ohno::{AppError, IntoAppError}; @@ -381,8 +381,8 @@ fn run_captured_with_spawner( let _ = command.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); let process = match timeout { Some(_) => timed_spawner(command), - None => command - .spawn() + None => ensure_reaper() + .and_then(|()| command.spawn()) .map(|child| CapturedProcess::Ordinary(Some(child))) .map_err(|error| error.to_string()), }; @@ -611,12 +611,15 @@ fn finish_ordinary_termination(child: Child, result: io::Result) -> finish_ordinary_termination_with(child, result, Child::try_wait, reap_later) } -fn finish_ordinary_termination_with( +fn finish_ordinary_termination_with( mut control: T, result: io::Result, observe: impl FnOnce(&mut T) -> io::Result>, - reap: impl FnOnce(T) -> io::Result<()>, -) -> io::Result { + reap: impl FnOnce(T) -> Result<(), E>, +) -> io::Result +where + E: fmt::Display, +{ match observe(&mut control) { Ok(Some(_status)) => result, Ok(None) | Err(_) => match reap(control) { @@ -639,12 +642,15 @@ fn finish_ordinary_wait(child: Child, outcome: TreeOutcome) -> TreeOutcome { finish_ordinary_wait_with(child, outcome, Child::try_wait, reap_later) } -fn finish_ordinary_wait_with( +fn finish_ordinary_wait_with( mut control: T, mut outcome: TreeOutcome, observe: impl FnOnce(&mut T) -> io::Result>, - reap: impl FnOnce(T) -> io::Result<()>, -) -> TreeOutcome { + reap: impl FnOnce(T) -> Result<(), E>, +) -> TreeOutcome +where + E: fmt::Display, +{ if !matches!(observe(&mut control), Ok(Some(_status))) && let Err(error) = reap(control) { @@ -1820,7 +1826,7 @@ mod tests { #[test] fn ordinary_child_handoff_preserves_primary_and_reaper_failures() { - let status = finish_ordinary_termination_with((), Ok(successful_status()), |()| Ok(None), |()| Ok(())) + let status = finish_ordinary_termination_with((), Ok(successful_status()), |()| Ok(None), |()| Ok::<(), io::Error>(())) .expect("a successful handoff preserves the termination status"); assert!(status.success()); diff --git a/crates/cargo-each/src/substitute.rs b/crates/cargo-each/src/substitute.rs index c471102a5..729d9efcf 100644 --- a/crates/cargo-each/src/substitute.rs +++ b/crates/cargo-each/src/substitute.rs @@ -87,14 +87,30 @@ impl Placeholders { } } -fn replace_workspace_rust_version(arg: String, placeholders: &Placeholders) -> Result { - if !arg.contains(WORKSPACE_RUST_VERSION_TOKEN) { - return Ok(arg); +fn replace_arg<'a>(arg: &str, placeholders: &'a Placeholders, mut replacements: Vec<(&'static str, &'a str)>) -> Result { + if arg.contains(WORKSPACE_RUST_VERSION_TOKEN) { + let version = placeholders.workspace_rust_version().ok_or_else(|| { + WorkspaceRustVersionError::new("the command uses the placeholder but its root value was not resolved".to_owned()) + })?; + replacements.push((WORKSPACE_RUST_VERSION_TOKEN, version)); } - let version = placeholders - .workspace_rust_version() - .ok_or_else(|| WorkspaceRustVersionError::new("the command uses the placeholder but its root value was not resolved".to_owned()))?; - Ok(arg.replace(WORKSPACE_RUST_VERSION_TOKEN, version)) + + let mut rest = arg; + let mut replaced = String::with_capacity(arg.len()); + while let Some((offset, token, value)) = replacements + .iter() + .filter_map(|&(token, value)| rest.find(token).map(|offset| (offset, token, value))) + .min_by_key(|&(offset, _, _)| offset) + { + let (literal, token_and_rest) = rest.split_at(offset); + let (_token, remaining) = token_and_rest.split_at(token.len()); + replaced.push_str(literal); + replaced.push_str(value); + rest = remaining; + } + replaced.push_str(rest); + + Ok(replaced) } /// Whether a command template uses the lazy workspace Rust-version token. @@ -175,18 +191,17 @@ pub(crate) fn substitute(args: &[String], placeholders: &Placeholders) -> Result manifest, .. } => { - let original = replace_workspace_rust_version(arg.clone(), placeholders)?; // The `{name}` / `{spec}` / … literals are cargo-each // placeholder tokens, not Rust format-string arguments. #[expect( clippy::literal_string_with_formatting_args, reason = "cargo-each placeholder tokens, not format args" )] - let replaced = original - .replace("{name}", name) - .replace("{spec}", spec) - .replace("{version}", version) - .replace("{manifest}", manifest); + let replaced = replace_arg( + arg, + placeholders, + vec![("{name}", name), ("{spec}", spec), ("{version}", version), ("{manifest}", manifest)], + )?; out.push(replaced); } Placeholders::Target { @@ -197,17 +212,21 @@ pub(crate) fn substitute(args: &[String], placeholders: &Placeholders) -> Result target, .. } => { - let original = replace_workspace_rust_version(arg.clone(), placeholders)?; #[expect( clippy::literal_string_with_formatting_args, reason = "cargo-each placeholder tokens, not format args" )] - let replaced = original - .replace("{name}", name) - .replace("{spec}", spec) - .replace("{version}", version) - .replace("{manifest}", manifest) - .replace(TARGET_TOKEN, target); + let replaced = replace_arg( + arg, + placeholders, + vec![ + ("{name}", name), + ("{spec}", spec), + ("{version}", version), + ("{manifest}", manifest), + (TARGET_TOKEN, target), + ], + )?; out.push(replaced); } Placeholders::Once { packages, .. } => { @@ -216,7 +235,7 @@ pub(crate) fn substitute(args: &[String], placeholders: &Placeholders) -> Result if arg == PACKAGES_TOKEN { out.extend(packages.iter().cloned()); } else { - out.push(replace_workspace_rust_version(arg.clone(), placeholders)?); + out.push(replace_arg(arg, placeholders, Vec::new())?); } } } @@ -348,6 +367,24 @@ mod tests { ); } + #[test] + fn unresolved_workspace_rust_version_is_reported_in_package_and_target_modes() { + let command = args(&["echo", "{workspace-rust-version}"]); + let package_error = substitute(&command, &pkg()).expect_err("package value is unresolved"); + assert!(package_error.to_string().contains("root value was not resolved"), "{package_error}"); + + let target = Placeholders::Target { + name: "crate".to_owned(), + spec: "crate@1.0.0".to_owned(), + version: "1.0.0".to_owned(), + manifest: "/ws/crate/Cargo.toml".to_owned(), + target: "example".to_owned(), + workspace_rust_version: None, + }; + let target_error = substitute(&command, &target).expect_err("target value is unresolved"); + assert!(target_error.to_string().contains("root value was not resolved"), "{target_error}"); + } + #[test] fn package_values_are_not_rescanned_for_workspace_tokens() { let placeholders = Placeholders::Package { @@ -380,6 +417,22 @@ mod tests { ); } + #[test] + fn manifest_values_are_not_rescanned_for_target_tokens() { + let placeholders = Placeholders::Target { + name: "crate".to_owned(), + spec: "crate@1.0.0".to_owned(), + version: "1.0.0".to_owned(), + manifest: "/ws/{target}/crate/Cargo.toml".to_owned(), + target: "example".to_owned(), + workspace_rust_version: None, + }; + assert_eq!( + substitute(&args(&["{manifest}:{target}"]), &placeholders).expect("substitute target placeholders"), + ["/ws/{target}/crate/Cargo.toml:example"] + ); + } + #[test] fn detects_workspace_rust_version_usage() { assert!(uses_workspace_rust_version(&args(&["tool", "v={workspace-rust-version}"]))); diff --git a/crates/cargo-gamma-process/docs/DESIGN.md b/crates/cargo-gamma-process/docs/DESIGN.md index 9c613e096..99984aa42 100644 --- a/crates/cargo-gamma-process/docs/DESIGN.md +++ b/crates/cargo-gamma-process/docs/DESIGN.md @@ -41,9 +41,12 @@ therefore covers the complete descendant tree. with the operating-system error so the caller can classify a transient resource-related spawn failure, back off, and retry; permanent launch failures are propagated. Success yields a distinct bundle coupling the child to its - boundary. Adoption consumes that bundle, so a successful launch cannot be - reused to create an earlier sibling awaiting adoption; abandoning the bundle - before adoption terminates and reaps the child. + boundary. Before creating that child, spawn also ensures the process-wide + detached reaper thread is running. A reaper thread-start failure therefore + returns the unchanged preparation before a repository-controlled process + exists. Adoption consumes the successful bundle, so a successful launch + cannot be reused to create an earlier sibling awaiting adoption; abandoning + the bundle before adoption terminates and reaps the child. - The contained `output` convenience mirrors `Command::output`: it disconnects stdin and captures stdout and stderr. Both pipes are drained concurrently while the child runs, avoiding pipe-capacity deadlocks. When the leader exits, @@ -73,11 +76,17 @@ therefore covers the complete descendant tree. for the caller-provided grace. A leader that remains running after a failed kill is transferred to a shared detached reaper rather than handed to an indefinite `wait` or Drop path. The reaper polls all retained leaders so one - survivor cannot block collection of the others; the containment handles - remain owned until the `ProcessTree` itself is dropped. An error while - polling the leader follows the same handoff before the observation error is - returned, because an observation failure does not prove the child was - reaped. + survivor cannot block collection of the others, remains alive while its queue + is empty, and accepts each handle only after its thread is known to exist. + Callers handing over children created outside `PreparedCommand` can preflight + the same durable thread; if a direct handoff must start it and startup fails, + the failure returns ownership of the unqueued child. The containment handles + remain owned until the `ProcessTree` itself is dropped. An error while polling + the leader follows the same handoff before the observation error is returned, + because an observation failure does not prove the child was reaped. If the + detached reaper itself later receives an observation error, it emits a warning + to stderr and permanently stops tracking that child. The released handle may + leave a zombie on Unix until this process exits. - Sealed containment uses a boundary that descendants cannot leave. A host that offers no sealed boundary at all silently uses best-effort process-group containment for an unmetered launch; absence of a warning does not establish diff --git a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md index 9f1a44e7b..2cd96555a 100644 --- a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md +++ b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md @@ -8,7 +8,10 @@ This guide records lifecycle mechanics behind [`DESIGN.md`](DESIGN.md). returns the same preparation in `SpawnFailure`; a successful spawn returns `SpawnedCommand`, which owns the child and containment until `ProcessTree::adopt` consumes it. Dropping the successful pre-adoption state terminates and reaps the -child. +child. Before calling the operating-system spawn, `PreparedCommand::spawn` +ensures the process-wide detached reaper is running. A failure to create that +thread therefore returns `SpawnFailure` with the preparation intact and no child +to recover. ## Output capture @@ -24,10 +27,15 @@ expires. It never follows a failed kill with blocking `wait`: at the deadline the leader handle moves to the shared detached reaper, the `ProcessTree` no longer owns a child that Drop could wait for, and the original cleanup failure is retained in the returned error. The reaper polls every retained child -without blocking on one leader, so later handoffs remain collectable even when -an earlier leader survives. A `try_wait` error also transfers the still-owned -leader handle before returning the observation error; only a successful -`Some(status)` proves that no later reaping is required. +without blocking on one leader and waits on a condition variable when its queue +is empty. It is durable: repository-controlled children are created only after +the thread exists, and direct `reap_later` startup failures return an unqueued +child in `ReapFailure` rather than abandoning ownership. A `try_wait` error also +transfers the still-owned leader handle before returning the observation error; +only a successful `Some(status)` proves that no later reaping is required. If a +later `try_wait` in the detached reaper fails, the reaper writes a warning to +stderr and permanently releases that child handle. On Unix, the child may +remain a zombie until this process exits. ## Platform composition diff --git a/crates/cargo-gamma-process/src/faults.rs b/crates/cargo-gamma-process/src/faults.rs index 66335ab58..67b6ce77e 100644 --- a/crates/cargo-gamma-process/src/faults.rs +++ b/crates/cargo-gamma-process/src/faults.rs @@ -34,6 +34,9 @@ pub enum Fault { /// Observing the leader after termination returns an operating-system error. Observe, + + /// Starting the shared detached child reaper is refused. + ReaperStart, } /// Arms `fault` on this thread until the returned value is dropped. @@ -114,6 +117,7 @@ mod tests { assert!(!fired(Fault::Kill)); assert!(!fired(Fault::Linger)); assert!(!fired(Fault::Observe)); + assert!(!fired(Fault::ReaperStart)); } #[test] diff --git a/crates/cargo-gamma-process/src/lib.rs b/crates/cargo-gamma-process/src/lib.rs index a0f6d06c8..8e9a8ba94 100644 --- a/crates/cargo-gamma-process/src/lib.rs +++ b/crates/cargo-gamma-process/src/lib.rs @@ -79,7 +79,8 @@ pub use memory_request::MemoryRequest; pub use memory_usage::MemoryUsage; #[doc(inline)] pub use process_tree::{ - OutputError, PreparedCommand, ProcessTree, SpawnFailure, SpawnedCommand, capacity, containment, output, prepare, reap_later, + OutputError, PreparedCommand, ProcessTree, ReapFailure, SpawnFailure, SpawnedCommand, capacity, containment, ensure_reaper, output, + prepare, reap_later, }; mod memory_request; diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index 17144412d..23a3399eb 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -42,19 +42,17 @@ static CHILD_REAPER: Mutex = Mutex::new(ChildReaper { }); static CHILD_REAPER_READY: Condvar = Condvar::new(); -/// Transfers a live child handle to the shared detached reaper. +/// Ensures the shared detached child reaper is ready before a child is spawned. /// -/// The reaper polls every retained child rather than blocking on one, so a -/// leader that survives termination cannot prevent unrelated leaders from -/// being collected. +/// The reaper is process-wide and remains alive after its queue becomes empty, +/// sleeping on a condition variable until another child is handed off. /// /// # Errors /// /// Returns the thread creation error when the shared reaper could not be -/// started. The child handle remains retained for a later start attempt. -pub fn reap_later(child: Child) -> io::Result<()> { +/// started. No child has been created or transferred by this operation. +pub fn ensure_reaper() -> io::Result<()> { let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - reaper.children.push(child); while reaper.starting { reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); } @@ -64,9 +62,19 @@ pub fn reap_later(child: Child) -> io::Result<()> { reaper.starting = true; drop(reaper); + #[cfg(any(test, feature = "fault-injection"))] + let spawned = if faults::fired(faults::Fault::ReaperStart) { + Err(io::Error::other("detached child reaper thread start failed as requested by a test")) + } else { + thread::Builder::new() + .name("cargo-gamma-child-reaper".to_owned()) + .spawn(child_reaper_loop) + }; + #[cfg(not(any(test, feature = "fault-injection")))] let spawned = thread::Builder::new() .name("cargo-gamma-child-reaper".to_owned()) .spawn(child_reaper_loop); + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); reaper.starting = false; match spawned { @@ -84,31 +92,84 @@ pub fn reap_later(child: Child) -> io::Result<()> { } } +/// Transfers a live child handle to the shared detached reaper. +/// +/// The reaper polls every retained child rather than blocking on one, so a +/// leader that survives termination cannot prevent unrelated leaders from +/// being collected. An observation error emits a warning to stderr and +/// permanently stops tracking that child. On Unix, the child may then remain a +/// zombie until this process exits. +/// +/// # Errors +/// +/// Returns [`ReapFailure`] when the shared reaper could not be started. The +/// failure retains the child handle so the caller can recover ownership. +pub fn reap_later(child: Child) -> Result<(), ReapFailure> { + if let Err(cause) = ensure_reaper() { + return Err(ReapFailure { cause, child }); + } + + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + reaper.children.push(child); + CHILD_REAPER_READY.notify_one(); + Ok(()) +} + +/// A failed detached-reaper handoff that retains ownership of the child. +#[derive(Debug)] +pub struct ReapFailure { + cause: io::Error, + child: Child, +} + +impl ReapFailure { + /// Borrows the operating-system error that prevented the reaper from starting. + #[must_use] + pub const fn cause(&self) -> &io::Error { + &self.cause + } + + /// Recovers the failure and the child that was not transferred. + #[must_use] + pub fn into_parts(self) -> (io::Error, Child) { + (self.cause, self.child) + } +} + +impl fmt::Display for ReapFailure { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.cause.fmt(f) + } +} + +impl std::error::Error for ReapFailure { + fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { + Some(&self.cause) + } +} + fn child_reaper_loop() { - { - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - while reaper.starting { - reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); - } + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + while reaper.starting { + reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); } loop { - let empty = { - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - reaper.children.retain_mut(|child| { - let id = child.id(); - retain_reaper_child(id, child.try_wait()) - }); - if reaper.children.is_empty() { - reaper.running = false; - true - } else { - false - } - }; - if empty { - return; + while reaper.children.is_empty() { + reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); + } + + reaper.children.retain_mut(|child| { + let id = child.id(); + retain_reaper_child(id, child.try_wait()) + }); + if reaper.children.is_empty() { + continue; } - thread::sleep(REAPER_PAUSE); + + let (next, _timeout) = CHILD_REAPER_READY + .wait_timeout(reaper, REAPER_PAUSE) + .unwrap_or_else(std::sync::PoisonError::into_inner); + reaper = next; } } @@ -381,9 +442,18 @@ impl PreparedCommand { /// /// # Errors /// - /// Returns whatever [`Command::spawn`] returns: the child does not exist, so there is nothing - /// here to clean up and the returned preparation remains valid for another attempt. + /// Returns the thread creation error when the durable detached reaper could + /// not be prepared, or whatever [`Command::spawn`] returns. In either case, + /// the child does not exist, so there is nothing here to clean up and the + /// returned preparation remains valid for another attempt. pub fn spawn(mut self) -> Result { + if let Err(cause) = ensure_reaper() { + return Err(SpawnFailure { + cause, + prepared: Box::new(self), + }); + } + match self.command.spawn() { Ok(child) => { let Self { command: _spawned, guard } = self; @@ -1411,17 +1481,19 @@ impl ProcessTree { /// /// Unlike [`Self::terminate`], this method never performs a blocking /// [`Child::wait`] after signalling. If the leader remains alive at the - /// deadline, its handle is transferred to the shared detached reaper and - /// this process tree is left without a child for [`Drop`] to wait on. The - /// surrounding cgroup or job handle remains owned by `self` and is released - /// normally when the process tree is dropped. + /// deadline, its handle is transferred to the shared detached reaper. A + /// successful handoff leaves this process tree without a child for [`Drop`] + /// to wait on. The surrounding cgroup or job handle remains owned by `self` + /// and is released normally when the process tree is dropped. /// /// # Errors /// /// Returns the observation error if `try_wait` fails, the termination /// error if the leader exits after signalling but cleanup had failed, or a /// timed-out error (including an earlier termination error, when present) - /// if the leader is still running after `grace`. + /// if the leader is still running after `grace`. If the detached reaper + /// cannot accept the child, the returned error includes that failure and + /// this process tree retains the child handle. pub fn terminate_bounded(&mut self, grace: Duration) -> io::Result { let started = Instant::now(); let mut child = self @@ -1469,7 +1541,9 @@ impl ProcessTree { Err(error) => { let kind = error.kind(); let mut message = error.to_string(); - if let Err(reaper) = reap_later(child) { + if let Err(failure) = reap_later(child) { + let (reaper, child) = failure.into_parts(); + self.child = Some(child); message = format!("{message}; the detached child reaper could not be started: {reaper}"); } return Err(io::Error::new(kind, message)); @@ -1482,7 +1556,9 @@ impl ProcessTree { || (io::ErrorKind::TimedOut, deadline_error.clone()), |error| (error.kind(), format!("{error}; {deadline_error}")), ); - if let Err(error) = reap_later(child) { + if let Err(failure) = reap_later(child) { + let (error, child) = failure.into_parts(); + self.child = Some(child); message = format!("{message}; the detached child reaper could not be started: {error}"); } return Err(io::Error::new(kind, message)); @@ -1761,12 +1837,10 @@ mod tests { use core::cell::RefCell; #[cfg(unix)] use core::mem; - #[cfg(unix)] - use std::env; use std::error::Error as _; - use std::fs; #[cfg(unix)] use std::io::{BufRead as _, Write as _}; + use std::{env, fs}; use camino::Utf8Path; @@ -2567,9 +2641,8 @@ mod tests { } #[test] - fn shared_reaper_collects_out_of_order_and_restarts_after_idle() { + fn shared_reaper_collects_out_of_order_and_remains_ready_after_idle() { let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - wait_for_reaper_to_stop(Duration::from_secs(2)); let long = spawn_reaper_probe(750); let long_id = long.id(); @@ -2584,14 +2657,70 @@ mod tests { "the shorter-lived later child must be collected before the earlier long-lived child" ); wait_for_reaper_to_collect(long_id, Duration::from_secs(2)); - wait_for_reaper_to_stop(Duration::from_secs(2)); + assert!(reaper_running(), "the durable shared reaper stopped when its queue became empty"); + + let after_idle = spawn_reaper_probe(50); + let after_idle_id = after_idle.id(); + reap_later(after_idle).expect("hand a child to the idle shared reaper"); + wait_for_reaper_to_collect(after_idle_id, Duration::from_secs(2)); + } - let restarted = spawn_reaper_probe(50); - let restarted_id = restarted.id(); - reap_later(restarted).expect("restart the shared reaper after its queue became empty"); - assert!(reaper_running(), "the shared reaper did not restart"); - wait_for_reaper_to_collect(restarted_id, Duration::from_secs(2)); - wait_for_reaper_to_stop(Duration::from_secs(2)); + #[test] + fn reaper_start_failure_preserves_child_and_preparation_ownership() { + isolated_run( + "a_reaper_start_failure_preserves_child_and_preparation_ownership", + "the failed reaper start preserved both owners", + ); + } + + #[test] + fn a_reaper_start_failure_preserves_child_and_preparation_ownership() { + if env::var_os(ISOLATED_CHILD).is_none() { + return; + } + + let child = Command::new(testing::helper_binary_path().as_std_path()) + .arg(testing::directive("sleep:1000")) + .stdin(Stdio::null()) + .stdout(Stdio::null()) + .stderr(Stdio::null()) + .spawn() + .expect("spawn isolated reaper probe"); + let child_id = child.id(); + let _failed_start = faults::arm(faults::Fault::ReaperStart); + let failure = reap_later(child).expect_err("the injected thread start failure must reject the handoff"); + assert!(!reaper_contains(child_id), "a child without a reaper must not enter the queue"); + assert!(failure.to_string().contains("thread start failed as requested"), "{failure}"); + assert!( + failure + .source() + .expect("the handoff failure retains its operating-system cause") + .to_string() + .contains("thread start failed as requested") + ); + let (error, mut child) = failure.into_parts(); + assert!(error.to_string().contains("thread start failed as requested"), "{error}"); + assert_eq!(child.id(), child_id, "the failed handoff returned a different child"); + child.kill().expect("terminate the recovered child"); + let _status = child.wait().expect("reap the recovered child"); + + let prepared = prepare(no_op_command(), MemoryRequest::default()).expect("containment"); + let _failed_start = faults::arm(faults::Fault::ReaperStart); + let failure = prepared + .spawn() + .expect_err("reaper startup must be proven before the child is spawned"); + assert!( + failure.cause().to_string().contains("thread start failed as requested"), + "{}", + failure.cause() + ); + let (_error, prepared) = failure.into_parts(); + let spawned = prepared + .spawn() + .expect("the unchanged preparation can retry after the transient failure"); + drop(spawned); + + println!("the failed reaper start preserved both owners"); } fn spawn_reaper_probe(sleep_millis: u64) -> Child { @@ -2609,14 +2738,6 @@ mod tests { assert!(!reaper_contains(id), "the shared reaper did not collect child {id}"); } - fn wait_for_reaper_to_stop(grace: Duration) { - let deadline = Instant::now() + grace; - while reaper_running() && Instant::now() < deadline { - thread::sleep(Duration::from_millis(10)); - } - assert!(!reaper_running(), "the shared reaper did not stop after its queue became empty"); - } - #[test] fn ordinary_termination_reports_post_reap_cleanup_failure() { let mut command = Command::new(testing::helper_binary_path().as_std_path()); @@ -2909,7 +3030,6 @@ mod tests { /// /// The name is on the child's environment rather than on its command line because the command /// line belongs to the test harness, which would reject an argument it does not know. - #[cfg(unix)] const ISOLATED_CHILD: &str = "GAMMA_ISOLATED_CHILD"; /// Runs one of the inner tests below in a process of its own, and pins what it reported. @@ -2923,7 +3043,6 @@ mod tests { /// /// The marker is required rather than the exit status alone, because a filter that matched /// nothing — a renamed inner test, say — is also a successful run of zero tests. - #[cfg(unix)] fn isolated_run(inner: &str, marker: &str) { let program = env::args().next().expect("this test binary"); let outcome = Command::new(&program) From c46004ffc55b2ee17e258540ef6739b68d69b0f3 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 16 Sep 2026 21:29:23 +0200 Subject: [PATCH 18/37] test(cargo-each): avoid requiring sealed CI containment Exercise the timed streamed runner through an injected test spawner so Linux hosts without delegated cgroups validate the logic without weakening production timeout refusal. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 26 ++++++++++++++++++++------ 1 file changed, 20 insertions(+), 6 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index ea10fca90..74ee09418 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -344,11 +344,19 @@ fn run_streamed(invocation: &Invocation) -> InvocationResult { } fn run_streamed_with_timeout(invocation: &Invocation, timeout: Duration) -> InvocationResult { + run_streamed_with_timeout_with(invocation, timeout, spawn_sealed_tree) +} + +fn run_streamed_with_timeout_with( + invocation: &Invocation, + timeout: Duration, + spawn: impl FnOnce(Command) -> Result, +) -> InvocationResult { let (program, command) = match command_for(invocation) { Ok(command) => command, Err(message) => return InvocationResult::Infrastructure(message), }; - let mut tree = match spawn_sealed_tree(command) { + let mut tree = match spawn(command) { Ok(tree) => tree, Err(error) => { return InvocationResult::Infrastructure(format!("failed to spawn `{program}`: {error}")); @@ -1358,10 +1366,10 @@ mod tests { Plan, ReaderCompletion, RunningWorker, SpillFile, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, failure_stops_launching, finish_ordinary_termination_with, finish_ordinary_wait_with, finish_output_reader, finish_wait_with_cleanup, panic_description, - parallel_failure_exit_code, run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, spawn_if_sealed, - spawn_output_reader, spawn_output_reader_with, spawn_tree, spawn_worker, terminate_ordinary_child, terminate_ordinary_with, - wait_for_captured_process, wait_for_tree, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, - with_cleanup_failure, + parallel_failure_exit_code, run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, + run_streamed_with_timeout_with, spawn_if_sealed, spawn_output_reader, spawn_output_reader_with, spawn_tree, spawn_worker, + terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, wait_for_tree, wait_for_tree_with, + wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, }; const ORDINARY_BOUNDARY: &str = "ordinary process tree"; @@ -2354,11 +2362,17 @@ mod tests { assert!(result_infrastructure_message(run_streamed_with_timeout(&missing, Duration::from_secs(1))).contains("failed to spawn")); assert!(infrastructure_message(run_captured(&missing, None)).contains("failed to spawn")); - let InvocationResult::Exited(status) = run_streamed_with_timeout(&invocation(&["rustc", "--version"]), Duration::from_secs(2)) + let InvocationResult::Exited(status) = + run_streamed_with_timeout_with(&invocation(&["rustc", "--version"]), Duration::from_secs(2), spawn_tree) else { panic!("the timed streamed runner must execute rustc"); }; assert!(status.success()); + + let spawn_failure = run_streamed_with_timeout_with(&invocation(&["rustc", "--version"]), Duration::from_secs(2), |_command| { + Err("injected timed-stream spawn failure".to_owned()) + }); + assert!(result_infrastructure_message(spawn_failure).contains("injected timed-stream spawn failure")); } #[test] From 0156a5f307971ea628ba7a836e408055ddbbe71a Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Thu, 17 Sep 2026 11:26:32 +0200 Subject: [PATCH 19/37] fix(cargo-gamma-process): retry interrupted reaping Keep children queued when try_wait is interrupted while continuing to warn and stop tracking on permanent observation failures. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-gamma-process/docs/DESIGN.md | 7 ++++--- crates/cargo-gamma-process/docs/IMPLEMENTATION.md | 7 ++++--- crates/cargo-gamma-process/src/process_tree.rs | 11 ++++++++--- 3 files changed, 16 insertions(+), 9 deletions(-) diff --git a/crates/cargo-gamma-process/docs/DESIGN.md b/crates/cargo-gamma-process/docs/DESIGN.md index 99984aa42..070db5ccd 100644 --- a/crates/cargo-gamma-process/docs/DESIGN.md +++ b/crates/cargo-gamma-process/docs/DESIGN.md @@ -84,9 +84,10 @@ therefore covers the complete descendant tree. remain owned until the `ProcessTree` itself is dropped. An error while polling the leader follows the same handoff before the observation error is returned, because an observation failure does not prove the child was reaped. If the - detached reaper itself later receives an observation error, it emits a warning - to stderr and permanently stops tracking that child. The released handle may - leave a zombie on Unix until this process exits. + detached reaper itself later receives an interrupted observation, it keeps the + child queued and retries. Any other observation error emits a warning to stderr + and permanently stops tracking that child. The released handle may leave a + zombie on Unix until this process exits. - Sealed containment uses a boundary that descendants cannot leave. A host that offers no sealed boundary at all silently uses best-effort process-group containment for an unmetered launch; absence of a warning does not establish diff --git a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md index 2cd96555a..458080fca 100644 --- a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md +++ b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md @@ -33,9 +33,10 @@ the thread exists, and direct `reap_later` startup failures return an unqueued child in `ReapFailure` rather than abandoning ownership. A `try_wait` error also transfers the still-owned leader handle before returning the observation error; only a successful `Some(status)` proves that no later reaping is required. If a -later `try_wait` in the detached reaper fails, the reaper writes a warning to -stderr and permanently releases that child handle. On Unix, the child may -remain a zombie until this process exits. +later `try_wait` in the detached reaper is interrupted, the child remains queued +for another attempt. Any other observation error writes a warning to stderr and +permanently releases that child handle. On Unix, the child may remain a zombie +until this process exits. ## Platform composition diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index 23a3399eb..02f6c2793 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -96,9 +96,9 @@ pub fn ensure_reaper() -> io::Result<()> { /// /// The reaper polls every retained child rather than blocking on one, so a /// leader that survives termination cannot prevent unrelated leaders from -/// being collected. An observation error emits a warning to stderr and -/// permanently stops tracking that child. On Unix, the child may then remain a -/// zombie until this process exits. +/// being collected. Interrupted observations are retried. Any other observation +/// error emits a warning to stderr and permanently stops tracking that child. +/// On Unix, the child may then remain a zombie until this process exits. /// /// # Errors /// @@ -177,6 +177,7 @@ fn retain_reaper_child(id: u32, observation: io::Result>) -> match observation { Ok(None) => true, Ok(Some(_status)) => false, + Err(error) if error.kind() == io::ErrorKind::Interrupted => true, Err(error) => { eprintln!("warning: detached child reaper stopped tracking process {id} after observation failed: {error}"); false @@ -1910,6 +1911,10 @@ mod tests { #[test] fn detached_reaper_drops_unobservable_children() { assert!(retain_reaper_child(17, Ok(None))); + assert!(retain_reaper_child( + 17, + Err(io::Error::new(io::ErrorKind::Interrupted, "wait interrupted")), + )); assert!(!retain_reaper_child( 17, Err(io::Error::new(io::ErrorKind::InvalidInput, "invalid child handle")), From d02990e14dce05815ea33496896038cc7f2dc650 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Fri, 18 Sep 2026 13:52:32 +0200 Subject: [PATCH 20/37] fix(cargo-each): preserve bounded execution contracts Use effective concurrency for stream mode, defer workspace-version resolution for empty plans, keep cancelled output recovery nonblocking, and make detached reaper diagnostics restart-safe. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/README.md | 22 +- crates/cargo-each/docs/design/README.md | 24 +- crates/cargo-each/src/cli.rs | 1 + crates/cargo-each/src/main.rs | 22 +- crates/cargo-each/src/plan.rs | 64 ++++-- crates/cargo-each/src/run.rs | 205 +++++++++++++----- crates/cargo-each/tests/cli.rs | 78 ++++++- crates/cargo-gamma-process/docs/DESIGN.md | 14 +- .../docs/IMPLEMENTATION.md | 11 +- crates/cargo-gamma-process/src/lib.rs | 4 + .../cargo-gamma-process/src/process_tree.rs | 182 ++++++++++++++-- 11 files changed, 508 insertions(+), 119 deletions(-) diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index 2a7dc54ee..bfbe99c7a 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -125,13 +125,16 @@ An empty resolved selection (via `--none`, or a filter that removes every member) is a **successful no-op**: `cargo-each` prints a one-line note and exits 0. This is what lets callers drop bespoke nothing-to-do guards. Workspace Rust-version validation is lazy: it runs only when the command -uses `{workspace-rust-version}`, then requires every member’s resolved -minimum to be present and no newer than the root floor. - -With `--jobs > 1`, including when `auto` resolves above one, the effective -worker count is capped by the plan size and scheduler capacity. Each -invocation’s output is buffered and complete blocks are emitted in -deterministic plan order. Fail-fast stops launching after the first +uses `{workspace-rust-version}` and the resolved plan has work, then requires +every member’s resolved minimum to be present and no newer than the root +floor. Placeholder mode validation still runs before an empty-plan no-op. + +The effective worker count is the requested `--jobs` value capped by plan +size and scheduler capacity. An effective count of one uses sequential +execution with inherited stdin, stdout, and stderr even when the requested +value was larger. A genuinely parallel count disconnects child stdin and +buffers stdout and stderr; complete blocks are emitted in deterministic +plan order. Fail-fast stops launching after the first observed failure, waits for running work, and chooses the final failure by plan order. `--keep-going` runs the complete plan. Worker panics and unexpected worker-channel disconnections become infrastructure-failure @@ -146,7 +149,10 @@ removed by RAII after deterministic plan-order emission. Reader failures are observed while the child is running and trigger bounded termination. Output drain is bounded after every completion: readers get one second to observe EOF, then readiness-polling capture is cancelled and -joined while partial bytes become an explicit infrastructure failure. +joined while partial bytes become an explicit infrastructure failure. If a +cancelled reader remains stalled while holding its capture mutex, output +recovery is nonblocking and any unavailable partial bytes are reported +rather than extending the drain bound. Timed-out tree termination likewise gets a bounded 250 ms leader-reap grace, after which the leader handle moves to a shared detached reaper so no wait or Drop path can defeat the timeout without abandoning reap ownership. diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index 88cb79745..2d9967d8f 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -285,7 +285,11 @@ floor. Missing values, a member requiring a newer compiler, or a non-Rust semantic version is a configuration error. Lower member minima are valid. This matches the meaning of one compiler selected for a complete workspace; it is not a per-package toolchain matrix. The validation is lazy: commands that do -not contain the placeholder do not require a workspace Rust version. +not contain the placeholder do not require a workspace Rust version, and a +resolved plan with no invocations does not resolve or validate the value even +when the template contains the placeholder. Placeholder mode validation still +runs before that no-op decision, so misuse remains an exit-2 usage error on an +empty set. Targets run in package-name order and then target-name order. A target matching more than one requested kind runs once. No matching targets is a successful @@ -315,7 +319,13 @@ no-op. block in deterministic plan order. Fail-fast stops launching new work after the first observed failure and waits for already-running children; `--keep-going` launches the complete plan. The final failure is chosen by - plan order, not scheduler timing. A worker panic is converted into an + plan order, not scheduler timing. Requested parallelism does not by itself + select this captured mode: when plan-size or process-capacity capping leaves + an effective worker count of one, cargo-each uses the sequential path and the + child inherits stdin, stdout, and stderr. With a genuinely parallel effective + worker count, child stdin is disconnected (`null`) so workers cannot race to + consume the caller's input; stdout and stderr are captured for deterministic + emission. A worker panic is converted into an infrastructure-failure outcome; each worker has a dedicated completion channel, so an unexpected exit is observable as disconnection rather than leaving the scheduler blocked forever. A worker-thread launch failure is @@ -337,9 +347,13 @@ no-op. EOF. Complete output is preserved when both pipes close within that grace. If a background or escaped descendant keeps a pipe open, capture stops retaining new bytes, cancels and joins the readiness-polling reader within a - bounded grace, emits the partial bytes already buffered, and reports an - explicit infrastructure failure rather than hanging, silently truncating, or - accumulating detached reader threads. + bounded grace, emits the partial bytes already buffered when their capture + mutex is immediately available, and reports an explicit infrastructure + failure rather than hanging, silently succeeding, or accumulating detached + reader threads. If cancellation expires while a reader is stalled inside a + spill operation with that mutex held, cargo-each detaches the reader and uses + a nonblocking acquisition; unavailable partial bytes are reported explicitly + instead of defeating the drain bound. - **Timeouts terminate trees.** A timed-out command is a failure. cargo-each terminates the child process tree rather than only the immediate process, so compiler or test descendants cannot continue mutating the target directory diff --git a/crates/cargo-each/src/cli.rs b/crates/cargo-each/src/cli.rs index 21fe5ee82..9162e8023 100644 --- a/crates/cargo-each/src/cli.rs +++ b/crates/cargo-each/src/cli.rs @@ -97,6 +97,7 @@ pub(crate) struct EachArgs { /// Run at most N per-package or per-target commands concurrently. Use /// `auto` to detect available parallelism once. Defaults to 1. Buffered /// output spills to unique system-temporary files beyond 1 MiB per stream. + /// Only an effective count above 1 disconnects child stdin for capture. #[arg(long, default_value_t = NonZeroUsize::MIN, value_name = "N|auto", value_parser = parse_jobs)] pub(crate) jobs: NonZeroUsize, diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 6c3105454..2b8259d88 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -115,13 +115,16 @@ //! member) is a **successful no-op**: `cargo-each` prints a one-line note and //! exits 0. This is what lets callers drop bespoke nothing-to-do guards. //! Workspace Rust-version validation is lazy: it runs only when the command -//! uses `{workspace-rust-version}`, then requires every member's resolved -//! minimum to be present and no newer than the root floor. -//! -//! With `--jobs > 1`, including when `auto` resolves above one, the effective -//! worker count is capped by the plan size and scheduler capacity. Each -//! invocation's output is buffered and complete blocks are emitted in -//! deterministic plan order. Fail-fast stops launching after the first +//! uses `{workspace-rust-version}` and the resolved plan has work, then requires +//! every member's resolved minimum to be present and no newer than the root +//! floor. Placeholder mode validation still runs before an empty-plan no-op. +//! +//! The effective worker count is the requested `--jobs` value capped by plan +//! size and scheduler capacity. An effective count of one uses sequential +//! execution with inherited stdin, stdout, and stderr even when the requested +//! value was larger. A genuinely parallel count disconnects child stdin and +//! buffers stdout and stderr; complete blocks are emitted in deterministic +//! plan order. Fail-fast stops launching after the first //! observed failure, waits for running work, and chooses the final failure by //! plan order. `--keep-going` runs the complete plan. Worker panics and //! unexpected worker-channel disconnections become infrastructure-failure @@ -136,7 +139,10 @@ //! Reader failures are observed while the child is running and trigger bounded //! termination. Output drain is bounded after every completion: readers get //! one second to observe EOF, then readiness-polling capture is cancelled and -//! joined while partial bytes become an explicit infrastructure failure. +//! joined while partial bytes become an explicit infrastructure failure. If a +//! cancelled reader remains stalled while holding its capture mutex, output +//! recovery is nonblocking and any unavailable partial bytes are reported +//! rather than extending the drain bound. //! Timed-out tree termination likewise gets a bounded 250 ms leader-reap grace, //! after which the leader handle moves to a shared detached reaper so no wait //! or Drop path can defeat the timeout without abandoning reap ownership. diff --git a/crates/cargo-each/src/plan.rs b/crates/cargo-each/src/plan.rs index 19bef8a68..db145d9b6 100644 --- a/crates/cargo-each/src/plan.rs +++ b/crates/cargo-each/src/plan.rs @@ -75,6 +75,28 @@ pub(crate) struct BuildOptions<'a> { } impl Plan { + /// Validate plan-wide configuration and report whether it produces no + /// invocations without expanding placeholders. + /// + /// This lets callers preserve usage validation while deferring lazy + /// workspace-scoped values until the resolved selection is known to + /// produce work. + /// + /// # Errors + /// + /// Returns [`EachError`] under the same invalid `chdir` and placeholder + /// combinations as [`Self::build`]. + pub(crate) fn is_empty(members: &[&Member], command: &[String], options: BuildOptions<'_>) -> Result { + validate_build(command, options)?; + Ok(match options.mode { + Mode::PerPackage | Mode::Once => members.is_empty(), + Mode::PerTarget => !members + .iter() + .flat_map(|member| &member.targets) + .any(|target| target_matches(target, options.target_kinds, options.target_required_features)), + }) + } + /// Build the plan. /// /// `chdir` runs each per-package or per-target invocation from the member's @@ -93,6 +115,10 @@ impl Plan { /// Returns [`EachError`] if `chdir` is combined with [`Mode::Once`], or if /// a placeholder in `command` is used in the wrong mode. pub(crate) fn build(members: &[&Member], command: &[String], options: BuildOptions<'_>) -> Result { + if Self::is_empty(members, command, options)? { + return Ok(Self { invocations: Vec::new() }); + } + let BuildOptions { mode, chdir, @@ -101,17 +127,6 @@ impl Plan { target_required_features, workspace_rust_version, } = options; - if chdir && mode == Mode::Once { - return Err(ChdirConflictsWithOnceError::new().into()); - } - // Validate placeholder/mode consistency up front — before the - // empty-set short-circuit — so a misused template (e.g. `{name}` under - // `--once`) is a usage error even when the selection resolves to no - // members, rather than silently passing until some tier is non-empty. - validate_placeholders(command, mode)?; - if members.is_empty() { - return Ok(Self { invocations: Vec::new() }); - } let invocations = match mode { Mode::PerPackage => members @@ -137,12 +152,7 @@ impl Plan { member .targets .iter() - .filter(|target| { - target.kinds.iter().any(|kind| target_kinds.contains(kind)) - && target_required_features - .iter() - .all(|feature| target.required_features.contains(feature)) - }) + .filter(|target| target_matches(target, target_kinds, target_required_features)) .map(|target| { let placeholders = Placeholders::Target { name: member.name.clone(), @@ -178,6 +188,26 @@ impl Plan { } } +fn validate_build(command: &[String], options: BuildOptions<'_>) -> Result<(), EachError> { + if options.chdir && options.mode == Mode::Once { + return Err(ChdirConflictsWithOnceError::new().into()); + } + // Validate placeholder/mode consistency before any empty-set short-circuit + // so an invalid template never depends on whether a computed tier has work. + validate_placeholders(command, options.mode) +} + +fn target_matches( + target: &crate::workspace::MemberTarget, + target_kinds: &BTreeSet, + target_required_features: &BTreeSet, +) -> bool { + target.kinds.iter().any(|kind| target_kinds.contains(kind)) + && target_required_features + .iter() + .all(|feature| target.required_features.contains(feature)) +} + /// The `{packages}` expansion: `--workspace` for the whole workspace, else an /// explicit `--package name@version` per member. fn packages_flags(members: &[&Member], packages: PackagesExpansion) -> Vec { diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 74ee09418..b4055b536 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -10,7 +10,7 @@ use std::num::NonZeroUsize; use std::panic::{self, AssertUnwindSafe, UnwindSafe}; use std::process::{Child, ChildStderr, ChildStdout, Command, ExitCode, ExitStatus, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::{Arc, Mutex, mpsc}; +use std::sync::{Arc, Mutex, TryLockError, mpsc}; use std::time::{Duration, Instant}; use std::{fmt, thread}; @@ -23,7 +23,7 @@ use crate::error::{InvalidTargetKindError, JobsConflictWithOnceError}; use crate::filter::Predicate; use crate::plan::{BuildOptions, Invocation, Mode, PackagesExpansion, Plan}; use crate::select::Selection; -use crate::substitute::{uses_workspace_rust_version, validate_placeholders}; +use crate::substitute::uses_workspace_rust_version; use crate::workspace::{Member, Workspace}; #[cfg(test)] @@ -64,9 +64,19 @@ pub(crate) fn run(args: &EachArgs) -> Result { return Err(JobsConflictWithOnceError::new()).into_app_err("invalid execution configuration"); } - // Validate mode-specific tokens before resolving the lazy workspace token - // so a malformed command reports its direct usage error first. - validate_placeholders(&args.command, mode).into_app_err("failed to build command plan")?; + let mut build_options = BuildOptions { + mode, + chdir: args.chdir, + packages, + target_kinds: &target_kinds, + target_required_features: &target_required_features, + workspace_rust_version: None, + }; + if Plan::is_empty(&members, &args.command, build_options).into_app_err("failed to build command plan")? { + eprintln!("cargo each: selection resolved to no work; nothing to do"); + return Ok(ExitCode::SUCCESS); + } + let workspace_rust_version = if uses_workspace_rust_version(&args.command) { Some( workspace @@ -76,25 +86,9 @@ pub(crate) fn run(args: &EachArgs) -> Result { } else { None }; + build_options.workspace_rust_version = workspace_rust_version.as_deref(); - let plan = Plan::build( - &members, - &args.command, - BuildOptions { - mode, - chdir: args.chdir, - packages, - target_kinds: &target_kinds, - target_required_features: &target_required_features, - workspace_rust_version: workspace_rust_version.as_deref(), - }, - ) - .into_app_err("failed to build command plan")?; - - if plan.invocations.is_empty() { - eprintln!("cargo each: selection resolved to no work; nothing to do"); - return Ok(ExitCode::SUCCESS); - } + let plan = Plan::build(&members, &args.command, build_options).into_app_err("failed to build command plan")?; if args.dry_run { for inv in &plan.invocations { @@ -103,6 +97,7 @@ pub(crate) fn run(args: &EachArgs) -> Result { None => println!("{}", shell_join(&inv.argv)), } } + return Ok(ExitCode::SUCCESS); } @@ -146,13 +141,19 @@ fn parse_target_kinds(kinds: &[String]) -> Result, AppError } fn execute(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: Option) -> Result { - if jobs.get() == 1 { + let worker_count = effective_worker_count(jobs, plan.invocations.len(), cargo_gamma_process::capacity()); + if worker_count.get() == 1 { Ok(execute_sequential(plan, keep_going, timeout)) } else { - execute_parallel(plan, keep_going, jobs, timeout) + execute_parallel(plan, keep_going, worker_count, timeout) } } +fn effective_worker_count(requested: NonZeroUsize, plan_size: usize, process_capacity: usize) -> NonZeroUsize { + NonZeroUsize::new(requested.get().min(plan_size).min(process_capacity.max(1))) + .expect("execution receives a nonempty plan and process capacity is clamped to at least one") +} + fn execute_sequential(plan: &Plan, keep_going: bool, timeout: Option) -> ExitCode { execute_sequential_with(plan, keep_going, timeout, |invocation, timeout| { if let Some(timeout) = timeout { @@ -200,16 +201,15 @@ fn execute_sequential_with( if any_failed { ExitCode::from(1) } else { ExitCode::SUCCESS } } -fn execute_parallel(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: Option) -> Result { +fn execute_parallel(plan: &Plan, keep_going: bool, worker_count: NonZeroUsize, timeout: Option) -> Result { let invocations = plan.invocations.clone(); - let worker_count = jobs.get().min(invocations.len()).min(cargo_gamma_process::capacity().max(1)); let mut pending: VecDeque<(usize, Invocation)> = invocations.iter().cloned().enumerate().collect(); - let mut workers = Vec::with_capacity(worker_count); + let mut workers = Vec::with_capacity(worker_count.get()); let mut outcomes = Vec::with_capacity(invocations.len()); let mut stop_launching = false; loop { - while !stop_launching && workers.len() < worker_count { + while !stop_launching && workers.len() < worker_count.get() { let Some((index, invocation)) = pending.pop_front() else { break; }; @@ -1033,22 +1033,46 @@ fn finish_output_reader(mut reader: OutputReader, stream: &str, grace: Duration, drop(reader.thread); } - match reader.output.lock() { - Ok(mut captured) => CapturedStream { - output: std::mem::replace(&mut *captured, CapturedOutput::empty()), - failure, - }, - Err(poisoned) => { - let mut captured = poisoned.into_inner(); - CapturedStream { - output: std::mem::replace(&mut *captured, CapturedOutput::empty()), - failure: Some(match failure { - Some(failure) => format!("{failure}; child {stream} capture buffer was poisoned"), - None => format!("child {stream} capture buffer was poisoned"), - }), + let output = take_reader_output(&reader.output, stream, thread_finished, &mut failure); + CapturedStream { output, failure } +} + +fn take_reader_output(output: &Mutex, stream: &str, thread_finished: bool, failure: &mut Option) -> CapturedOutput { + let mut captured = if thread_finished { + match output.lock() { + Ok(captured) => captured, + Err(poisoned) => { + append_failure(failure, format!("child {stream} capture buffer was poisoned")); + poisoned.into_inner() } } - } + } else { + match output.try_lock() { + Ok(captured) => captured, + Err(TryLockError::Poisoned(poisoned)) => { + append_failure(failure, format!("child {stream} capture buffer was poisoned")); + poisoned.into_inner() + } + Err(TryLockError::WouldBlock) => { + append_failure( + failure, + format!( + "child {stream} capture buffer remained locked after reader cancellation; \ + partial output could not be recovered without exceeding the drain bound" + ), + ); + return CapturedOutput::empty(); + } + } + }; + std::mem::replace(&mut *captured, CapturedOutput::empty()) +} + +fn append_failure(failure: &mut Option, additional: String) { + *failure = Some(match failure.take() { + Some(failure) => format!("{failure}; {additional}"), + None => additional, + }); } fn reader_failure(completion: &ReaderCompletion, stream: &str, grace: Duration, boundary: &str) -> Option { @@ -1358,18 +1382,18 @@ mod tests { use std::process::{Command, ExitCode, ExitStatus, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Condvar, Mutex, mpsc}; - use std::time::Duration; + use std::time::{Duration, Instant}; use std::{io, thread}; use super::{ BufferedOutcome, CapturedOutput, CapturedProcess, CapturedStream, Invocation, InvocationResult, OutputEmitError, OutputReader, Plan, ReaderCompletion, RunningWorker, SpillFile, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, - combine_captured_output, display_duration, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, failure_stops_launching, - finish_ordinary_termination_with, finish_ordinary_wait_with, finish_output_reader, finish_wait_with_cleanup, panic_description, - parallel_failure_exit_code, run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, - run_streamed_with_timeout_with, spawn_if_sealed, spawn_output_reader, spawn_output_reader_with, spawn_tree, spawn_worker, - terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, wait_for_tree, wait_for_tree_with, - wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, + combine_captured_output, display_duration, effective_worker_count, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, + failure_stops_launching, finish_ordinary_termination_with, finish_ordinary_wait_with, finish_output_reader, + finish_wait_with_cleanup, panic_description, parallel_failure_exit_code, run_captured, run_captured_with_spawner, run_streamed, + run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_if_sealed, spawn_output_reader, spawn_output_reader_with, + spawn_tree, spawn_worker, terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, wait_for_tree, + wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, }; const ORDINARY_BOUNDARY: &str = "ordinary process tree"; @@ -1394,6 +1418,29 @@ mod tests { dropped: Option>, } + struct DropSignalReader { + bytes: Option<&'static [u8]>, + dropped: Option>, + } + + impl io::Read for DropSignalReader { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + let Some(bytes) = self.bytes.take() else { + return Ok(0); + }; + buf[..bytes.len()].copy_from_slice(bytes); + Ok(bytes.len()) + } + } + + impl Drop for DropSignalReader { + fn drop(&mut self) { + if let Some(dropped) = self.dropped.take() { + let _receiver_gone = dropped.send(()); + } + } + } + impl io::Read for PendingPipe { fn read(&mut self, _buf: &mut [u8]) -> io::Result { Err(io::Error::from(io::ErrorKind::WouldBlock)) @@ -1724,6 +1771,17 @@ mod tests { assert!(!failure_stops_launching(true, false)); } + #[test] + fn effective_worker_count_caps_requested_parallelism() { + let four = NonZeroUsize::new(4).expect("literal four is nonzero"); + assert_eq!(effective_worker_count(four, 1, 8), NonZeroUsize::MIN); + assert_eq!( + effective_worker_count(four, 8, 2), + NonZeroUsize::new(2).expect("literal two is nonzero") + ); + assert_eq!(effective_worker_count(four, 8, 0), NonZeroUsize::MIN); + } + #[test] fn sequential_timeout_policy_distinguishes_fail_fast_from_keep_going() { let plan = Plan { @@ -1974,6 +2032,53 @@ mod tests { assert!(!failure.contains("remained open"), "{failure}"); } + #[test] + fn cancellation_does_not_wait_for_a_reader_holding_the_capture_mutex() { + let (factory_started, factory_is_started) = mpsc::sync_channel(0); + let release = Arc::new((Mutex::new(false), Condvar::new())); + let factory_release = Arc::clone(&release); + let (reader_dropped, reader_is_dropped) = mpsc::channel(); + let reader = spawn_output_reader_with( + DropSignalReader { + bytes: Some(b"stalled append"), + dropped: Some(reader_dropped), + }, + "stalled-append-reader", + 0, + Box::new(move || { + factory_started.send(()).expect("the finisher waits for the stalled append"); + let (lock, condition) = &*factory_release; + let mut released = lock.lock().expect("the test owns the release mutex without panicking"); + while !*released { + released = condition.wait(released).expect("the test owns the release mutex without panicking"); + } + Ok(Box::new(io::Cursor::new(Vec::new())) as Box) + }), + ) + .expect("create stalled append reader"); + factory_is_started + .recv_timeout(Duration::from_secs(1)) + .expect("the reader entered spill creation while holding the capture mutex"); + + let started = Instant::now(); + let mut captured = finish_output_reader(reader, "stdout", Duration::ZERO, ORDINARY_BOUNDARY); + assert!( + started.elapsed() < Duration::from_millis(500), + "capture cleanup blocked on the stalled output mutex" + ); + assert!(output_bytes(&mut captured.output).is_empty()); + let failure = captured.failure.expect("the unavailable partial output must be explicit"); + assert!(failure.contains("reader did not stop"), "{failure}"); + assert!(failure.contains("capture buffer remained locked"), "{failure}"); + + let (lock, condition) = &*release; + *lock.lock().expect("the test owns the release mutex without panicking") = true; + condition.notify_all(); + reader_is_dropped + .recv_timeout(Duration::from_secs(1)) + .expect("the cancelled detached reader exits after the stalled append is released"); + } + #[test] fn normal_completion_does_not_join_a_stubborn_descendant_pipe() { let release = Arc::new((Mutex::new(false), Condvar::new())); diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index c8897950e..836ac6a01 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -166,7 +166,7 @@ fn compile_execution_probe(directory: &Path) -> PathBuf { r#" use std::env; use std::fs::{self, OpenOptions}; -use std::io::Write as _; +use std::io::{Read as _, Write as _}; use std::process::{self, Command}; use std::thread; use std::time::{Duration, Instant}; @@ -235,6 +235,11 @@ fn main() { stderr.write_all(&vec![stderr_byte; size]).expect("write large stderr"); process::exit(if name == "alpha" { 7 } else { 0 }); } + "stdin" => { + let mut input = String::new(); + std::io::stdin().read_to_string(&mut input).expect("read stdin"); + println!("{}:stdin:{input}", args[2]); + } "timeout-fail-fast" => { if args[2] == "alpha" { thread::sleep(Duration::from_secs(5)); @@ -431,7 +436,7 @@ fn present_empty_package_file_is_an_explicit_empty_selection() { each(&manifest) .arg("--package-file") .arg(packages) - .args(["--dry-run", "--", "echo", "{name}"]) + .args(["--dry-run", "--", "echo", "{name}:{workspace-rust-version}"]) .assert() .success() .stdout(predicate::str::is_empty()) @@ -530,9 +535,38 @@ fn package_file_input_errors_fail_loudly() { fn none_is_a_successful_noop() { let (_tmp, manifest) = fixture(); each(&manifest) - .args(["--none", "--once", "--dry-run", "--", "cargo", "test", "{packages}"]) + .args([ + "--none", + "--once", + "--dry-run", + "--", + "cargo", + "test", + "{packages}", + "{workspace-rust-version}", + ]) + .assert() + .success() + .stderr(predicate::str::contains("nothing to do")); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn empty_filtered_selection_skips_workspace_rust_version_resolution() { + let (_tmp, manifest) = fixture(); + each(&manifest) + .args([ + "--workspace", + "--filter", + "metadata:does-not-exist", + "--dry-run", + "--", + "echo", + "{name}:{workspace-rust-version}", + ]) .assert() .success() + .stdout(predicate::str::is_empty()) .stderr(predicate::str::contains("nothing to do")); } @@ -543,7 +577,7 @@ fn none_with_misused_placeholder_is_a_usage_error() { // usage error (exit 2), not a silent no-op. let (_tmp, manifest) = fixture(); each(&manifest) - .args(["--none", "--once", "--dry-run", "--", "echo", "{name}"]) + .args(["--none", "--once", "--dry-run", "--", "echo", "{name}", "{workspace-rust-version}"]) .assert() .failure() .code(2) @@ -1371,6 +1405,40 @@ fn default_jobs_runs_one_invocation_at_a_time() { ); } +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn requested_parallelism_with_one_invocation_preserves_inherited_stdin() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + each(&manifest) + .args(["-p", "alpha", "--jobs", "2", "--"]) + .arg(probe) + .args(["stdin", "{name}"]) + .write_stdin("inherited-input") + .assert() + .success() + .stdout(predicate::str::contains("alpha:stdin:inherited-input")); +} + +#[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] +#[test] +fn genuinely_parallel_children_receive_null_stdin() { + let (tmp, manifest) = fixture(); + let probe = compile_execution_probe(tmp.path()); + each(&manifest) + .args(["-p", "alpha", "-p", "beta", "--jobs", "2", "--"]) + .arg(probe) + .args(["stdin", "{name}"]) + .write_stdin("must-not-reach-children") + .assert() + .success() + .stdout( + predicate::str::contains("alpha:stdin:\n") + .and(predicate::str::contains("beta:stdin:\n")) + .and(predicate::str::contains("must-not-reach-children").not()), + ); +} + #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] #[test] fn parallel_output_is_buffered_in_plan_order() { @@ -1509,7 +1577,7 @@ fn parallel_open_descendant_pipe_returns_after_bounded_drain() { .arg("each") .arg("--manifest-path") .arg(&manifest) - .args(["-p", "alpha", "--jobs", "2", "--"]) + .args(["-p", "alpha", "-p", "beta", "--jobs", "2", "--"]) .arg(probe) .arg("stubborn-background-parent") .arg(&marker); diff --git a/crates/cargo-gamma-process/docs/DESIGN.md b/crates/cargo-gamma-process/docs/DESIGN.md index 070db5ccd..f1741719e 100644 --- a/crates/cargo-gamma-process/docs/DESIGN.md +++ b/crates/cargo-gamma-process/docs/DESIGN.md @@ -77,7 +77,9 @@ therefore covers the complete descendant tree. kill is transferred to a shared detached reaper rather than handed to an indefinite `wait` or Drop path. The reaper polls all retained leaders so one survivor cannot block collection of the others, remains alive while its queue - is empty, and accepts each handle only after its thread is known to exist. + is empty, and accepts each handle only after its thread is known to exist. A + handoff rechecks that state under the queue lock and retries startup if a + previous loop exited between the readiness check and transfer. Callers handing over children created outside `PreparedCommand` can preflight the same durable thread; if a direct handoff must start it and startup fails, the failure returns ownership of the unqueued child. The containment handles @@ -85,9 +87,13 @@ therefore covers the complete descendant tree. the leader follows the same handoff before the observation error is returned, because an observation failure does not prove the child was reaped. If the detached reaper itself later receives an interrupted observation, it keeps the - child queued and retries. Any other observation error emits a warning to stderr - and permanently stops tracking that child. The released handle may leave a - zombie on Unix until this process exits. + child queued and retries. Any other observation error emits a warning to + stderr and permanently stops tracking that child. Warning formatting happens + while the queue is locked, but the fallible stderr write happens after + releasing the lock and its error is discarded. Every loop exit or unwind + clears the running state and notifies waiters, allowing a later + `ensure_reaper` or `reap_later` call to restart it. The released handle may + leave a zombie on Unix until this process exits. - Sealed containment uses a boundary that descendants cannot leave. A host that offers no sealed boundary at all silently uses best-effort process-group containment for an unmetered launch; absence of a warning does not establish diff --git a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md index 458080fca..9b73baf65 100644 --- a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md +++ b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md @@ -30,13 +30,18 @@ is retained in the returned error. The reaper polls every retained child without blocking on one leader and waits on a condition variable when its queue is empty. It is durable: repository-controlled children are created only after the thread exists, and direct `reap_later` startup failures return an unqueued -child in `ReapFailure` rather than abandoning ownership. A `try_wait` error also +child in `ReapFailure` rather than abandoning ownership. Handoff rechecks the +running state while holding the queue lock and retries startup after a raced +loop exit. A `try_wait` error also transfers the still-owned leader handle before returning the observation error; only a successful `Some(status)` proves that no later reaping is required. If a later `try_wait` in the detached reaper is interrupted, the child remains queued for another attempt. Any other observation error writes a warning to stderr and -permanently releases that child handle. On Unix, the child may remain a zombie -until this process exits. +permanently releases that child handle. The write is fallible, its error is +discarded, and it occurs outside the queue mutex. A lifecycle guard clears the +running state and wakes readiness waiters whenever the loop exits or unwinds, +so later callers can start a replacement. On Unix, the child may remain a +zombie until this process exits. ## Platform composition diff --git a/crates/cargo-gamma-process/src/lib.rs b/crates/cargo-gamma-process/src/lib.rs index 8e9a8ba94..d92a7f63e 100644 --- a/crates/cargo-gamma-process/src/lib.rs +++ b/crates/cargo-gamma-process/src/lib.rs @@ -66,6 +66,10 @@ //! [`Command::output`](std::process::Command::output). It drains stdout and stderr concurrently, //! then sweeps descendants before waiting for inherited pipe handles to close. //! +//! Bounded termination can transfer a still-running leader to a shared detached reaper. The reaper +//! writes observation warnings outside its global queue lock through a fallible stderr path, and +//! any loop exit or unwind clears readiness state so a later handoff can start a replacement. +//! //! A terminal delivers `Ctrl-C` to the whole foreground process group, so a child sharing this //! process's group dies with it automatically while a child leading its own group does not. Windows //! normally preserves that guarantee through a dedicated job that dies with its last handle. Unix diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index 02f6c2793..63a89c7c3 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -8,7 +8,7 @@ use core::time::Duration; use std::io; use std::process::{Child, ChildStderr, ChildStdout, Command, ExitStatus, Output, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::{Arc, Condvar, Mutex}; +use std::sync::{Arc, Condvar, Mutex, MutexGuard}; use std::thread::{self, JoinHandle}; use std::time::Instant; @@ -45,7 +45,9 @@ static CHILD_REAPER_READY: Condvar = Condvar::new(); /// Ensures the shared detached child reaper is ready before a child is spawned. /// /// The reaper is process-wide and remains alive after its queue becomes empty, -/// sleeping on a condition variable until another child is handed off. +/// sleeping on a condition variable until another child is handed off. If its +/// loop exits or unwinds, readiness is cleared and a later call starts a +/// replacement. /// /// # Errors /// @@ -97,22 +99,31 @@ pub fn ensure_reaper() -> io::Result<()> { /// The reaper polls every retained child rather than blocking on one, so a /// leader that survives termination cannot prevent unrelated leaders from /// being collected. Interrupted observations are retried. Any other observation -/// error emits a warning to stderr and permanently stops tracking that child. -/// On Unix, the child may then remain a zombie until this process exits. +/// error emits a best-effort warning to stderr outside the global queue lock and +/// permanently stops tracking that child. Diagnostic write failures are +/// ignored. On Unix, the child may then remain a zombie until this process +/// exits. The handoff retries if the previous reaper loop exits while readiness +/// is being checked. /// /// # Errors /// /// Returns [`ReapFailure`] when the shared reaper could not be started. The /// failure retains the child handle so the caller can recover ownership. pub fn reap_later(child: Child) -> Result<(), ReapFailure> { - if let Err(cause) = ensure_reaper() { - return Err(ReapFailure { cause, child }); - } + loop { + if let Err(cause) = ensure_reaper() { + return Err(ReapFailure { cause, child }); + } - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - reaper.children.push(child); - CHILD_REAPER_READY.notify_one(); - Ok(()) + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + if reaper.running { + reaper.children.push(child); + CHILD_REAPER_READY.notify_one(); + return Ok(()); + } + // The previous loop exited between ensure_reaper's observation and + // this handoff. Retry so the child is queued only behind a live loop. + } } /// A failed detached-reaper handoff that retains ownership of the child. @@ -149,18 +160,26 @@ impl std::error::Error for ReapFailure { } fn child_reaper_loop() { - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - while reaper.starting { - reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); + let mut startup = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + while startup.starting { + startup = CHILD_REAPER_READY.wait(startup).unwrap_or_else(std::sync::PoisonError::into_inner); } + drop(startup); + + let _running = ReaperRunningGuard; + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); loop { while reaper.children.is_empty() { reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); } + let mut warnings = Vec::new(); reaper.children.retain_mut(|child| { let id = child.id(); - retain_reaper_child(id, child.try_wait()) + retain_reaper_child(id, child.try_wait(), &mut warnings) + }); + reaper = report_reaper_warnings(reaper, warnings, |warning| { + emit_reaper_warning_to(io::stderr().lock(), warning); }); if reaper.children.is_empty() { continue; @@ -173,18 +192,51 @@ fn child_reaper_loop() { } } -fn retain_reaper_child(id: u32, observation: io::Result>) -> bool { +struct ReaperRunningGuard; + +impl Drop for ReaperRunningGuard { + fn drop(&mut self) { + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + reaper.running = false; + CHILD_REAPER_READY.notify_all(); + } +} + +fn retain_reaper_child(id: u32, observation: io::Result>, warnings: &mut Vec) -> bool { match observation { Ok(None) => true, Ok(Some(_status)) => false, Err(error) if error.kind() == io::ErrorKind::Interrupted => true, Err(error) => { - eprintln!("warning: detached child reaper stopped tracking process {id} after observation failed: {error}"); + warnings.push(format!( + "warning: detached child reaper stopped tracking process {id} after observation failed: {error}" + )); false } } } +fn report_reaper_warnings( + reaper: MutexGuard<'static, ChildReaper>, + warnings: Vec, + mut report: impl FnMut(&str), +) -> MutexGuard<'static, ChildReaper> { + let mut warnings = warnings.into_iter(); + let Some(first) = warnings.next() else { + return reaper; + }; + drop(reaper); + report(&first); + for warning in warnings { + report(&warning); + } + CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner) +} + +fn emit_reaper_warning_to(mut destination: impl io::Write, message: &str) { + let _ignored = writeln!(destination, "{message}"); +} + #[cfg(test)] fn reaper_contains(id: u32) -> bool { CHILD_REAPER @@ -1840,7 +1892,7 @@ mod tests { use core::mem; use std::error::Error as _; #[cfg(unix)] - use std::io::{BufRead as _, Write as _}; + use std::io::BufRead as _; use std::{env, fs}; use camino::Utf8Path; @@ -1850,6 +1902,21 @@ mod tests { static REAPER_TEST_LOCK: Mutex<()> = Mutex::new(()); + struct FailingDiagnosticWriter { + attempted: Arc, + } + + impl io::Write for FailingDiagnosticWriter { + fn write(&mut self, _buf: &[u8]) -> io::Result { + self.attempted.store(true, Ordering::Release); + Err(io::Error::other("injected diagnostic failure")) + } + + fn flush(&mut self) -> io::Result<()> { + Ok(()) + } + } + struct PausedReader { reads: usize, ready: std::sync::mpsc::SyncSender<()>, @@ -1910,15 +1977,20 @@ mod tests { #[test] fn detached_reaper_drops_unobservable_children() { - assert!(retain_reaper_child(17, Ok(None))); + let mut warnings = Vec::new(); + assert!(retain_reaper_child(17, Ok(None), &mut warnings)); assert!(retain_reaper_child( 17, Err(io::Error::new(io::ErrorKind::Interrupted, "wait interrupted")), + &mut warnings, )); assert!(!retain_reaper_child( 17, Err(io::Error::new(io::ErrorKind::InvalidInput, "invalid child handle")), + &mut warnings, )); + assert_eq!(warnings.len(), 1); + assert!(warnings[0].contains("invalid child handle")); } struct FailingReader; @@ -2670,6 +2742,78 @@ mod tests { wait_for_reaper_to_collect(after_idle_id, Duration::from_secs(2)); } + #[test] + fn reaper_diagnostics_release_the_lock_and_an_unwind_can_restart() { + isolated_run( + "a_reaper_diagnostic_releases_the_lock_and_an_unwind_can_restart", + "the reaper restarted after a diagnostic unwind", + ); + } + + #[test] + fn a_reaper_diagnostic_releases_the_lock_and_an_unwind_can_restart() { + if env::var_os(ISOLATED_CHILD).is_none() { + return; + } + + let diagnostic_attempted = Arc::new(AtomicBool::new(false)); + emit_reaper_warning_to( + FailingDiagnosticWriter { + attempted: Arc::clone(&diagnostic_attempted), + }, + "injected warning", + ); + assert!( + diagnostic_attempted.load(Ordering::Acquire), + "the fallible diagnostic path did not attempt the stderr write" + ); + + CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner).running = true; + let (reporting, reported) = std::sync::mpsc::sync_channel(0); + let (release, released) = std::sync::mpsc::sync_channel(0); + let blocked_reporter = thread::spawn(move || { + let _running = ReaperRunningGuard; + let reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + let _reaper = report_reaper_warnings(reaper, vec!["blocked warning".to_owned()], |_warning| { + reporting.send(()).expect("the lock probe is waiting"); + released.recv().expect("the lock probe releases the diagnostic"); + }); + }); + reported + .recv_timeout(Duration::from_secs(1)) + .expect("the diagnostic callback started"); + assert!( + CHILD_REAPER.try_lock().is_ok(), + "a blocked diagnostic writer retained the global reaper lock" + ); + release.send(()).expect("release the blocked diagnostic"); + blocked_reporter.join().expect("the diagnostic reporter exits"); + assert!(!reaper_running(), "loop exit did not reset the running state"); + + CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner).running = true; + let panicked = std::panic::catch_unwind(|| { + let _running = ReaperRunningGuard; + let reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + let _reaper = report_reaper_warnings(reaper, vec!["panicking warning".to_owned()], |_warning| { + panic!("injected diagnostic panic"); + }); + }); + assert!(panicked.is_err(), "the diagnostic panic was not injected"); + assert!(!reaper_running(), "diagnostic unwind left the reaper marked running"); + assert!( + CHILD_REAPER.try_lock().is_ok(), + "diagnostic unwind left the global reaper lock unavailable" + ); + + ensure_reaper().expect("a later caller restarts the reaper"); + let child = spawn_reaper_probe(50); + let child_id = child.id(); + reap_later(child).expect("the restarted reaper accepts a child"); + wait_for_reaper_to_collect(child_id, Duration::from_secs(2)); + + println!("the reaper restarted after a diagnostic unwind"); + } + #[test] fn reaper_start_failure_preserves_child_and_preparation_ownership() { isolated_run( From 76d92a8ce69e53a75983f6e0cf56e370e3ba7753 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Fri, 18 Sep 2026 14:37:21 +0200 Subject: [PATCH 21/37] fix(cargo-each): restore cross-platform validation Restore the Unix test Write import and use repository-approved standard-input terminology in public documentation. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/README.md | 4 ++-- crates/cargo-each/docs/design/README.md | 8 ++++---- crates/cargo-each/src/cli.rs | 2 +- crates/cargo-each/src/main.rs | 4 ++-- crates/cargo-gamma-process/src/process_tree.rs | 2 +- 5 files changed, 10 insertions(+), 10 deletions(-) diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index bfbe99c7a..437ef1452 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -131,8 +131,8 @@ floor. Placeholder mode validation still runs before an empty-plan no-op. The effective worker count is the requested `--jobs` value capped by plan size and scheduler capacity. An effective count of one uses sequential -execution with inherited stdin, stdout, and stderr even when the requested -value was larger. A genuinely parallel count disconnects child stdin and +execution with inherited standard input, output, and error even when the +requested value was larger. A genuinely parallel count disconnects child input and buffers stdout and stderr; complete blocks are emitted in deterministic plan order. Fail-fast stops launching after the first observed failure, waits for running work, and chooses the final failure by diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index 2d9967d8f..a19ad2f41 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -322,10 +322,10 @@ no-op. plan order, not scheduler timing. Requested parallelism does not by itself select this captured mode: when plan-size or process-capacity capping leaves an effective worker count of one, cargo-each uses the sequential path and the - child inherits stdin, stdout, and stderr. With a genuinely parallel effective - worker count, child stdin is disconnected (`null`) so workers cannot race to - consume the caller's input; stdout and stderr are captured for deterministic - emission. A worker panic is converted into an + child inherits standard input, output, and error. With a genuinely parallel + effective worker count, child standard input is disconnected (`null`) so + workers cannot race to consume the caller's input; output and error are + captured for deterministic emission. A worker panic is converted into an infrastructure-failure outcome; each worker has a dedicated completion channel, so an unexpected exit is observable as disconnection rather than leaving the scheduler blocked forever. A worker-thread launch failure is diff --git a/crates/cargo-each/src/cli.rs b/crates/cargo-each/src/cli.rs index 9162e8023..b04b6b83d 100644 --- a/crates/cargo-each/src/cli.rs +++ b/crates/cargo-each/src/cli.rs @@ -97,7 +97,7 @@ pub(crate) struct EachArgs { /// Run at most N per-package or per-target commands concurrently. Use /// `auto` to detect available parallelism once. Defaults to 1. Buffered /// output spills to unique system-temporary files beyond 1 MiB per stream. - /// Only an effective count above 1 disconnects child stdin for capture. + /// Only an effective count above 1 disconnects child standard input for capture. #[arg(long, default_value_t = NonZeroUsize::MIN, value_name = "N|auto", value_parser = parse_jobs)] pub(crate) jobs: NonZeroUsize, diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 2b8259d88..a100b1e7a 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -121,8 +121,8 @@ //! //! The effective worker count is the requested `--jobs` value capped by plan //! size and scheduler capacity. An effective count of one uses sequential -//! execution with inherited stdin, stdout, and stderr even when the requested -//! value was larger. A genuinely parallel count disconnects child stdin and +//! execution with inherited standard input, output, and error even when the +//! requested value was larger. A genuinely parallel count disconnects child input and //! buffers stdout and stderr; complete blocks are emitted in deterministic //! plan order. Fail-fast stops launching after the first //! observed failure, waits for running work, and chooses the final failure by diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index 63a89c7c3..9bad3f4bb 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -1892,7 +1892,7 @@ mod tests { use core::mem; use std::error::Error as _; #[cfg(unix)] - use std::io::BufRead as _; + use std::io::{BufRead as _, Write as _}; use std::{env, fs}; use camino::Utf8Path; From 561206352763eb18affdcd729a66e045af3c3cdf Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Fri, 18 Sep 2026 15:55:54 +0200 Subject: [PATCH 22/37] fix(process): retain failed reaper handoffs Move recovered children into a process-wide retry owner so ordinary and contained cleanup paths remain bounded without dropping the last wait handle. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/Cargo.toml | 1 + crates/cargo-each/README.md | 5 +- crates/cargo-each/docs/design/README.md | 8 + crates/cargo-each/src/main.rs | 5 +- crates/cargo-each/src/run.rs | 227 +++++++++++------- crates/cargo-gamma-process/docs/DESIGN.md | 32 ++- .../docs/IMPLEMENTATION.md | 16 +- crates/cargo-gamma-process/src/faults.rs | 6 + crates/cargo-gamma-process/src/lib.rs | 10 +- .../cargo-gamma-process/src/process_tree.rs | 91 ++++++- 10 files changed, 281 insertions(+), 120 deletions(-) diff --git a/crates/cargo-each/Cargo.toml b/crates/cargo-each/Cargo.toml index 30bbdbc31..3459b51dc 100644 --- a/crates/cargo-each/Cargo.toml +++ b/crates/cargo-each/Cargo.toml @@ -28,6 +28,7 @@ toml = { workspace = true, features = ["parse", "serde"] } [dev-dependencies] assert_cmd = { workspace = true } +cargo-gamma-process = { workspace = true, features = ["fault-injection"] } predicates = { workspace = true } # >>> anvil-managed: anvil-lints diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index 437ef1452..f99ce1d4d 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -144,7 +144,10 @@ parallel commands retain ordinary direct-child semantics and do not kill background descendants. Each output stream retains at most 1 MiB in memory before spilling to a unique system-temporary file owned by the invocation outcome; spill failures are infrastructure failures and spill files are -removed by RAII after deterministic plan-order emission. +removed by RAII after deterministic plan-order emission. If a later +ordinary-child reaper handoff fails, cargo-each explicitly recovers the +child and transfers it to the process-wide retry queue; no returning cleanup +or local Drop path waits for it without a bound. Reader failures are observed while the child is running and trigger bounded termination. Output drain is bounded after every completion: readers get diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index a19ad2f41..d6f781344 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -340,6 +340,14 @@ no-op. so every success, failure, and panic path removes it through RAII. Spill creation, write, seek, or read failures are infrastructure failures; output is never intentionally truncated on a successful path. +- **Failed ordinary-child handoffs preserve ownership.** Captured parallel + children are preflighted against the detached reaper before spawn. If a later + bounded cleanup still cannot hand a live ordinary child to that reaper, + cargo-each explicitly recovers both the error and `Child` from + `ReapFailure`, transfers the handle to the process-wide retry queue, and + reports the infrastructure failure. The retry queue is drained by the live + reaper or its next successful restart; neither the cleanup return nor a local + Drop path performs an unbounded wait. - **Output capture and drain are bounded.** Reader failures are observed while the leader is still running; cargo-each terminates the invocation and reports the infrastructure failure instead of waiting indefinitely with an diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index a100b1e7a..752b8bf9c 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -134,7 +134,10 @@ //! background descendants. Each output stream retains at most 1 MiB in memory //! before spilling to a unique system-temporary file owned by the invocation //! outcome; spill failures are infrastructure failures and spill files are -//! removed by RAII after deterministic plan-order emission. +//! removed by RAII after deterministic plan-order emission. If a later +//! ordinary-child reaper handoff fails, cargo-each explicitly recovers the +//! child and transfers it to the process-wide retry queue; no returning cleanup +//! or local Drop path waits for it without a bound. //! //! Reader failures are observed while the child is running and trigger bounded //! termination. Output drain is bounded after every completion: readers get diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index b4055b536..0563c06cb 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -14,7 +14,9 @@ use std::sync::{Arc, Mutex, TryLockError, mpsc}; use std::time::{Duration, Instant}; use std::{fmt, thread}; -use cargo_gamma_process::{InterruptiblePipe, MemoryRequest, PreparedCommand, ProcessTree, ensure_reaper, prepare, reap_later}; +use cargo_gamma_process::{ + InterruptiblePipe, MemoryRequest, PreparedCommand, ProcessTree, ensure_reaper, prepare, reap_later, retain_for_reaper_retry, +}; use cargo_metadata::TargetKind; use ohno::{AppError, IntoAppError}; @@ -614,54 +616,34 @@ impl CapturedProcess { } } -#[mutants::skip] // Thin Child adapter; the generic helper below carries and directly tests every ownership branch. -fn finish_ordinary_termination(child: Child, result: io::Result) -> io::Result { - finish_ordinary_termination_with(child, result, Child::try_wait, reap_later) -} - -fn finish_ordinary_termination_with( - mut control: T, - result: io::Result, - observe: impl FnOnce(&mut T) -> io::Result>, - reap: impl FnOnce(T) -> Result<(), E>, -) -> io::Result -where - E: fmt::Display, -{ - match observe(&mut control) { +fn finish_ordinary_termination(mut child: Child, result: io::Result) -> io::Result { + match child.try_wait() { Ok(Some(_status)) => result, - Ok(None) | Err(_) => match reap(control) { + Ok(None) | Err(_) => match reap_later(child) { Ok(()) => result, - Err(reaper) => match result { - Ok(_status) => Err(io::Error::other(format!( - "the detached child reaper could not be started: {reaper}" - ))), - Err(error) => Err(io::Error::new( - error.kind(), - format!("{error}; the detached child reaper could not be started: {reaper}"), - )), - }, + Err(failure) => { + let (reaper, child) = failure.into_parts(); + retain_for_reaper_retry(child); + match result { + Ok(_status) => Err(io::Error::other(format!( + "the detached child reaper could not be started: {reaper}" + ))), + Err(error) => Err(io::Error::new( + error.kind(), + format!("{error}; the detached child reaper could not be started: {reaper}"), + )), + } + } }, } } -#[mutants::skip] // Thin Child adapter; the generic helper below carries and directly tests every ownership branch. -fn finish_ordinary_wait(child: Child, outcome: TreeOutcome) -> TreeOutcome { - finish_ordinary_wait_with(child, outcome, Child::try_wait, reap_later) -} - -fn finish_ordinary_wait_with( - mut control: T, - mut outcome: TreeOutcome, - observe: impl FnOnce(&mut T) -> io::Result>, - reap: impl FnOnce(T) -> Result<(), E>, -) -> TreeOutcome -where - E: fmt::Display, -{ - if !matches!(observe(&mut control), Ok(Some(_status))) - && let Err(error) = reap(control) +fn finish_ordinary_wait(mut child: Child, mut outcome: TreeOutcome) -> TreeOutcome { + if !matches!(child.try_wait(), Ok(Some(_status))) + && let Err(failure) = reap_later(child) { + let (error, child) = failure.into_parts(); + retain_for_reaper_retry(child); outcome.result = add_infrastructure_failure(outcome.result, format!("the detached child reaper could not be started: {error}")); } outcome @@ -1385,14 +1367,17 @@ mod tests { use std::time::{Duration, Instant}; use std::{io, thread}; + use cargo_gamma_process::ensure_reaper; + use cargo_gamma_process::faults::{self, Fault}; + use super::{ BufferedOutcome, CapturedOutput, CapturedProcess, CapturedStream, Invocation, InvocationResult, OutputEmitError, OutputReader, Plan, ReaderCompletion, RunningWorker, SpillFile, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, combine_captured_output, display_duration, effective_worker_count, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, - failure_stops_launching, finish_ordinary_termination_with, finish_ordinary_wait_with, finish_output_reader, - finish_wait_with_cleanup, panic_description, parallel_failure_exit_code, run_captured, run_captured_with_spawner, run_streamed, - run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_if_sealed, spawn_output_reader, spawn_output_reader_with, - spawn_tree, spawn_worker, terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, wait_for_tree, + failure_stops_launching, finish_ordinary_termination, finish_ordinary_wait, finish_output_reader, finish_wait_with_cleanup, + panic_description, parallel_failure_exit_code, run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, + run_streamed_with_timeout_with, spawn_if_sealed, spawn_output_reader, spawn_output_reader_with, spawn_tree, spawn_worker, + take_reader_output, terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, wait_for_tree, wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, }; @@ -1606,15 +1591,27 @@ mod tests { } fn sleeping_test_command() -> Command { + sleeping_test_command_for(Duration::from_secs(30)) + } + + fn sleeping_test_command_for(duration: Duration) -> Command { let mut command = Command::new(std::env::current_exe().expect("the test binary knows its path")); let _ = command .args(["--exact", "run::tests::ordinary_child_sleep_probe", "--nocapture"]) - .env("CARGO_EACH_ORDINARY_CHILD_PROBE", "1") + .env("CARGO_EACH_ORDINARY_CHILD_PROBE", duration.as_millis().to_string()) .stdout(Stdio::null()) .stderr(Stdio::null()); command } + fn wait_for_reaper_owner_to_release(id: u32) { + let deadline = Instant::now() + Duration::from_secs(2); + while faults::reaper_owns(id) && Instant::now() < deadline { + thread::sleep(Duration::from_millis(10)); + } + assert!(!faults::reaper_owns(id), "the detached reaper did not release child {id}"); + } + fn spawn_ordinary_capture(mut command: Command) -> Result { command .spawn() @@ -1891,52 +1888,75 @@ mod tests { } #[test] - fn ordinary_child_handoff_preserves_primary_and_reaper_failures() { - let status = finish_ordinary_termination_with((), Ok(successful_status()), |()| Ok(None), |()| Ok::<(), io::Error>(())) - .expect("a successful handoff preserves the termination status"); - assert!(status.success()); - - let error = finish_ordinary_termination_with( - (), - Ok(successful_status()), - |()| Ok(None), - |()| Err(io::Error::other("reaper unavailable")), + #[cfg_attr(miri, ignore = "spawns ordinary child processes")] + fn ordinary_child_handoffs_retain_a_bounded_reap_owner() { + ensure_reaper().expect("preflight the production reaper"); + let handed_off = sleeping_test_command_for(Duration::from_millis(500)) + .spawn() + .expect("spawn successful handoff probe"); + let handed_off_id = handed_off.id(); + let error = finish_ordinary_termination( + handed_off, + Err(io::Error::new(io::ErrorKind::PermissionDenied, "primary cleanup failed")), ) - .expect_err("a failed handoff must replace a successful cleanup result"); - assert!(error.to_string().contains("reaper unavailable"), "{error}"); + .expect_err("a successful handoff preserves the primary cleanup error"); + assert_eq!(error.kind(), io::ErrorKind::PermissionDenied); + assert!(faults::reaper_owns(handed_off_id), "the successful handoff lost its wait owner"); - let error = finish_ordinary_termination_with( - (), + let successful_cleanup = sleeping_test_command_for(Duration::from_millis(500)) + .spawn() + .expect("spawn successful-cleanup handoff probe"); + let successful_cleanup_id = successful_cleanup.id(); + let _failed_start = faults::arm(Fault::ReaperStart); + let error = finish_ordinary_termination(successful_cleanup, Ok(successful_status())) + .expect_err("a failed handoff replaces a successful cleanup result"); + assert!(error.to_string().contains("reaper thread start failed"), "{error}"); + assert!( + faults::reaper_owns(successful_cleanup_id), + "the successful-cleanup handoff lost its wait owner" + ); + + let termination_child = sleeping_test_command_for(Duration::from_millis(350)) + .spawn() + .expect("spawn termination handoff probe"); + let termination_id = termination_child.id(); + let _failed_start = faults::arm(Fault::ReaperStart); + let started = Instant::now(); + let error = finish_ordinary_termination( + termination_child, Err(io::Error::new(io::ErrorKind::PermissionDenied, "termination failed")), - |()| Err(io::Error::other("observation failed")), - |()| Err(io::Error::other("reaper unavailable")), ) - .expect_err("the termination and handoff failures must both be retained"); + .expect_err("the failed cleanup and handoff must be reported"); + assert!(started.elapsed() < Duration::from_millis(200), "failed handoff blocked its caller"); assert_eq!(error.kind(), io::ErrorKind::PermissionDenied); assert!(error.to_string().contains("termination failed"), "{error}"); - assert!(error.to_string().contains("reaper unavailable"), "{error}"); + assert!(error.to_string().contains("reaper thread start failed"), "{error}"); + assert!( + faults::reaper_owns(termination_id), + "the failed termination handoff lost its wait owner" + ); - let outcome = finish_ordinary_wait_with( - (), - TreeOutcome::new(InvocationResult::Exited(successful_status())), - |()| Ok(None), - |()| Err(io::Error::other("wait reaper unavailable")), + let wait_child = sleeping_test_command_for(Duration::from_millis(350)) + .spawn() + .expect("spawn wait handoff probe"); + let wait_id = wait_child.id(); + let _failed_start = faults::arm(Fault::ReaperStart); + let started = Instant::now(); + let outcome = finish_ordinary_wait(wait_child, TreeOutcome::new(InvocationResult::Exited(successful_status()))); + assert!( + started.elapsed() < Duration::from_millis(200), + "failed wait handoff blocked its caller" ); assert!( - result_infrastructure_message(outcome.result).contains("wait reaper unavailable"), - "a post-wait handoff failure must become infrastructure failure" + result_infrastructure_message(outcome.result).contains("reaper thread start failed"), + "a post-wait handoff failure must become an infrastructure failure" ); + assert!(faults::reaper_owns(wait_id), "the failed wait handoff lost its wait owner"); - let outcome = finish_ordinary_wait_with( - (), - TreeOutcome::new(InvocationResult::Exited(successful_status())), - |()| Ok(Some(successful_status())), - |()| -> io::Result<()> { panic!("an already reaped child must not be handed off") }, - ); - let InvocationResult::Exited(status) = outcome.result else { - panic!("an already reaped child must preserve its exit status"); - }; - assert!(status.success()); + wait_for_reaper_owner_to_release(handed_off_id); + wait_for_reaper_owner_to_release(successful_cleanup_id); + wait_for_reaper_owner_to_release(termination_id); + wait_for_reaper_owner_to_release(wait_id); } #[test] @@ -2079,6 +2099,20 @@ mod tests { .expect("the cancelled detached reader exits after the stalled append is released"); } + #[test] + fn detached_output_recovery_preserves_a_poisoned_available_buffer() { + let output = poisoned_buffer(); + let mut failure = None; + + let mut captured = take_reader_output(&output, "stdout", false, &mut failure); + + assert_eq!(output_bytes(&mut captured), b"poisoned bytes"); + assert!( + failure.is_some_and(|failure| failure.contains("capture buffer was poisoned")), + "the detached poisoned-buffer branch did not report its infrastructure failure" + ); + } + #[test] fn normal_completion_does_not_join_a_stubborn_descendant_pipe() { let release = Arc::new((Mutex::new(false), Condvar::new())); @@ -2237,8 +2271,9 @@ mod tests { #[test] fn ordinary_child_sleep_probe() { - if std::env::var_os("CARGO_EACH_ORDINARY_CHILD_PROBE").is_some() { - thread::sleep(Duration::from_secs(30)); + if let Some(duration) = std::env::var_os("CARGO_EACH_ORDINARY_CHILD_PROBE") { + let millis = duration.to_string_lossy().parse().expect("the parent passes a millisecond count"); + thread::sleep(Duration::from_millis(millis)); } } @@ -2480,6 +2515,30 @@ mod tests { assert!(result_infrastructure_message(spawn_failure).contains("injected timed-stream spawn failure")); } + #[test] + #[cfg_attr(miri, ignore = "spawns a contained child process")] + fn captured_runner_uses_the_production_timeout_spawner() { + let mut outcome = run_captured(&invocation(&["rustc", "--version"]), Some(Duration::from_secs(2))); + + match &outcome.result { + InvocationResult::Exited(status) => { + assert!(status.success()); + assert!( + String::from_utf8(output_bytes(&mut outcome.stdout)) + .expect("rustc output is UTF-8") + .contains("rustc") + ); + } + InvocationResult::Infrastructure(message) => { + assert!( + message.contains("timeout requires sealed process-tree containment"), + "unexpected production timeout-spawn failure: {message}" + ); + } + InvocationResult::TimedOut(duration) => panic!("rustc --version timed out after {duration:?}"), + } + } + #[test] fn captured_output_combines_every_reader_result_shape() { let mut success = combine_captured_output( diff --git a/crates/cargo-gamma-process/docs/DESIGN.md b/crates/cargo-gamma-process/docs/DESIGN.md index f1741719e..94cdbebf3 100644 --- a/crates/cargo-gamma-process/docs/DESIGN.md +++ b/crates/cargo-gamma-process/docs/DESIGN.md @@ -82,18 +82,26 @@ therefore covers the complete descendant tree. previous loop exited between the readiness check and transfer. Callers handing over children created outside `PreparedCommand` can preflight the same durable thread; if a direct handoff must start it and startup fails, - the failure returns ownership of the unqueued child. The containment handles - remain owned until the `ProcessTree` itself is dropped. An error while polling - the leader follows the same handoff before the observation error is returned, - because an observation failure does not prove the child was reaped. If the - detached reaper itself later receives an interrupted observation, it keeps the - child queued and retries. Any other observation error emits a warning to - stderr and permanently stops tracking that child. Warning formatting happens - while the queue is locked, but the fallible stderr write happens after - releasing the lock and its error is discarded. Every loop exit or unwind - clears the running state and notifies waiters, allowing a later - `ensure_reaper` or `reap_later` call to restart it. The released handle may - leave a zombie on Unix until this process exits. + the failure returns ownership of the unqueued child. A caller that cannot + retain the child locally without making its own Drop path blocking transfers + it to a separate process-wide retry queue after explicitly recovering it from + `ReapFailure`. That queue owns the handle without claiming a live reaper: the + current loop drains it when available, or the next successful startup drains + it after an earlier startup failure. `ProcessTree::terminate_bounded` uses + this fallback and never restores a live child into itself, so its later Drop + remains bounded. The containment handles remain owned until the + `ProcessTree` itself is dropped. An error while polling the leader follows the + same handoff before the observation error is returned, because an observation + failure does not prove the child was reaped. If the detached reaper itself + later receives an interrupted observation, it keeps the child queued and + retries. Any other observation error emits a warning to stderr and + permanently stops tracking that child. Warning formatting happens while the + queue is locked, but the fallible stderr write happens after releasing the + lock and its error is discarded. Every loop exit or unwind moves retained + active children back to the retry queue, clears the running state, and + notifies waiters, allowing a later `ensure_reaper` or `reap_later` call to + restart it. The released handle may leave a zombie on Unix until this process + exits. - Sealed containment uses a boundary that descendants cannot leave. A host that offers no sealed boundary at all silently uses best-effort process-group containment for an unmetered launch; absence of a warning does not establish diff --git a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md index 9b73baf65..57cd92a52 100644 --- a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md +++ b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md @@ -30,18 +30,22 @@ is retained in the returned error. The reaper polls every retained child without blocking on one leader and waits on a condition variable when its queue is empty. It is durable: repository-controlled children are created only after the thread exists, and direct `reap_later` startup failures return an unqueued -child in `ReapFailure` rather than abandoning ownership. Handoff rechecks the -running state while holding the queue lock and retries startup after a raced -loop exit. A `try_wait` error also +child in `ReapFailure` rather than abandoning ownership. Callers explicitly +recover that child with `ReapFailure::into_parts`; those whose local Drop path +would block transfer it to the process-wide retry queue. The retry queue is a +distinct owner, not evidence that a reaper is running. A live loop drains it +after notification, or a later successful startup drains it after startup +failure. Handoff rechecks the running state while holding the queue lock and +retries startup after a raced loop exit. A `try_wait` error also transfers the still-owned leader handle before returning the observation error; only a successful `Some(status)` proves that no later reaping is required. If a later `try_wait` in the detached reaper is interrupted, the child remains queued for another attempt. Any other observation error writes a warning to stderr and permanently releases that child handle. The write is fallible, its error is discarded, and it occurs outside the queue mutex. A lifecycle guard clears the -running state and wakes readiness waiters whenever the loop exits or unwinds, -so later callers can start a replacement. On Unix, the child may remain a -zombie until this process exits. +running state, returns active handles to the retry queue, and wakes readiness +waiters whenever the loop exits or unwinds, so later callers can start a +replacement. On Unix, the child may remain a zombie until this process exits. ## Platform composition diff --git a/crates/cargo-gamma-process/src/faults.rs b/crates/cargo-gamma-process/src/faults.rs index 67b6ce77e..f615d71c5 100644 --- a/crates/cargo-gamma-process/src/faults.rs +++ b/crates/cargo-gamma-process/src/faults.rs @@ -51,6 +51,12 @@ pub fn arm_late(fault: Fault, delay: Duration) -> Armed { ripe_at(fault, Instant::now() + delay) } +/// Reports whether the process-wide reaper or its retry queue owns `id`. +#[must_use] +pub fn reaper_owns(id: u32) -> bool { + crate::process_tree::reaper_contains(id) +} + fn ripe_at(fault: Fault, ripe: Instant) -> Armed { ARMED.with_borrow_mut(|armed| armed.push((fault, ripe))); diff --git a/crates/cargo-gamma-process/src/lib.rs b/crates/cargo-gamma-process/src/lib.rs index d92a7f63e..96d76e763 100644 --- a/crates/cargo-gamma-process/src/lib.rs +++ b/crates/cargo-gamma-process/src/lib.rs @@ -66,9 +66,11 @@ //! [`Command::output`](std::process::Command::output). It drains stdout and stderr concurrently, //! then sweeps descendants before waiting for inherited pipe handles to close. //! -//! Bounded termination can transfer a still-running leader to a shared detached reaper. The reaper -//! writes observation warnings outside its global queue lock through a fallible stderr path, and -//! any loop exit or unwind clears readiness state so a later handoff can start a replacement. +//! Bounded termination can transfer a still-running leader to a shared detached reaper. A failed +//! handoff moves the recovered child into a separate process-wide retry queue rather than restoring +//! it to a blocking Drop path. The reaper writes observation warnings outside its global queue lock +//! through a fallible stderr path, and any loop exit or unwind preserves retained handles for a +//! later replacement. //! //! A terminal delivers `Ctrl-C` to the whole foreground process group, so a child sharing this //! process's group dies with it automatically while a child leading its own group does not. Windows @@ -84,7 +86,7 @@ pub use memory_usage::MemoryUsage; #[doc(inline)] pub use process_tree::{ OutputError, PreparedCommand, ProcessTree, ReapFailure, SpawnFailure, SpawnedCommand, capacity, containment, ensure_reaper, output, - prepare, reap_later, + prepare, reap_later, retain_for_reaper_retry, }; mod memory_request; diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index 9bad3f4bb..146f8ea7f 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -31,12 +31,14 @@ const REAPER_PAUSE: Duration = Duration::from_millis(25); #[derive(Debug, Default)] struct ChildReaper { children: Vec, + retry_children: Vec, starting: bool, running: bool, } static CHILD_REAPER: Mutex = Mutex::new(ChildReaper { children: Vec::new(), + retry_children: Vec::new(), starting: false, running: false, }); @@ -110,6 +112,14 @@ pub fn ensure_reaper() -> io::Result<()> { /// Returns [`ReapFailure`] when the shared reaper could not be started. The /// failure retains the child handle so the caller can recover ownership. pub fn reap_later(child: Child) -> Result<(), ReapFailure> { + #[cfg(any(test, feature = "fault-injection"))] + if faults::fired(faults::Fault::ReaperStart) { + return Err(ReapFailure { + cause: io::Error::other("detached child reaper thread start failed as requested by a test"), + child, + }); + } + loop { if let Err(cause) = ensure_reaper() { return Err(ReapFailure { cause, child }); @@ -126,6 +136,19 @@ pub fn reap_later(child: Child) -> Result<(), ReapFailure> { } } +/// Retains a child from a failed [`reap_later`] handoff for a later retry. +/// +/// The process-wide retry queue owns the handle without claiming that a reaper +/// thread is live. A running reaper drains it when notified; otherwise the next +/// successfully started reaper drains it. This function never waits for the +/// child. It is intended for callers that have explicitly recovered the child +/// with [`ReapFailure::into_parts`]. +pub fn retain_for_reaper_retry(child: Child) { + let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + reaper.retry_children.push(child); + CHILD_REAPER_READY.notify_one(); +} + /// A failed detached-reaper handoff that retains ownership of the child. #[derive(Debug)] pub struct ReapFailure { @@ -169,9 +192,11 @@ fn child_reaper_loop() { let _running = ReaperRunningGuard; let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); loop { - while reaper.children.is_empty() { + while reaper.children.is_empty() && reaper.retry_children.is_empty() { reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); } + let mut retries = std::mem::take(&mut reaper.retry_children); + reaper.children.append(&mut retries); let mut warnings = Vec::new(); reaper.children.retain_mut(|child| { @@ -197,6 +222,8 @@ struct ReaperRunningGuard; impl Drop for ReaperRunningGuard { fn drop(&mut self) { let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + let mut active = std::mem::take(&mut reaper.children); + reaper.retry_children.append(&mut active); reaper.running = false; CHILD_REAPER_READY.notify_all(); } @@ -237,13 +264,13 @@ fn emit_reaper_warning_to(mut destination: impl io::Write, message: &str) { let _ignored = writeln!(destination, "{message}"); } -#[cfg(test)] -fn reaper_contains(id: u32) -> bool { - CHILD_REAPER - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) +#[cfg(any(test, feature = "fault-injection"))] +pub(crate) fn reaper_contains(id: u32) -> bool { + let reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + reaper .children .iter() + .chain(reaper.retry_children.iter()) .any(|child| child.id() == id) } @@ -1546,7 +1573,8 @@ impl ProcessTree { /// timed-out error (including an earlier termination error, when present) /// if the leader is still running after `grace`. If the detached reaper /// cannot accept the child, the returned error includes that failure and - /// this process tree retains the child handle. + /// the child handle moves to the process-wide retry queue. This process tree + /// remains empty so its [`Drop`] path cannot wait for the live child. pub fn terminate_bounded(&mut self, grace: Duration) -> io::Result { let started = Instant::now(); let mut child = self @@ -1596,7 +1624,7 @@ impl ProcessTree { let mut message = error.to_string(); if let Err(failure) = reap_later(child) { let (reaper, child) = failure.into_parts(); - self.child = Some(child); + retain_for_reaper_retry(child); message = format!("{message}; the detached child reaper could not be started: {reaper}"); } return Err(io::Error::new(kind, message)); @@ -1611,7 +1639,7 @@ impl ProcessTree { ); if let Err(failure) = reap_later(child) { let (error, child) = failure.into_parts(); - self.child = Some(child); + retain_for_reaper_retry(child); message = format!("{message}; the detached child reaper could not be started: {error}"); } return Err(io::Error::new(kind, message)); @@ -2717,6 +2745,44 @@ mod tests { wait_for_reaper_to_collect(leader, Duration::from_secs(2)); } + #[test] + fn bounded_termination_reaper_start_failures_keep_return_and_drop_bounded() { + let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + + for fail_observation in [false, true] { + let mut command = Command::new(testing::helper_binary_path().as_std_path()); + let _ = command.arg(testing::directive("sleep:350")); + let prepared = prepare(command, MemoryRequest::default()).expect("containment"); + let spawned = prepared.spawn().expect("spawn"); + let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); + let leader = subtree.child.as_ref().expect("the adopted subtree owns its leader").id(); + let _ignored_kill = faults::arm(faults::Fault::Linger); + let _failed_observation = fail_observation.then(|| faults::arm(faults::Fault::Observe)); + let _failed_reaper_start = faults::arm(faults::Fault::ReaperStart); + + let started = Instant::now(); + let error = subtree + .terminate_bounded(Duration::from_millis(25)) + .expect_err("the injected reaper start failure must be reported"); + assert!( + started.elapsed() < Duration::from_millis(200), + "bounded termination exceeded its grace after reaper startup failed" + ); + assert!(error.to_string().contains("reaper thread start failed"), "{error}"); + assert!(subtree.child.is_none(), "the live child was restored to the Drop path"); + assert!(reaper_contains(leader), "the retry queue did not retain the live child"); + + let drop_started = Instant::now(); + drop(subtree); + assert!( + drop_started.elapsed() < Duration::from_millis(200), + "ProcessTree::drop waited for the stubborn child" + ); + + wait_for_reaper_to_collect(leader, Duration::from_secs(2)); + } + } + #[test] fn shared_reaper_collects_out_of_order_and_remains_ready_after_idle() { let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); @@ -2847,11 +2913,11 @@ mod tests { .to_string() .contains("thread start failed as requested") ); - let (error, mut child) = failure.into_parts(); + let (error, child) = failure.into_parts(); assert!(error.to_string().contains("thread start failed as requested"), "{error}"); assert_eq!(child.id(), child_id, "the failed handoff returned a different child"); - child.kill().expect("terminate the recovered child"); - let _status = child.wait().expect("reap the recovered child"); + retain_for_reaper_retry(child); + assert!(reaper_contains(child_id), "the retry queue did not take ownership"); let prepared = prepare(no_op_command(), MemoryRequest::default()).expect("containment"); let _failed_start = faults::arm(faults::Fault::ReaperStart); @@ -2868,6 +2934,7 @@ mod tests { .spawn() .expect("the unchanged preparation can retry after the transient failure"); drop(spawned); + wait_for_reaper_to_collect(child_id, Duration::from_secs(2)); println!("the failed reaper start preserved both owners"); } From 78844666c054563c62e90ed3efb6ec5c9e2bbc05 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Mon, 21 Sep 2026 20:38:26 +0200 Subject: [PATCH 23/37] refactor(cargo-each): own process-group execution Remove all cargo-gamma dependencies and changes, use command-group for Unix process groups and Windows jobs, and keep timeout, capture, spill, and bounded fallback behavior local to cargo-each. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- .spelling | 1 - Cargo.lock | 25 +- crates/cargo-each/Cargo.toml | 3 +- crates/cargo-each/README.md | 39 +- crates/cargo-each/docs/design/README.md | 62 +- crates/cargo-each/src/cli.rs | 7 +- crates/cargo-each/src/main.rs | 39 +- crates/cargo-each/src/run.rs | 2650 ++++++----------- crates/cargo-each/tests/cli.rs | 49 +- crates/cargo-gamma-process/docs/DESIGN.md | 41 +- .../docs/IMPLEMENTATION.md | 35 +- crates/cargo-gamma-process/src/faults.rs | 23 - crates/cargo-gamma-process/src/lib.rs | 12 +- .../cargo-gamma-process/src/process_tree.rs | 808 +---- crates/cargo-gamma-unsafe/Cargo.toml | 1 - crates/cargo-gamma-unsafe/docs/DESIGN.md | 6 - .../cargo-gamma-unsafe/docs/IMPLEMENTATION.md | 11 - crates/cargo-gamma-unsafe/src/lib.rs | 5 +- crates/cargo-gamma-unsafe/src/pipe.rs | 230 -- 19 files changed, 1057 insertions(+), 2990 deletions(-) delete mode 100644 crates/cargo-gamma-unsafe/src/pipe.rs diff --git a/.spelling b/.spelling index 6fe2043d8..127d8068f 100644 --- a/.spelling +++ b/.spelling @@ -931,7 +931,6 @@ symlinked uninherited closers unmanaged -interruptible nonblocking SHA upserts diff --git a/Cargo.lock b/Cargo.lock index 3f0ac5e0b..ff19e440b 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -538,9 +538,9 @@ name = "cargo-each" version = "0.2.0" dependencies = [ "assert_cmd", - "cargo-gamma-process", "cargo_metadata", "clap", + "command-group", "mutants", "ohno", "predicates", @@ -892,6 +892,16 @@ dependencies = [ "memchr", ] +[[package]] +name = "command-group" +version = "5.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a68fa787550392a9d58f44c21a3022cfb3ea3e2458b7f85d3b399d0ceeccf409" +dependencies = [ + "nix 0.27.1", + "winapi", +] + [[package]] name = "compact_str" version = "0.10.0" @@ -3090,7 +3100,7 @@ dependencies = [ "combine", "libc", "mach2", - "nix", + "nix 0.30.1", "sysctl", "thiserror 2.0.20", "widestring", @@ -3112,6 +3122,17 @@ version = "0.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "add0ac067452ff1aca8c5002111bd6b1c895baee6e45fcbc44e0193aea17be56" +[[package]] +name = "nix" +version = "0.27.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2eb04e9c688eff1c89d72b407f168cf79bb9e867a9d3323ed6c01519eb9cc053" +dependencies = [ + "bitflags 2.13.1", + "cfg-if", + "libc", +] + [[package]] name = "nix" version = "0.30.1" diff --git a/crates/cargo-each/Cargo.toml b/crates/cargo-each/Cargo.toml index 3459b51dc..9e4014462 100644 --- a/crates/cargo-each/Cargo.toml +++ b/crates/cargo-each/Cargo.toml @@ -17,9 +17,9 @@ homepage.workspace = true repository = "https://github.com/microsoft/ox-tools/tree/main/crates/cargo-each" [dependencies] -cargo-gamma-process = { workspace = true } cargo_metadata = { workspace = true } clap = { workspace = true, features = ["derive", "std", "help", "usage", "error-context"] } +command-group = "=5.0.1" mutants = { workspace = true } ohno = { workspace = true, features = ["app-err"] } serde_json = { workspace = true, features = ["std"] } @@ -28,7 +28,6 @@ toml = { workspace = true, features = ["parse", "serde"] } [dev-dependencies] assert_cmd = { workspace = true } -cargo-gamma-process = { workspace = true, features = ["fault-injection"] } predicates = { workspace = true } # >>> anvil-managed: anvil-lints diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index f99ce1d4d..a33ccff45 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -89,11 +89,10 @@ can be double-quoted. Expression atoms: per-target work. Omitting it runs exactly one invocation at a time; `auto` resolves once to the machine’s available parallelism. Detection failure is reported explicitly without falling back. `--timeout ` terminates -each invocation and its process tree independently (`250ms`, `30s`, or -`2m`). -Timeouts require sealed process-tree containment; on a host that only -offers best-effort containment, cargo-each reports an unsupported -infrastructure failure before starting the child. +each invocation’s Windows job object or Unix process group independently +(`250ms`, `30s`, or `2m`). Unix descendants can escape a process group by +starting a new session, so timeout cleanup is best-effort for those escaped +descendants. `--chdir` runs each per-package or per-target command from that member crate root; `--dry-run` prints commands without running them. @@ -130,7 +129,7 @@ every member’s resolved minimum to be present and no newer than the root floor. Placeholder mode validation still runs before an empty-plan no-op. The effective worker count is the requested `--jobs` value capped by plan -size and scheduler capacity. An effective count of one uses sequential +size. An effective count of one uses sequential execution with inherited standard input, output, and error even when the requested value was larger. A genuinely parallel count disconnects child input and buffers stdout and stderr; complete blocks are emitted in deterministic @@ -140,25 +139,25 @@ plan order. `--keep-going` runs the complete plan. Worker panics and unexpected worker-channel disconnections become infrastructure-failure outcomes instead of blocking the scheduler. Worker launch failures retain output already collected at earlier plan indices. Without `--timeout`, -parallel commands retain ordinary direct-child semantics and do not kill -background descendants. Each output stream retains at most 1 MiB in memory +parallel commands are launched in a job or process group, but cargo-each +observes only the leader and does not kill background descendants. Each +output stream retains at most 1 MiB in memory before spilling to a unique system-temporary file owned by the invocation outcome; spill failures are infrastructure failures and spill files are -removed by RAII after deterministic plan-order emission. If a later -ordinary-child reaper handoff fails, cargo-each explicitly recovers the -child and transfers it to the process-wide retry queue; no returning cleanup -or local Drop path waits for it without a bound. +removed by RAII after deterministic plan-order emission. Reader failures are observed while the child is running and trigger bounded termination. Output drain is bounded after every completion: readers get -one second to observe EOF, then readiness-polling capture is cancelled and -joined while partial bytes become an explicit infrastructure failure. If a -cancelled reader remains stalled while holding its capture mutex, output -recovery is nonblocking and any unavailable partial bytes are reported -rather than extending the drain bound. -Timed-out tree termination likewise gets a bounded 250 ms leader-reap grace, -after which the leader handle moves to a shared detached reaper so no wait -or Drop path can defeat the timeout without abandoning reap ownership. +one second to observe EOF, then cargo-each stops retaining bytes and gives +each reader a bounded join opportunity. A reader blocked in a pipe read is +detached and may remain until an escaped or background descendant closes +the pipe. Partial output is recovered only through a nonblocking capture +mutex acquisition; unavailable bytes and every detached-reader case are +explicit infrastructure failures. +Timed-out group termination gets a bounded 250 ms reap grace. If the group +still has not completed, its handle moves to a detached cargo-each reaper +thread whose blocking wait cannot delay the caller; failure to start that +thread is also reported as an infrastructure failure. Child commands inherit `PATH` explicitly. On Windows this makes relative program lookup honor the inherited `PATH` order instead of preferring an unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index d6f781344..66374d9f3 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -250,8 +250,8 @@ filtered set is empty, `cargo-each` exits 0, exactly like an empty selection. | `--each-target ` | **per-target**: run once for each selected member target of `KIND`. Repeatable; kinds are OR-combined and each target runs at most once. Mutually exclusive with `--once`. | | `--target-required-feature ` | In per-target mode, retain targets whose `required-features` contains `FEATURE`. Repeatable; values are AND-combined. Requires `--each-target`. | | `--keep-going` | Don't stop at the first failing command; run them all and exit non-zero if any failed. Default is fail-fast (exit with the first failure's code). | -| `--jobs ` | Run at most the positive integer `N` per-package or per-target commands concurrently. When omitted, the default is exactly `1`. `auto` resolves once during CLI parsing via `std::thread::available_parallelism()`; detection failure is an explicit usage error with no fallback. The effective worker count remains capped by the plan size and scheduler capacity. With `--once`, resolved values other than `1` are a usage error. | -| `--timeout ` | Terminate an invocation and its child process tree when it exceeds the positive duration, such as `30s` or `2m`. Applies independently to every invocation, including `--once`. Requires sealed process-tree containment; unsupported hosts fail before the child starts. No timeout by default. | +| `--jobs ` | Run at most the positive integer `N` per-package or per-target commands concurrently. When omitted, the default is exactly `1`. `auto` resolves once during CLI parsing via `std::thread::available_parallelism()`; detection failure is an explicit usage error with no fallback. The effective worker count remains capped by the plan size. With `--once`, resolved values other than `1` are a usage error. | +| `--timeout ` | Terminate an invocation's Windows job object or Unix process group when it exceeds the positive duration, such as `30s` or `2m`. Applies independently to every invocation, including `--once`. Unix descendants can escape by starting a new session, so termination is best-effort for those escaped descendants. No timeout by default. | | `--chdir` | Run each per-package or per-target command from that member's crate root (the directory containing its `Cargo.toml`) instead of the caller's CWD. Combined with `--once` it is a usage error (exit 2). Placeholders stay absolute, so only *relative* args in the command shift to the member dir. | | `--manifest-path ` | Workspace root `Cargo.toml`. Defaults to auto-detection from CWD. | | `--dry-run` | Print the fully-substituted commands that *would* run, one per line, without executing. | @@ -314,13 +314,13 @@ no-op. invocation. A positive integer requests that fixed limit; `auto` resolves exactly once during CLI parsing to the machine's available parallelism and fails explicitly if detection is unavailable. The scheduler caps every - request by the plan size and its process capacity. With an effective job + request by the plan size. With an effective job count above one, output from each invocation is buffered and emitted as one block in deterministic plan order. Fail-fast stops launching new work after the first observed failure and waits for already-running children; `--keep-going` launches the complete plan. The final failure is chosen by plan order, not scheduler timing. Requested parallelism does not by itself - select this captured mode: when plan-size or process-capacity capping leaves + select this captured mode: when plan-size capping leaves an effective worker count of one, cargo-each uses the sequential path and the child inherits standard input, output, and error. With a genuinely parallel effective worker count, child standard input is disconnected (`null`) so @@ -331,48 +331,40 @@ no-op. leaving the scheduler blocked forever. A worker-thread launch failure is represented as an infrastructure outcome at that invocation's plan index, so output already collected from earlier invocations is still emitted. - Without `--timeout`, parallel commands use the ordinary direct-child - lifecycle: - cargo-each waits for the launched leader but does not contain or kill - background descendants. Buffering is memory-bounded per stream: after 1 MiB, + Without `--timeout`, parallel commands are still launched in a Windows job + or Unix process group, but cargo-each observes only the launched leader and + does not kill background descendants. Buffering is memory-bounded per stream: + after 1 MiB, output spills to a unique file in the system temporary directory. The invocation outcome owns that file through deterministic plan-order emission, so every success, failure, and panic path removes it through RAII. Spill creation, write, seek, or read failures are infrastructure failures; output is never intentionally truncated on a successful path. -- **Failed ordinary-child handoffs preserve ownership.** Captured parallel - children are preflighted against the detached reaper before spawn. If a later - bounded cleanup still cannot hand a live ordinary child to that reaper, - cargo-each explicitly recovers both the error and `Child` from - `ReapFailure`, transfers the handle to the process-wide retry queue, and - reports the infrastructure failure. The retry queue is drained by the live - reaper or its next successful restart; neither the cleanup return nor a local - Drop path performs an unbounded wait. - **Output capture and drain are bounded.** Reader failures are observed while the leader is still running; cargo-each terminates the invocation and reports the infrastructure failure instead of waiting indefinitely with an unconsumed pipe. After leader completion, readers get one second to observe EOF. Complete output is preserved when both pipes close within that grace. If a background or escaped descendant keeps a pipe open, capture stops - retaining new bytes, cancels and joins the readiness-polling reader within a - bounded grace, emits the partial bytes already buffered when their capture - mutex is immediately available, and reports an explicit infrastructure - failure rather than hanging, silently succeeding, or accumulating detached - reader threads. If cancellation expires while a reader is stalled inside a - spill operation with that mutex held, cargo-each detaches the reader and uses - a nonblocking acquisition; unavailable partial bytes are reported explicitly - instead of defeating the drain bound. -- **Timeouts terminate trees.** A timed-out command is a failure. cargo-each - terminates the child process tree rather than only the immediate process, so - compiler or test descendants cannot continue mutating the target directory - after cargo-each returns. Termination gets a bounded 250 ms grace to reap the - leader. If signalling fails and the leader is still running at that deadline, - its handle is transferred to a shared detached reaper so neither termination - nor Drop can defeat the invocation timeout while the leader still has a - wait/reap owner; cargo-each reports the infrastructure failure. A timeout is - accepted only when launch preparation reports a sealed cgroup or job - boundary. On a host with best-effort process-group containment, cargo-each - reports that timeout is unsupported and does not spawn the command. + retaining new bytes and gives the reader a bounded join opportunity. A + blocking pipe read cannot be forcibly interrupted without platform-specific + unsafe code, so a reader that does not finish is detached and may remain + until the escaped or background descendant closes the pipe. cargo-each emits + partial bytes only when their capture mutex is immediately available and + reports an explicit infrastructure failure, including when partial bytes + cannot be recovered without blocking. It never blocks on that mutex after + detaching a reader. +- **Timeouts terminate jobs or process groups.** A timed-out command is a + failure. `command-group` creates a job object on Windows and a process group + on Unix. cargo-each kills that boundary, polls completion with bounded sleeps, + and allows 250 ms for the group to finish. If it still has not completed, the + `GroupChild` moves to a detached cargo-each-local reaper thread whose + blocking wait cannot delay the caller. A failed kill, observation, bounded + reap, or reaper-thread launch is an infrastructure failure. Unix process + groups are not sealed containment: a descendant can escape by creating a new + session, and timeout cleanup is best-effort for such descendants. An escaped + descendant can also keep an inherited output pipe open, in which case the + bounded drain behavior above applies. - **Child executable resolution follows `PATH`.** `cargo-each` explicitly copies an inherited `PATH` onto every child command. This is equivalent to ordinary inheritance on other platforms and makes Windows resolve a relative diff --git a/crates/cargo-each/src/cli.rs b/crates/cargo-each/src/cli.rs index b04b6b83d..9939a2114 100644 --- a/crates/cargo-each/src/cli.rs +++ b/crates/cargo-each/src/cli.rs @@ -101,10 +101,9 @@ pub(crate) struct EachArgs { #[arg(long, default_value_t = NonZeroUsize::MIN, value_name = "N|auto", value_parser = parse_jobs)] pub(crate) jobs: NonZeroUsize, - /// Terminate each invocation and its process tree after this duration. - /// Requires sealed process-tree containment; unsupported hosts fail before - /// starting the child. Accepts a positive integer followed by `ms`, `s`, - /// or `m`. + /// Terminate each invocation's Windows job or Unix process group after this + /// duration. Unix descendants can escape by starting a new session. Accepts + /// a positive integer followed by `ms`, `s`, or `m`. #[arg(long, value_name = "DURATION", value_parser = parse_duration)] pub(crate) timeout: Option, diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 752b8bf9c..0f88254e6 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -79,11 +79,10 @@ //! per-target work. Omitting it runs exactly one invocation at a time; `auto` //! resolves once to the machine's available parallelism. Detection failure is //! reported explicitly without falling back. `--timeout ` terminates -//! each invocation and its process tree independently (`250ms`, `30s`, or -//! `2m`). -//! Timeouts require sealed process-tree containment; on a host that only -//! offers best-effort containment, cargo-each reports an unsupported -//! infrastructure failure before starting the child. +//! each invocation's Windows job object or Unix process group independently +//! (`250ms`, `30s`, or `2m`). Unix descendants can escape a process group by +//! starting a new session, so timeout cleanup is best-effort for those escaped +//! descendants. //! `--chdir` runs each per-package or per-target command from that member crate //! root; `--dry-run` prints commands without running them. //! @@ -120,7 +119,7 @@ //! floor. Placeholder mode validation still runs before an empty-plan no-op. //! //! The effective worker count is the requested `--jobs` value capped by plan -//! size and scheduler capacity. An effective count of one uses sequential +//! size. An effective count of one uses sequential //! execution with inherited standard input, output, and error even when the //! requested value was larger. A genuinely parallel count disconnects child input and //! buffers stdout and stderr; complete blocks are emitted in deterministic @@ -130,25 +129,25 @@ //! unexpected worker-channel disconnections become infrastructure-failure //! outcomes instead of blocking the scheduler. Worker launch failures retain //! output already collected at earlier plan indices. Without `--timeout`, -//! parallel commands retain ordinary direct-child semantics and do not kill -//! background descendants. Each output stream retains at most 1 MiB in memory +//! parallel commands are launched in a job or process group, but cargo-each +//! observes only the leader and does not kill background descendants. Each +//! output stream retains at most 1 MiB in memory //! before spilling to a unique system-temporary file owned by the invocation //! outcome; spill failures are infrastructure failures and spill files are -//! removed by RAII after deterministic plan-order emission. If a later -//! ordinary-child reaper handoff fails, cargo-each explicitly recovers the -//! child and transfers it to the process-wide retry queue; no returning cleanup -//! or local Drop path waits for it without a bound. +//! removed by RAII after deterministic plan-order emission. //! //! Reader failures are observed while the child is running and trigger bounded //! termination. Output drain is bounded after every completion: readers get -//! one second to observe EOF, then readiness-polling capture is cancelled and -//! joined while partial bytes become an explicit infrastructure failure. If a -//! cancelled reader remains stalled while holding its capture mutex, output -//! recovery is nonblocking and any unavailable partial bytes are reported -//! rather than extending the drain bound. -//! Timed-out tree termination likewise gets a bounded 250 ms leader-reap grace, -//! after which the leader handle moves to a shared detached reaper so no wait -//! or Drop path can defeat the timeout without abandoning reap ownership. +//! one second to observe EOF, then cargo-each stops retaining bytes and gives +//! each reader a bounded join opportunity. A reader blocked in a pipe read is +//! detached and may remain until an escaped or background descendant closes +//! the pipe. Partial output is recovered only through a nonblocking capture +//! mutex acquisition; unavailable bytes and every detached-reader case are +//! explicit infrastructure failures. +//! Timed-out group termination gets a bounded 250 ms reap grace. If the group +//! still has not completed, its handle moves to a detached cargo-each reaper +//! thread whose blocking wait cannot delay the caller; failure to start that +//! thread is also reported as an infrastructure failure. //! Child commands inherit `PATH` explicitly. On Windows this makes relative //! program lookup honor the inherited `PATH` order instead of preferring an //! unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 0563c06cb..dae8f3533 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -8,16 +8,14 @@ use std::collections::{BTreeSet, VecDeque}; use std::io::{self, Read as _, Seek as _, SeekFrom, Write as _}; use std::num::NonZeroUsize; use std::panic::{self, AssertUnwindSafe, UnwindSafe}; -use std::process::{Child, ChildStderr, ChildStdout, Command, ExitCode, ExitStatus, Stdio}; +use std::process::{ChildStderr, ChildStdout, Command, ExitCode, ExitStatus, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Mutex, TryLockError, mpsc}; use std::time::{Duration, Instant}; use std::{fmt, thread}; -use cargo_gamma_process::{ - InterruptiblePipe, MemoryRequest, PreparedCommand, ProcessTree, ensure_reaper, prepare, reap_later, retain_for_reaper_retry, -}; use cargo_metadata::TargetKind; +use command_group::{CommandGroup as _, GroupChild}; use ohno::{AppError, IntoAppError}; use crate::cli::EachArgs; @@ -143,7 +141,7 @@ fn parse_target_kinds(kinds: &[String]) -> Result, AppError } fn execute(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: Option) -> Result { - let worker_count = effective_worker_count(jobs, plan.invocations.len(), cargo_gamma_process::capacity()); + let worker_count = effective_worker_count(jobs, plan.invocations.len()); if worker_count.get() == 1 { Ok(execute_sequential(plan, keep_going, timeout)) } else { @@ -151,9 +149,8 @@ fn execute(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: Option NonZeroUsize { - NonZeroUsize::new(requested.get().min(plan_size).min(process_capacity.max(1))) - .expect("execution receives a nonempty plan and process capacity is clamped to at least one") +fn effective_worker_count(requested: NonZeroUsize, plan_size: usize) -> NonZeroUsize { + NonZeroUsize::new(requested.get().min(plan_size)).expect("execution receives a nonempty plan") } fn execute_sequential(plan: &Plan, keep_going: bool, timeout: Option) -> ExitCode { @@ -346,25 +343,33 @@ fn run_streamed(invocation: &Invocation) -> InvocationResult { } fn run_streamed_with_timeout(invocation: &Invocation, timeout: Duration) -> InvocationResult { - run_streamed_with_timeout_with(invocation, timeout, spawn_sealed_tree) + run_streamed_with_timeout_with(invocation, timeout, spawn_group) } fn run_streamed_with_timeout_with( invocation: &Invocation, timeout: Duration, - spawn: impl FnOnce(Command) -> Result, + spawn: impl FnOnce(Command) -> Result, ) -> InvocationResult { let (program, command) = match command_for(invocation) { Ok(command) => command, Err(message) => return InvocationResult::Infrastructure(message), }; - let mut tree = match spawn(command) { + let tree = match spawn(command) { Ok(tree) => tree, Err(error) => { return InvocationResult::Infrastructure(format!("failed to spawn `{program}`: {error}")); } }; - wait_for_tree(&mut tree, timeout).result + wait_for_process( + tree, + Some(timeout), + GroupChild::try_wait, + || None, + terminate_group_bounded, + "observe child process group", + ) + .result } fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedOutcome { @@ -374,86 +379,86 @@ fn run_captured(invocation: &Invocation, timeout: Option) -> BufferedO "injected worker panic" ); - run_captured_with_spawner(invocation, timeout, |command| { - spawn_sealed_tree(command).map(CapturedProcess::Contained) - }) + run_captured_with( + invocation, + timeout, + spawn_group, + spawn_child_stdout_reader, + spawn_child_stderr_reader, + ) } -fn run_captured_with_spawner( +fn run_captured_with( invocation: &Invocation, timeout: Option, - timed_spawner: impl FnOnce(Command) -> Result, + spawner: impl FnOnce(Command) -> Result, + stdout_spawner: impl FnOnce(ChildStdout, &'static str) -> io::Result, + stderr_spawner: impl FnOnce(ChildStderr, &'static str) -> io::Result, ) -> BufferedOutcome { let (program, mut command) = match command_for(invocation) { Ok(command) => command, Err(message) => return BufferedOutcome::infrastructure(message), }; let _ = command.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); - let process = match timeout { - Some(_) => timed_spawner(command), - None => ensure_reaper() - .and_then(|()| command.spawn()) - .map(|child| CapturedProcess::Ordinary(Some(child))) - .map_err(|error| error.to_string()), - }; - let mut process = match process { + let mut process = match spawner(command) { Ok(process) => process, Err(error) => { return BufferedOutcome::infrastructure(format!("failed to spawn `{program}`: {error}")); } }; - let drain_boundary = process.drain_boundary(); - let capture_fault = capture_fault(invocation); - - let stdout = if capture_fault == Some(CaptureFault::MissingStdout) { - None - } else { - process.take_stdout() - }; + let stdout = process.inner().stdout.take(); let Some(stdout) = stdout else { - let cleanup = process.terminate_bounded(); + let cleanup = terminate_group_bounded(process); return BufferedOutcome::infrastructure(with_cleanup_failure("failed to capture child stdout".to_owned(), &cleanup)); }; - let mut stdout_reader = match if capture_fault == Some(CaptureFault::StdoutReader) { - Err(io::Error::other("injected stdout reader failure")) - } else { - spawn_child_stdout_reader(stdout, "cargo-each-stdout") - } { + let mut stdout_reader = match stdout_spawner(stdout, "cargo-each-stdout") { Ok(reader) => reader, Err(error) => { - let cleanup = process.terminate_bounded(); + let cleanup = terminate_group_bounded(process); return BufferedOutcome::infrastructure(with_cleanup_failure(format!("failed to create stdout reader: {error}"), &cleanup)); } }; - let stderr = if capture_fault == Some(CaptureFault::MissingStderr) { - None - } else { - process.take_stderr() - }; + let stderr = process.inner().stderr.take(); let Some(stderr) = stderr else { - let cleanup = process.terminate_bounded(); - return BufferedOutcome::from_reader_failure("failed to capture child stderr".to_owned(), stdout_reader, &cleanup, drain_boundary); + let cleanup = terminate_group_bounded(process); + return BufferedOutcome::from_reader_failure("failed to capture child stderr".to_owned(), stdout_reader, &cleanup); }; - let mut stderr_reader = match if capture_fault == Some(CaptureFault::StderrReader) { - Err(io::Error::other("injected stderr reader failure")) - } else { - spawn_child_stderr_reader(stderr, "cargo-each-stderr") - } { + let mut stderr_reader = match stderr_spawner(stderr, "cargo-each-stderr") { Ok(reader) => reader, Err(error) => { - let cleanup = process.terminate_bounded(); - return BufferedOutcome::from_reader_failure( - format!("failed to create stderr reader: {error}"), - stdout_reader, - &cleanup, - drain_boundary, - ); + let cleanup = terminate_group_bounded(process); + return BufferedOutcome::from_reader_failure(format!("failed to create stderr reader: {error}"), stdout_reader, &cleanup); } }; - let process_outcome = process.wait(timeout, capture_fault, &mut stdout_reader, &mut stderr_reader); - drop(process); - let (stdout, stderr) = finish_output_readers(stdout_reader, stderr_reader, OUTPUT_DRAIN_GRACE, drain_boundary); + let observe_group = timeout.is_some(); + let process_outcome = wait_for_process( + process, + timeout, + move |process| { + if observe_group { + process.try_wait() + } else { + process.inner().try_wait() + } + }, + || { + let failure = [stdout_reader.take_failure("stdout"), stderr_reader.take_failure("stderr")] + .into_iter() + .flatten() + .collect::>() + .join("; "); + (!failure.is_empty()).then_some(failure) + }, + terminate_group_bounded, + if observe_group { + "observe child process group" + } else { + "observe child process leader" + }, + ); + let boundary = if observe_group { "process group" } else { "process leader" }; + let (stdout, stderr) = finish_output_readers(stdout_reader, stderr_reader, OUTPUT_DRAIN_GRACE, boundary); combine_captured_output(stdout, stderr, process_outcome.result) } @@ -499,182 +504,31 @@ fn command_for(invocation: &Invocation) -> Result<(&str, Command), String> { Ok((program, command)) } -#[cfg(test)] -fn spawn_tree(command: Command) -> Result { - let prepared = prepare_tree(command)?; - spawn_prepared_tree(prepared) -} - -fn spawn_sealed_tree(command: Command) -> Result { - let prepared = prepare_tree(command)?; - spawn_if_sealed(prepared, PreparedCommand::sealed, spawn_prepared_tree) -} - -fn prepare_tree(command: Command) -> Result { - prepare(command, MemoryRequest::default()).map_err(|error| format!("could not prepare process-tree containment: {error}")) -} - -fn spawn_prepared_tree(prepared: PreparedCommand) -> Result { - let spawned = prepared.spawn().map_err(|failure| failure.to_string())?; - ProcessTree::adopt(spawned).map_err(|error| format!("could not adopt child into process-tree containment: {error}")) -} - -fn spawn_if_sealed(prepared: T, sealed: impl FnOnce(&T) -> bool, spawn: impl FnOnce(T) -> Result) -> Result { - if !sealed(&prepared) { - return Err( - "timeout requires sealed process-tree containment, but this host only provides best-effort containment; the child was not started" - .to_owned(), - ); - } - spawn(prepared) -} - -enum CapturedProcess { - Ordinary(Option), - Contained(ProcessTree), -} - -impl CapturedProcess { - fn drain_boundary(&self) -> &'static str { - match self { - Self::Ordinary(_) => "ordinary process tree", - Self::Contained(_) => "contained process tree", - } - } - - fn take_stdout(&mut self) -> Option { - match self { - Self::Ordinary(child) => child.as_mut()?.stdout.take(), - Self::Contained(tree) => tree.take_stdout(), - } - } - - fn take_stderr(&mut self) -> Option { - match self { - Self::Ordinary(child) => child.as_mut()?.stderr.take(), - Self::Contained(tree) => tree.take_stderr(), - } - } - - fn terminate_bounded(&mut self) -> io::Result { - match self { - Self::Ordinary(child) => { - let mut child = child - .take() - .ok_or_else(|| io::Error::other("ordinary child was already reaped or detached"))?; - let result = terminate_ordinary_child(&mut child, TERMINATION_GRACE); - finish_ordinary_termination(child, result) - } - Self::Contained(tree) => tree.terminate_bounded(TERMINATION_GRACE), - } - } - - fn wait( - &mut self, - timeout: Option, - capture_fault: Option, - stdout: &mut OutputReader, - stderr: &mut OutputReader, - ) -> TreeOutcome { - match (self, timeout) { - (Self::Ordinary(child), None) => { - let Some(mut child) = child.take() else { - return TreeOutcome::new(InvocationResult::Infrastructure( - "ordinary child was already reaped or detached".to_owned(), - )); - }; - let outcome = wait_for_captured_process( - &mut child, - None, - stdout, - stderr, - "wait for child process", - |child| { - if capture_fault == Some(CaptureFault::WaitFailure) { - Err(io::Error::other("injected child wait failure")) - } else { - child.try_wait() - } - }, - |child| terminate_ordinary_child(child, TERMINATION_GRACE), - ); - finish_ordinary_wait(child, outcome) - } - (Self::Contained(tree), Some(timeout)) => wait_for_captured_process( - tree, - Some(timeout), - stdout, - stderr, - "observe child process tree", - ProcessTree::observe, - |tree| tree.terminate_bounded(TERMINATION_GRACE), - ), - (Self::Ordinary(_), Some(_)) | (Self::Contained(_), None) => TreeOutcome::new(InvocationResult::Infrastructure( - "internal capture mode did not match timeout configuration".to_owned(), - )), - } - } -} - -fn finish_ordinary_termination(mut child: Child, result: io::Result) -> io::Result { - match child.try_wait() { - Ok(Some(_status)) => result, - Ok(None) | Err(_) => match reap_later(child) { - Ok(()) => result, - Err(failure) => { - let (reaper, child) = failure.into_parts(); - retain_for_reaper_retry(child); - match result { - Ok(_status) => Err(io::Error::other(format!( - "the detached child reaper could not be started: {reaper}" - ))), - Err(error) => Err(io::Error::new( - error.kind(), - format!("{error}; the detached child reaper could not be started: {reaper}"), - )), - } - } - }, - } -} - -fn finish_ordinary_wait(mut child: Child, mut outcome: TreeOutcome) -> TreeOutcome { - if !matches!(child.try_wait(), Ok(Some(_status))) - && let Err(failure) = reap_later(child) - { - let (error, child) = failure.into_parts(); - retain_for_reaper_retry(child); - outcome.result = add_infrastructure_failure(outcome.result, format!("the detached child reaper could not be started: {error}")); - } - outcome +fn spawn_group(mut command: Command) -> Result { + command.group_spawn().map_err(|error| error.to_string()) } -fn wait_for_captured_process( - control: &mut T, +fn wait_for_process( + mut control: T, timeout: Option, - stdout: &mut OutputReader, - stderr: &mut OutputReader, - operation: &str, mut observe: impl FnMut(&mut T) -> io::Result>, - mut terminate: impl FnMut(&mut T) -> io::Result, + mut reader_failure: impl FnMut() -> Option, + terminate: impl FnOnce(T) -> io::Result, + operation: &str, ) -> TreeOutcome { let started = Instant::now(); + let mut terminate = Some(terminate); loop { - let reader_failure = [stdout.take_failure("stdout"), stderr.take_failure("stderr")] - .into_iter() - .flatten() - .collect::>() - .join("; "); - if !reader_failure.is_empty() { - let cleanup = terminate(control); + if let Some(reader_failure) = reader_failure() { + let cleanup = terminate.take().expect("termination is consumed only on a returning branch")(control); return TreeOutcome::new(InvocationResult::Infrastructure(with_cleanup_failure(reader_failure, &cleanup))); } - match observe(control) { + match observe(&mut control) { Ok(Some(status)) => return TreeOutcome::new(InvocationResult::Exited(status)), Ok(None) => {} Err(error) => { - let cleanup = terminate(control); + let cleanup = terminate.take().expect("termination is consumed only on a returning branch")(control); return TreeOutcome::new(InvocationResult::Infrastructure(with_cleanup_failure( format!("failed to {operation}: {error}"), &cleanup, @@ -685,10 +539,10 @@ fn wait_for_captured_process( if let Some(timeout) = timeout && timeout.checked_sub(started.elapsed()).is_none() { - return match terminate(control) { + return match terminate.take().expect("termination is consumed only on a returning branch")(control) { Ok(_) => TreeOutcome::new(InvocationResult::TimedOut(timeout)), Err(error) => TreeOutcome::new(InvocationResult::Infrastructure(format!( - "invocation timed out after {}; process-tree termination failed: {error}", + "invocation timed out after {}; process-group termination failed: {error}", display_duration(timeout) ))), }; @@ -701,58 +555,73 @@ fn wait_for_captured_process( } } -#[cfg(test)] -#[cfg_attr(coverage_nightly, coverage(off))] -fn finish_wait_with_cleanup( - control: &mut T, - waited: io::Result, - cleanup: impl FnOnce(&mut T) -> io::Result, -) -> TreeOutcome { - match waited { - Ok(status) => TreeOutcome::new(InvocationResult::Exited(status)), +fn terminate_group_bounded(child: GroupChild) -> io::Result { + terminate_group_with( + child, + TERMINATION_GRACE, + GroupChild::kill, + GroupChild::try_wait, + detach_group_reaper, + ) +} + +fn terminate_group_with( + mut child: T, + grace: Duration, + kill: impl FnOnce(&mut T) -> io::Result<()>, + mut try_wait: impl FnMut(&mut T) -> io::Result>, + detach: impl FnOnce(T) -> io::Result<()>, +) -> io::Result { + let kill_error = kill(&mut child).err(); + let observed = poll_process_exit(&mut child, grace, &mut try_wait); + match observed { + Ok(Some(status)) => match kill_error { + Some(error) => Err(error), + None => Ok(status), + }, + Ok(None) => { + let reaper = detach(child); + let message = kill_error.map_or_else( + || format!("process group did not exit within {} ms after termination", grace.as_millis()), + |error| { + format!( + "{error}; process group did not exit within {} ms after termination", + grace.as_millis() + ) + }, + ); + Err(io::Error::new(io::ErrorKind::WouldBlock, with_reaper_handoff(&message, &reaper))) + } Err(error) => { - let cleanup = cleanup(control); - TreeOutcome::new(InvocationResult::Infrastructure(with_cleanup_failure( - format!("failed to wait for child process: {error}"), - &cleanup, - ))) + let reaper = detach(child); + Err(io::Error::new( + error.kind(), + with_reaper_handoff(&format!("failed to observe process group after termination: {error}"), &reaper), + )) } } } -fn terminate_ordinary_child(child: &mut Child, grace: Duration) -> io::Result { - terminate_ordinary_with(child, grace, Child::kill, Child::try_wait) +fn detach_group_reaper(mut child: GroupChild) -> io::Result<()> { + thread::Builder::new() + .name("cargo-each-process-reaper".to_owned()) + .spawn(move || { + let _ignored = child.wait(); + }) + .map(drop) } -fn terminate_ordinary_with( - control: &mut T, - grace: Duration, - kill: impl FnOnce(&mut T) -> io::Result<()>, - mut try_wait: impl FnMut(&mut T) -> io::Result>, -) -> io::Result { - let kill_error = kill(control).err(); - let Some(status) = poll_child_exit(control, grace, &mut try_wait)? else { - let message = kill_error.map_or_else( - || format!("ordinary child did not exit within {} ms after termination", grace.as_millis()), - |error| { - format!( - "{error}; ordinary child did not exit within {} ms after termination", - grace.as_millis() - ) - }, - ); - return Err(io::Error::new(io::ErrorKind::WouldBlock, message)); - }; - match kill_error { - Some(error) => Err(error), - None => Ok(status), +fn with_reaper_handoff(message: &str, reaper: &io::Result<()>) -> String { + match reaper { + Ok(()) => format!("{message}; the process group was moved to a detached reaper thread"), + Err(error) => format!("{message}; failed to start the detached process-group reaper: {error}"), } } -fn poll_child_exit( +fn poll_process_exit( control: &mut T, grace: Duration, - try_wait: &mut impl FnMut(&mut T) -> io::Result>, + mut try_wait: impl FnMut(&mut T) -> io::Result>, ) -> io::Result> { let started = Instant::now(); loop { @@ -766,104 +635,9 @@ fn poll_child_exit( } } -fn wait_for_tree(tree: &mut ProcessTree, timeout: Duration) -> TreeOutcome { - wait_for_tree_with(tree, timeout, ProcessTree::observe, |tree| { - tree.terminate_bounded(TERMINATION_GRACE) - }) -} - -fn wait_for_tree_with( - control: &mut T, - timeout: Duration, - mut observe: impl FnMut(&mut T) -> io::Result>, - mut terminate: impl FnMut(&mut T) -> io::Result, -) -> TreeOutcome { - let started = Instant::now(); - loop { - match observe(control) { - Ok(Some(status)) => return TreeOutcome::new(InvocationResult::Exited(status)), - Ok(None) => {} - Err(error) => { - let cleanup = terminate(control); - return match cleanup { - Ok(_) => TreeOutcome::new(InvocationResult::Infrastructure(format!( - "failed to observe child process tree: {error}" - ))), - Err(cleanup) => TreeOutcome::new(InvocationResult::Infrastructure(format!( - "failed to observe child process tree: {error}; cleanup also failed: {cleanup}" - ))), - }; - } - } - let Some(remaining) = timeout.checked_sub(started.elapsed()) else { - return match terminate(control) { - Ok(_) => TreeOutcome::new(InvocationResult::TimedOut(timeout)), - Err(error) => TreeOutcome::new(InvocationResult::Infrastructure(format!( - "invocation timed out after {}; process-tree termination failed: {error}", - display_duration(timeout) - ))), - }; - }; - thread::sleep(remaining.min(Duration::from_millis(10))); - } -} - -#[cfg(test)] -fn wait_for_tree_without_timeout_with( - control: &mut T, - mut observe: impl FnMut(&mut T) -> io::Result>, - mut terminate: impl FnMut(&mut T) -> io::Result, -) -> TreeOutcome { - loop { - match observe(control) { - Ok(Some(status)) => return TreeOutcome::new(InvocationResult::Exited(status)), - Ok(None) => thread::sleep(Duration::from_millis(10)), - Err(error) => { - let cleanup = terminate(control); - return match cleanup { - Ok(_) => TreeOutcome::new(InvocationResult::Infrastructure(format!( - "failed to observe child process tree: {error}" - ))), - Err(cleanup) => TreeOutcome::new(InvocationResult::Infrastructure(format!( - "failed to observe child process tree: {error}; cleanup also failed: {cleanup}" - ))), - }; - } - } - } -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -enum CaptureFault { - MissingStdout, - StdoutReader, - MissingStderr, - StderrReader, - WaitFailure, -} - -fn capture_fault(invocation: &Invocation) -> Option { - #[cfg(test)] - { - match invocation.label.as_deref() { - Some("__cargo_each_missing_stdout") => Some(CaptureFault::MissingStdout), - Some("__cargo_each_stdout_reader_failure") => Some(CaptureFault::StdoutReader), - Some("__cargo_each_missing_stderr") => Some(CaptureFault::MissingStderr), - Some("__cargo_each_stderr_reader_failure") => Some(CaptureFault::StderrReader), - Some("__cargo_each_wait_failure") => Some(CaptureFault::WaitFailure), - _ => None, - } - } - #[cfg(not(test))] - { - let _ = invocation; - None - } -} - fn spawn_child_stdout_reader(stream: ChildStdout, name: &'static str) -> io::Result { spawn_output_reader_inner( - InterruptiblePipe::stdout(stream)?, + stream, name, OUTPUT_MEMORY_LIMIT, Box::new(|| tempfile::tempfile().map(|file| Box::new(file) as Box)), @@ -872,7 +646,7 @@ fn spawn_child_stdout_reader(stream: ChildStdout, name: &'static str) -> io::Res fn spawn_child_stderr_reader(stream: ChildStderr, name: &'static str) -> io::Result { spawn_output_reader_inner( - InterruptiblePipe::stderr(stream)?, + stream, name, OUTPUT_MEMORY_LIMIT, Box::new(|| tempfile::tempfile().map(|file| Box::new(file) as Box)), @@ -1083,7 +857,7 @@ fn finish_output_readers(stdout: OutputReader, stderr: OutputReader, grace: Dura fn with_cleanup_failure(message: String, cleanup: &io::Result) -> String { match cleanup { Ok(_) => message, - Err(error) => format!("{message}; process-tree cleanup also failed: {error}"), + Err(error) => format!("{message}; process-group cleanup also failed: {error}"), } } @@ -1301,10 +1075,10 @@ impl BufferedOutcome { } } - fn from_reader_failure(message: String, stdout_reader: OutputReader, cleanup: &io::Result, boundary: &str) -> Self { + fn from_reader_failure(message: String, stdout_reader: OutputReader, cleanup: &io::Result) -> Self { let message = with_cleanup_failure(message, cleanup); combine_captured_output( - finish_output_reader(stdout_reader, "stdout", OUTPUT_DRAIN_GRACE, boundary), + finish_output_reader(stdout_reader, "stdout", OUTPUT_DRAIN_GRACE, "process group"), CapturedStream { output: CapturedOutput::empty(), failure: None, @@ -1362,196 +1136,167 @@ mod tests { #[cfg(windows)] use std::os::windows::process::ExitStatusExt as _; use std::process::{Command, ExitCode, ExitStatus, Stdio}; - use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Condvar, Mutex, mpsc}; use std::time::{Duration, Instant}; use std::{io, thread}; - use cargo_gamma_process::ensure_reaper; - use cargo_gamma_process::faults::{self, Fault}; - use super::{ - BufferedOutcome, CapturedOutput, CapturedProcess, CapturedStream, Invocation, InvocationResult, OutputEmitError, OutputReader, - Plan, ReaderCompletion, RunningWorker, SpillFile, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, - combine_captured_output, display_duration, effective_worker_count, emit_buffered, emit_buffered_to, execute_parallel, exit_byte, - failure_stops_launching, finish_ordinary_termination, finish_ordinary_wait, finish_output_reader, finish_wait_with_cleanup, - panic_description, parallel_failure_exit_code, run_captured, run_captured_with_spawner, run_streamed, run_streamed_with_timeout, - run_streamed_with_timeout_with, spawn_if_sealed, spawn_output_reader, spawn_output_reader_with, spawn_tree, spawn_worker, - take_reader_output, terminate_ordinary_child, terminate_ordinary_with, wait_for_captured_process, wait_for_tree, - wait_for_tree_with, wait_for_tree_without_timeout_with, wait_for_worker, with_cleanup_failure, + BufferedOutcome, CapturedOutput, CapturedStream, Invocation, InvocationResult, OutputEmitError, OutputReader, Plan, + ReaderCompletion, RunningWorker, SpillFile, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, + add_infrastructure_failure, combine_captured_output, detach_group_reaper, display_duration, effective_worker_count, + emit_buffered_to, execute_parallel, exit_byte, failure_stops_launching, finish_output_reader, panic_description, + parallel_failure_exit_code, poll_process_exit, run_captured, run_captured_with, run_streamed, run_streamed_with_timeout, + run_streamed_with_timeout_with, spawn_child_stderr_reader, spawn_child_stdout_reader, spawn_group, spawn_output_reader, + spawn_output_reader_with, spawn_worker, take_reader_output, terminate_group_bounded, terminate_group_with, wait_for_process, + wait_for_worker, with_cleanup_failure, with_reaper_handoff, }; - const ORDINARY_BOUNDARY: &str = "ordinary process tree"; - const CONTAINED_BOUNDARY: &str = "contained process tree"; + const LEADER_BOUNDARY: &str = "process leader"; + const GROUP_BOUNDARY: &str = "process group"; - struct StubbornPipe { - read: bool, - blocked: Option>, - finished: mpsc::Sender<()>, - release: Arc<(Mutex, Condvar)>, + fn invocation(argv: &[&str]) -> Invocation { + Invocation { + label: None, + argv: argv.iter().map(|value| (*value).to_owned()).collect(), + work_dir: None, + } } - struct FailingReader; + fn successful_status() -> ExitStatus { + ExitStatus::from_raw(0) + } - impl io::Read for FailingReader { - fn read(&mut self, _buf: &mut [u8]) -> io::Result { - Err(io::Error::other("injected read failure")) - } + #[cfg(unix)] + fn failed_status(code: i32) -> ExitStatus { + ExitStatus::from_raw(code << 8) } - struct PendingPipe { - dropped: Option>, + #[cfg(windows)] + fn failed_status(code: i32) -> ExitStatus { + ExitStatus::from_raw(u32::try_from(code).expect("test exit code is nonnegative")) } - struct DropSignalReader { - bytes: Option<&'static [u8]>, - dropped: Option>, + fn result_infrastructure_message(result: InvocationResult) -> String { + let InvocationResult::Infrastructure(message) = result else { + panic!("the test expects an infrastructure outcome"); + }; + message } - impl io::Read for DropSignalReader { - fn read(&mut self, buf: &mut [u8]) -> io::Result { - let Some(bytes) = self.bytes.take() else { - return Ok(0); - }; - buf[..bytes.len()].copy_from_slice(bytes); - Ok(bytes.len()) - } + fn infrastructure_message(outcome: BufferedOutcome) -> String { + result_infrastructure_message(outcome.result) } - impl Drop for DropSignalReader { - fn drop(&mut self) { - if let Some(dropped) = self.dropped.take() { - let _receiver_gone = dropped.send(()); + fn output_bytes(output: &mut CapturedOutput) -> Vec { + let mut bytes = Vec::new(); + match output.emit_to(&mut bytes) { + Ok(()) => bytes, + Err(OutputEmitError::Source(error) | OutputEmitError::Destination(error)) => { + panic!("captured test output cannot be read: {error}") } } } - impl io::Read for PendingPipe { - fn read(&mut self, _buf: &mut [u8]) -> io::Result { - Err(io::Error::from(io::ErrorKind::WouldBlock)) + fn captured(bytes: &[u8], failure: Option<&str>) -> CapturedStream { + CapturedStream { + output: CapturedOutput::Memory(bytes.to_vec()), + failure: failure.map(str::to_owned), } } - impl Drop for PendingPipe { - fn drop(&mut self) { - if let Some(dropped) = self.dropped.take() { - let _receiver_gone = dropped.send(()); - } - } + struct FakeProcess { + observations: VecDeque>>, + termination: Option>, } - struct PanickingReader; - - impl io::Read for PanickingReader { - fn read(&mut self, _buf: &mut [u8]) -> io::Result { - panic!("injected reader panic"); + impl FakeProcess { + fn observe(&mut self) -> io::Result> { + self.observations.pop_front().unwrap_or(Ok(None)) } - } - struct EofThenPanicReader { - reached_eof: bool, + fn terminate(mut self) -> io::Result { + self.termination + .take() + .unwrap_or_else(|| Err(io::Error::other("unexpected termination"))) + } } - struct InterruptedThenDataReader { - state: u8, + struct PendingReader { + dropped: Option>, } - struct ChunkedReader { - chunks: VecDeque>, + impl io::Read for PendingReader { + fn read(&mut self, _buf: &mut [u8]) -> io::Result { + Err(io::Error::from(io::ErrorKind::WouldBlock)) + } } - impl io::Read for ChunkedReader { - fn read(&mut self, buf: &mut [u8]) -> io::Result { - let Some(chunk) = self.chunks.pop_front() else { - return Ok(0); - }; - buf[..chunk.len()].copy_from_slice(&chunk); - Ok(chunk.len()) + impl Drop for PendingReader { + fn drop(&mut self) { + if let Some(dropped) = self.dropped.take() { + let _receiver_gone = dropped.send(()); + } } } - #[derive(Debug)] - struct FaultySpill { - cursor: io::Cursor>, - fail_write: bool, - fail_read: bool, + struct BlockingReader { + first_read: bool, + blocked: Option>, + finished: mpsc::Sender<()>, + release: Arc<(Mutex, Condvar)>, } - struct FailingWriter; - - impl io::Write for FailingWriter { - fn write(&mut self, _buf: &[u8]) -> io::Result { - Err(io::Error::other("injected destination write failure")) - } - - fn flush(&mut self) -> io::Result<()> { - Ok(()) - } - } - - impl io::Write for FaultySpill { - fn write(&mut self, buf: &[u8]) -> io::Result { - if self.fail_write { - Err(io::Error::other("injected spill write failure")) - } else { - io::Write::write(&mut self.cursor, buf) - } - } - - fn flush(&mut self) -> io::Result<()> { - io::Write::flush(&mut self.cursor) - } - } - - impl io::Read for FaultySpill { + impl io::Read for BlockingReader { fn read(&mut self, buf: &mut [u8]) -> io::Result { - if self.fail_read { - Err(io::Error::other("injected spill read failure")) - } else { - io::Read::read(&mut self.cursor, buf) + if !self.first_read { + self.first_read = true; + let bytes = b"captured-before-detach"; + buf[..bytes.len()].copy_from_slice(bytes); + return Ok(bytes.len()); + } + if let Some(blocked) = self.blocked.take() { + let _receiver_gone = blocked.send(()); + } + let (lock, condition) = &*self.release; + let mut released = lock.lock().expect("the test owns the release mutex without panicking"); + while !*released { + released = condition.wait(released).expect("the test owns the release mutex without panicking"); } + let _receiver_gone = self.finished.send(()); + Ok(0) } } - impl io::Seek for FaultySpill { - fn seek(&mut self, pos: io::SeekFrom) -> io::Result { - io::Seek::seek(&mut self.cursor, pos) - } + struct DropSignalReader { + bytes: Option<&'static [u8]>, + dropped: Option>, } - impl io::Read for InterruptedThenDataReader { + impl io::Read for DropSignalReader { fn read(&mut self, buf: &mut [u8]) -> io::Result { - match self.state { - 0 => { - self.state = 1; - Err(io::Error::from(io::ErrorKind::Interrupted)) - } - 1 => { - self.state = 2; - let content = b"after interrupt"; - buf[..content.len()].copy_from_slice(content); - Ok(content.len()) - } - _ => Ok(0), - } + let Some(bytes) = self.bytes.take() else { + return Ok(0); + }; + buf[..bytes.len()].copy_from_slice(bytes); + Ok(bytes.len()) } } - impl io::Read for EofThenPanicReader { - fn read(&mut self, _buf: &mut [u8]) -> io::Result { - assert!(!self.reached_eof, "the output reader must stop after the first EOF"); - self.reached_eof = true; - Ok(0) + impl Drop for DropSignalReader { + fn drop(&mut self) { + if let Some(dropped) = self.dropped.take() { + let _receiver_gone = dropped.send(()); + } } } - struct LateDataPipe { + struct LateDataReader { started: Option>, finished: Option>, release: Arc<(Mutex, Condvar)>, } - impl io::Read for LateDataPipe { + impl io::Read for LateDataReader { fn read(&mut self, buf: &mut [u8]) -> io::Result { if let Some(started) = self.started.take() { let _receiver_gone = started.send(()); @@ -1561,13 +1306,13 @@ mod tests { while !*released { released = condition.wait(released).expect("the test owns the release mutex without panicking"); } - let content = b"late data"; - buf[..content.len()].copy_from_slice(content); - Ok(content.len()) + let bytes = b"late data"; + buf[..bytes.len()].copy_from_slice(bytes); + Ok(bytes.len()) } } - impl Drop for LateDataPipe { + impl Drop for LateDataReader { fn drop(&mut self) { if let Some(finished) = self.finished.take() { let _receiver_gone = finished.send(()); @@ -1575,498 +1320,600 @@ mod tests { } } - fn invocation(argv: &[&str]) -> Invocation { - Invocation { - label: None, - argv: argv.iter().map(|value| (*value).to_owned()).collect(), - work_dir: None, - } + struct InterruptedThenData { + interrupted: bool, + emitted: bool, } - fn labelled_invocation(label: &str, argv: &[&str]) -> Invocation { - Invocation { - label: Some(label.to_owned()), - ..invocation(argv) + impl io::Read for InterruptedThenData { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + if !self.interrupted { + self.interrupted = true; + return Err(io::Error::from(io::ErrorKind::Interrupted)); + } + if self.emitted { + return Ok(0); + } + self.emitted = true; + let bytes = b"after interrupt"; + buf[..bytes.len()].copy_from_slice(bytes); + Ok(bytes.len()) } } - fn sleeping_test_command() -> Command { - sleeping_test_command_for(Duration::from_secs(30)) - } - - fn sleeping_test_command_for(duration: Duration) -> Command { - let mut command = Command::new(std::env::current_exe().expect("the test binary knows its path")); - let _ = command - .args(["--exact", "run::tests::ordinary_child_sleep_probe", "--nocapture"]) - .env("CARGO_EACH_ORDINARY_CHILD_PROBE", duration.as_millis().to_string()) - .stdout(Stdio::null()) - .stderr(Stdio::null()); - command - } + struct FailingReader; - fn wait_for_reaper_owner_to_release(id: u32) { - let deadline = Instant::now() + Duration::from_secs(2); - while faults::reaper_owns(id) && Instant::now() < deadline { - thread::sleep(Duration::from_millis(10)); + impl io::Read for FailingReader { + fn read(&mut self, _buf: &mut [u8]) -> io::Result { + Err(io::Error::other("injected read failure")) } - assert!(!faults::reaper_owns(id), "the detached reaper did not release child {id}"); } - fn spawn_ordinary_capture(mut command: Command) -> Result { - command - .spawn() - .map(|child| CapturedProcess::Ordinary(Some(child))) - .map_err(|error| error.to_string()) + struct PanickingReader; + + impl io::Read for PanickingReader { + fn read(&mut self, _buf: &mut [u8]) -> io::Result { + panic!("injected reader panic"); + } } - fn result_infrastructure_message(result: InvocationResult) -> String { - let InvocationResult::Infrastructure(message) = result else { - panic!("the test expects an infrastructure outcome"); - }; - message + struct ChunkedReader { + chunks: VecDeque>, } - struct FakeProcess { - observations: VecDeque>>, - termination: Option>, + impl io::Read for ChunkedReader { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + let Some(chunk) = self.chunks.pop_front() else { + return Ok(0); + }; + buf[..chunk.len()].copy_from_slice(&chunk); + Ok(chunk.len()) + } } - struct FakeOrdinaryChild { - kill_error: Option, - observations: VecDeque>>, + #[derive(Debug)] + struct FaultySpill { + cursor: io::Cursor>, + fail_write: bool, + fail_read: bool, } - impl FakeOrdinaryChild { - fn kill_child(&mut self) -> io::Result<()> { - match self.kill_error.take() { - Some(error) => Err(error), - None => Ok(()), + impl io::Write for FaultySpill { + fn write(&mut self, buf: &[u8]) -> io::Result { + if self.fail_write { + Err(io::Error::other("injected spill write failure")) + } else { + io::Write::write(&mut self.cursor, buf) } } - fn try_wait_child(&mut self) -> io::Result> { - self.observations.pop_front().unwrap_or(Ok(None)) + fn flush(&mut self) -> io::Result<()> { + io::Write::flush(&mut self.cursor) } } - impl FakeProcess { - fn observe(&mut self) -> io::Result> { - self.observations.pop_front().unwrap_or(Ok(None)) - } - - fn terminate(&mut self) -> io::Result { - self.termination - .take() - .unwrap_or_else(|| Err(io::Error::other("unexpected termination"))) + impl io::Read for FaultySpill { + fn read(&mut self, buf: &mut [u8]) -> io::Result { + if self.fail_read { + Err(io::Error::other("injected spill read failure")) + } else { + io::Read::read(&mut self.cursor, buf) + } } } - fn successful_status() -> ExitStatus { - ExitStatus::from_raw(0) - } - - #[cfg(unix)] - fn failed_status(code: i32) -> ExitStatus { - ExitStatus::from_raw(code << 8) - } - - #[cfg(windows)] - fn failed_status(code: i32) -> ExitStatus { - ExitStatus::from_raw(u32::try_from(code).expect("test exit code is nonnegative")) + impl io::Seek for FaultySpill { + fn seek(&mut self, pos: io::SeekFrom) -> io::Result { + io::Seek::seek(&mut self.cursor, pos) + } } - fn infrastructure_message(outcome: BufferedOutcome) -> String { - result_infrastructure_message(outcome.result) - } + struct FailingWriter; - fn captured(bytes: &[u8], failure: Option<&str>) -> CapturedStream { - CapturedStream { - output: CapturedOutput::Memory(bytes.to_vec()), - failure: failure.map(str::to_owned), + impl io::Write for FailingWriter { + fn write(&mut self, _buffer: &[u8]) -> io::Result { + Err(io::Error::other("injected destination write failure")) } - } - fn output_bytes(output: &mut CapturedOutput) -> Vec { - let mut bytes = Vec::new(); - match output.emit_to(&mut bytes) { - Ok(()) => bytes, - Err(OutputEmitError::Source(error) | OutputEmitError::Destination(error)) => { - panic!("captured test output cannot be read: {error}") - } + fn flush(&mut self) -> io::Result<()> { + Ok(()) } } - fn poisoned_buffer() -> Arc> { - let bytes = Arc::new(Mutex::new(CapturedOutput::Memory(b"poisoned bytes".to_vec()))); - let poisoned = Arc::clone(&bytes); - let _panic = thread::spawn(move || { - let _guard = poisoned.lock().expect("the fresh capture mutex is available"); - panic!("poison capture buffer"); - }) - .join(); - bytes + fn sleeping_test_command() -> Command { + let mut command = Command::new(std::env::current_exe().expect("the test binary knows its path")); + let _ = command + .args(["--exact", "run::tests::child_sleep_probe", "--nocapture"]) + .env("CARGO_EACH_CHILD_SLEEP_MS", "30000") + .stdout(Stdio::null()) + .stderr(Stdio::null()); + command } - impl io::Read for StubbornPipe { - fn read(&mut self, buf: &mut [u8]) -> io::Result { - if !self.read { - self.read = true; - let content = b"captured-before-timeout"; - buf[..content.len()].copy_from_slice(content); - return Ok(content.len()); - } - - if let Some(blocked) = self.blocked.take() { - let _receiver_gone = blocked.send(()); - } - let (lock, condition) = &*self.release; - let mut released = lock.lock().expect("the test owns the release mutex without panicking"); - while !*released { - released = condition.wait(released).expect("the test owns the release mutex without panicking"); - } - let _receiver_gone = self.finished.send(()); - Ok(0) + #[test] + fn child_sleep_probe() { + if let Some(duration) = std::env::var_os("CARGO_EACH_CHILD_SLEEP_MS") { + let millis = duration.to_string_lossy().parse().expect("the parent passes milliseconds"); + thread::sleep(Duration::from_millis(millis)); } } #[test] - fn signal_terminated_child_maps_to_one() { + fn exit_codes_and_durations_follow_the_cli_contract() { assert_eq!(exit_byte(None), 1); - } - - #[test] - fn in_range_codes_pass_through() { - assert_eq!(exit_byte(Some(1)), 1); - assert_eq!(exit_byte(Some(2)), 2); - assert_eq!(exit_byte(Some(255)), 255); - } - - #[test] - fn wide_codes_reduce_to_low_byte() { + assert_eq!(exit_byte(Some(7)), 7); assert_eq!(exit_byte(Some(259)), 3); - assert_eq!(exit_byte(Some(257)), 1); - } - - #[test] - fn nonzero_code_with_zero_low_byte_maps_to_one() { assert_eq!(exit_byte(Some(256)), 1); - assert_eq!(exit_byte(Some(512)), 1); - } - - #[test] - fn durations_have_compact_diagnostics() { assert_eq!(display_duration(Duration::from_millis(250)), "250ms"); assert_eq!(display_duration(Duration::from_secs(30)), "30s"); assert_eq!(display_duration(Duration::from_mins(2)), "2m"); } #[test] - fn failure_launch_policy_distinguishes_fail_fast_from_keep_going() { + fn worker_count_and_launch_policy_are_plan_bounded() { + let four = NonZeroUsize::new(4).expect("literal four is nonzero"); + assert_eq!(effective_worker_count(four, 1), NonZeroUsize::MIN); + assert_eq!(effective_worker_count(four, 8), four); assert!(failure_stops_launching(false, true)); assert!(!failure_stops_launching(false, false)); assert!(!failure_stops_launching(true, true)); - assert!(!failure_stops_launching(true, false)); - } - - #[test] - fn effective_worker_count_caps_requested_parallelism() { - let four = NonZeroUsize::new(4).expect("literal four is nonzero"); - assert_eq!(effective_worker_count(four, 1, 8), NonZeroUsize::MIN); - assert_eq!( - effective_worker_count(four, 8, 2), - NonZeroUsize::new(2).expect("literal two is nonzero") - ); - assert_eq!(effective_worker_count(four, 8, 0), NonZeroUsize::MIN); } #[test] - fn sequential_timeout_policy_distinguishes_fail_fast_from_keep_going() { + fn sequential_and_parallel_failures_preserve_their_taxonomy() { let plan = Plan { invocations: vec![invocation(&["first"]), invocation(&["second"])], }; - let mut fail_fast_results = VecDeque::from([ + let mut results = VecDeque::from([ InvocationResult::TimedOut(Duration::from_millis(50)), InvocationResult::Exited(successful_status()), ]); - let mut fail_fast_calls = 0; + let mut calls = 0; let fail_fast = super::execute_sequential_with(&plan, false, Some(Duration::from_millis(50)), |_, timeout| { assert_eq!(timeout, Some(Duration::from_millis(50))); - fail_fast_calls += 1; - fail_fast_results.pop_front().expect("one result per launched invocation") + calls += 1; + results.pop_front().expect("one result per launched invocation") }); assert_eq!(fail_fast, ExitCode::from(1)); - assert_eq!(fail_fast_calls, 1, "fail-fast must stop after the timeout"); - - let mut keep_going_results = VecDeque::from([ - InvocationResult::TimedOut(Duration::from_millis(50)), - InvocationResult::Exited(successful_status()), - ]); - let mut keep_going_calls = 0; - let keep_going = super::execute_sequential_with(&plan, true, None, |_, timeout| { - assert_eq!(timeout, None); - keep_going_calls += 1; - keep_going_results.pop_front().expect("one result per launched invocation") - }); - assert_eq!(keep_going, ExitCode::from(1)); - assert_eq!(keep_going_calls, 2, "keep-going must launch after a timeout"); - - let infrastructure = super::execute_sequential_with( - &Plan { - invocations: vec![invocation(&["only"])], - }, - false, - None, - |_, _| InvocationResult::Infrastructure("injected infrastructure failure".to_owned()), - ); - assert_eq!(infrastructure, ExitCode::from(2)); - } - - #[test] - fn parallel_failure_exit_codes_preserve_failure_class() { + assert_eq!(calls, 1); assert_eq!( parallel_failure_exit_code(&InvocationResult::Exited(failed_status(7))), ExitCode::from(7) ); - assert_eq!( - parallel_failure_exit_code(&InvocationResult::TimedOut(Duration::from_secs(1))), - ExitCode::from(1) - ); assert_eq!( parallel_failure_exit_code(&InvocationResult::Infrastructure("capture failed".to_owned())), ExitCode::from(2) ); + assert_eq!( + parallel_failure_exit_code(&InvocationResult::TimedOut(Duration::from_secs(1))), + ExitCode::from(1) + ); } #[test] - fn cancellation_joins_a_reader_waiting_for_pipe_readiness() { - let (dropped_tx, dropped_rx) = mpsc::channel(); - let reader = spawn_output_reader(PendingPipe { dropped: Some(dropped_tx) }, "pending-test-pipe") - .expect("the readiness-polling reader can be created"); - - let captured = finish_output_reader(reader, "stdout", Duration::from_millis(25), ORDINARY_BOUNDARY); + fn scheduler_surfaces_worker_launch_failures_in_both_policies() { + let fail_fast = Plan { + invocations: vec![invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM]), invocation(&["rustc", "--version"])], + }; + let code = execute_parallel(&fail_fast, false, NonZeroUsize::new(2).expect("literal two is nonzero"), None) + .expect("worker launch failure is an invocation outcome"); + assert_eq!(code, ExitCode::from(2)); - dropped_rx - .recv_timeout(Duration::from_secs(1)) - .expect("cancellation drops the pipe before the bounded finish returns"); - let failure = captured - .failure - .expect("an open pipe past the drain grace is an infrastructure failure"); - assert!(failure.contains("remained open"), "{failure}"); - assert!(!failure.contains("did not stop"), "{failure}"); + let keep_going = Plan { + invocations: vec![invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM]), invocation(&["rustc", "--version"])], + }; + let code = execute_parallel(&keep_going, true, NonZeroUsize::new(2).expect("literal two is nonzero"), None) + .expect("keep-going retains worker launch failures"); + assert_eq!(code, ExitCode::from(1)); } #[test] - fn reader_failure_terminates_an_untimed_running_process() { - let mut process = FakeProcess { - observations: VecDeque::from([Ok(Some(successful_status()))]), - termination: Some(Ok(successful_status())), + fn process_waiting_handles_completion_timeout_and_cleanup_failures() { + let completed = FakeProcess { + observations: VecDeque::from([Ok(None), Ok(Some(successful_status()))]), + termination: None, + }; + let outcome = wait_for_process( + completed, + None, + FakeProcess::observe, + || None, + FakeProcess::terminate, + "observe fake process", + ); + assert!(matches!(outcome.result, InvocationResult::Exited(status) if status.success())); + + let timed_out = FakeProcess { + observations: VecDeque::from([Ok(None)]), + termination: Some(Ok(successful_status())), }; - let mut stdout = spawn_output_reader(FailingReader, "early-failing-reader").expect("create failing stdout reader"); - let mut stderr = spawn_output_reader(io::empty(), "empty-stderr-reader").expect("create empty stderr reader"); - stdout.reported = Some( - stdout - .completion - .recv_timeout(Duration::from_secs(1)) - .expect("the injected read failure is ready before process observation"), + let outcome = wait_for_process( + timed_out, + Some(Duration::ZERO), + FakeProcess::observe, + || None, + FakeProcess::terminate, + "observe fake process", ); + assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())); - let outcome = wait_for_captured_process( - &mut process, + let failed = FakeProcess { + observations: VecDeque::from([Err(io::Error::other("observation failed"))]), + termination: Some(Err(io::Error::other("cleanup failed"))), + }; + let outcome = wait_for_process( + failed, None, - &mut stdout, - &mut stderr, + FakeProcess::observe, + || None, + FakeProcess::terminate, "observe fake process", + ); + let message = result_infrastructure_message(outcome.result); + assert!(message.contains("observation failed")); + assert!(message.contains("cleanup failed")); + + let failed_timeout_cleanup = FakeProcess { + observations: VecDeque::from([Ok(None)]), + termination: Some(Err(io::Error::other("timeout cleanup failed"))), + }; + let outcome = wait_for_process( + failed_timeout_cleanup, + Some(Duration::ZERO), FakeProcess::observe, + || None, FakeProcess::terminate, + "observe fake process", ); + assert!(result_infrastructure_message(outcome.result).contains("timeout cleanup failed")); + } - let message = result_infrastructure_message(outcome.result); - assert!(message.contains("injected read failure"), "{message}"); - assert!(process.termination.is_none(), "reader failure must trigger process cleanup"); - let _stdout = finish_output_reader(stdout, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); - let _stderr = finish_output_reader(stderr, "stderr", Duration::from_secs(1), ORDINARY_BOUNDARY); + #[test] + fn reader_failure_terminates_the_process_through_the_local_wait_seam() { + let process = FakeProcess { + observations: VecDeque::from([Ok(None)]), + termination: Some(Ok(successful_status())), + }; + let outcome = wait_for_process( + process, + None, + FakeProcess::observe, + || Some("failed to read child stdout".to_owned()), + FakeProcess::terminate, + "observe fake process", + ); + assert!(result_infrastructure_message(outcome.result).contains("failed to read child stdout")); } #[test] - #[cfg_attr(miri, ignore = "spawns ordinary child processes")] - fn ordinary_child_handoffs_retain_a_bounded_reap_owner() { - ensure_reaper().expect("preflight the production reaper"); - let handed_off = sleeping_test_command_for(Duration::from_millis(500)) - .spawn() - .expect("spawn successful handoff probe"); - let handed_off_id = handed_off.id(); - let error = finish_ordinary_termination( - handed_off, - Err(io::Error::new(io::ErrorKind::PermissionDenied, "primary cleanup failed")), - ) - .expect_err("a successful handoff preserves the primary cleanup error"); - assert_eq!(error.kind(), io::ErrorKind::PermissionDenied); - assert!(faults::reaper_owns(handed_off_id), "the successful handoff lost its wait owner"); - - let successful_cleanup = sleeping_test_command_for(Duration::from_millis(500)) - .spawn() - .expect("spawn successful-cleanup handoff probe"); - let successful_cleanup_id = successful_cleanup.id(); - let _failed_start = faults::arm(Fault::ReaperStart); - let error = finish_ordinary_termination(successful_cleanup, Ok(successful_status())) - .expect_err("a failed handoff replaces a successful cleanup result"); - assert!(error.to_string().contains("reaper thread start failed"), "{error}"); + fn bounded_polling_stops_on_exit_error_or_deadline() { + let mut exited = VecDeque::from([Ok(None), Ok(Some(successful_status()))]); + let status = poll_process_exit(&mut exited, Duration::from_secs(1), |observations| { + observations.pop_front().expect("the fake has enough observations") + }) + .expect("polling succeeds") + .expect("the fake exits"); + assert!(status.success()); + + let mut failed = VecDeque::from([Err(io::Error::other("poll failed"))]); + let error = poll_process_exit(&mut failed, Duration::from_secs(1), |observations| { + observations.pop_front().expect("the fake has one observation") + }) + .expect_err("polling failure propagates"); + assert!(error.to_string().contains("poll failed")); + + let mut running = (); assert!( - faults::reaper_owns(successful_cleanup_id), - "the successful-cleanup handoff lost its wait owner" + poll_process_exit(&mut running, Duration::ZERO, |()| Ok(None)) + .expect("deadline is not an I/O failure") + .is_none() ); + } - let termination_child = sleeping_test_command_for(Duration::from_millis(350)) - .spawn() - .expect("spawn termination handoff probe"); - let termination_id = termination_child.id(); - let _failed_start = faults::arm(Fault::ReaperStart); - let started = Instant::now(); - let error = finish_ordinary_termination( - termination_child, - Err(io::Error::new(io::ErrorKind::PermissionDenied, "termination failed")), + #[test] + fn bounded_group_termination_reports_every_local_failure_shape() { + let status = successful_status(); + let kill_error = terminate_group_with( + (), + Duration::from_secs(1), + |()| Err(io::Error::new(io::ErrorKind::PermissionDenied, "kill failed")), + move |()| Ok(Some(status)), + |()| Ok(()), ) - .expect_err("the failed cleanup and handoff must be reported"); - assert!(started.elapsed() < Duration::from_millis(200), "failed handoff blocked its caller"); - assert_eq!(error.kind(), io::ErrorKind::PermissionDenied); - assert!(error.to_string().contains("termination failed"), "{error}"); - assert!(error.to_string().contains("reaper thread start failed"), "{error}"); - assert!( - faults::reaper_owns(termination_id), - "the failed termination handoff lost its wait owner" - ); + .expect_err("a kill error is not hidden by later completion"); + assert_eq!(kill_error.kind(), io::ErrorKind::PermissionDenied); + + let deadline = terminate_group_with((), Duration::ZERO, |()| Ok(()), |()| Ok(None), |()| Ok(())) + .expect_err("an unreaped group reaches the deadline"); + assert_eq!(deadline.kind(), io::ErrorKind::WouldBlock); + assert!(deadline.to_string().contains("detached reaper thread")); + + let failed_handoff = terminate_group_with( + (), + Duration::ZERO, + |()| Err(io::Error::other("kill failed")), + |()| Ok(None), + |()| Err(io::Error::other("reaper failed")), + ) + .expect_err("kill and handoff failures are both reported"); + assert!(failed_handoff.to_string().contains("kill failed")); + assert!(failed_handoff.to_string().contains("reaper failed")); + + let failed_observation = terminate_group_with( + (), + Duration::from_secs(1), + |()| Ok(()), + |()| Err(io::Error::other("observation failed")), + |()| Ok(()), + ) + .expect_err("post-kill observation failure is reported"); + assert!(failed_observation.to_string().contains("observation failed")); + assert!(failed_observation.to_string().contains("detached reaper thread")); + } + + #[test] + #[cfg_attr(miri, ignore = "spawns process groups")] + fn real_group_execution_observes_completion_and_timeout() { + let success = run_streamed_with_timeout(&invocation(&["rustc", "--version"]), Duration::from_secs(5)); + assert!(matches!(success, InvocationResult::Exited(status) if status.success())); - let wait_child = sleeping_test_command_for(Duration::from_millis(350)) - .spawn() - .expect("spawn wait handoff probe"); - let wait_id = wait_child.id(); - let _failed_start = faults::arm(Fault::ReaperStart); + let group = spawn_group(sleeping_test_command()).expect("spawn sleeping process group"); let started = Instant::now(); - let outcome = finish_ordinary_wait(wait_child, TreeOutcome::new(InvocationResult::Exited(successful_status()))); + let error = terminate_group_bounded(group).expect("killed process group is reaped"); + assert!(!error.success()); + assert!(started.elapsed() < Duration::from_secs(2)); + + let mut quick = Command::new("rustc"); + let _ = quick.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); + let group = spawn_group(quick).expect("spawn quick process group"); + detach_group_reaper(group).expect("detach the local process-group reaper"); + } + + #[test] + #[cfg_attr(miri, ignore = "spawns and captures process groups")] + fn captured_runner_uses_group_control_and_keeps_output() { + let mut untimed = run_captured(&invocation(&["rustc", "--version"]), None); + assert!(matches!(untimed.result, InvocationResult::Exited(status) if status.success())); assert!( - started.elapsed() < Duration::from_millis(200), - "failed wait handoff blocked its caller" + String::from_utf8(output_bytes(&mut untimed.stdout)) + .expect("rustc output is UTF-8") + .contains("rustc") ); - assert!( - result_infrastructure_message(outcome.result).contains("reaper thread start failed"), - "a post-wait handoff failure must become an infrastructure failure" + + let timed = run_captured(&invocation(&["rustc", "--version"]), Some(Duration::from_secs(5))); + assert!(matches!(timed.result, InvocationResult::Exited(status) if status.success())); + } + + #[test] + fn direct_runners_report_empty_and_unspawnable_commands() { + let empty = invocation(&[]); + assert!(matches!(run_streamed(&empty), InvocationResult::Infrastructure(message) if message.contains("empty argument vector"))); + assert!(infrastructure_message(run_captured(&empty, None)).contains("empty argument vector")); + assert!(matches!( + run_streamed_with_timeout(&empty, Duration::from_secs(1)), + InvocationResult::Infrastructure(message) if message.contains("empty argument vector") + )); + + let missing = invocation(&["__cargo_each_missing_program_for_unit_test__"]); + assert!(matches!(run_streamed(&missing), InvocationResult::Infrastructure(message) if message.contains("failed to spawn"))); + assert!(infrastructure_message(run_captured(&missing, None)).contains("failed to spawn")); + + let injected = run_streamed_with_timeout_with(&invocation(&["rustc", "--version"]), Duration::from_secs(1), |_| { + Err("injected group spawn failure".to_owned()) + }); + assert!(matches!( + injected, + InvocationResult::Infrastructure(message) if message.contains("injected group spawn failure") + )); + } + + #[test] + #[cfg_attr(miri, ignore = "spawns process groups with injected capture seams")] + fn captured_setup_failures_use_local_reader_and_process_seams() { + let invocation = invocation(&["rustc", "--version"]); + + let missing_stdout = run_captured_with( + &invocation, + None, + |command| { + let mut group = spawn_group(command)?; + drop(group.inner().stdout.take()); + Ok(group) + }, + spawn_child_stdout_reader, + spawn_child_stderr_reader, + ); + assert!(infrastructure_message(missing_stdout).contains("failed to capture child stdout")); + + let stdout_reader = run_captured_with( + &invocation, + None, + spawn_group, + |_stream, _name| Err(io::Error::other("injected stdout reader failure")), + spawn_child_stderr_reader, + ); + assert!(infrastructure_message(stdout_reader).contains("injected stdout reader failure")); + + let missing_stderr = run_captured_with( + &invocation, + None, + |command| { + let mut group = spawn_group(command)?; + drop(group.inner().stderr.take()); + Ok(group) + }, + spawn_child_stdout_reader, + spawn_child_stderr_reader, ); - assert!(faults::reaper_owns(wait_id), "the failed wait handoff lost its wait owner"); + assert!(infrastructure_message(missing_stderr).contains("failed to capture child stderr")); - wait_for_reaper_owner_to_release(handed_off_id); - wait_for_reaper_owner_to_release(successful_cleanup_id); - wait_for_reaper_owner_to_release(termination_id); - wait_for_reaper_owner_to_release(wait_id); + let stderr_reader = run_captured_with(&invocation, None, spawn_group, spawn_child_stdout_reader, |_stream, _name| { + Err(io::Error::other("injected stderr reader failure")) + }); + assert!(infrastructure_message(stderr_reader).contains("injected stderr reader failure")); } #[test] - fn captured_wait_classifies_timeout_cleanup_results() { - for (termination, expected) in [ - (Ok(successful_status()), None), - (Err(io::Error::other("timeout cleanup failed")), Some("timeout cleanup failed")), - ] { - let mut process = FakeProcess { - observations: VecDeque::from([Ok(None)]), - termination: Some(termination), - }; - let mut stdout = spawn_output_reader(io::empty(), "timeout-empty-stdout").expect("create empty stdout reader"); - let mut stderr = spawn_output_reader(io::empty(), "timeout-empty-stderr").expect("create empty stderr reader"); - - let outcome = wait_for_captured_process( - &mut process, - Some(Duration::ZERO), - &mut stdout, - &mut stderr, - "observe fake process", - FakeProcess::observe, - FakeProcess::terminate, - ); - match expected { - None => assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())), - Some(expected) => assert!(result_infrastructure_message(outcome.result).contains(expected)), - } - let _stdout = finish_output_reader(stdout, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); - let _stderr = finish_output_reader(stderr, "stderr", Duration::from_secs(1), ORDINARY_BOUNDARY); - } + fn output_reader_retries_interrupts_and_reports_failures() { + let interrupted = spawn_output_reader( + InterruptedThenData { + interrupted: false, + emitted: false, + }, + "interrupted-reader", + ) + .expect("spawn interrupted reader"); + let mut interrupted = finish_output_reader(interrupted, "stdout", Duration::from_secs(1), GROUP_BOUNDARY); + assert_eq!(output_bytes(&mut interrupted.output), b"after interrupt"); + assert!(interrupted.failure.is_none()); + + let failed = spawn_output_reader(FailingReader, "failing-reader").expect("spawn failing reader"); + let failed = finish_output_reader(failed, "stdout", Duration::from_secs(1), GROUP_BOUNDARY); + assert!(failed.failure.is_some_and(|failure| failure.contains("injected read failure"))); + + let panicked = spawn_output_reader(PanickingReader, "panicking-reader").expect("spawn panicking reader"); + let panicked = finish_output_reader(panicked, "stderr", Duration::from_secs(1), GROUP_BOUNDARY); + assert!(panicked.failure.is_some_and(|failure| failure.contains("injected reader panic"))); } #[test] - fn reader_completion_observation_is_idempotent() { - let (sender, completion) = mpsc::channel::(); - drop(sender); - let mut reader = OutputReader { - thread: thread::spawn(|| {}), + fn reader_finish_reports_post_completion_panics_and_failed_cancellation() { + let (sender, completion) = mpsc::channel(); + let thread = thread::spawn(move || { + sender.send(ReaderCompletion::Finished(Ok(()))).expect("the receiver remains alive"); + panic!("panic after completion"); + }); + let reader = OutputReader { + thread, completion, output: Arc::new(Mutex::new(CapturedOutput::empty())), - retaining: Arc::new(AtomicBool::new(true)), + retaining: Arc::new(std::sync::atomic::AtomicBool::new(true)), + reported: None, + failure_claimed: false, + }; + let failure = finish_output_reader(reader, "stdout", Duration::from_secs(1), LEADER_BOUNDARY) + .failure + .expect("the join panic is reported"); + assert!(failure.contains("panic after completion")); + + let (failed_sender, failed_completion) = mpsc::channel(); + let failed_thread = thread::spawn(move || { + failed_sender + .send(ReaderCompletion::Finished(Err(io::Error::other("read failed")))) + .expect("the receiver remains alive"); + panic!("panic after failed completion"); + }); + let failed_reader = OutputReader { + thread: failed_thread, + completion: failed_completion, + output: Arc::new(Mutex::new(CapturedOutput::empty())), + retaining: Arc::new(std::sync::atomic::AtomicBool::new(true)), reported: None, failure_claimed: false, }; + let failure = finish_output_reader(failed_reader, "stdout", Duration::from_secs(1), LEADER_BOUNDARY) + .failure + .expect("read and join failures are reported"); + assert!(failure.contains("read failed")); + assert!(failure.contains("panic after failed completion")); - assert!( - reader - .take_failure("stdout") - .is_some_and(|failure| failure.contains("without reporting completion")) - ); - assert!(reader.take_failure("stdout").is_none()); - let captured = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); - assert!(captured.failure.is_none(), "the process outcome already claimed the reader failure"); + let (sender, completion) = mpsc::channel::(); + let reader = OutputReader { + thread: thread::spawn(|| {}), + completion, + output: Arc::new(Mutex::new(CapturedOutput::empty())), + retaining: Arc::new(std::sync::atomic::AtomicBool::new(true)), + reported: None, + failure_claimed: true, + }; + let failure = finish_output_reader(reader, "stderr", Duration::ZERO, LEADER_BOUNDARY) + .failure + .expect("failed cancellation is reported"); + assert!(failure.contains("reader did not stop")); + drop(sender); } #[test] - fn reader_finish_reports_join_panics_and_failed_cancellation() { - for completion in [ - ReaderCompletion::Finished(Ok(())), - ReaderCompletion::Finished(Err(io::Error::other("reported read failure"))), - ] { - let (sender, receiver) = mpsc::channel(); - let thread = thread::spawn(move || { - sender.send(completion).expect("the synthetic completion receiver remains alive"); - panic!("panic after reader completion"); - }); - let reader = OutputReader { - thread, - completion: receiver, - output: Arc::new(Mutex::new(CapturedOutput::empty())), - retaining: Arc::new(AtomicBool::new(true)), - reported: None, - failure_claimed: false, - }; - let captured = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); - let failure = captured.failure.expect("the join panic must be reported"); - assert!(failure.contains("panic after reader completion"), "{failure}"); - } - - let (_sender, completion) = mpsc::channel(); + fn disconnected_reader_completion_is_reported_by_the_finisher() { + let (sender, completion) = mpsc::channel::(); + drop(sender); let reader = OutputReader { thread: thread::spawn(|| {}), completion, output: Arc::new(Mutex::new(CapturedOutput::empty())), - retaining: Arc::new(AtomicBool::new(true)), + retaining: Arc::new(std::sync::atomic::AtomicBool::new(true)), reported: None, - failure_claimed: true, + failure_claimed: false, }; - let captured = finish_output_reader(reader, "stderr", Duration::ZERO, ORDINARY_BOUNDARY); - let failure = captured.failure.expect("failed cancellation must be reported"); - assert!(failure.contains("reader did not stop"), "{failure}"); - assert!(!failure.contains("remained open"), "{failure}"); + let failure = finish_output_reader(reader, "stdout", Duration::from_secs(1), LEADER_BOUNDARY) + .failure + .expect("disconnection is reported"); + assert!(failure.contains("without reporting completion")); + } + + #[test] + fn cancellable_reader_stops_within_the_join_grace() { + let (dropped_tx, dropped_rx) = mpsc::channel(); + let reader = spawn_output_reader(PendingReader { dropped: Some(dropped_tx) }, "pending-reader").expect("spawn pending reader"); + let captured = finish_output_reader(reader, "stdout", Duration::from_millis(25), LEADER_BOUNDARY); + dropped_rx + .recv_timeout(Duration::from_secs(1)) + .expect("the polling reader observes capture cancellation"); + let failure = captured.failure.expect("an open pipe is an infrastructure failure"); + assert!(failure.contains("remained open")); + assert!(!failure.contains("did not stop")); + } + + #[test] + fn blocking_reader_is_detached_and_partial_output_is_recovered_without_blocking() { + let release = Arc::new((Mutex::new(false), Condvar::new())); + let (blocked_tx, blocked_rx) = mpsc::channel(); + let (finished_tx, finished_rx) = mpsc::channel(); + let reader = spawn_output_reader( + BlockingReader { + first_read: false, + blocked: Some(blocked_tx), + finished: finished_tx, + release: Arc::clone(&release), + }, + "blocking-reader", + ) + .expect("spawn blocking reader"); + blocked_rx + .recv_timeout(Duration::from_secs(1)) + .expect("the reader reaches its blocking read"); + + let started = Instant::now(); + let mut captured = finish_output_reader(reader, "stdout", Duration::from_millis(25), LEADER_BOUNDARY); + assert!(started.elapsed() < Duration::from_millis(500)); + assert_eq!(output_bytes(&mut captured.output), b"captured-before-detach"); + let failure = captured.failure.expect("detachment is an infrastructure failure"); + assert!(failure.contains("remained open")); + assert!(failure.contains("did not stop")); + + let (lock, condition) = &*release; + *lock.lock().expect("the test owns the release mutex without panicking") = true; + condition.notify_all(); + finished_rx + .recv_timeout(Duration::from_secs(1)) + .expect("the detached reader exits after the pipe closes"); } #[test] - fn cancellation_does_not_wait_for_a_reader_holding_the_capture_mutex() { + fn detached_reader_never_blocks_on_a_locked_capture_buffer() { let (factory_started, factory_is_started) = mpsc::sync_channel(0); let release = Arc::new((Mutex::new(false), Condvar::new())); let factory_release = Arc::clone(&release); - let (reader_dropped, reader_is_dropped) = mpsc::channel(); + let (dropped_tx, dropped_rx) = mpsc::channel(); let reader = spawn_output_reader_with( DropSignalReader { bytes: Some(b"stalled append"), - dropped: Some(reader_dropped), + dropped: Some(dropped_tx), }, - "stalled-append-reader", + "locked-buffer-reader", 0, Box::new(move || { - factory_started.send(()).expect("the finisher waits for the stalled append"); + factory_started.send(()).expect("the finisher waits for spill creation"); let (lock, condition) = &*factory_release; let mut released = lock.lock().expect("the test owns the release mutex without panicking"); while !*released { @@ -2075,728 +1922,158 @@ mod tests { Ok(Box::new(io::Cursor::new(Vec::new())) as Box) }), ) - .expect("create stalled append reader"); + .expect("spawn locked-buffer reader"); factory_is_started .recv_timeout(Duration::from_secs(1)) - .expect("the reader entered spill creation while holding the capture mutex"); + .expect("the reader holds the capture mutex"); let started = Instant::now(); - let mut captured = finish_output_reader(reader, "stdout", Duration::ZERO, ORDINARY_BOUNDARY); - assert!( - started.elapsed() < Duration::from_millis(500), - "capture cleanup blocked on the stalled output mutex" - ); + let mut captured = finish_output_reader(reader, "stdout", Duration::ZERO, LEADER_BOUNDARY); + assert!(started.elapsed() < Duration::from_millis(500)); assert!(output_bytes(&mut captured.output).is_empty()); - let failure = captured.failure.expect("the unavailable partial output must be explicit"); - assert!(failure.contains("reader did not stop"), "{failure}"); - assert!(failure.contains("capture buffer remained locked"), "{failure}"); - - let (lock, condition) = &*release; - *lock.lock().expect("the test owns the release mutex without panicking") = true; - condition.notify_all(); - reader_is_dropped - .recv_timeout(Duration::from_secs(1)) - .expect("the cancelled detached reader exits after the stalled append is released"); - } - - #[test] - fn detached_output_recovery_preserves_a_poisoned_available_buffer() { - let output = poisoned_buffer(); - let mut failure = None; - - let mut captured = take_reader_output(&output, "stdout", false, &mut failure); - - assert_eq!(output_bytes(&mut captured), b"poisoned bytes"); - assert!( - failure.is_some_and(|failure| failure.contains("capture buffer was poisoned")), - "the detached poisoned-buffer branch did not report its infrastructure failure" - ); - } - - #[test] - fn normal_completion_does_not_join_a_stubborn_descendant_pipe() { - let release = Arc::new((Mutex::new(false), Condvar::new())); - let (blocked_tx, blocked_rx) = mpsc::channel(); - let (reader_finished_tx, reader_finished_rx) = mpsc::channel(); - let reader = spawn_output_reader( - StubbornPipe { - read: false, - blocked: Some(blocked_tx), - finished: reader_finished_tx, - release: Arc::clone(&release), - }, - "stubborn-test-pipe", - ) - .expect("the test reader thread can be created"); - blocked_rx - .recv_timeout(Duration::from_secs(1)) - .expect("the reader reaches the simulated descendant's open pipe"); - - let (finished_tx, finished_rx) = mpsc::channel(); - let finisher = thread::spawn(move || { - let result = finish_output_reader(reader, "stdout", Duration::from_millis(25), ORDINARY_BOUNDARY); - let _receiver_gone = finished_tx.send(result); - }); - let result = finished_rx.recv_timeout(Duration::from_millis(500)); + let failure = captured.failure.expect("unavailable partial bytes are explicit"); + assert!(failure.contains("capture buffer remained locked")); let (lock, condition) = &*release; *lock.lock().expect("the test owns the release mutex without panicking") = true; condition.notify_all(); - - let mut captured = result.expect("normal completion must not wait for an escaped descendant to close its pipe"); - finisher.join().expect("the bounded finisher thread does not panic"); - reader_finished_rx + dropped_rx .recv_timeout(Duration::from_secs(1)) - .expect("the detached reader exits after the test releases its simulated pipe"); - assert_eq!(output_bytes(&mut captured.output), b"captured-before-timeout"); - assert!( - captured.failure.as_deref().is_some_and(|failure| failure.contains("remained open")), - "grace expiry must be an explicit infrastructure failure" - ); + .expect("the detached reader exits after spill creation resumes"); } #[test] - fn unproven_cleanup_discards_data_that_arrives_after_capture_stops() { + fn data_arriving_after_capture_cancellation_is_discarded() { let release = Arc::new((Mutex::new(false), Condvar::new())); let (started_tx, started_rx) = mpsc::channel(); - let (reader_finished_tx, reader_finished_rx) = mpsc::channel(); + let (finished_tx, finished_rx) = mpsc::channel(); let reader = spawn_output_reader( - LateDataPipe { + LateDataReader { started: Some(started_tx), - finished: Some(reader_finished_tx), + finished: Some(finished_tx), release: Arc::clone(&release), }, - "late-data-test-pipe", + "late-data-reader", ) - .expect("create late-data reader"); + .expect("spawn late-data reader"); started_rx .recv_timeout(Duration::from_secs(1)) - .expect("the late-data reader starts its blocking read"); + .expect("the reader blocks before producing data"); - let mut captured = finish_output_reader(reader, "stdout", Duration::from_millis(25), ORDINARY_BOUNDARY); + let mut captured = finish_output_reader(reader, "stdout", Duration::ZERO, LEADER_BOUNDARY); assert!(output_bytes(&mut captured.output).is_empty()); assert!(captured.failure.is_some()); let (lock, condition) = &*release; *lock.lock().expect("the test owns the release mutex without panicking") = true; condition.notify_all(); - reader_finished_rx + finished_rx .recv_timeout(Duration::from_secs(1)) .expect("the detached reader discards late data and exits"); } #[test] - fn worker_panic_becomes_an_outcome_without_deadlocking_the_scheduler() { - let plan = Plan { - invocations: vec![Invocation { - label: Some("panic-probe".to_owned()), - argv: vec![WORKER_PANIC_TEST_PROGRAM.to_owned()], - work_dir: None, - }], - }; - let (finished_tx, finished_rx) = mpsc::channel(); - let scheduler = thread::spawn(move || { - let result = execute_parallel(&plan, false, NonZeroUsize::new(2).expect("literal two is nonzero"), None); - let exit_two = result.is_ok_and(|code| code == ExitCode::from(2)); - let _receiver_gone = finished_tx.send(exit_two); - }); - - assert!( - finished_rx - .recv_timeout(Duration::from_secs(2)) - .expect("the panic-safe worker reports before the scheduler deadline"), - "a worker panic must become an infrastructure exit" - ); - scheduler.join().expect("the scheduler handles the worker panic without panicking"); + fn output_spills_after_the_memory_limit_and_cleans_up_by_raii() { + let named = tempfile::NamedTempFile::new().expect("create named spill file"); + let path = named.path().to_path_buf(); + let mut named = Some(named); + let reader = spawn_output_reader_with( + ChunkedReader { + chunks: VecDeque::from([b"abcd".to_vec(), b"efgh".to_vec(), b"ijkl".to_vec()]), + }, + "spill-reader", + 4, + Box::new(move || Ok(Box::new(named.take().expect("capture creates one spill")) as Box)), + ) + .expect("spawn spill reader"); + let mut captured = finish_output_reader(reader, "stdout", Duration::from_secs(1), LEADER_BOUNDARY); + assert!(captured.output.is_spilled()); + assert_eq!(output_bytes(&mut captured.output), b"abcdefghijkl"); + assert!(path.exists()); + drop(captured); + assert!(!path.exists()); } #[test] - fn disconnected_worker_channel_becomes_an_infrastructure_outcome() { - let (sender, receiver) = mpsc::channel::(); - drop(sender); - let thread = thread::spawn(|| {}); - let outcome = wait_for_worker(&mut vec![RunningWorker { - index: 4, - receiver, - thread, - }]) - .expect("the disconnected worker is observable"); - assert_eq!(outcome.index, 4); - let InvocationResult::Infrastructure(message) = outcome.outcome.result else { - panic!("a disconnected worker must produce an infrastructure outcome"); - }; - assert!(message.contains("without reporting")); - } + fn output_at_the_memory_limit_does_not_create_a_spill() { + let mut output = CapturedOutput::empty(); + let mut spill_factory: super::SpillFactory = Box::new(|| panic!("output at the limit must stay in memory")); - #[test] - fn worker_panic_descriptions_preserve_string_payloads() { - let borrowed: Box = Box::new("borrowed panic"); - let owned: Box = Box::new("owned panic".to_owned()); - let other: Box = Box::new(7_u8); - assert_eq!(panic_description(borrowed.as_ref()), "borrowed panic"); - assert_eq!(panic_description(owned.as_ref()), "owned panic"); - assert_eq!(panic_description(other.as_ref()), "non-string panic payload"); - } + output.append(b"abcd", 4, &mut spill_factory).expect("in-memory append succeeds"); - #[test] - fn cleanup_failure_context_preserves_both_outcomes() { - assert_eq!(with_cleanup_failure("primary".to_owned(), &Ok::<_, io::Error>(())), "primary"); - assert_eq!( - with_cleanup_failure("primary".to_owned(), &Err::<(), _>(io::Error::other("cleanup"))), - "primary; process-tree cleanup also failed: cleanup" - ); + assert!(!output.is_spilled()); + assert_eq!(output_bytes(&mut output), b"abcd"); } #[test] - fn ordinary_termination_polls_until_successful_reap() { - let mut child = FakeOrdinaryChild { - kill_error: None, - observations: VecDeque::from([Ok(None), Ok(Some(successful_status()))]), - }; - - let cleanup = terminate_ordinary_with( - &mut child, - Duration::from_secs(1), - FakeOrdinaryChild::kill_child, - FakeOrdinaryChild::try_wait_child, - ); - - assert!(cleanup.as_ref().is_ok_and(ExitStatus::success)); - assert!(child.observations.is_empty(), "termination returned after only one immediate poll"); - assert_eq!( - with_cleanup_failure("capture setup failed".to_owned(), &cleanup), - "capture setup failed" - ); - } - - #[test] - fn ordinary_child_sleep_probe() { - if let Some(duration) = std::env::var_os("CARGO_EACH_ORDINARY_CHILD_PROBE") { - let millis = duration.to_string_lossy().parse().expect("the parent passes a millisecond count"); - thread::sleep(Duration::from_millis(millis)); - } - } - - #[test] - #[cfg_attr(miri, ignore = "spawns and terminates a child process")] - fn ordinary_termination_kills_polls_and_reaps_a_real_child() { - let mut child = sleeping_test_command().spawn().expect("spawn ordinary child probe"); - - let cleanup = terminate_ordinary_child(&mut child, Duration::from_secs(1)); - - let status = cleanup.expect("a successfully killed child must be reaped within the grace"); - assert!(!status.success(), "a killed child must not report successful completion"); - assert!( - child.try_wait().expect("the reaped child remains observable").is_some(), - "bounded termination returned without reaping the child" - ); - } - - #[test] - fn ordinary_termination_reports_deadline_kill_and_observation_failures() { - let mut deadline = FakeOrdinaryChild { - kill_error: None, - observations: VecDeque::from([Ok(None)]), - }; - let error = terminate_ordinary_with( - &mut deadline, - Duration::ZERO, - FakeOrdinaryChild::kill_child, - FakeOrdinaryChild::try_wait_child, - ) - .expect_err("an unreaped child reaches the deadline"); - assert_eq!(error.kind(), io::ErrorKind::WouldBlock); - - let mut failed_kill_deadline = FakeOrdinaryChild { - kill_error: Some(io::Error::new(io::ErrorKind::PermissionDenied, "kill failed")), - observations: VecDeque::from([Ok(None)]), - }; - let error = terminate_ordinary_with( - &mut failed_kill_deadline, - Duration::ZERO, - FakeOrdinaryChild::kill_child, - FakeOrdinaryChild::try_wait_child, - ) - .expect_err("a failed kill with an unreaped child reaches the deadline"); - assert_eq!(error.kind(), io::ErrorKind::WouldBlock); - assert!(error.to_string().contains("kill failed")); - - let mut failed_kill = FakeOrdinaryChild { - kill_error: Some(io::Error::new(io::ErrorKind::PermissionDenied, "kill failed")), - observations: VecDeque::from([Ok(Some(successful_status()))]), - }; - let error = terminate_ordinary_with( - &mut failed_kill, - Duration::from_secs(1), - FakeOrdinaryChild::kill_child, - FakeOrdinaryChild::try_wait_child, - ) - .expect_err("reaping must not hide a failed kill"); - assert_eq!(error.kind(), io::ErrorKind::PermissionDenied); - - let mut failed_observation = FakeOrdinaryChild { - kill_error: None, - observations: VecDeque::from([Err(io::Error::other("observation failed"))]), - }; - let error = terminate_ordinary_with( - &mut failed_observation, - Duration::from_secs(1), - FakeOrdinaryChild::kill_child, - FakeOrdinaryChild::try_wait_child, - ) - .expect_err("try_wait failure must be preserved"); - assert!(error.to_string().contains("observation failed")); - } - - #[test] - fn wait_error_runs_bounded_cleanup_and_preserves_both_errors() { - let mut cleaned = false; - let outcome = finish_wait_with_cleanup(&mut cleaned, Err(io::Error::other("wait failed")), |cleaned| { - *cleaned = true; - Ok(successful_status()) - }); - assert!(cleaned, "wait failure did not invoke bounded cleanup"); - let message = result_infrastructure_message(outcome.result); - assert!(message.contains("wait failed")); - assert!(!message.contains("cleanup also failed")); - - let mut attempted = false; - let outcome = finish_wait_with_cleanup(&mut attempted, Err(io::Error::other("wait failed")), |attempted| { - *attempted = true; - Err(io::Error::other("cleanup failed")) - }); - assert!(attempted, "failed cleanup was not attempted"); - let message = result_infrastructure_message(outcome.result); - assert!(message.contains("wait failed")); - assert!(message.contains("cleanup also failed: cleanup failed")); - } - - #[test] - fn unsealed_timeout_is_refused_before_spawn() { - struct FakePrepared { - sealed: bool, - } - - let spawned = Arc::new(AtomicBool::new(false)); - let spawn_observed = Arc::clone(&spawned); - let error = spawn_if_sealed( - FakePrepared { sealed: false }, - |prepared| prepared.sealed, - move |_prepared| { - spawn_observed.store(true, Ordering::SeqCst); - Ok::<_, String>(()) - }, - ) - .expect_err("an unsealed timeout launch must be refused"); - - assert!( - !spawned.load(Ordering::SeqCst), - "the child spawn path ran despite unsealed containment" - ); - assert!(error.contains("timeout requires sealed process-tree containment")); - assert!(error.contains("child was not started")); - - let spawned = Arc::new(AtomicBool::new(false)); - let spawn_observed = Arc::clone(&spawned); - spawn_if_sealed( - FakePrepared { sealed: true }, - |prepared| prepared.sealed, - move |_prepared| { - spawn_observed.store(true, Ordering::SeqCst); - Ok::<_, String>(()) - }, - ) - .expect("sealed containment permits the spawn"); - assert!(spawned.load(Ordering::SeqCst)); - } - - #[test] - fn panicked_worker_without_a_report_becomes_an_infrastructure_outcome() { - let (sender, receiver) = mpsc::channel::(); - drop(sender); - let thread = thread::spawn(|| panic!("panic before reporting")); - let outcome = wait_for_worker(&mut vec![RunningWorker { - index: 5, - receiver, - thread, - }]) - .expect("the panicked worker is observable"); - let InvocationResult::Infrastructure(message) = outcome.outcome.result else { - panic!("a panicked worker must produce an infrastructure outcome"); - }; - assert!(message.contains("panic before reporting")); - } - - #[test] - fn panic_after_a_worker_report_overrides_the_report() { - let (sender, receiver) = mpsc::channel(); - let thread = thread::spawn(move || { - sender - .send(BufferedOutcome::infrastructure("premature report".to_owned())) - .expect("the scheduler receiver remains alive"); - panic!("panic after reporting"); - }); - let outcome = wait_for_worker(&mut vec![RunningWorker { - index: 6, - receiver, - thread, - }]) - .expect("the reported worker is observable"); - let InvocationResult::Infrastructure(message) = outcome.outcome.result else { - panic!("a worker panic must override its premature report"); - }; - assert!(message.contains("panic after reporting")); - assert!(wait_for_worker(&mut Vec::new()).is_none()); - } - - #[test] - #[cfg_attr(miri, ignore = "spawns child processes through the scheduler")] - fn worker_spawn_failures_are_reported_during_initial_and_replacement_launches() { - let initial = Plan { - invocations: vec![invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM])], - }; - let code = execute_parallel(&initial, false, NonZeroUsize::new(2).expect("literal two is nonzero"), None) - .expect("worker launch failure is represented as an invocation outcome"); - assert_eq!(code, ExitCode::from(2)); - - let replacement = Plan { - invocations: vec![invocation(&["rustc", "--version"]), invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM])], - }; - let code = execute_parallel(&replacement, false, NonZeroUsize::new(1).expect("literal one is nonzero"), None) - .expect("replacement launch failure is emitted after the completed outcome"); - assert_eq!(code, ExitCode::from(2)); - - let keep_going = Plan { - invocations: vec![invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM]), invocation(&["rustc", "--version"])], - }; - let code = execute_parallel(&keep_going, true, NonZeroUsize::new(1).expect("literal one is nonzero"), None) - .expect("keep-going continues after a worker launch outcome"); - assert_eq!(code, ExitCode::from(1)); - } - - #[test] - #[cfg_attr(miri, ignore = "spawns a child process through the worker")] - fn worker_spawn_injection_only_rejects_the_named_program() { - let worker = spawn_worker(0, invocation(&["rustc", "--version"]), None).expect("ordinary worker launch must succeed"); - let outcome = wait_for_worker(&mut vec![worker]).expect("ordinary worker reports its outcome"); - let InvocationResult::Exited(status) = outcome.outcome.result else { - panic!("the ordinary worker must execute rustc"); - }; - assert!(status.success()); - - let error = spawn_worker(1, invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM]), None) - .expect_err("the named test program injects worker spawn failure"); - assert!(error.to_string().contains("injected worker spawn failure")); - } - - #[test] - #[cfg_attr(miri, ignore = "spawns child processes")] - fn direct_runners_report_empty_and_unspawnable_commands() { - let empty = invocation(&[]); - assert!(result_infrastructure_message(run_streamed(&empty)).contains("empty argument vector")); - assert!(result_infrastructure_message(run_streamed_with_timeout(&empty, Duration::from_secs(1))).contains("empty argument vector")); - assert!(infrastructure_message(run_captured(&empty, None)).contains("empty argument vector")); - - let missing = invocation(&["cargo-each-no-such-program-for-unit-test"]); - assert!(result_infrastructure_message(run_streamed(&missing)).contains("failed to spawn")); - assert!(result_infrastructure_message(run_streamed_with_timeout(&missing, Duration::from_secs(1))).contains("failed to spawn")); - assert!(infrastructure_message(run_captured(&missing, None)).contains("failed to spawn")); - - let InvocationResult::Exited(status) = - run_streamed_with_timeout_with(&invocation(&["rustc", "--version"]), Duration::from_secs(2), spawn_tree) - else { - panic!("the timed streamed runner must execute rustc"); - }; - assert!(status.success()); - - let spawn_failure = run_streamed_with_timeout_with(&invocation(&["rustc", "--version"]), Duration::from_secs(2), |_command| { - Err("injected timed-stream spawn failure".to_owned()) - }); - assert!(result_infrastructure_message(spawn_failure).contains("injected timed-stream spawn failure")); - } - - #[test] - #[cfg_attr(miri, ignore = "spawns a contained child process")] - fn captured_runner_uses_the_production_timeout_spawner() { - let mut outcome = run_captured(&invocation(&["rustc", "--version"]), Some(Duration::from_secs(2))); - - match &outcome.result { - InvocationResult::Exited(status) => { - assert!(status.success()); - assert!( - String::from_utf8(output_bytes(&mut outcome.stdout)) - .expect("rustc output is UTF-8") - .contains("rustc") - ); - } - InvocationResult::Infrastructure(message) => { - assert!( - message.contains("timeout requires sealed process-tree containment"), - "unexpected production timeout-spawn failure: {message}" - ); - } - InvocationResult::TimedOut(duration) => panic!("rustc --version timed out after {duration:?}"), - } - } - - #[test] - fn captured_output_combines_every_reader_result_shape() { - let mut success = combine_captured_output( - captured(b"stdout", None), - captured(b"stderr", None), - InvocationResult::Infrastructure("primary".to_owned()), - ); - assert_eq!(output_bytes(&mut success.stdout), b"stdout"); - assert_eq!(output_bytes(&mut success.stderr), b"stderr"); - assert_eq!(infrastructure_message(success), "primary"); - - let mut stdout_failed = combine_captured_output( - captured(b"partial stdout", Some("stdout failed")), - captured(b"stderr", None), - InvocationResult::Infrastructure("primary".to_owned()), - ); - assert_eq!(output_bytes(&mut stdout_failed.stdout), b"partial stdout"); - assert_eq!(output_bytes(&mut stdout_failed.stderr), b"stderr"); - assert_eq!(infrastructure_message(stdout_failed), "primary; stdout failed"); - - let mut stderr_failed = combine_captured_output( - captured(b"stdout", None), - captured(b"partial stderr", Some("stderr failed")), - InvocationResult::Infrastructure("primary".to_owned()), - ); - assert_eq!(output_bytes(&mut stderr_failed.stdout), b"stdout"); - assert_eq!(output_bytes(&mut stderr_failed.stderr), b"partial stderr"); - assert_eq!(infrastructure_message(stderr_failed), "primary; stderr failed"); - - let mut both_failed = combine_captured_output( - captured(b"partial stdout", Some("stdout failed")), - captured(b"partial stderr", Some("stderr failed")), - InvocationResult::Infrastructure("primary".to_owned()), - ); - assert_eq!(output_bytes(&mut both_failed.stdout), b"partial stdout"); - assert_eq!(output_bytes(&mut both_failed.stderr), b"partial stderr"); - assert_eq!(infrastructure_message(both_failed), "primary; stdout failed; stderr failed"); - - let timed_out = combine_captured_output( - captured(b"partial stdout", Some("stdout still open")), - captured(b"", None), - InvocationResult::TimedOut(Duration::from_secs(2)), - ); - assert_eq!( - infrastructure_message(timed_out), - "invocation timed out after 2s; stdout still open" - ); - - let exited = combine_captured_output( - captured(b"partial stdout", Some("stdout still open")), - captured(b"", None), - InvocationResult::Exited(successful_status()), - ); - assert_eq!(infrastructure_message(exited), "stdout still open"); - } - - #[test] - fn output_reader_surfaces_read_and_panic_failures() { - let failed = spawn_output_reader(FailingReader, "failing-reader").expect("create failing reader"); - let failed = finish_output_reader(failed, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); - assert!( - failed - .failure - .as_deref() - .is_some_and(|failure| failure.contains("injected read failure")) - ); - - let panicked = spawn_output_reader(PanickingReader, "panicking-reader").expect("create panicking reader"); - let panicked = finish_output_reader(panicked, "stderr", Duration::from_secs(1), CONTAINED_BOUNDARY); - assert!( - panicked - .failure - .as_deref() - .is_some_and(|failure| failure.contains("injected reader panic")) - ); - - let eof = spawn_output_reader(EofThenPanicReader { reached_eof: false }, "eof-reader").expect("create EOF reader"); - let mut eof = finish_output_reader(eof, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); - assert!(output_bytes(&mut eof.output).is_empty()); - assert!(eof.failure.is_none()); - - let interrupted = - spawn_output_reader(InterruptedThenDataReader { state: 0 }, "interrupted-reader").expect("create interrupted reader"); - let mut interrupted = finish_output_reader(interrupted, "stdout", Duration::from_secs(1), CONTAINED_BOUNDARY); - assert_eq!(output_bytes(&mut interrupted.output), b"after interrupt"); - assert!(interrupted.failure.is_none()); - - let (completion_sender, completion) = mpsc::channel(); - let sealed_timeout = OutputReader { - thread: thread::spawn(|| {}), - completion, - output: Arc::new(Mutex::new(CapturedOutput::empty())), - retaining: Arc::new(AtomicBool::new(true)), - reported: None, - failure_claimed: false, - }; - let sealed_timeout = finish_output_reader(sealed_timeout, "stdout", Duration::ZERO, CONTAINED_BOUNDARY); - drop(completion_sender); - assert!( - sealed_timeout - .failure - .as_deref() - .is_some_and(|failure| failure.contains("contained process tree")) - ); - - let (completion_sender, completion) = mpsc::channel::(); - drop(completion_sender); - let disconnected = OutputReader { - thread: thread::spawn(|| {}), - completion, - output: Arc::new(Mutex::new(CapturedOutput::empty())), - retaining: Arc::new(AtomicBool::new(true)), - reported: None, - failure_claimed: false, - }; - let disconnected = finish_output_reader(disconnected, "stderr", Duration::from_secs(1), ORDINARY_BOUNDARY); - assert!( - disconnected - .failure - .as_deref() - .is_some_and(|failure| failure.contains("without reporting completion")) - ); - - for completion_result in [ - ReaderCompletion::Finished(Ok(())), - ReaderCompletion::Finished(Err(io::Error::other("read failed"))), - ] { - let (completion_sender, completion) = mpsc::channel(); - completion_sender - .send(completion_result) - .expect("the synthetic reader completion receiver is alive"); - let poisoned = OutputReader { - thread: thread::spawn(|| {}), - completion, - output: poisoned_buffer(), - retaining: Arc::new(AtomicBool::new(true)), - reported: None, - failure_claimed: false, - }; - let mut poisoned = finish_output_reader(poisoned, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); - assert_eq!(output_bytes(&mut poisoned.output), b"poisoned bytes"); - assert!( - poisoned - .failure - .as_deref() - .is_some_and(|failure| failure.contains("capture buffer was poisoned")) - ); - } - } - - #[test] - #[cfg_attr(miri, ignore = "creates a named system-temporary spill file")] - fn output_spills_after_threshold_and_cleans_up_with_its_outcome() { - let named = tempfile::NamedTempFile::new().expect("create named spill file"); - let path = named.path().to_path_buf(); - let mut named = Some(named); + fn spill_failures_become_infrastructure_failures() { let reader = spawn_output_reader_with( ChunkedReader { - chunks: VecDeque::from([b"abcd".to_vec(), b"efgh".to_vec()]), + chunks: VecDeque::from([b"abc".to_vec(), b"def".to_vec()]), }, - "spill-threshold-reader", - 4, - Box::new(move || Ok(Box::new(named.take().expect("the capture creates at most one spill file")) as Box)), - ) - .expect("create threshold reader"); - let mut captured = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); - - assert!(captured.output.is_spilled()); - assert!(path.exists(), "the spill must live while its outcome owns it"); - assert_eq!(output_bytes(&mut captured.output), b"abcdefgh"); - - drop(captured); - assert!(!path.exists(), "dropping the outcome must remove its spill file"); - } - - #[test] - fn spill_create_and_write_failures_preserve_memory_and_become_infrastructure_failures() { - for (factory, expected) in [ - ( - Box::new(|| Err(io::Error::other("injected spill create failure"))) as super::SpillFactory, - "injected spill create failure", - ), - ( - Box::new(|| { - Ok(Box::new(FaultySpill { - cursor: io::Cursor::new(Vec::new()), - fail_write: true, - fail_read: false, - }) as Box) - }) as super::SpillFactory, - "injected spill write failure", - ), - ] { - let reader = spawn_output_reader_with( - ChunkedReader { - chunks: VecDeque::from([b"abc".to_vec(), b"def".to_vec()]), - }, - "failing-spill-reader", - 4, - factory, - ) - .expect("create failing spill reader"); - let mut captured_stream = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); - assert_eq!(output_bytes(&mut captured_stream.output), b"abc"); - let outcome = combine_captured_output(captured_stream, captured(b"", None), InvocationResult::Exited(successful_status())); - - assert!(infrastructure_message(outcome).contains(expected)); - } - } - - #[test] - fn spill_read_failure_becomes_an_infrastructure_failure_during_emission() { - let reader = spawn_output_reader_with( - io::Cursor::new(b"abcdefgh".to_vec()), - "read-failing-spill", + "write-failing-spill", 4, Box::new(|| { Ok(Box::new(FaultySpill { cursor: io::Cursor::new(Vec::new()), - fail_write: false, - fail_read: true, + fail_write: true, + fail_read: false, }) as Box) }), ) - .expect("create read-failing spill reader"); - let captured_stream = finish_output_reader(reader, "stdout", Duration::from_secs(1), ORDINARY_BOUNDARY); - assert!(captured_stream.output.is_spilled()); - let mut outcome = combine_captured_output(captured_stream, captured(b"", None), InvocationResult::Exited(successful_status())); - - emit_buffered(&invocation(&["probe"]), &mut outcome).expect("destination output remains writable"); + .expect("spawn write-failing reader"); + let captured_stream = finish_output_reader(reader, "stdout", Duration::from_secs(1), LEADER_BOUNDARY); + let outcome = combine_captured_output(captured_stream, captured(b"", None), InvocationResult::Exited(successful_status())); + assert!(infrastructure_message(outcome).contains("injected spill write failure")); - assert!(infrastructure_message(outcome).contains("injected spill read failure")); - } - - #[test] - fn stderr_spill_read_and_destination_write_failures_are_reported() { - let read_failing_spill = || { - CapturedOutput::Spill(Box::new(FaultySpill { + let mut outcome = BufferedOutcome { + stdout: CapturedOutput::Spill(Box::new(FaultySpill { cursor: io::Cursor::new(Vec::new()), fail_write: false, fail_read: true, - })) + })), + stderr: CapturedOutput::empty(), + result: InvocationResult::Exited(successful_status()), }; - let mut outcome = BufferedOutcome { + emit_buffered_to(&invocation(&["probe"]), &mut outcome, &mut Vec::new(), &mut Vec::new()).expect("destinations remain writable"); + assert!(infrastructure_message(outcome).contains("injected spill read failure")); + + let mut stderr_outcome = BufferedOutcome { stdout: CapturedOutput::empty(), - stderr: read_failing_spill(), + stderr: CapturedOutput::Spill(Box::new(FaultySpill { + cursor: io::Cursor::new(Vec::new()), + fail_write: false, + fail_read: true, + })), result: InvocationResult::Exited(successful_status()), }; - emit_buffered_to(&invocation(&["probe"]), &mut outcome, &mut Vec::new(), &mut Vec::new()).expect("destinations remain writable"); - assert!(infrastructure_message(outcome).contains("failed to read spilled child stderr")); + emit_buffered_to(&invocation(&["probe"]), &mut stderr_outcome, &mut Vec::new(), &mut Vec::new()) + .expect("destinations remain writable"); + assert!(infrastructure_message(stderr_outcome).contains("injected spill read failure")); + } + + #[test] + fn capture_and_emission_preserve_all_failure_context() { + for (stdout, stderr, expected) in [ + (captured(b"out", Some("stdout failed")), captured(b"err", None), "stdout failed"), + (captured(b"out", None), captured(b"err", Some("stderr failed")), "stderr failed"), + ( + captured(b"out", Some("stdout failed")), + captured(b"err", Some("stderr failed")), + "stdout failed; stderr failed", + ), + ] { + let outcome = combine_captured_output(stdout, stderr, InvocationResult::Exited(successful_status())); + assert_eq!(infrastructure_message(outcome), expected); + } - let mut stdout_failure = BufferedOutcome { + let mut outcome = BufferedOutcome { stdout: CapturedOutput::Memory(b"stdout".to_vec()), stderr: CapturedOutput::empty(), result: InvocationResult::Exited(successful_status()), }; - let error = emit_buffered_to(&invocation(&["probe"]), &mut stdout_failure, &mut FailingWriter, &mut Vec::new()) - .expect_err("stdout destination failure must propagate"); + let error = emit_buffered_to(&invocation(&["probe"]), &mut outcome, &mut FailingWriter, &mut Vec::new()) + .expect_err("destination failure propagates"); assert!(error.to_string().contains("injected destination write failure")); let mut stderr_failure = BufferedOutcome { @@ -2805,249 +2082,196 @@ mod tests { result: InvocationResult::Exited(successful_status()), }; let error = emit_buffered_to(&invocation(&["probe"]), &mut stderr_failure, &mut Vec::new(), &mut FailingWriter) - .expect_err("stderr destination failure must propagate"); + .expect_err("stderr destination failure propagates"); assert!(error.to_string().contains("injected destination write failure")); - for (result, diagnostic) in [ - (InvocationResult::TimedOut(Duration::from_millis(250)), "timed out after 250ms"), - ( - InvocationResult::Infrastructure("capture infrastructure failed".to_owned()), - "capture infrastructure failed", - ), + for result in [ + InvocationResult::TimedOut(Duration::from_millis(10)), + InvocationResult::Infrastructure("infrastructure".to_owned()), ] { - let mut outcome = BufferedOutcome { + let mut diagnostic = BufferedOutcome { stdout: CapturedOutput::empty(), stderr: CapturedOutput::empty(), result, }; let mut stderr = Vec::new(); - emit_buffered_to(&invocation(&["probe"]), &mut outcome, &mut Vec::new(), &mut stderr) - .expect("memory destinations remain writable"); - assert!(String::from_utf8(stderr).expect("diagnostics are UTF-8").contains(diagnostic)); + emit_buffered_to(&invocation(&["probe"]), &mut diagnostic, &mut Vec::new(), &mut stderr).expect("memory output succeeds"); + assert!(!stderr.is_empty()); } } #[test] - fn reader_setup_failure_preserves_cleanup_and_reader_errors() { - let successful_reader = - spawn_output_reader(io::Cursor::new(b"partial".to_vec()), "successful-reader").expect("create successful reader"); - let mut outcome = BufferedOutcome::from_reader_failure( - "stderr unavailable".to_owned(), - successful_reader, - &Ok::<_, io::Error>(()), - ORDINARY_BOUNDARY, + fn infrastructure_failure_merging_preserves_primary_context() { + assert!(matches!( + add_infrastructure_failure(InvocationResult::Exited(successful_status()), String::new()), + InvocationResult::Exited(status) if status.success() + )); + assert_eq!( + result_infrastructure_message(add_infrastructure_failure( + InvocationResult::TimedOut(Duration::from_millis(10)), + "drain failed".to_owned(), + )), + "invocation timed out after 10ms; drain failed" ); - assert_eq!(output_bytes(&mut outcome.stdout), b"partial"); - assert_eq!(infrastructure_message(outcome), "stderr unavailable"); + assert_eq!( + result_infrastructure_message(add_infrastructure_failure( + InvocationResult::Infrastructure("wait failed".to_owned()), + "drain failed".to_owned(), + )), + "wait failed; drain failed" + ); + } - let failing_reader = spawn_output_reader(FailingReader, "failing-reader").expect("create failing reader"); - let outcome = BufferedOutcome::from_reader_failure( + #[test] + fn reader_setup_failure_keeps_partial_output_and_cleanup_context() { + let reader = spawn_output_reader(io::Cursor::new(b"partial".to_vec()), "partial-reader").expect("spawn partial reader"); + let mut outcome = BufferedOutcome::from_reader_failure( "stderr unavailable".to_owned(), - failing_reader, + reader, &Err::<(), _>(io::Error::other("cleanup failed")), - ORDINARY_BOUNDARY, ); + assert_eq!(output_bytes(&mut outcome.stdout), b"partial"); let message = infrastructure_message(outcome); assert!(message.contains("stderr unavailable")); assert!(message.contains("cleanup failed")); - - let failing_reader = spawn_output_reader(FailingReader, "failing-reader").expect("create failing reader"); - let outcome = BufferedOutcome::from_reader_failure( - "stderr unavailable".to_owned(), - failing_reader, - &Ok::<_, io::Error>(()), - ORDINARY_BOUNDARY, - ); - assert!(infrastructure_message(outcome).contains("injected read failure")); } #[test] - fn tree_outcome_retains_the_invocation_result() { - let outcome = TreeOutcome::new(InvocationResult::Infrastructure("outcome".to_owned())); - assert_eq!( - infrastructure_message(BufferedOutcome { - stdout: CapturedOutput::empty(), - stderr: CapturedOutput::empty(), - result: outcome.result, - }), - "outcome" + fn reader_state_reports_disconnection_and_never_repeats_a_failure() { + let (sender, completion) = mpsc::channel::(); + drop(sender); + let mut reader = OutputReader { + thread: thread::spawn(|| {}), + completion, + output: Arc::new(Mutex::new(CapturedOutput::empty())), + retaining: Arc::new(std::sync::atomic::AtomicBool::new(true)), + reported: None, + failure_claimed: false, + }; + assert!( + reader + .take_failure("stdout") + .is_some_and(|failure| failure.contains("without reporting completion")) + ); + assert!(reader.take_failure("stdout").is_none()); + assert!( + finish_output_reader(reader, "stdout", Duration::from_secs(1), LEADER_BOUNDARY) + .failure + .is_none() ); } #[test] - fn process_waiting_classifies_observation_and_cleanup_failures() { - let mut cleaned = FakeProcess { - observations: VecDeque::from([Err(io::Error::other("observe failed"))]), - termination: Some(Ok(successful_status())), - }; - let outcome = wait_for_tree_with(&mut cleaned, Duration::from_secs(1), FakeProcess::observe, FakeProcess::terminate); - assert!(result_infrastructure_message(outcome.result).contains("observe failed")); - - let mut uncleaned = FakeProcess { - observations: VecDeque::from([Err(io::Error::other("observe failed"))]), - termination: Some(Err(io::Error::other("cleanup failed"))), - }; - let outcome = wait_for_tree_without_timeout_with(&mut uncleaned, FakeProcess::observe, FakeProcess::terminate); - let message = result_infrastructure_message(outcome.result); - assert!(message.contains("observe failed")); - assert!(message.contains("cleanup failed")); - - let mut timed_uncleaned = FakeProcess { - observations: VecDeque::from([Err(io::Error::other("timed observe failed"))]), - termination: Some(Err(io::Error::other("timed cleanup failed"))), - }; - let outcome = wait_for_tree_with( - &mut timed_uncleaned, - Duration::from_secs(1), - FakeProcess::observe, - FakeProcess::terminate, - ); - let message = result_infrastructure_message(outcome.result); - assert!(message.contains("timed observe failed")); - assert!(message.contains("timed cleanup failed")); - - let mut untimed_cleaned = FakeProcess { - observations: VecDeque::from([Err(io::Error::other("untimed observe failed"))]), - termination: Some(Ok(successful_status())), - }; - let outcome = wait_for_tree_without_timeout_with(&mut untimed_cleaned, FakeProcess::observe, FakeProcess::terminate); - assert!(result_infrastructure_message(outcome.result).contains("untimed observe failed")); - - let mut completed = FakeProcess { - observations: VecDeque::from([Ok(None), Ok(Some(successful_status()))]), - termination: None, - }; - let outcome = wait_for_tree_without_timeout_with(&mut completed, FakeProcess::observe, FakeProcess::terminate); - let InvocationResult::Exited(status) = outcome.result else { - panic!("untimed waiting must return the completed status"); - }; - assert!(status.success()); + fn detached_output_recovery_uses_only_nonblocking_mutex_acquisition() { + let output = Arc::new(Mutex::new(CapturedOutput::Memory(b"partial".to_vec()))); + let held = output.lock().expect("the fresh capture mutex is available"); + let mut failure = None; + let mut captured = take_reader_output(&output, "stdout", false, &mut failure); + assert!(output_bytes(&mut captured).is_empty()); + assert!(failure.is_some_and(|message| message.contains("partial output could not be recovered"))); + drop(held); } #[test] - fn timed_process_waiting_covers_completion_and_failed_termination() { - let mut completed = FakeProcess { - observations: VecDeque::from([Ok(Some(successful_status()))]), - termination: None, - }; - let outcome = wait_for_tree_with(&mut completed, Duration::from_secs(1), FakeProcess::observe, FakeProcess::terminate); - let InvocationResult::Exited(status) = outcome.result else { - panic!("a completed process must retain its exit status"); - }; - assert!(status.success()); - - let mut timed_out = FakeProcess { - observations: VecDeque::from([Ok(None)]), - termination: Some(Ok(successful_status())), - }; - let outcome = wait_for_tree_with(&mut timed_out, Duration::ZERO, FakeProcess::observe, FakeProcess::terminate); - assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())); + fn poisoned_capture_buffers_are_recovered_in_joined_and_detached_modes() { + fn poisoned_output() -> Arc> { + let output = Arc::new(Mutex::new(CapturedOutput::Memory(b"poisoned".to_vec()))); + let poisoned = Arc::clone(&output); + let _panic = thread::spawn(move || { + let _guard = poisoned.lock().expect("the fresh mutex is available"); + panic!("poison output"); + }) + .join(); + output + } - let mut uncleaned = FakeProcess { - observations: VecDeque::from([Ok(None)]), - termination: Some(Err(io::Error::other("termination failed"))), - }; - let outcome = wait_for_tree_with(&mut uncleaned, Duration::ZERO, FakeProcess::observe, FakeProcess::terminate); - assert!(result_infrastructure_message(outcome.result).contains("termination failed")); + for thread_finished in [true, false] { + let output = poisoned_output(); + let mut failure = None; + let mut captured = take_reader_output(&output, "stdout", thread_finished, &mut failure); + assert_eq!(output_bytes(&mut captured), b"poisoned"); + assert!(failure.is_some_and(|message| message.contains("capture buffer was poisoned"))); + } } #[test] - #[cfg_attr(miri, ignore = "spawns a contained child process")] - fn process_tree_control_delegates_real_observation_and_termination() { - let mut command = Command::new("rustc"); - let _ = command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); - let mut tree = spawn_tree(command).expect("spawn contained rustc probe"); - let outcome = wait_for_tree(&mut tree, Duration::from_secs(2)); - let InvocationResult::Exited(status) = outcome.result else { - panic!("the contained rustc probe must exit before its deadline"); + fn worker_panics_and_disconnects_become_infrastructure_outcomes() { + let plan = Plan { + invocations: vec![Invocation { + label: Some("panic-probe".to_owned()), + argv: vec![WORKER_PANIC_TEST_PROGRAM.to_owned()], + work_dir: None, + }], }; - assert!(status.success()); + let code = execute_parallel(&plan, false, NonZeroUsize::new(2).expect("literal two is nonzero"), None) + .expect("worker panic is represented as an outcome"); + assert_eq!(code, ExitCode::from(2)); - let mut tree = spawn_tree(sleeping_test_command()).expect("spawn contained sleep probe"); - let outcome = wait_for_tree(&mut tree, Duration::ZERO); - assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())); - } + let (sender, receiver) = mpsc::channel::(); + drop(sender); + let outcome = wait_for_worker(&mut vec![RunningWorker { + index: 4, + receiver, + thread: thread::spawn(|| {}), + }]) + .expect("disconnection is observable"); + assert_eq!(outcome.index, 4); + assert!(result_infrastructure_message(outcome.outcome.result).contains("without reporting")); - #[test] - #[cfg_attr(miri, ignore = "spawns and captures a contained child process")] - fn captured_contained_process_delegates_pipes_and_waiting() { - let mut outcome = run_captured_with_spawner(&invocation(&["rustc", "--version"]), Some(Duration::from_secs(2)), |command| { - spawn_tree(command).map(CapturedProcess::Contained) - }); - let InvocationResult::Exited(status) = outcome.result else { - panic!("the captured contained rustc probe must exit"); - }; - assert!(status.success()); - assert!( - String::from_utf8(output_bytes(&mut outcome.stdout)) - .expect("rustc output is UTF-8") - .contains("rustc") - ); + let (sender, receiver) = mpsc::channel(); + let outcome = wait_for_worker(&mut vec![RunningWorker { + index: 5, + receiver, + thread: thread::spawn(move || { + sender + .send(BufferedOutcome::infrastructure("premature".to_owned())) + .expect("the receiver remains alive"); + panic!("panic after report"); + }), + }]) + .expect("reported panic is observable"); + assert!(result_infrastructure_message(outcome.outcome.result).contains("panic after report")); + + let (sender, receiver) = mpsc::channel::(); + drop(sender); + let outcome = wait_for_worker(&mut vec![RunningWorker { + index: 6, + receiver, + thread: thread::spawn(|| panic!("panic before report")), + }]) + .expect("unreported panic is observable"); + assert!(result_infrastructure_message(outcome.outcome.result).contains("panic before report")); } #[test] - #[cfg_attr(miri, ignore = "spawns ordinary and contained child processes")] - fn captured_runner_reports_stream_setup_failures() { - for (label, expected) in [ - ("__cargo_each_missing_stdout", "failed to capture child stdout"), - ("__cargo_each_stdout_reader_failure", "injected stdout reader failure"), - ("__cargo_each_missing_stderr", "failed to capture child stderr"), - ("__cargo_each_stderr_reader_failure", "injected stderr reader failure"), - ("__cargo_each_wait_failure", "injected child wait failure"), - ] { - let outcome = run_captured(&labelled_invocation(label, &["rustc", "--version"]), None); - assert!(infrastructure_message(outcome).contains(expected), "{label}"); - } + fn worker_spawn_errors_are_local_and_scheduler_visible() { + let error = + spawn_worker(0, invocation(&[WORKER_SPAWN_ERROR_TEST_PROGRAM]), None).expect_err("the local seam rejects only its sentinel"); + assert!(error.to_string().contains("injected worker spawn failure")); - for (label, expected) in [ - ("__cargo_each_missing_stdout", "failed to capture child stdout"), - ("__cargo_each_stdout_reader_failure", "injected stdout reader failure"), - ("__cargo_each_missing_stderr", "failed to capture child stderr"), - ("__cargo_each_stderr_reader_failure", "injected stderr reader failure"), - ] { - let outcome = run_captured_with_spawner( - &labelled_invocation(label, &["rustc", "--version"]), - Some(Duration::from_secs(1)), - spawn_ordinary_capture, - ); - assert!(infrastructure_message(outcome).contains(expected), "contained {label}"); - } + let worker = spawn_worker(1, invocation(&["rustc", "--version"]), None).expect("ordinary program launches"); + let outcome = wait_for_worker(&mut vec![worker]).expect("ordinary worker reports"); + assert_eq!(outcome.index, 1); + assert!(!outcome.outcome.result.failed()); } #[test] - #[cfg_attr(miri, ignore = "spawns ordinary and contained child processes")] - fn captured_process_rejects_mismatched_timeout_modes() { - let mut ordinary_command = Command::new("rustc"); - let _ = ordinary_command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); - let mut ordinary = CapturedProcess::Ordinary(Some(ordinary_command.spawn().expect("spawn ordinary rustc"))); - let mut stdout = spawn_output_reader(io::empty(), "ordinary-empty-stdout").expect("spawn empty stdout reader"); - let mut stderr = spawn_output_reader(io::empty(), "ordinary-empty-stderr").expect("spawn empty stderr reader"); - assert_eq!(ordinary.drain_boundary(), "ordinary process tree"); - assert!( - result_infrastructure_message(ordinary.wait(Some(Duration::from_secs(1)), None, &mut stdout, &mut stderr).result) - .contains("did not match timeout configuration") - ); - let first_wait = ordinary.wait(None, None, &mut stdout, &mut stderr); - let InvocationResult::Exited(status) = first_wait.result else { - panic!("the ordinary child must be reaped by the matching wait mode"); - }; - assert!(status.success()); - assert!( - result_infrastructure_message(ordinary.wait(None, None, &mut stdout, &mut stderr).result) - .contains("already reaped or detached") + fn helper_diagnostics_preserve_cleanup_and_reaper_context() { + assert_eq!(with_cleanup_failure("primary".to_owned(), &Ok::<_, io::Error>(())), "primary"); + assert_eq!( + with_cleanup_failure("primary".to_owned(), &Err::<(), _>(io::Error::other("cleanup"))), + "primary; process-group cleanup also failed: cleanup" ); + assert!(with_reaper_handoff("deadline", &Ok(())).contains("detached reaper")); + assert!(with_reaper_handoff("deadline", &Err(io::Error::other("thread unavailable"))).contains("thread unavailable")); + assert_eq!(panic_description(&"borrowed panic"), "borrowed panic"); + assert_eq!(panic_description(&"owned panic".to_owned()), "owned panic"); + assert_eq!(panic_description(&7_u8), "non-string panic payload"); + } - let mut contained = CapturedProcess::Contained(spawn_tree(sleeping_test_command()).expect("spawn contained sleep probe")); - assert_eq!(contained.drain_boundary(), "contained process tree"); - assert!( - result_infrastructure_message(contained.wait(None, None, &mut stdout, &mut stderr).result) - .contains("did not match timeout configuration") - ); - let outcome = contained.wait(Some(Duration::ZERO), None, &mut stdout, &mut stderr); - assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())); - assert!( - contained.terminate_bounded().is_err(), - "a fabricated successful second termination must not be accepted" - ); + #[test] + fn tree_outcome_retains_the_invocation_result() { + let outcome = TreeOutcome::new(InvocationResult::Infrastructure("outcome".to_owned())); + assert_eq!(result_infrastructure_message(outcome.result), "outcome"); } } diff --git a/crates/cargo-each/tests/cli.rs b/crates/cargo-each/tests/cli.rs index 836ac6a01..14146db3f 100644 --- a/crates/cargo-each/tests/cli.rs +++ b/crates/cargo-each/tests/cli.rs @@ -111,13 +111,6 @@ fn each(manifest: &Path) -> Command { cmd } -const TIMEOUT_REFUSAL: &str = - "timeout requires sealed process-tree containment, but this host only provides best-effort containment; the child was not started"; - -fn is_timeout_refusal(output: &std::process::Output) -> bool { - String::from_utf8_lossy(&output.stderr).contains(TIMEOUT_REFUSAL) -} - fn rust_version_fixture(root_floor: Option<&str>, members: &[(&str, Option<&str>)]) -> (TempDir, PathBuf) { let tmp = tempfile::tempdir().expect("tempdir"); let root = tmp.path(); @@ -1617,16 +1610,9 @@ fn sequential_timeout_fail_fast_does_not_run_later_members() { .arg(&later_marker) .output() .expect("run cargo-each timeout fail-fast"); - if is_timeout_refusal(&output) { - assert!(!later_marker.exists(), "pre-spawn refusal must not launch any member"); - } else { - assert_eq!(output.status.code(), Some(1), "stderr: {}", String::from_utf8_lossy(&output.stderr)); - assert!(String::from_utf8_lossy(&output.stderr).contains("timed out after 50ms")); - } - assert!( - !later_marker.exists(), - "fail-fast or pre-spawn refusal must not launch the later member" - ); + assert_eq!(output.status.code(), Some(1), "stderr: {}", String::from_utf8_lossy(&output.stderr)); + assert!(String::from_utf8_lossy(&output.stderr).contains("timed out after 50ms")); + assert!(!later_marker.exists(), "fail-fast must not launch the later member"); } #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] @@ -1642,24 +1628,17 @@ fn sequential_timeout_keep_going_runs_later_members() { .arg(&later_marker) .output() .expect("run cargo-each timeout keep-going"); - if is_timeout_refusal(&output) { - assert!( - !later_marker.exists(), - "unsealed containment must refuse every timed child before spawn" - ); - } else { - assert_eq!(output.status.code(), Some(1), "stderr: {}", String::from_utf8_lossy(&output.stderr)); - assert!(String::from_utf8_lossy(&output.stderr).contains("timed out after 1s")); - assert!( - later_marker.exists(), - "--keep-going must launch the member after a timed-out invocation" - ); - } + assert_eq!(output.status.code(), Some(1), "stderr: {}", String::from_utf8_lossy(&output.stderr)); + assert!(String::from_utf8_lossy(&output.stderr).contains("timed out after 1s")); + assert!( + later_marker.exists(), + "--keep-going must launch the member after a timed-out invocation" + ); } #[cfg_attr(miri, ignore = "spawns the cargo-each binary and cargo subprocesses; miri supports neither")] #[test] -fn timeout_terminates_the_complete_process_tree() { +fn timeout_kills_an_ordinary_descendant_in_the_process_group() { let (tmp, manifest) = fixture(); let probe = compile_execution_probe(tmp.path()); let marker = tmp.path().join("grandchild-survived"); @@ -1670,13 +1649,11 @@ fn timeout_terminates_the_complete_process_tree() { .arg(&marker) .output() .expect("run cargo-each tree timeout"); - if !is_timeout_refusal(&output) { - assert_eq!(output.status.code(), Some(1), "stderr: {}", String::from_utf8_lossy(&output.stderr)); - assert!(String::from_utf8_lossy(&output.stderr).contains("timed out after 50ms")); - } + assert_eq!(output.status.code(), Some(1), "stderr: {}", String::from_utf8_lossy(&output.stderr)); + assert!(String::from_utf8_lossy(&output.stderr).contains("timed out after 50ms")); std::thread::sleep(std::time::Duration::from_millis(700)); assert!( !marker.exists(), - "a timed-out grandchild must be terminated, and an unsupported timed child must never start" + "a timed-out ordinary descendant must be terminated with its process group" ); } diff --git a/crates/cargo-gamma-process/docs/DESIGN.md b/crates/cargo-gamma-process/docs/DESIGN.md index afad4eb7b..96f351422 100644 --- a/crates/cargo-gamma-process/docs/DESIGN.md +++ b/crates/cargo-gamma-process/docs/DESIGN.md @@ -41,12 +41,9 @@ therefore covers the complete descendant tree. with the operating-system error so the caller can classify a transient resource-related spawn failure, back off, and retry; permanent launch failures are propagated. Success yields a distinct bundle coupling the child to its - boundary. Before creating that child, spawn also ensures the process-wide - detached reaper thread is running. A reaper thread-start failure therefore - returns the unchanged preparation before a repository-controlled process - exists. Adoption consumes the successful bundle, so a successful launch - cannot be reused to create an earlier sibling awaiting adoption; abandoning - the bundle before adoption terminates and reaps the child. + boundary. Adoption consumes that bundle, so a successful launch cannot be + reused to create an earlier sibling awaiting adoption; abandoning the bundle + before adoption terminates and reaps the child. - The contained `output` convenience mirrors `Command::output`: it disconnects stdin and captures stdout and stderr. Both pipes are drained concurrently while the child runs, avoiding pipe-capacity deadlocks. When the leader exits, @@ -71,38 +68,6 @@ therefore covers the complete descendant tree. nested assignment, the spawn is rejected: an inherited job does not provide a handle through which this process can later terminate the child's descendants. Failure to create a job is likewise a refusal rather than a degraded launch. -- Callers with an external deadline use bounded termination. It signals the - same process-tree boundary as ordinary termination but polls the leader only - for the caller-provided grace. A leader that remains running after a failed - kill is transferred to a shared detached reaper rather than handed to an - indefinite `wait` or Drop path. The reaper polls all retained leaders so one - survivor cannot block collection of the others, remains alive while its queue - is empty, and accepts each handle only after its thread is known to exist. A - handoff rechecks that state under the queue lock and retries startup if a - previous loop exited between the readiness check and transfer. - Callers handing over children created outside `PreparedCommand` can preflight - the same durable thread; if a direct handoff must start it and startup fails, - the failure returns ownership of the unqueued child. A caller that cannot - retain the child locally without making its own Drop path blocking transfers - it to a separate process-wide retry queue after explicitly recovering it from - `ReapFailure`. That queue owns the handle without claiming a live reaper: the - current loop drains it when available, or the next successful startup drains - it after an earlier startup failure. `ProcessTree::terminate_bounded` uses - this fallback and never restores a live child into itself, so its later Drop - remains bounded. The containment handles remain owned until the - `ProcessTree` itself is dropped. An error while polling the leader follows the - same handoff before the observation error is returned, because an observation - failure does not prove the child was reaped. If the detached reaper itself - later receives an interrupted observation, it keeps the child queued and - retries. Any other observation error emits a warning to stderr and - permanently stops tracking that child. Warning formatting happens while the - queue is locked, but the fallible stderr write happens after releasing the - lock and its error is discarded. Every loop exit or unwind moves retained - active children back to the retry queue, clears the running state, and - notifies waiters. If any child remains, the lifecycle guard immediately - starts a replacement reaper; a handoff accepted while diagnostics were - outside the lock therefore never depends on an unrelated later caller. The - released handle may leave a zombie on Unix until this process exits. - Sealed containment uses a boundary that descendants cannot leave. A host that offers no sealed boundary at all silently uses best-effort process-group containment for an unmetered launch; absence of a warning does not establish diff --git a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md index fee493a33..7b690b871 100644 --- a/crates/cargo-gamma-process/docs/IMPLEMENTATION.md +++ b/crates/cargo-gamma-process/docs/IMPLEMENTATION.md @@ -8,10 +8,7 @@ This guide records lifecycle mechanics behind [`DESIGN.md`](DESIGN.md). returns the same preparation in `SpawnFailure`; a successful spawn returns `SpawnedCommand`, which owns the child and containment until `ProcessTree::adopt` consumes it. Dropping the successful pre-adoption state terminates and reaps the -child. Before calling the operating-system spawn, `PreparedCommand::spawn` -ensures the process-wide detached reaper is running. A failure to create that -thread therefore returns `SpawnFailure` with the preparation intact and no child -to recover. +child. ## Output capture @@ -19,36 +16,6 @@ The contained output path takes stdout and stderr exactly once and drains them concurrently. It sweeps descendants after the leader exits so inherited write ends do not keep readers open indefinitely. -## Bounded termination - -`ProcessTree::terminate_bounded` requests the same subtree and leader kills as -ordinary termination, then polls `try_wait` until a caller-provided grace -expires. It never follows a failed kill with blocking `wait`: at the deadline -the leader handle moves to the shared detached reaper, the `ProcessTree` no -longer owns a child that Drop could wait for, and the original cleanup failure -is retained in the returned error. The reaper polls every retained child -without blocking on one leader and waits on a condition variable when its queue -is empty. It is durable: repository-controlled children are created only after -the thread exists, and direct `reap_later` startup failures return an unqueued -child in `ReapFailure` rather than abandoning ownership. Callers explicitly -recover that child with `ReapFailure::into_parts`; those whose local Drop path -would block transfer it to the process-wide retry queue. The retry queue is a -distinct owner, not evidence that a reaper is running. A live loop drains it -after notification, or a later successful startup drains it after startup -failure. Handoff rechecks the running state while holding the queue lock and -retries startup after a raced loop exit. A `try_wait` error also -transfers the still-owned leader handle before returning the observation error; -only a successful `Some(status)` proves that no later reaping is required. If a -later `try_wait` in the detached reaper is interrupted, the child remains queued -for another attempt. Any other observation error writes a warning to stderr and -permanently releases that child handle. The write is fallible, its error is -discarded, and it occurs outside the queue mutex. A lifecycle guard clears the -running state, returns active handles to the retry queue, and wakes readiness -waiters whenever the loop exits or unwinds. When retained handles remain, the -guard immediately starts a replacement, closing the window in which a handoff -could observe the old loop as running while diagnostics were outside the lock. -On Unix, the child may remain a zombie until this process exits. - ## Platform composition Unix launch preparation holds the interrupt spawn window only across child diff --git a/crates/cargo-gamma-process/src/faults.rs b/crates/cargo-gamma-process/src/faults.rs index f615d71c5..e3df9a730 100644 --- a/crates/cargo-gamma-process/src/faults.rs +++ b/crates/cargo-gamma-process/src/faults.rs @@ -24,19 +24,6 @@ pub enum Fault { /// Terminating a contained subtree reports a cleanup failure. Terminate, - - /// The direct leader and surrounding subtree refuse the termination - /// signal, leaving the leader running. - Kill, - - /// The termination request reports success without signalling the leader. - Linger, - - /// Observing the leader after termination returns an operating-system error. - Observe, - - /// Starting the shared detached child reaper is refused. - ReaperStart, } /// Arms `fault` on this thread until the returned value is dropped. @@ -51,12 +38,6 @@ pub fn arm_late(fault: Fault, delay: Duration) -> Armed { ripe_at(fault, Instant::now() + delay) } -/// Reports whether the process-wide reaper or its retry queue owns `id`. -#[must_use] -pub fn reaper_owns(id: u32) -> bool { - crate::process_tree::reaper_contains(id) -} - fn ripe_at(fault: Fault, ripe: Instant) -> Armed { ARMED.with_borrow_mut(|armed| armed.push((fault, ripe))); @@ -120,10 +101,6 @@ mod tests { assert!(!fired(Fault::Boundary)); assert!(!fired(Fault::Window)); assert!(!fired(Fault::Terminate)); - assert!(!fired(Fault::Kill)); - assert!(!fired(Fault::Linger)); - assert!(!fired(Fault::Observe)); - assert!(!fired(Fault::ReaperStart)); } #[test] diff --git a/crates/cargo-gamma-process/src/lib.rs b/crates/cargo-gamma-process/src/lib.rs index 96d76e763..be01a7070 100644 --- a/crates/cargo-gamma-process/src/lib.rs +++ b/crates/cargo-gamma-process/src/lib.rs @@ -66,28 +66,18 @@ //! [`Command::output`](std::process::Command::output). It drains stdout and stderr concurrently, //! then sweeps descendants before waiting for inherited pipe handles to close. //! -//! Bounded termination can transfer a still-running leader to a shared detached reaper. A failed -//! handoff moves the recovered child into a separate process-wide retry queue rather than restoring -//! it to a blocking Drop path. The reaper writes observation warnings outside its global queue lock -//! through a fallible stderr path, and any loop exit or unwind preserves retained handles for a -//! later replacement. -//! //! A terminal delivers `Ctrl-C` to the whole foreground process group, so a child sharing this //! process's group dies with it automatically while a child leading its own group does not. Windows //! normally preserves that guarantee through a dedicated job that dies with its last handle. Unix //! installs explicit interruption handling through `cargo-gamma-unsafe`. -pub use cargo_gamma_unsafe::pipe::InterruptiblePipe; pub use cargo_gamma_unsafe::{PlatformError, Situation, support}; #[doc(inline)] pub use memory_request::MemoryRequest; #[doc(inline)] pub use memory_usage::MemoryUsage; #[doc(inline)] -pub use process_tree::{ - OutputError, PreparedCommand, ProcessTree, ReapFailure, SpawnFailure, SpawnedCommand, capacity, containment, ensure_reaper, output, - prepare, reap_later, retain_for_reaper_retry, -}; +pub use process_tree::{OutputError, PreparedCommand, ProcessTree, SpawnFailure, SpawnedCommand, capacity, containment, output, prepare}; mod memory_request; mod memory_usage; diff --git a/crates/cargo-gamma-process/src/process_tree.rs b/crates/cargo-gamma-process/src/process_tree.rs index 415fd339e..e3a41d310 100644 --- a/crates/cargo-gamma-process/src/process_tree.rs +++ b/crates/cargo-gamma-process/src/process_tree.rs @@ -8,9 +8,8 @@ use core::time::Duration; use std::io; use std::process::{Child, ChildStderr, ChildStdout, Command, ExitStatus, Output, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::{Arc, Condvar, Mutex, MutexGuard}; +use std::sync::{Arc, Mutex}; use std::thread::{self, JoinHandle}; -use std::time::Instant; #[cfg(target_os = "linux")] use cargo_gamma_unsafe::cgroup::Cgroup; @@ -26,269 +25,6 @@ use cargo_gamma_unsafe::{PlatformError, Situation}; use crate::faults; use crate::{MemoryRequest, MemoryUsage}; -const REAPER_PAUSE: Duration = Duration::from_millis(25); - -#[derive(Debug, Default)] -struct ChildReaper { - children: Vec, - retry_children: Vec, - starting: bool, - running: bool, -} - -static CHILD_REAPER: Mutex = Mutex::new(ChildReaper { - children: Vec::new(), - retry_children: Vec::new(), - starting: false, - running: false, -}); -static CHILD_REAPER_READY: Condvar = Condvar::new(); - -/// Ensures the shared detached child reaper is ready before a child is spawned. -/// -/// The reaper is process-wide and remains alive after its queue becomes empty, -/// sleeping on a condition variable until another child is handed off. If its -/// loop exits or unwinds, readiness is cleared and a later call starts a -/// replacement. -/// -/// # Errors -/// -/// Returns the thread creation error when the shared reaper could not be -/// started. No child has been created or transferred by this operation. -pub fn ensure_reaper() -> io::Result<()> { - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - while reaper.starting { - reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); - } - if reaper.running { - return Ok(()); - } - reaper.starting = true; - drop(reaper); - - #[cfg(any(test, feature = "fault-injection"))] - let spawned = if faults::fired(faults::Fault::ReaperStart) { - Err(io::Error::other("detached child reaper thread start failed as requested by a test")) - } else { - thread::Builder::new() - .name("cargo-gamma-child-reaper".to_owned()) - .spawn(child_reaper_loop) - }; - #[cfg(not(any(test, feature = "fault-injection")))] - let spawned = thread::Builder::new() - .name("cargo-gamma-child-reaper".to_owned()) - .spawn(child_reaper_loop); - - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - reaper.starting = false; - match spawned { - Ok(thread) => { - reaper.running = true; - CHILD_REAPER_READY.notify_all(); - drop(reaper); - drop(thread); - Ok(()) - } - Err(error) => { - CHILD_REAPER_READY.notify_all(); - Err(error) - } - } -} - -/// Transfers a live child handle to the shared detached reaper. -/// -/// The reaper polls every retained child rather than blocking on one, so a -/// leader that survives termination cannot prevent unrelated leaders from -/// being collected. Interrupted observations are retried. Any other observation -/// error emits a best-effort warning to stderr outside the global queue lock and -/// permanently stops tracking that child. Diagnostic write failures are -/// ignored. On Unix, the child may then remain a zombie until this process -/// exits. The handoff retries if the previous reaper loop exits while readiness -/// is being checked. -/// -/// # Errors -/// -/// Returns [`ReapFailure`] when the shared reaper could not be started. The -/// failure retains the child handle so the caller can recover ownership. -pub fn reap_later(child: Child) -> Result<(), ReapFailure> { - #[cfg(any(test, feature = "fault-injection"))] - if faults::fired(faults::Fault::ReaperStart) { - return Err(ReapFailure { - cause: io::Error::other("detached child reaper thread start failed as requested by a test"), - child, - }); - } - - loop { - if let Err(cause) = ensure_reaper() { - return Err(ReapFailure { cause, child }); - } - - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - if reaper.running { - reaper.children.push(child); - CHILD_REAPER_READY.notify_one(); - return Ok(()); - } - // The previous loop exited between ensure_reaper's observation and - // this handoff. Retry so the child is queued only behind a live loop. - } -} - -/// Retains a child from a failed [`reap_later`] handoff for a later retry. -/// -/// The process-wide retry queue owns the handle without claiming that a reaper -/// thread is live. A running reaper drains it when notified; otherwise the next -/// successfully started reaper drains it. This function never waits for the -/// child. It is intended for callers that have explicitly recovered the child -/// with [`ReapFailure::into_parts`]. -pub fn retain_for_reaper_retry(child: Child) { - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - reaper.retry_children.push(child); - CHILD_REAPER_READY.notify_one(); -} - -/// A failed detached-reaper handoff that retains ownership of the child. -#[derive(Debug)] -pub struct ReapFailure { - cause: io::Error, - child: Child, -} - -impl ReapFailure { - /// Borrows the operating-system error that prevented the reaper from starting. - #[must_use] - pub const fn cause(&self) -> &io::Error { - &self.cause - } - - /// Recovers the failure and the child that was not transferred. - #[must_use] - pub fn into_parts(self) -> (io::Error, Child) { - (self.cause, self.child) - } -} - -impl fmt::Display for ReapFailure { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - self.cause.fmt(f) - } -} - -impl std::error::Error for ReapFailure { - fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { - Some(&self.cause) - } -} - -fn child_reaper_loop() { - let mut startup = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - while startup.starting { - startup = CHILD_REAPER_READY.wait(startup).unwrap_or_else(std::sync::PoisonError::into_inner); - } - drop(startup); - - let _running = ReaperRunningGuard; - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - loop { - while reaper.children.is_empty() && reaper.retry_children.is_empty() { - reaper = CHILD_REAPER_READY.wait(reaper).unwrap_or_else(std::sync::PoisonError::into_inner); - } - let mut retries = std::mem::take(&mut reaper.retry_children); - reaper.children.append(&mut retries); - - let mut warnings = Vec::new(); - reaper.children.retain_mut(|child| { - let id = child.id(); - retain_reaper_child(id, child.try_wait(), &mut warnings) - }); - reaper = report_reaper_warnings(reaper, warnings, |warning| { - emit_reaper_warning_to(io::stderr().lock(), warning); - }); - if reaper.children.is_empty() { - continue; - } - - let (next, _timeout) = CHILD_REAPER_READY - .wait_timeout(reaper, REAPER_PAUSE) - .unwrap_or_else(std::sync::PoisonError::into_inner); - reaper = next; - } -} - -struct ReaperRunningGuard; - -impl Drop for ReaperRunningGuard { - fn drop(&mut self) { - let mut reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - let mut active = std::mem::take(&mut reaper.children); - reaper.retry_children.append(&mut active); - reaper.running = false; - let restart = !reaper.retry_children.is_empty(); - CHILD_REAPER_READY.notify_all(); - drop(reaper); - - // A handoff can race an unwind after warning reporting released the - // queue lock but before this guard cleared `running`. Restart here so - // a child accepted during that window never waits for an unrelated - // future caller to revive the queue. - if restart { - let _ignored = ensure_reaper(); - } - } -} - -fn retain_reaper_child(id: u32, observation: io::Result>, warnings: &mut Vec) -> bool { - match observation { - Ok(None) => true, - Ok(Some(_status)) => false, - Err(error) if error.kind() == io::ErrorKind::Interrupted => true, - Err(error) => { - warnings.push(format!( - "warning: detached child reaper stopped tracking process {id} after observation failed: {error}" - )); - false - } - } -} - -fn report_reaper_warnings( - reaper: MutexGuard<'static, ChildReaper>, - warnings: Vec, - mut report: impl FnMut(&str), -) -> MutexGuard<'static, ChildReaper> { - let mut warnings = warnings.into_iter(); - let Some(first) = warnings.next() else { - return reaper; - }; - drop(reaper); - report(&first); - for warning in warnings { - report(&warning); - } - CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner) -} - -fn emit_reaper_warning_to(mut destination: impl io::Write, message: &str) { - let _ignored = writeln!(destination, "{message}"); -} - -#[cfg(any(test, feature = "fault-injection"))] -pub(crate) fn reaper_contains(id: u32) -> bool { - let reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - reaper - .children - .iter() - .chain(reaper.retry_children.iter()) - .any(|child| child.id() == id) -} - -#[cfg(test)] -fn reaper_running() -> bool { - CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner).running -} - /// How many concurrent child subtrees can be watched for terminal interruption. #[cfg(unix)] #[must_use] @@ -533,18 +269,9 @@ impl PreparedCommand { /// /// # Errors /// - /// Returns the thread creation error when the durable detached reaper could - /// not be prepared, or whatever [`Command::spawn`] returns. In either case, - /// the child does not exist, so there is nothing here to clean up and the - /// returned preparation remains valid for another attempt. + /// Returns whatever [`Command::spawn`] returns: the child does not exist, so there is nothing + /// here to clean up and the returned preparation remains valid for another attempt. pub fn spawn(mut self) -> Result { - if let Err(cause) = ensure_reaper() { - return Err(SpawnFailure { - cause, - prepared: Box::new(self), - }); - } - match self.command.spawn() { Ok(child) => { let Self { command: _spawned, guard } = self; @@ -1571,98 +1298,6 @@ impl ProcessTree { Ok(reaped) } - /// Requests termination, then waits no longer than `grace` for the leader - /// to exit. - /// - /// Unlike [`Self::terminate`], this method never performs a blocking - /// [`Child::wait`] after signalling. If the leader remains alive at the - /// deadline, its handle is transferred to the shared detached reaper. A - /// successful handoff leaves this process tree without a child for [`Drop`] - /// to wait on. The surrounding cgroup or job handle remains owned by `self` - /// and is released normally when the process tree is dropped. - /// - /// # Errors - /// - /// Returns the observation error if `try_wait` fails, the termination - /// error if the leader exits after signalling but cleanup had failed, or a - /// timed-out error (including an earlier termination error, when present) - /// if the leader is still running after `grace`. If the detached reaper - /// cannot accept the child, the returned error includes that failure and - /// the child handle moves to the process-wide retry queue. This process tree - /// remains empty so its [`Drop`] path cannot wait for the live child. - pub fn terminate_bounded(&mut self, grace: Duration) -> io::Result { - let started = Instant::now(); - let mut child = self - .child - .take() - .ok_or_else(|| io::Error::other("the subtree leader was already reaped"))?; - - #[cfg(any(test, feature = "fault-injection"))] - let killed = if faults::fired(faults::Fault::Kill) { - Err(io::Error::other("subtree termination was refused as requested by a test")) - } else if faults::fired(faults::Fault::Linger) { - Ok(()) - } else { - self.kill(&mut child) - }; - #[cfg(not(any(test, feature = "fault-injection")))] - let killed = self.kill(&mut child); - let mut kill_error = killed.err(); - self.release(); - - loop { - #[cfg(any(test, feature = "fault-injection"))] - let observed = if faults::fired(faults::Fault::Observe) { - Err(io::Error::other("subtree observation failed as requested by a test")) - } else { - child.try_wait() - }; - #[cfg(not(any(test, feature = "fault-injection")))] - let observed = child.try_wait(); - - match observed { - Ok(Some(status)) => { - if let Some(error) = kill_error.take() { - return Err(error); - } - - #[cfg(any(test, feature = "fault-injection"))] - if faults::fired(faults::Fault::Terminate) { - return Err(io::Error::other("subtree termination failed as requested by a test")); - } - - return Ok(status); - } - Ok(None) => {} - Err(error) => { - let kind = error.kind(); - let mut message = error.to_string(); - if let Err(failure) = reap_later(child) { - let (reaper, child) = failure.into_parts(); - retain_for_reaper_retry(child); - message = format!("{message}; the detached child reaper could not be started: {reaper}"); - } - return Err(io::Error::new(kind, message)); - } - } - - let Some(remaining) = grace.checked_sub(started.elapsed()) else { - let deadline_error = format!("subtree leader did not exit within {} ms after termination", grace.as_millis()); - let (kind, mut message) = kill_error.take().map_or_else( - || (io::ErrorKind::TimedOut, deadline_error.clone()), - |error| (error.kind(), format!("{error}; {deadline_error}")), - ); - if let Err(failure) = reap_later(child) { - let (error, child) = failure.into_parts(); - retain_for_reaper_retry(child); - message = format!("{message}; the detached child reaper could not be started: {error}"); - } - return Err(io::Error::new(kind, message)); - }; - thread::sleep(remaining.min(Duration::from_millis(10))); - } - } - /// Ends descendants while their leader's process-group id is still reserved. /// /// An exited leader can leave servers and inherited pipe handles behind. This private @@ -1950,33 +1585,19 @@ mod tests { use core::cell::RefCell; #[cfg(unix)] use core::mem; + #[cfg(unix)] + use std::env; use std::error::Error as _; + use std::fs; #[cfg(unix)] use std::io::{BufRead as _, Write as _}; - use std::{env, fs}; + use std::time::Instant; use camino::Utf8Path; use super::*; use crate::testing; - static REAPER_TEST_LOCK: Mutex<()> = Mutex::new(()); - - struct FailingDiagnosticWriter { - attempted: Arc, - } - - impl io::Write for FailingDiagnosticWriter { - fn write(&mut self, _buf: &[u8]) -> io::Result { - self.attempted.store(true, Ordering::Release); - Err(io::Error::other("injected diagnostic failure")) - } - - fn flush(&mut self) -> io::Result<()> { - Ok(()) - } - } - struct PausedReader { reads: usize, ready: std::sync::mpsc::SyncSender<()>, @@ -2036,24 +1657,6 @@ mod tests { ); } - #[test] - fn detached_reaper_drops_unobservable_children() { - let mut warnings = Vec::new(); - assert!(retain_reaper_child(17, Ok(None), &mut warnings)); - assert!(retain_reaper_child( - 17, - Err(io::Error::new(io::ErrorKind::Interrupted, "wait interrupted")), - &mut warnings, - )); - assert!(!retain_reaper_child( - 17, - Err(io::Error::new(io::ErrorKind::InvalidInput, "invalid child handle")), - &mut warnings, - )); - assert_eq!(warnings.len(), 1); - assert!(warnings[0].contains("invalid child handle")); - } - struct FailingReader; impl io::Read for FailingReader { @@ -2814,401 +2417,6 @@ mod tests { assert!(!finished.exists(), "the grandchild kept working after the subtree was killed"); } - #[test] - fn bounded_termination_does_not_wait_forever_after_a_failed_kill() { - let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - let work = testing::workdir("gamma-bounded-termination"); - let finished = Utf8Path::from_path(work.path()) - .expect("the temporary path is UTF-8") - .join("finished"); - let mut command = Command::new(testing::helper_binary_path().as_std_path()); - let _ = command.args([ - testing::directive("sleep:250"), - testing::directive(format_args!("touch:{finished}")), - ]); - let prepared = prepare(command, MemoryRequest::default()).expect("containment"); - let spawned = prepared.spawn().expect("spawn"); - let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); - let leader = subtree.child.as_ref().expect("the adopted subtree owns its leader").id(); - let _failed_kill = faults::arm(faults::Fault::Kill); - - let started = Instant::now(); - let error = subtree - .terminate_bounded(Duration::from_millis(25)) - .expect_err("the injected failed kill must reach its deadline"); - - assert!( - started.elapsed() < Duration::from_millis(200), - "bounded termination exceeded its grace" - ); - assert!(error.to_string().contains("termination was refused"), "{error}"); - assert!(error.to_string().contains("did not exit within 25 ms"), "{error}"); - assert!(subtree.child.is_none(), "Drop must have no leader left to wait for"); - - thread::sleep(Duration::from_millis(350)); - assert!(finished.exists(), "the deliberately un-killed leader did not finish on its own"); - let deadline = Instant::now() + Duration::from_secs(1); - while reaper_contains(leader) && Instant::now() < deadline { - thread::sleep(Duration::from_millis(10)); - } - assert!( - !reaper_contains(leader), - "the detached reaper retained the naturally exited leader without reaping it" - ); - } - - #[test] - fn bounded_termination_reaps_a_killed_leader() { - let mut command = Command::new(testing::helper_binary_path().as_std_path()); - let _ = command.arg(testing::directive("sleep:30000")); - let prepared = prepare(command, MemoryRequest::default()).expect("containment"); - let spawned = prepared.spawn().expect("spawn"); - let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); - - let status = subtree - .terminate_bounded(Duration::from_secs(1)) - .expect("the killed leader exits within the grace"); - - assert!(!status.success(), "a killed leader must not report success"); - assert!(subtree.child.is_none(), "the leader was not reaped"); - assert!( - subtree.terminate_bounded(Duration::ZERO).is_err(), - "an already-reaped leader must fail" - ); - } - - #[test] - fn bounded_termination_preserves_a_failed_kill_after_natural_exit() { - let mut command = Command::new(testing::helper_binary_path().as_std_path()); - let _ = command.arg(testing::directive("sleep:50")); - let prepared = prepare(command, MemoryRequest::default()).expect("containment"); - let spawned = prepared.spawn().expect("spawn"); - let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); - let _failed_kill = faults::arm(faults::Fault::Kill); - - let error = subtree - .terminate_bounded(Duration::from_secs(1)) - .expect_err("natural exit must not hide the failed termination request"); - - assert!(error.to_string().contains("termination was refused"), "{error}"); - } - - #[test] - fn bounded_termination_reports_post_reap_cleanup_failure() { - let mut command = Command::new(testing::helper_binary_path().as_std_path()); - let _ = command.arg(testing::directive("sleep:30000")); - let prepared = prepare(command, MemoryRequest::default()).expect("containment"); - let spawned = prepared.spawn().expect("spawn"); - let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); - let _failed_cleanup = faults::arm(faults::Fault::Terminate); - - let error = subtree - .terminate_bounded(Duration::from_secs(1)) - .expect_err("the injected post-reap cleanup failure must be preserved"); - - assert!(error.to_string().contains("failed as requested by a test"), "{error}"); - } - - #[test] - fn bounded_termination_times_out_when_a_successful_signal_is_ignored() { - let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - let work = testing::workdir("gamma-bounded-linger"); - let finished = Utf8Path::from_path(work.path()) - .expect("the temporary path is UTF-8") - .join("finished"); - let mut command = Command::new(testing::helper_binary_path().as_std_path()); - let _ = command.args([ - testing::directive("sleep:250"), - testing::directive(format_args!("touch:{finished}")), - ]); - let prepared = prepare(command, MemoryRequest::default()).expect("containment"); - let spawned = prepared.spawn().expect("spawn"); - let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); - let _ignored_kill = faults::arm(faults::Fault::Linger); - - let error = subtree - .terminate_bounded(Duration::from_millis(25)) - .expect_err("an ignored successful signal must time out"); - - assert_eq!(error.kind(), io::ErrorKind::TimedOut); - assert!(error.to_string().contains("did not exit within 25 ms"), "{error}"); - thread::sleep(Duration::from_millis(350)); - assert!(finished.exists(), "the deliberately un-signalled leader did not finish"); - } - - #[test] - fn bounded_termination_hands_an_observation_error_to_the_reaper() { - let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - let mut command = Command::new(testing::helper_binary_path().as_std_path()); - let _ = command.arg(testing::directive("sleep:30000")); - let prepared = prepare(command, MemoryRequest::default()).expect("containment"); - let spawned = prepared.spawn().expect("spawn"); - let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); - let leader = subtree.child.as_ref().expect("the adopted subtree owns its leader").id(); - let _failed_observation = faults::arm(faults::Fault::Observe); - - let error = subtree - .terminate_bounded(Duration::from_secs(1)) - .expect_err("the injected observation failure must be reported"); - - assert!(error.to_string().contains("observation failed as requested"), "{error}"); - assert!(subtree.child.is_none(), "the failed observation must transfer leader ownership"); - wait_for_reaper_to_collect(leader, Duration::from_secs(2)); - } - - #[test] - fn bounded_termination_reaper_start_failures_keep_return_and_drop_bounded() { - let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - - for fail_observation in [false, true] { - let mut command = Command::new(testing::helper_binary_path().as_std_path()); - let _ = command.arg(testing::directive("sleep:350")); - let prepared = prepare(command, MemoryRequest::default()).expect("containment"); - let spawned = prepared.spawn().expect("spawn"); - let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); - let leader = subtree.child.as_ref().expect("the adopted subtree owns its leader").id(); - let _ignored_kill = faults::arm(faults::Fault::Linger); - let _failed_observation = fail_observation.then(|| faults::arm(faults::Fault::Observe)); - let _failed_reaper_start = faults::arm(faults::Fault::ReaperStart); - - let started = Instant::now(); - let error = subtree - .terminate_bounded(Duration::from_millis(25)) - .expect_err("the injected reaper start failure must be reported"); - assert!( - started.elapsed() < Duration::from_millis(200), - "bounded termination exceeded its grace after reaper startup failed" - ); - assert!(error.to_string().contains("reaper thread start failed"), "{error}"); - assert!(subtree.child.is_none(), "the live child was restored to the Drop path"); - assert!(reaper_contains(leader), "the retry queue did not retain the live child"); - - let drop_started = Instant::now(); - drop(subtree); - assert!( - drop_started.elapsed() < Duration::from_millis(200), - "ProcessTree::drop waited for the stubborn child" - ); - - wait_for_reaper_to_collect(leader, Duration::from_secs(2)); - } - } - - #[test] - fn shared_reaper_collects_out_of_order_and_remains_ready_after_idle() { - let _reaper_test = REAPER_TEST_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - - let long = spawn_reaper_probe(750); - let long_id = long.id(); - let short = spawn_reaper_probe(50); - let short_id = short.id(); - reap_later(long).expect("start the shared reaper"); - reap_later(short).expect("add a second child without starting another reaper"); - - wait_for_reaper_to_collect(short_id, Duration::from_secs(2)); - assert!( - reaper_contains(long_id), - "the shorter-lived later child must be collected before the earlier long-lived child" - ); - wait_for_reaper_to_collect(long_id, Duration::from_secs(2)); - assert!(reaper_running(), "the durable shared reaper stopped when its queue became empty"); - - let after_idle = spawn_reaper_probe(50); - let after_idle_id = after_idle.id(); - reap_later(after_idle).expect("hand a child to the idle shared reaper"); - wait_for_reaper_to_collect(after_idle_id, Duration::from_secs(2)); - } - - #[test] - fn reaper_diagnostics_release_the_lock_and_an_unwind_can_restart() { - isolated_run( - "a_reaper_diagnostic_releases_the_lock_and_an_unwind_can_restart", - "the reaper restarted after a diagnostic unwind", - ); - } - - #[test] - fn a_reaper_diagnostic_releases_the_lock_and_an_unwind_can_restart() { - if env::var_os(ISOLATED_CHILD).is_none() { - return; - } - - let diagnostic_attempted = Arc::new(AtomicBool::new(false)); - emit_reaper_warning_to( - FailingDiagnosticWriter { - attempted: Arc::clone(&diagnostic_attempted), - }, - "injected warning", - ); - assert!( - diagnostic_attempted.load(Ordering::Acquire), - "the fallible diagnostic path did not attempt the stderr write" - ); - - CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner).running = true; - let (reporting, reported) = std::sync::mpsc::sync_channel(0); - let (release, released) = std::sync::mpsc::sync_channel(0); - let blocked_reporter = thread::spawn(move || { - let _running = ReaperRunningGuard; - let reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - let _reaper = report_reaper_warnings(reaper, vec!["blocked warning".to_owned()], |_warning| { - reporting.send(()).expect("the lock probe is waiting"); - released.recv().expect("the lock probe releases the diagnostic"); - }); - }); - reported - .recv_timeout(Duration::from_secs(1)) - .expect("the diagnostic callback started"); - assert!( - CHILD_REAPER.try_lock().is_ok(), - "a blocked diagnostic writer retained the global reaper lock" - ); - release.send(()).expect("release the blocked diagnostic"); - blocked_reporter.join().expect("the diagnostic reporter exits"); - assert!(!reaper_running(), "loop exit did not reset the running state"); - - CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner).running = true; - let (accepted, accepted_child) = std::sync::mpsc::channel(); - let panicked = std::panic::catch_unwind(|| { - let _running = ReaperRunningGuard; - let reaper = CHILD_REAPER.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - let _reaper = report_reaper_warnings(reaper, vec!["panicking warning".to_owned()], |_warning| { - let child = spawn_reaper_probe(50); - let child_id = child.id(); - reap_later(child).expect("the handoff observes the reporting reaper as running"); - accepted.send(child_id).expect("the parent waits for the raced handoff"); - panic!("injected diagnostic panic"); - }); - }); - assert!(panicked.is_err(), "the diagnostic panic was not injected"); - assert!( - CHILD_REAPER.try_lock().is_ok(), - "diagnostic unwind left the global reaper lock unavailable" - ); - let child_id = accepted_child - .recv_timeout(Duration::from_secs(1)) - .expect("the warning callback accepted a child during the handoff window"); - wait_for_reaper_to_collect(child_id, Duration::from_secs(2)); - - println!("the reaper restarted after a diagnostic unwind"); - } - - #[test] - fn reaper_start_failure_preserves_child_and_preparation_ownership() { - isolated_run( - "a_reaper_start_failure_preserves_child_and_preparation_ownership", - "the failed reaper start preserved both owners", - ); - } - - #[test] - fn a_reaper_start_failure_preserves_child_and_preparation_ownership() { - if env::var_os(ISOLATED_CHILD).is_none() { - return; - } - - let child = Command::new(testing::helper_binary_path().as_std_path()) - .arg(testing::directive("sleep:1000")) - .stdin(Stdio::null()) - .stdout(Stdio::null()) - .stderr(Stdio::null()) - .spawn() - .expect("spawn isolated reaper probe"); - let child_id = child.id(); - let _failed_start = faults::arm(faults::Fault::ReaperStart); - let failure = reap_later(child).expect_err("the injected thread start failure must reject the handoff"); - assert!(!reaper_contains(child_id), "a child without a reaper must not enter the queue"); - assert!(failure.to_string().contains("thread start failed as requested"), "{failure}"); - assert!( - failure - .source() - .expect("the handoff failure retains its operating-system cause") - .to_string() - .contains("thread start failed as requested") - ); - let (error, child) = failure.into_parts(); - assert!(error.to_string().contains("thread start failed as requested"), "{error}"); - assert_eq!(child.id(), child_id, "the failed handoff returned a different child"); - retain_for_reaper_retry(child); - assert!(reaper_contains(child_id), "the retry queue did not take ownership"); - - let prepared = prepare(no_op_command(), MemoryRequest::default()).expect("containment"); - let _failed_start = faults::arm(faults::Fault::ReaperStart); - let failure = prepared - .spawn() - .expect_err("reaper startup must be proven before the child is spawned"); - assert!( - failure.cause().to_string().contains("thread start failed as requested"), - "{}", - failure.cause() - ); - let (_error, prepared) = failure.into_parts(); - let spawned = prepared - .spawn() - .expect("the unchanged preparation can retry after the transient failure"); - drop(spawned); - wait_for_reaper_to_collect(child_id, Duration::from_secs(2)); - - println!("the failed reaper start preserved both owners"); - } - - fn spawn_reaper_probe(sleep_millis: u64) -> Child { - Command::new(testing::helper_binary_path().as_std_path()) - .arg(testing::directive(format_args!("sleep:{sleep_millis}"))) - .spawn() - .expect("spawn reaper probe") - } - - fn wait_for_reaper_to_collect(id: u32, grace: Duration) { - let deadline = Instant::now() + grace; - while reaper_contains(id) && Instant::now() < deadline { - thread::sleep(Duration::from_millis(10)); - } - assert!(!reaper_contains(id), "the shared reaper did not collect child {id}"); - } - - #[test] - fn ordinary_termination_reports_post_reap_cleanup_failure() { - let mut command = Command::new(testing::helper_binary_path().as_std_path()); - let _ = command.arg(testing::directive("sleep:30000")); - let prepared = prepare(command, MemoryRequest::default()).expect("containment"); - let spawned = prepared.spawn().expect("spawn"); - let mut subtree = ProcessTree::adopt(spawned).expect("adoption"); - let _failed_cleanup = faults::arm(faults::Fault::Terminate); - - let error = subtree - .terminate() - .expect_err("the ordinary termination fault must be preserved after reaping"); - - assert!(error.to_string().contains("failed as requested by a test"), "{error}"); - } - - #[test] - fn observation_cleanup_classifies_pending_cleanup_and_dual_failures() { - let pending: Observation<()> = cleanup_after_observation(false, || unreachable!(), || unreachable!()); - assert!(matches!(pending, Observation::Pending)); - - let cleanup_failed = cleanup_after_observation( - true, - || Err(io::Error::new(io::ErrorKind::PermissionDenied, "cleanup failed")), - || Ok("reaped"), - ); - assert!(matches!( - cleanup_failed, - Observation::CleanupFailed(error) if error.kind() == io::ErrorKind::PermissionDenied - )); - - let both_failed = cleanup_after_observation( - true, - || Err(io::Error::other("cleanup failed")), - || Err::<(), _>(io::Error::new(io::ErrorKind::Interrupted, "reap failed")), - ); - assert!(matches!( - both_failed, - Observation::ReapFailed(error) if error.kind() == io::ErrorKind::Interrupted - )); - } - /// Killing a subtree reaches a grandchild that left the process group. /// /// This is the escape a process group has no answer to. `setsid` and `setpgid` cost one @@ -3459,6 +2667,7 @@ mod tests { /// /// The name is on the child's environment rather than on its command line because the command /// line belongs to the test harness, which would reject an argument it does not know. + #[cfg(unix)] const ISOLATED_CHILD: &str = "GAMMA_ISOLATED_CHILD"; /// Runs one of the inner tests below in a process of its own, and pins what it reported. @@ -3472,6 +2681,7 @@ mod tests { /// /// The marker is required rather than the exit status alone, because a filter that matched /// nothing — a renamed inner test, say — is also a successful run of zero tests. + #[cfg(unix)] fn isolated_run(inner: &str, marker: &str) { let program = env::args().next().expect("this test binary"); let outcome = Command::new(&program) diff --git a/crates/cargo-gamma-unsafe/Cargo.toml b/crates/cargo-gamma-unsafe/Cargo.toml index e1324c1d2..9d527b404 100644 --- a/crates/cargo-gamma-unsafe/Cargo.toml +++ b/crates/cargo-gamma-unsafe/Cargo.toml @@ -51,7 +51,6 @@ windows-sys = { workspace = true, features = [ "Win32_System_Diagnostics_ToolHelp", "Win32_System_IO", "Win32_System_JobObjects", - "Win32_System_Pipes", "Win32_System_SystemServices", "Win32_System_Threading", ] } diff --git a/crates/cargo-gamma-unsafe/docs/DESIGN.md b/crates/cargo-gamma-unsafe/docs/DESIGN.md index 9f3b8c4f1..806ca523b 100644 --- a/crates/cargo-gamma-unsafe/docs/DESIGN.md +++ b/crates/cargo-gamma-unsafe/docs/DESIGN.md @@ -30,12 +30,6 @@ crate's dependency graph and must remain dependency-free. closes it. A safe caller has no way to release it late, twice, or not at all. - Linux controller delegation moves only cargo-gamma itself and refuses a cgroup shared with any other process. -- Child stdout and stderr capture uses interruptible pipe reads. On Unix, the - owned child descriptor is switched to nonblocking mode and an unavailable - read reports `WouldBlock`; EOF remains a successful zero-byte read. On - Windows, anonymous-pipe readiness and closure are observed with - `PeekNamedPipe` before the ordinary read. Other hosts reject construction as - unsupported rather than exposing an uninterruptible safe wrapper. - Policy and limit calculation remain in `cargo-gamma-lib`; this crate only reports and applies platform capabilities. For example, choosing a memory ceiling is testable arithmetic in `cargo-gamma-lib`; this crate answers only diff --git a/crates/cargo-gamma-unsafe/docs/IMPLEMENTATION.md b/crates/cargo-gamma-unsafe/docs/IMPLEMENTATION.md index 356925d29..b29973f6d 100644 --- a/crates/cargo-gamma-unsafe/docs/IMPLEMENTATION.md +++ b/crates/cargo-gamma-unsafe/docs/IMPLEMENTATION.md @@ -25,14 +25,3 @@ distinguished from live ownership. Private backend functions isolate Win32 calls from the ownership algorithms. Unit tests inject per-thread failures at selected calls while retaining the same job, process, completion-port, and handle lifetimes as production. - -## Interruptible child pipes - -`InterruptiblePipe` owns a `ChildStdout` or `ChildStderr`, so callers cannot -retain an independently blocking copy through this API. Unix construction adds -`O_NONBLOCK` to the descriptor's shared open-file description; reads propagate -data and EOF normally and return `WouldBlock` while no data is ready. Windows -uses `PeekNamedPipe` to distinguish pending data, a closed writer, and a -readiness error before delegating to `Read`. Platforms with neither Unix -nonblocking descriptors nor Windows anonymous-pipe readiness return -`Unsupported` during construction. diff --git a/crates/cargo-gamma-unsafe/src/lib.rs b/crates/cargo-gamma-unsafe/src/lib.rs index d7526cb45..2d4083aa1 100644 --- a/crates/cargo-gamma-unsafe/src/lib.rs +++ b/crates/cargo-gamma-unsafe/src/lib.rs @@ -11,9 +11,7 @@ //! Two things the tool does have no safe expression in `std`: killing a whole process subtree (a //! process group on Unix, a job object on Windows) and bounding what that subtree allocates (a //! cgroup leaf on Linux, the same job object on Windows). Neither is a case of reaching for -//! `unsafe` to go faster — there is no safe version to prefer. The same applies to interruptible -//! reads from anonymous child pipes, which output capture uses to stop readers after a bounded -//! drain grace. +//! `unsafe` to go faster — there is no safe version to prefer. //! //! Concentrating those calls here is what lets every other crate in the workspace carry //! `#![forbid(unsafe_code)]`, which turns "we reviewed the unsafe code" into a property the @@ -36,7 +34,6 @@ pub mod identity; pub mod interrupt; #[cfg(windows)] pub mod job; -pub mod pipe; #[cfg(all(windows, test))] mod native_faults; diff --git a/crates/cargo-gamma-unsafe/src/pipe.rs b/crates/cargo-gamma-unsafe/src/pipe.rs deleted file mode 100644 index 26f877682..000000000 --- a/crates/cargo-gamma-unsafe/src/pipe.rs +++ /dev/null @@ -1,230 +0,0 @@ -// Copyright (c) Microsoft Corporation. -// Licensed under the MIT License. - -//! Interruptible reads from child stdout and stderr pipes. -//! -//! A blocking `Read` owned by a capture thread cannot be interrupted by -//! dropping its `JoinHandle`. This wrapper exposes the platform's nonblocking -//! pipe observation behind an ordinary safe [`Read`] implementation: no data -//! currently available is reported as [`io::ErrorKind::WouldBlock`]. - -use std::io::{self, Read}; -use std::process::{ChildStderr, ChildStdout}; - -/// A child pipe whose reads report [`io::ErrorKind::WouldBlock`] instead of -/// waiting indefinitely for another process to write or close the pipe. -#[derive(Debug)] -pub struct InterruptiblePipe { - inner: R, -} - -impl InterruptiblePipe { - /// Takes ownership of a child's stdout pipe and makes its reads interruptible. - /// - /// On Unix, this sets `O_NONBLOCK` on the pipe's shared open-file - /// description. Callers must not retain duplicated descriptors whose - /// blocking mode they expect to remain unchanged. On Windows, - /// [`ChildStdout`] supplies the readable anonymous pipe required by - /// `PeekNamedPipe`. - /// - /// # Errors - /// - /// Returns the operating system's error when pipe readiness cannot be - /// configured. - pub fn stdout(inner: ChildStdout) -> io::Result { - interruptible(inner) - } -} - -impl InterruptiblePipe { - /// Takes ownership of a child's stderr pipe and makes its reads interruptible. - /// - /// The platform preconditions are the same as for child stdout. - /// - /// # Errors - /// - /// Returns the operating system's error when pipe readiness cannot be - /// configured. - pub fn stderr(inner: ChildStderr) -> io::Result { - interruptible(inner) - } -} - -#[cfg(unix)] -fn interruptible(inner: R) -> io::Result> -where - R: std::os::fd::AsFd, -{ - use std::os::fd::AsRawFd as _; - - let descriptor = inner.as_fd().as_raw_fd(); - - // SAFETY: `AsFd` proves the descriptor remains live for this borrow. - // `F_GETFL` reads its flags and does not access caller memory. - let flags = unsafe { libc::fcntl(descriptor, libc::F_GETFL) }; - if flags == -1 { - return Err(io::Error::last_os_error()); - } - if flags & libc::O_NONBLOCK == 0 { - // SAFETY: `AsFd` keeps the same descriptor live through this call. - // Preserving every existing flag avoids changing any property other - // than nonblocking I/O. - if unsafe { libc::fcntl(descriptor, libc::F_SETFL, flags | libc::O_NONBLOCK) } == -1 { - return Err(io::Error::last_os_error()); - } - } - - Ok(InterruptiblePipe { inner }) -} - -#[cfg(unix)] -impl Read for InterruptiblePipe -where - R: Read, -{ - fn read(&mut self, buf: &mut [u8]) -> io::Result { - self.inner.read(buf) - } -} - -#[cfg(all(test, any(unix, windows)))] -mod tests { - use std::io::{Read as _, Write as _}; - use std::process::{Command, Stdio}; - use std::thread; - use std::time::{Duration, Instant}; - - use super::InterruptiblePipe; - - #[test] - #[cfg_attr(miri, ignore = "spawns a child process; Miri isolation does not support process creation")] - fn child_pipe_reads_are_pending_data_then_eof() { - let mut command = pipe_writer(); - let _ = command.stdin(Stdio::piped()).stdout(Stdio::piped()).stderr(Stdio::null()); - let mut child = command.spawn().expect("spawn pipe writer"); - let stdout = child.stdout.take().expect("capture child stdout"); - let mut pipe = InterruptiblePipe::stdout(stdout).expect("make child pipe interruptible"); - let mut buffer = [0_u8; 64]; - - let pending = pipe.read(&mut buffer).expect_err("the blocked writer has not produced output"); - assert_eq!(pending.kind(), std::io::ErrorKind::WouldBlock); - - child - .stdin - .take() - .expect("capture child stdin") - .write_all(b"go\n") - .expect("release child writer"); - - let deadline = Instant::now() + Duration::from_secs(5); - let mut output = Vec::new(); - loop { - match pipe.read(&mut buffer) { - Ok(0) => break, - Ok(read) => output.extend_from_slice(&buffer[..read]), - Err(error) if error.kind() == std::io::ErrorKind::WouldBlock && Instant::now() < deadline => { - thread::sleep(Duration::from_millis(5)); - } - Err(error) => panic!("failed to read child pipe: {error}"), - } - } - - assert!(child.wait().expect("wait for pipe writer").success()); - assert_eq!(output, b"payload"); - } - - #[cfg(unix)] - fn pipe_writer() -> Command { - let mut command = Command::new("sh"); - let _ = command.args(["-c", "read gate; printf payload"]); - command - } - - #[cfg(windows)] - fn pipe_writer() -> Command { - let mut command = Command::new("pwsh"); - let _ = command.args([ - "-NoLogo", - "-NoProfile", - "-NonInteractive", - "-Command", - "$null = [Console]::In.ReadLine(); [Console]::Out.Write('payload')", - ]); - command - } -} - -#[cfg(windows)] -#[expect( - clippy::unnecessary_wraps, - reason = "Unix construction configures descriptor flags and can fail; the cross-platform constructor keeps one signature" -)] -fn interruptible(inner: R) -> io::Result> { - Ok(InterruptiblePipe { inner }) -} - -#[cfg(windows)] -impl Read for InterruptiblePipe -where - R: Read + std::os::windows::io::AsHandle, -{ - fn read(&mut self, buf: &mut [u8]) -> io::Result { - use std::os::windows::io::AsRawHandle as _; - - use windows_sys::Win32::Foundation::{ERROR_BROKEN_PIPE, ERROR_NO_DATA, ERROR_PIPE_NOT_CONNECTED, HANDLE}; - use windows_sys::Win32::System::Pipes::PeekNamedPipe; - - if buf.is_empty() { - return Ok(0); - } - - let mut available = 0_u32; - // SAFETY: the handle is borrowed from the live pipe for the duration - // of this call. Null buffer arguments request only the available-byte - // count, which is written to the valid `available` pointer. - let ready = unsafe { - PeekNamedPipe( - self.inner.as_handle().as_raw_handle().cast::() as HANDLE, - core::ptr::null_mut(), - 0, - core::ptr::null_mut(), - core::ptr::from_mut(&mut available), - core::ptr::null_mut(), - ) - }; - if ready == 0 { - let error = io::Error::last_os_error(); - return match error.raw_os_error().and_then(|code| u32::try_from(code).ok()) { - Some(ERROR_BROKEN_PIPE | ERROR_NO_DATA | ERROR_PIPE_NOT_CONNECTED) => Ok(0), - _ => Err(error), - }; - } - if available == 0 { - return Err(io::Error::from(io::ErrorKind::WouldBlock)); - } - - let available = usize::try_from(available).unwrap_or(usize::MAX); - let readable = buf.len().min(available); - self.inner.read(&mut buf[..readable]) - } -} - -#[cfg(not(any(unix, windows)))] -fn interruptible(inner: R) -> io::Result> { - // No bounded child-pipe readiness primitive is available. - drop(inner); - Err(io::Error::new( - io::ErrorKind::Unsupported, - "interruptible child-pipe reads require Unix nonblocking descriptors or Windows pipe readiness", - )) -} - -#[cfg(not(any(unix, windows)))] -impl Read for InterruptiblePipe -where - R: Read, -{ - fn read(&mut self, buf: &mut [u8]) -> io::Result { - self.inner.read(buf) - } -} From 96c1e266eead3d1e0577c19b15bae6e795494868 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Mon, 21 Sep 2026 20:40:40 +0200 Subject: [PATCH 24/37] fix(cargo-each): enforce timeout before completion Check the deadline before accepting a later process observation so completion discovered after the configured timeout remains a timeout. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 41 +++++++++++++++++++++++++----------- 1 file changed, 29 insertions(+), 12 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index dae8f3533..41f7b5c10 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -524,18 +524,6 @@ fn wait_for_process( return TreeOutcome::new(InvocationResult::Infrastructure(with_cleanup_failure(reader_failure, &cleanup))); } - match observe(&mut control) { - Ok(Some(status)) => return TreeOutcome::new(InvocationResult::Exited(status)), - Ok(None) => {} - Err(error) => { - let cleanup = terminate.take().expect("termination is consumed only on a returning branch")(control); - return TreeOutcome::new(InvocationResult::Infrastructure(with_cleanup_failure( - format!("failed to {operation}: {error}"), - &cleanup, - ))); - } - } - if let Some(timeout) = timeout && timeout.checked_sub(started.elapsed()).is_none() { @@ -548,6 +536,18 @@ fn wait_for_process( }; } + match observe(&mut control) { + Ok(Some(status)) => return TreeOutcome::new(InvocationResult::Exited(status)), + Ok(None) => {} + Err(error) => { + let cleanup = terminate.take().expect("termination is consumed only on a returning branch")(control); + return TreeOutcome::new(InvocationResult::Infrastructure(with_cleanup_failure( + format!("failed to {operation}: {error}"), + &cleanup, + ))); + } + } + let pause = timeout .and_then(|timeout| timeout.checked_sub(started.elapsed())) .map_or(Duration::from_millis(10), |remaining| remaining.min(Duration::from_millis(10))); @@ -1537,6 +1537,23 @@ mod tests { ); assert!(matches!(outcome.result, InvocationResult::TimedOut(duration) if duration.is_zero())); + let late_success = FakeProcess { + observations: VecDeque::from([Ok(None), Ok(Some(successful_status()))]), + termination: Some(Ok(successful_status())), + }; + let outcome = wait_for_process( + late_success, + Some(Duration::from_millis(1)), + FakeProcess::observe, + || None, + FakeProcess::terminate, + "observe fake process", + ); + assert!( + matches!(outcome.result, InvocationResult::TimedOut(duration) if duration == Duration::from_millis(1)), + "completion observed after the deadline must not win the race" + ); + let failed = FakeProcess { observations: VecDeque::from([Err(io::Error::other("observation failed"))]), termination: Some(Err(io::Error::other("cleanup failed"))), From a1b883bee3ebf1dee207d3a94c5bae6bbf57ecad Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 22 Sep 2026 10:23:31 +0200 Subject: [PATCH 25/37] chore(deps): centralize command-group version Declare command-group in workspace dependencies with the repository's normal compatible-version syntax and consume it from cargo-each via workspace inheritance. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- Cargo.toml | 1 + crates/cargo-each/Cargo.toml | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/Cargo.toml b/Cargo.toml index 9fe00d78c..35df70590 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -48,6 +48,7 @@ cel = { version = "0.14.4", default-features = false } chrono = { version = "0.4.45", default-features = false } clap = { version = "4.6.6", default-features = false } clap_complete = { version = "4.6.9", default-features = false } +command-group = { version = "5.0.1", default-features = false } compact_str = { version = "0.10.0", default-features = false } csv = { version = "1.4.0", default-features = false } duct = { version = "1.1.1", default-features = false } diff --git a/crates/cargo-each/Cargo.toml b/crates/cargo-each/Cargo.toml index 9e4014462..58cb0df67 100644 --- a/crates/cargo-each/Cargo.toml +++ b/crates/cargo-each/Cargo.toml @@ -19,7 +19,7 @@ repository = "https://github.com/microsoft/ox-tools/tree/main/crates/cargo-each" [dependencies] cargo_metadata = { workspace = true } clap = { workspace = true, features = ["derive", "std", "help", "usage", "error-context"] } -command-group = "=5.0.1" +command-group = { workspace = true } mutants = { workspace = true } ohno = { workspace = true, features = ["app-err"] } serde_json = { workspace = true, features = ["std"] } From ac9d80c01b78a71e1627a6fda7c187a5ec0d7b57 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 22 Sep 2026 13:53:30 +0200 Subject: [PATCH 26/37] fix(cargo-each): poll failed reaper handoffs Retain disconnected process-group handoffs before starting an emergency polling reaper, so startup failure never drops the wait handle and later handoffs can retry collection. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/README.md | 3 +- crates/cargo-each/docs/design/README.md | 12 ++-- crates/cargo-each/src/main.rs | 3 +- crates/cargo-each/src/run.rs | 93 +++++++++++++++++++++++-- 4 files changed, 99 insertions(+), 12 deletions(-) diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index 9b32a6027..0031f08ab 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -160,7 +160,8 @@ reaper started before any command. The reaper checks every retained group without blocking on one child, remains the wait owner after the caller returns, and exits after all senders disconnect and retained groups are collected. Reaper startup and handoff failures are explicit infrastructure -failures; a failed handoff retains the group handle in a persistent fallback. +failures; a failed handoff retains the group handle in a persistent fallback +queue and starts an emergency polling reaper. Child commands inherit `PATH` explicitly. On Windows this makes relative program lookup honor the inherited `PATH` order instead of preferring an unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index dd3445f1d..86e210980 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -366,11 +366,13 @@ no-op. after the caller returns, and shuts down only after all senders disconnect and its retained set is empty. Startup failure therefore aborts before command launch. A disconnected handoff reports an infrastructure failure and - places the recovered handle in a persistent fallback instead of dropping - ownership. A failed kill, observation, bounded reap, reaper startup, or - handoff is an infrastructure failure. Unix process groups are not sealed - containment: a descendant can escape by creating a new session, so timeout - cleanup remains best-effort for escaped descendants. + places the recovered handle in a persistent fallback queue before starting an + emergency polling reaper. If that thread cannot start, the queue retains + ownership and a later failed handoff retries startup. A failed kill, + observation, bounded reap, reaper startup, or handoff is an infrastructure + failure. Unix process groups are not sealed containment: a descendant can + escape by creating a new session, so timeout cleanup remains best-effort for + escaped descendants. - **Child executable resolution follows `PATH`.** `cargo-each` explicitly copies an inherited `PATH` onto every child command. This is equivalent to ordinary inheritance on other platforms and makes Windows resolve a relative diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 02607b15f..042f2ebd8 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -150,7 +150,8 @@ //! without blocking on one child, remains the wait owner after the caller //! returns, and exits after all senders disconnect and retained groups are //! collected. Reaper startup and handoff failures are explicit infrastructure -//! failures; a failed handoff retains the group handle in a persistent fallback. +//! failures; a failed handoff retains the group handle in a persistent fallback +//! queue and starts an emergency polling reaper. //! Child commands inherit `PATH` explicitly. On Windows this makes relative //! program lookup honor the inherited `PATH` order instead of preferring an //! unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 1e9d53ba9..47a1752dd 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -5,10 +5,11 @@ //! apply filters, build the plan, and run it. use std::collections::BTreeSet; -use std::io::{self, Read as _, Seek as _, SeekFrom}; +use std::io::{self, Read as _, Seek as _, SeekFrom, Write as _}; use std::num::NonZeroUsize; use std::panic::{self, UnwindSafe}; use std::process::{Command, ExitCode, ExitStatus, Stdio}; +use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Mutex, OnceLock, mpsc}; use std::time::{Duration, Instant}; use std::{fmt, thread}; @@ -639,8 +640,52 @@ impl GroupReaper { } fn handoff(&self, child: GroupChild) -> io::Result<()> { - static FAILED_HANDOFFS: OnceLock>> = OnceLock::new(); - handoff_group(&self.sender, FAILED_HANDOFFS.get_or_init(|| Mutex::new(Vec::new())), child) + let retained = failed_handoffs(); + let handoff = handoff_group(&self.sender, retained, child); + if handoff.is_err() + && let Err(error) = start_failed_handoff_reaper(retained) + { + return Err(io::Error::new( + io::ErrorKind::BrokenPipe, + format!("process-group reaper channel disconnected; the fallback retained the wait handle but failed to start: {error}"), + )); + } + handoff + } +} + +fn failed_handoffs() -> &'static Mutex> { + static FAILED_HANDOFFS: OnceLock>> = OnceLock::new(); + FAILED_HANDOFFS.get_or_init(|| Mutex::new(Vec::new())) +} + +fn start_failed_handoff_reaper(retained: &'static Mutex>) -> io::Result<()> { + static RUNNING: AtomicBool = AtomicBool::new(false); + if RUNNING.compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire).is_err() { + return Ok(()); + } + match thread::Builder::new() + .name("cargo-each-fallback-reaper".to_owned()) + .spawn(move || poll_failed_handoffs(retained, &RUNNING)) + { + Ok(_) => Ok(()), + Err(error) => { + RUNNING.store(false, Ordering::Release); + Err(error) + } + } +} + +fn poll_failed_handoffs(retained: &Mutex>, running: &AtomicBool) { + loop { + let mut children = retained.lock().unwrap_or_else(std::sync::PoisonError::into_inner); + children.retain_mut(|child| !matches!(child.try_wait(), Ok(Some(_)))); + if children.is_empty() { + running.store(false, Ordering::Release); + return; + } + drop(children); + thread::sleep(REAPER_POLL_INTERVAL); } } @@ -665,7 +710,15 @@ struct ReaperEntry { #[cfg_attr(coverage_nightly, coverage(off))] #[mutants::skip] // Process-thread diagnostic for an OS observation failure; behavior is covered through the injected reporter seam. fn report_reaper_failure(error: &io::Error) { - eprintln!("cargo each: process-group reaper failed to observe a retained group: {error}"); + let message = error.to_string(); + let _ = thread::Builder::new() + .name("cargo-each-reaper-diagnostic".to_owned()) + .spawn(move || { + let _ = writeln!( + io::stderr().lock(), + "cargo each: process-group reaper failed to observe a retained group: {message}" + ); + }); } #[mutants::skip] // Deleting the disconnected-and-empty shutdown arm hangs by definition; deterministic tests cover polling and exit. @@ -984,7 +1037,7 @@ mod tests { use super::{ BufferedOutcome, CapturedOutput, CapturedStream, GroupReaper, Invocation, InvocationResult, OutputEmitError, Plan, RunningWorker, SnapshotSource, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, add_infrastructure_failure, - combine_captured_output, display_duration, effective_worker_count, emit_buffered_to, execute_parallel, exit_byte, + combine_captured_output, display_duration, effective_worker_count, emit_buffered_to, execute_parallel, exit_byte, failed_handoffs, failure_stops_launching, finish_capture, handoff_group, panic_description, parallel_failure_exit_code, poll_process_exit, poll_reaper, run_captured, run_captured_with, run_streamed, run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_group, spawn_worker, terminate_group_bounded, terminate_group_with, wait_for_process, wait_for_worker, with_cleanup_failure, @@ -1442,6 +1495,36 @@ mod tests { ); } + #[test] + #[cfg_attr(miri, ignore = "spawns a process group and a fallback reaper thread")] + fn disconnected_reaper_handoff_is_eventually_collected() { + let (sender, receiver) = mpsc::channel(); + drop(receiver); + let reaper = GroupReaper { sender }; + let mut command = Command::new("rustc"); + let _ = command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); + let child = spawn_group(command).expect("spawn a short-lived process group"); + + let error = reaper.handoff(child).expect_err("the disconnected primary reaper is reported"); + assert_eq!(error.kind(), io::ErrorKind::BrokenPipe); + let deadline = Instant::now() + Duration::from_secs(2); + while !failed_handoffs() + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .is_empty() + && Instant::now() < deadline + { + thread::sleep(Duration::from_millis(10)); + } + assert!( + failed_handoffs() + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .is_empty(), + "the fallback reaper must eventually collect a recovered handoff" + ); + } + #[test] #[cfg_attr(miri, ignore = "spawns process groups")] fn real_group_execution_observes_completion_and_timeout() { From b9baffdf24749f9ade081d655b6621b03ad755a8 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 22 Sep 2026 15:33:49 +0200 Subject: [PATCH 27/37] test(cargo-each): cover fallback reaper states Inject fallback startup and polling seams, verify stable ownership storage, and exercise independent snapshot seek behavior so coverage and mutation checks pin the exceptional paths. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 180 ++++++++++++++++++++++++++++------- 1 file changed, 147 insertions(+), 33 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 47a1752dd..fbd15ebb4 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -562,15 +562,10 @@ fn wait_for_process( #[cfg_attr(coverage_nightly, coverage(off))] fn terminate_group_bounded(child: GroupChild, reaper: &GroupReaper) -> io::Result { terminate_group_with(child, TERMINATION_GRACE, GroupChild::kill, GroupChild::try_wait, |child| { - handoff_terminated_group(reaper, child) + reaper.handoff(child) }) } -#[cfg_attr(coverage_nightly, coverage(off))] -fn handoff_terminated_group(reaper: &GroupReaper, child: GroupChild) -> io::Result<()> { - reaper.handoff(child) -} - fn terminate_group_with( mut child: T, grace: Duration, @@ -641,16 +636,7 @@ impl GroupReaper { fn handoff(&self, child: GroupChild) -> io::Result<()> { let retained = failed_handoffs(); - let handoff = handoff_group(&self.sender, retained, child); - if handoff.is_err() - && let Err(error) = start_failed_handoff_reaper(retained) - { - return Err(io::Error::new( - io::ErrorKind::BrokenPipe, - format!("process-group reaper channel disconnected; the fallback retained the wait handle but failed to start: {error}"), - )); - } - handoff + handoff_group_with_fallback(&self.sender, retained, child, || start_failed_handoff_reaper(retained)) } } @@ -661,25 +647,40 @@ fn failed_handoffs() -> &'static Mutex> { fn start_failed_handoff_reaper(retained: &'static Mutex>) -> io::Result<()> { static RUNNING: AtomicBool = AtomicBool::new(false); - if RUNNING.compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire).is_err() { + start_failed_handoff_reaper_with(retained, &RUNNING, GroupChild::try_wait, |job| { + thread::Builder::new() + .name("cargo-each-fallback-reaper".to_owned()) + .spawn(job) + .map(drop) + }) +} + +fn start_failed_handoff_reaper_with( + retained: &'static Mutex>, + running: &'static AtomicBool, + try_wait: fn(&mut T) -> io::Result>, + spawn: impl FnOnce(ReaperJob) -> io::Result<()>, +) -> io::Result<()> { + if running.compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire).is_err() { return Ok(()); } - match thread::Builder::new() - .name("cargo-each-fallback-reaper".to_owned()) - .spawn(move || poll_failed_handoffs(retained, &RUNNING)) - { - Ok(_) => Ok(()), + match spawn(Box::new(move || poll_failed_handoffs(retained, running, try_wait))) { + Ok(()) => Ok(()), Err(error) => { - RUNNING.store(false, Ordering::Release); + running.store(false, Ordering::Release); Err(error) } } } -fn poll_failed_handoffs(retained: &Mutex>, running: &AtomicBool) { +fn poll_failed_handoffs( + retained: &Mutex>, + running: &AtomicBool, + mut try_wait: impl FnMut(&mut T) -> io::Result>, +) { loop { let mut children = retained.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - children.retain_mut(|child| !matches!(child.try_wait(), Ok(Some(_)))); + children.retain_mut(|child| !matches!(try_wait(child), Ok(Some(_)))); if children.is_empty() { running.store(false, Ordering::Release); return; @@ -702,6 +703,24 @@ fn handoff_group(sender: &mpsc::Sender, retained: &Mutex>, child: T } } +fn handoff_group_with_fallback( + sender: &mpsc::Sender, + retained: &Mutex>, + child: T, + start_fallback: impl FnOnce() -> io::Result<()>, +) -> io::Result<()> { + let handoff = handoff_group(sender, retained, child); + if handoff.is_err() + && let Err(error) = start_fallback() + { + return Err(io::Error::new( + io::ErrorKind::BrokenPipe, + format!("process-group reaper channel disconnected; the fallback retained the wait handle but failed to start: {error}"), + )); + } + handoff +} + struct ReaperEntry { child: T, failure_reported: bool, @@ -1023,25 +1042,26 @@ fn exit_byte(raw: Option) -> u8 { #[cfg_attr(coverage_nightly, coverage(off))] mod tests { use std::collections::VecDeque; + use std::io::{Read as _, Seek as _}; use std::num::NonZeroUsize; #[cfg(unix)] use std::os::unix::process::ExitStatusExt as _; #[cfg(windows)] use std::os::windows::process::ExitStatusExt as _; use std::process::{Command, ExitCode, ExitStatus, Stdio}; - use std::sync::atomic::{AtomicUsize, Ordering}; + use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; use std::sync::{Arc, Mutex, mpsc}; use std::time::{Duration, Instant}; use std::{io, thread}; use super::{ BufferedOutcome, CapturedOutput, CapturedStream, GroupReaper, Invocation, InvocationResult, OutputEmitError, Plan, RunningWorker, - SnapshotSource, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, add_infrastructure_failure, - combine_captured_output, display_duration, effective_worker_count, emit_buffered_to, execute_parallel, exit_byte, failed_handoffs, - failure_stops_launching, finish_capture, handoff_group, panic_description, parallel_failure_exit_code, poll_process_exit, - poll_reaper, run_captured, run_captured_with, run_streamed, run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_group, - spawn_worker, terminate_group_bounded, terminate_group_with, wait_for_process, wait_for_worker, with_cleanup_failure, - with_reaper_handoff, + SnapshotSource, TemporarySnapshot, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, + add_infrastructure_failure, combine_captured_output, display_duration, effective_worker_count, emit_buffered_to, execute_parallel, + exit_byte, failed_handoffs, failure_stops_launching, finish_capture, handoff_group, handoff_group_with_fallback, panic_description, + parallel_failure_exit_code, poll_process_exit, poll_reaper, run_captured, run_captured_with, run_streamed, + run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_group, spawn_worker, start_failed_handoff_reaper_with, + terminate_group_bounded, terminate_group_with, wait_for_process, wait_for_worker, with_cleanup_failure, with_reaper_handoff, }; fn invocation(argv: &[&str]) -> Invocation { @@ -1481,7 +1501,7 @@ mod tests { let (connected_sender, connected_receiver) = mpsc::channel(); let retained = Mutex::new(Vec::new()); handoff_group(&connected_sender, &retained, "delivered group").expect("connected handoff succeeds"); - assert_eq!(connected_receiver.recv().expect("group is delivered"), "delivered group"); + assert_eq!(connected_receiver.try_recv().expect("group is delivered"), "delivered group"); assert!(retained.lock().expect("fallback ownership mutex is not poisoned").is_empty()); let (sender, receiver) = mpsc::channel(); @@ -1495,6 +1515,83 @@ mod tests { ); } + #[test] + fn fallback_handoff_reports_startup_failure_without_losing_ownership() { + let (sender, receiver) = mpsc::channel(); + drop(receiver); + let retained = Mutex::new(Vec::new()); + let error = handoff_group_with_fallback(&sender, &retained, "owned group", || { + Err(io::Error::other("injected fallback startup failure")) + }) + .expect_err("fallback startup failure is reported"); + assert_eq!(error.kind(), io::ErrorKind::BrokenPipe); + assert!(error.to_string().contains("injected fallback startup failure")); + assert_eq!( + retained.lock().expect("fallback ownership mutex is not poisoned").as_slice(), + ["owned group"] + ); + } + + #[test] + fn fallback_reaper_startup_and_polling_cover_every_state() { + struct FakeGroup { + errors_remaining: usize, + polls_remaining: usize, + collected: Arc, + } + + fn observe(group: &mut FakeGroup) -> io::Result> { + if group.errors_remaining > 0 { + group.errors_remaining -= 1; + Err(io::Error::other("injected fallback observation failure")) + } else if group.polls_remaining == 0 { + group.collected.fetch_add(1, Ordering::SeqCst); + Ok(Some(successful_status())) + } else { + group.polls_remaining -= 1; + Ok(None) + } + } + + let retained: &'static Mutex> = Box::leak(Box::new(Mutex::new(Vec::new()))); + let running: &'static AtomicBool = Box::leak(Box::new(AtomicBool::new(true))); + start_failed_handoff_reaper_with(retained, running, observe, |_| { + panic!("an already-running fallback must not spawn another thread"); + }) + .expect("an already-running fallback accepts more work"); + + running.store(false, Ordering::Release); + let collected = Arc::new(AtomicUsize::new(0)); + retained.lock().expect("fallback ownership mutex is not poisoned").push(FakeGroup { + errors_remaining: 1, + polls_remaining: 1, + collected: Arc::clone(&collected), + }); + let error = start_failed_handoff_reaper_with(retained, running, observe, |job| { + drop(job); + Err(io::Error::other("injected fallback thread failure")) + }) + .expect_err("fallback thread failure is reported"); + assert!(error.to_string().contains("injected fallback thread failure")); + assert!(!running.load(Ordering::Acquire)); + assert_eq!(retained.lock().expect("fallback ownership mutex is not poisoned").len(), 1); + + start_failed_handoff_reaper_with(retained, running, observe, |job| thread::Builder::new().spawn(job).map(drop)) + .expect("fallback polling thread starts"); + let deadline = Instant::now() + Duration::from_secs(1); + while running.load(Ordering::Acquire) && Instant::now() < deadline { + thread::sleep(Duration::from_millis(1)); + } + assert!(!running.load(Ordering::Acquire), "fallback polling thread did not finish"); + assert_eq!(collected.load(Ordering::SeqCst), 1); + assert!(retained.lock().expect("fallback ownership mutex is not poisoned").is_empty()); + } + + #[test] + fn failed_handoff_storage_is_process_stable() { + assert!(std::ptr::eq(failed_handoffs(), failed_handoffs())); + } + #[test] #[cfg_attr(miri, ignore = "spawns a process group and a fallback reaper thread")] fn disconnected_reaper_handoff_is_eventually_collected() { @@ -1678,6 +1775,23 @@ mod tests { ); } + #[test] + fn temporary_snapshot_seek_rewinds_the_independent_reader() { + let temporary = tempfile::NamedTempFile::new().expect("create named temporary capture"); + std::fs::write(temporary.path(), b"snapshot").expect("write temporary capture"); + let reader = temporary.reopen().expect("reopen temporary capture reader"); + let path = temporary.into_temp_path(); + let mut snapshot = TemporarySnapshot { _path: path, reader }; + let mut first = [0_u8; 1]; + snapshot.read_exact(&mut first).expect("read first snapshot byte"); + assert_eq!(first, [b's']); + + snapshot.seek(io::SeekFrom::Start(0)).expect("rewind snapshot reader"); + let mut output = Vec::new(); + snapshot.read_to_end(&mut output).expect("read rewound snapshot"); + assert_eq!(output, b"snapshot"); + } + #[test] fn capture_source_failures_become_infrastructure_failures() { let mut outcome = BufferedOutcome { From fa63c34a23d9b64871564e1a5b546f3e0ca1a9b9 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 22 Sep 2026 18:00:36 +0200 Subject: [PATCH 28/37] test(cargo-each): keep fallback seams Miri-clean Use static test storage instead of leaked allocations and exclude only the filesystem-backed snapshot regression under isolated Miri. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 79 +++++++++++++++++++----------------- 1 file changed, 42 insertions(+), 37 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index fbd15ebb4..387fff4d0 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -1050,7 +1050,7 @@ mod tests { use std::os::windows::process::ExitStatusExt as _; use std::process::{Command, ExitCode, ExitStatus, Stdio}; use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; - use std::sync::{Arc, Mutex, mpsc}; + use std::sync::{Arc, Mutex, OnceLock, mpsc}; use std::time::{Duration, Instant}; use std::{io, thread}; @@ -1086,6 +1086,27 @@ mod tests { ExitStatus::from_raw(u32::try_from(code).expect("test exit code is nonnegative")) } + static FALLBACK_TEST_GROUPS: OnceLock>> = OnceLock::new(); + static FALLBACK_TEST_RUNNING: AtomicBool = AtomicBool::new(false); + static FALLBACK_TEST_COLLECTED: AtomicUsize = AtomicUsize::new(0); + + fn observe_fallback_test_group(state: &mut usize) -> io::Result> { + match *state { + 2 => { + *state = 1; + Err(io::Error::other("injected fallback observation failure")) + } + 1 => { + *state = 0; + Ok(None) + } + _ => { + FALLBACK_TEST_COLLECTED.fetch_add(1, Ordering::SeqCst); + Ok(Some(successful_status())) + } + } + } + fn result_infrastructure_message(result: InvocationResult) -> String { let InvocationResult::Infrastructure(message) = result else { panic!("the test expects an infrastructure outcome"); @@ -1534,56 +1555,39 @@ mod tests { #[test] fn fallback_reaper_startup_and_polling_cover_every_state() { - struct FakeGroup { - errors_remaining: usize, - polls_remaining: usize, - collected: Arc, - } - - fn observe(group: &mut FakeGroup) -> io::Result> { - if group.errors_remaining > 0 { - group.errors_remaining -= 1; - Err(io::Error::other("injected fallback observation failure")) - } else if group.polls_remaining == 0 { - group.collected.fetch_add(1, Ordering::SeqCst); - Ok(Some(successful_status())) - } else { - group.polls_remaining -= 1; - Ok(None) - } - } - - let retained: &'static Mutex> = Box::leak(Box::new(Mutex::new(Vec::new()))); - let running: &'static AtomicBool = Box::leak(Box::new(AtomicBool::new(true))); - start_failed_handoff_reaper_with(retained, running, observe, |_| { + let retained = FALLBACK_TEST_GROUPS.get_or_init(|| Mutex::new(Vec::new())); + retained.lock().expect("fallback ownership mutex is not poisoned").clear(); + FALLBACK_TEST_RUNNING.store(true, Ordering::Release); + FALLBACK_TEST_COLLECTED.store(0, Ordering::SeqCst); + start_failed_handoff_reaper_with(retained, &FALLBACK_TEST_RUNNING, observe_fallback_test_group, |_| { panic!("an already-running fallback must not spawn another thread"); }) .expect("an already-running fallback accepts more work"); - running.store(false, Ordering::Release); - let collected = Arc::new(AtomicUsize::new(0)); - retained.lock().expect("fallback ownership mutex is not poisoned").push(FakeGroup { - errors_remaining: 1, - polls_remaining: 1, - collected: Arc::clone(&collected), - }); - let error = start_failed_handoff_reaper_with(retained, running, observe, |job| { + FALLBACK_TEST_RUNNING.store(false, Ordering::Release); + retained.lock().expect("fallback ownership mutex is not poisoned").push(2); + let error = start_failed_handoff_reaper_with(retained, &FALLBACK_TEST_RUNNING, observe_fallback_test_group, |job| { drop(job); Err(io::Error::other("injected fallback thread failure")) }) .expect_err("fallback thread failure is reported"); assert!(error.to_string().contains("injected fallback thread failure")); - assert!(!running.load(Ordering::Acquire)); + assert!(!FALLBACK_TEST_RUNNING.load(Ordering::Acquire)); assert_eq!(retained.lock().expect("fallback ownership mutex is not poisoned").len(), 1); - start_failed_handoff_reaper_with(retained, running, observe, |job| thread::Builder::new().spawn(job).map(drop)) - .expect("fallback polling thread starts"); + start_failed_handoff_reaper_with(retained, &FALLBACK_TEST_RUNNING, observe_fallback_test_group, |job| { + thread::Builder::new().spawn(job).map(drop) + }) + .expect("fallback polling thread starts"); let deadline = Instant::now() + Duration::from_secs(1); - while running.load(Ordering::Acquire) && Instant::now() < deadline { + while FALLBACK_TEST_RUNNING.load(Ordering::Acquire) && Instant::now() < deadline { thread::sleep(Duration::from_millis(1)); } - assert!(!running.load(Ordering::Acquire), "fallback polling thread did not finish"); - assert_eq!(collected.load(Ordering::SeqCst), 1); + assert!( + !FALLBACK_TEST_RUNNING.load(Ordering::Acquire), + "fallback polling thread did not finish" + ); + assert_eq!(FALLBACK_TEST_COLLECTED.load(Ordering::SeqCst), 1); assert!(retained.lock().expect("fallback ownership mutex is not poisoned").is_empty()); } @@ -1776,6 +1780,7 @@ mod tests { } #[test] + #[cfg_attr(miri, ignore = "uses filesystem-backed temporary files; Miri isolation forbids them")] fn temporary_snapshot_seek_rewinds_the_independent_reader() { let temporary = tempfile::NamedTempFile::new().expect("create named temporary capture"); std::fs::write(temporary.path(), b"snapshot").expect("write temporary capture"); From 1efa75083bf524ae3ed72a2ad43e9eeb331e9d39 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 22 Sep 2026 19:57:36 +0200 Subject: [PATCH 29/37] fix(cargo-each): close parallel cleanup gaps Propagate capture-read failures into scheduler state, observe the direct leader during timeout cleanup, stop retaining terminal reaper errors, and document storage retained by preserved descendants. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/README.md | 14 +- crates/cargo-each/docs/design/README.md | 22 ++- crates/cargo-each/src/main.rs | 14 +- crates/cargo-each/src/run.rs | 189 ++++++++++++++++++------ 4 files changed, 176 insertions(+), 63 deletions(-) diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index 0031f08ab..e04aeb99c 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -150,18 +150,20 @@ cannot move descendant write positions. Cargo-each records each file’s current length when the leader completes (or after timeout cleanup), then reads exactly that finite snapshot in plan order without loading unbounded output into memory. Later writes by background or escaped descendants are -outside the snapshot, and inherited file handles cannot hold capture open. -Capture create, reopen, length, seek, and read failures are infrastructure -failures; files are removed by RAII. +outside the snapshot. RAII removes cargo-each’s directory entry, but a +preserved descendant can keep the backing storage allocated and growing +until its inherited writer closes. Capture create, reopen, length, seek, and +read failures are infrastructure failures. Timed-out group termination gets a bounded 250 ms reap grace. If the group still has not completed, its handle moves to a cargo-each-local polling reaper started before any command. The reaper checks every retained group without blocking on one child, remains the wait owner after the caller returns, and exits after all senders disconnect and retained groups are -collected. Reaper startup and handoff failures are explicit infrastructure -failures; a failed handoff retains the group handle in a persistent fallback -queue and starts an emergency polling reaper. +collected. Interrupted observations are retried; terminal observation +errors are reported and removed. Reaper startup and handoff failures are +explicit infrastructure failures; a failed handoff retains the group handle +in a persistent fallback queue and starts an emergency polling reaper. Child commands inherit `PATH` explicitly. On Windows this makes relative program lookup honor the inherited `PATH` order instead of preferring an unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index 86e210980..c40a40efa 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -349,17 +349,21 @@ no-op. intentionally truncated. The plan-contiguous wave bound also caps the number of retained capture files. A background or escaped descendant can continue and can append through an - inherited handle, but it cannot hold capture open and bytes written after - finalization are outside the finite snapshot. Capture create, handle-reopen, - length, seek, or read failures are infrastructure failures. Each invocation - owns both files through emission and removes them through RAII. + inherited handle; bytes written after finalization are outside the finite + snapshot. RAII removes cargo-each's directory entry after emission, but an + untimed descendant that preserves the inherited writer can keep the backing + storage allocated and continue growing it until that handle closes. The wave + bound therefore limits cargo-each-owned files, not storage retained by + preserved descendants. There is no portable way to revoke an inherited file + handle without terminating that descendant. Capture create, handle-reopen, + length, seek, or read failures are infrastructure failures. - **Timeouts terminate jobs or process groups.** A timed-out command is a failure. `command-group` creates a job object on Windows and a process group on Unix. Both timed streamed and captured execution observe the launched leader directly, preserving its exit status even while an ordinary background group member remains. The group handle remains available solely for deadline termination. At the deadline cargo-each kills that boundary, - polls completion with bounded sleeps, and allows 250 ms for the group to + polls the direct leader with bounded sleeps, and allows 250 ms for it to finish. If it still has not completed, the `GroupChild` moves to one cargo-each-local polling reaper started before any child process. The reaper polls every retained group rather than blocking forever on one, owns groups @@ -370,9 +374,11 @@ no-op. emergency polling reaper. If that thread cannot start, the queue retains ownership and a later failed handoff retries startup. A failed kill, observation, bounded reap, reaper startup, or handoff is an infrastructure - failure. Unix process groups are not sealed containment: a descendant can - escape by creating a new session, so timeout cleanup remains best-effort for - escaped descendants. + failure. Reaper polling retries interrupted observations; a terminal + observation error is reported asynchronously and the unobservable handle is + no longer retained forever. Unix process groups are not sealed containment: + a descendant can escape by creating a new session, so timeout cleanup remains + best-effort for escaped descendants. - **Child executable resolution follows `PATH`.** `cargo-each` explicitly copies an inherited `PATH` onto every child command. This is equivalent to ordinary inheritance on other platforms and makes Windows resolve a relative diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 042f2ebd8..8a72d4792 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -140,18 +140,20 @@ //! current length when the leader completes (or after timeout cleanup), then //! reads exactly that finite snapshot in plan order without loading unbounded //! output into memory. Later writes by background or escaped descendants are -//! outside the snapshot, and inherited file handles cannot hold capture open. -//! Capture create, reopen, length, seek, and read failures are infrastructure -//! failures; files are removed by RAII. +//! outside the snapshot. RAII removes cargo-each's directory entry, but a +//! preserved descendant can keep the backing storage allocated and growing +//! until its inherited writer closes. Capture create, reopen, length, seek, and +//! read failures are infrastructure failures. //! //! Timed-out group termination gets a bounded 250 ms reap grace. If the group //! still has not completed, its handle moves to a cargo-each-local polling //! reaper started before any command. The reaper checks every retained group //! without blocking on one child, remains the wait owner after the caller //! returns, and exits after all senders disconnect and retained groups are -//! collected. Reaper startup and handoff failures are explicit infrastructure -//! failures; a failed handoff retains the group handle in a persistent fallback -//! queue and starts an emergency polling reaper. +//! collected. Interrupted observations are retried; terminal observation +//! errors are reported and removed. Reaper startup and handoff failures are +//! explicit infrastructure failures; a failed handoff retains the group handle +//! in a persistent fallback queue and starts an emergency polling reaper. //! Child commands inherit `PATH` explicitly. On Windows this makes relative //! program lookup honor the inherited `PATH` order instead of preferring an //! unrelated executable beside `cargo-each`. diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 387fff4d0..a98540c5f 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -250,9 +250,13 @@ fn execute_parallel( outcomes.sort_by_key(|outcome| outcome.index); for indexed in &mut outcomes { emit_buffered(&invocations[indexed.index], &mut indexed.outcome).into_app_err("failed to emit buffered command output")?; - if first_failure.is_none() && indexed.outcome.result.failed() { - first_failure = Some(parallel_failure_exit_code(&indexed.outcome.result)); - } + record_emitted_failure( + &indexed.outcome.result, + keep_going, + &mut any_failed, + &mut stop_launching, + &mut first_failure, + ); } outcomes.clear(); if stop_launching { @@ -271,6 +275,25 @@ fn execute_parallel( } } +fn record_emitted_failure( + result: &InvocationResult, + keep_going: bool, + any_failed: &mut bool, + stop_launching: &mut bool, + first_failure: &mut Option, +) { + if !result.failed() { + return; + } + *any_failed = true; + if failure_stops_launching(keep_going, true) { + *stop_launching = true; + } + if first_failure.is_none() { + *first_failure = Some(parallel_failure_exit_code(result)); + } +} + fn parallel_failure_exit_code(result: &InvocationResult) -> ExitCode { match result { InvocationResult::Exited(status) => ExitCode::from(exit_byte(status.code())), @@ -561,9 +584,13 @@ fn wait_for_process( #[cfg_attr(coverage_nightly, coverage(off))] fn terminate_group_bounded(child: GroupChild, reaper: &GroupReaper) -> io::Result { - terminate_group_with(child, TERMINATION_GRACE, GroupChild::kill, GroupChild::try_wait, |child| { - reaper.handoff(child) - }) + terminate_group_with( + child, + TERMINATION_GRACE, + GroupChild::kill, + |child| child.inner().try_wait(), + |child| reaper.handoff(child), + ) } fn terminate_group_with( @@ -647,7 +674,7 @@ fn failed_handoffs() -> &'static Mutex> { fn start_failed_handoff_reaper(retained: &'static Mutex>) -> io::Result<()> { static RUNNING: AtomicBool = AtomicBool::new(false); - start_failed_handoff_reaper_with(retained, &RUNNING, GroupChild::try_wait, |job| { + start_failed_handoff_reaper_with(retained, &RUNNING, GroupChild::try_wait, report_reaper_failure, |job| { thread::Builder::new() .name("cargo-each-fallback-reaper".to_owned()) .spawn(job) @@ -659,12 +686,15 @@ fn start_failed_handoff_reaper_with( retained: &'static Mutex>, running: &'static AtomicBool, try_wait: fn(&mut T) -> io::Result>, + report_failure: fn(&io::Error), spawn: impl FnOnce(ReaperJob) -> io::Result<()>, ) -> io::Result<()> { if running.compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire).is_err() { return Ok(()); } - match spawn(Box::new(move || poll_failed_handoffs(retained, running, try_wait))) { + match spawn(Box::new(move || { + poll_failed_handoffs(retained, running, try_wait, report_failure); + })) { Ok(()) => Ok(()), Err(error) => { running.store(false, Ordering::Release); @@ -677,10 +707,11 @@ fn poll_failed_handoffs( retained: &Mutex>, running: &AtomicBool, mut try_wait: impl FnMut(&mut T) -> io::Result>, + mut report_failure: impl FnMut(&io::Error), ) { loop { let mut children = retained.lock().unwrap_or_else(std::sync::PoisonError::into_inner); - children.retain_mut(|child| !matches!(try_wait(child), Ok(Some(_)))); + children.retain_mut(|child| retain_after_reaper_observation(try_wait(child), &mut report_failure)); if children.is_empty() { running.store(false, Ordering::Release); return; @@ -721,11 +752,6 @@ fn handoff_group_with_fallback( handoff } -struct ReaperEntry { - child: T, - failure_reported: bool, -} - #[cfg_attr(coverage_nightly, coverage(off))] #[mutants::skip] // Process-thread diagnostic for an OS observation failure; behavior is covered through the injected reporter seam. fn report_reaper_failure(error: &io::Error) { @@ -746,31 +772,18 @@ fn poll_reaper( mut try_wait: impl FnMut(&mut T) -> io::Result>, mut report_failure: impl FnMut(&io::Error), ) { - let mut retained: Vec> = Vec::new(); + let mut retained: Vec = Vec::new(); let mut connected = true; loop { if connected { match receiver.recv_timeout(REAPER_POLL_INTERVAL) { - Ok(child) => retained.push(ReaperEntry { - child, - failure_reported: false, - }), + Ok(child) => retained.push(child), Err(mpsc::RecvTimeoutError::Timeout) => {} Err(mpsc::RecvTimeoutError::Disconnected) => connected = false, } } - retained.retain_mut(|entry| match try_wait(&mut entry.child) { - Ok(Some(_)) => false, - Ok(None) => true, - Err(error) => { - if !entry.failure_reported { - report_failure(&error); - entry.failure_reported = true; - } - true - } - }); + retained.retain_mut(|child| retain_after_reaper_observation(try_wait(child), &mut report_failure)); match (connected, retained.is_empty()) { (false, true) => return, @@ -779,6 +792,18 @@ fn poll_reaper( } } +fn retain_after_reaper_observation(observation: io::Result>, report_failure: &mut impl FnMut(&io::Error)) -> bool { + match observation { + Ok(Some(_)) => false, + Ok(None) => true, + Err(error) if error.kind() == io::ErrorKind::Interrupted => true, + Err(error) => { + report_failure(&error); + false + } + } +} + fn with_reaper_handoff(message: &str, reaper: &io::Result<()>) -> String { match reaper { Ok(()) => format!("{message}; the process group was moved to the local polling reaper"), @@ -1059,7 +1084,7 @@ mod tests { SnapshotSource, TemporarySnapshot, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, add_infrastructure_failure, combine_captured_output, display_duration, effective_worker_count, emit_buffered_to, execute_parallel, exit_byte, failed_handoffs, failure_stops_launching, finish_capture, handoff_group, handoff_group_with_fallback, panic_description, - parallel_failure_exit_code, poll_process_exit, poll_reaper, run_captured, run_captured_with, run_streamed, + parallel_failure_exit_code, poll_process_exit, poll_reaper, record_emitted_failure, run_captured, run_captured_with, run_streamed, run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_group, spawn_worker, start_failed_handoff_reaper_with, terminate_group_bounded, terminate_group_with, wait_for_process, wait_for_worker, with_cleanup_failure, with_reaper_handoff, }; @@ -1089,12 +1114,17 @@ mod tests { static FALLBACK_TEST_GROUPS: OnceLock>> = OnceLock::new(); static FALLBACK_TEST_RUNNING: AtomicBool = AtomicBool::new(false); static FALLBACK_TEST_COLLECTED: AtomicUsize = AtomicUsize::new(0); + static FALLBACK_TEST_REPORTED: AtomicUsize = AtomicUsize::new(0); fn observe_fallback_test_group(state: &mut usize) -> io::Result> { match *state { + 3 => Err(io::Error::other("injected terminal fallback observation failure")), 2 => { *state = 1; - Err(io::Error::other("injected fallback observation failure")) + Err(io::Error::new( + io::ErrorKind::Interrupted, + "injected interrupted fallback observation", + )) } 1 => { *state = 0; @@ -1107,6 +1137,10 @@ mod tests { } } + fn report_fallback_test_error(_error: &io::Error) { + FALLBACK_TEST_REPORTED.fetch_add(1, Ordering::SeqCst); + } + fn result_infrastructure_message(result: InvocationResult) -> String { let InvocationResult::Infrastructure(message) = result else { panic!("the test expects an infrastructure outcome"); @@ -1249,6 +1283,46 @@ mod tests { assert!(!failure_stops_launching(true, true)); } + #[test] + fn emitted_capture_failures_update_parallel_scheduler_state() { + let result = InvocationResult::Infrastructure("capture read failed".to_owned()); + let mut any_failed = false; + let mut stop_launching = false; + let mut first_failure = None; + record_emitted_failure( + &InvocationResult::Exited(successful_status()), + false, + &mut any_failed, + &mut stop_launching, + &mut first_failure, + ); + assert!(!any_failed); + assert!(!stop_launching); + assert!(first_failure.is_none()); + + record_emitted_failure(&result, false, &mut any_failed, &mut stop_launching, &mut first_failure); + assert!(any_failed); + assert!(stop_launching); + assert_eq!(first_failure, Some(ExitCode::from(2))); + + let mut keep_going_failed = false; + let mut keep_going_stop = false; + let mut keep_going_first = None; + record_emitted_failure(&result, true, &mut keep_going_failed, &mut keep_going_stop, &mut keep_going_first); + assert!(keep_going_failed); + assert!(!keep_going_stop); + assert_eq!(keep_going_first, Some(ExitCode::from(2))); + + record_emitted_failure( + &InvocationResult::Exited(failed_status(7)), + true, + &mut keep_going_failed, + &mut keep_going_stop, + &mut keep_going_first, + ); + assert_eq!(keep_going_first, Some(ExitCode::from(2)), "the first plan-order failure wins"); + } + #[test] fn sequential_and_parallel_failures_preserve_their_taxonomy() { let plan = Plan { @@ -1458,6 +1532,7 @@ mod tests { fn polling_reaper_checks_every_retained_group_and_eventually_collects_them() { struct FakeGroup { errors_remaining: usize, + error_kind: io::ErrorKind, polls_remaining: usize, collected: Arc, } @@ -1467,16 +1542,25 @@ mod tests { let reports = Arc::new(AtomicUsize::new(0)); let first = FakeGroup { errors_remaining: 2, + error_kind: io::ErrorKind::Interrupted, polls_remaining: 20, collected: Arc::clone(&collected), }; let second = FakeGroup { errors_remaining: 0, + error_kind: io::ErrorKind::Other, + polls_remaining: 0, + collected: Arc::clone(&collected), + }; + let terminal = FakeGroup { + errors_remaining: 1, + error_kind: io::ErrorKind::Other, polls_remaining: 0, collected: Arc::clone(&collected), }; sender.send(first).expect("reaper receiver is connected"); sender.send(second).expect("reaper receiver is connected"); + sender.send(terminal).expect("reaper receiver is connected"); drop(sender); let worker = thread::spawn({ @@ -1487,7 +1571,7 @@ mod tests { |group| { if group.errors_remaining > 0 { group.errors_remaining -= 1; - return Err(io::Error::other("injected reaper observation failure")); + return Err(io::Error::new(group.error_kind, "injected reaper observation failure")); } if group.polls_remaining == 0 { group.collected.fetch_add(1, Ordering::SeqCst); @@ -1559,25 +1643,43 @@ mod tests { retained.lock().expect("fallback ownership mutex is not poisoned").clear(); FALLBACK_TEST_RUNNING.store(true, Ordering::Release); FALLBACK_TEST_COLLECTED.store(0, Ordering::SeqCst); - start_failed_handoff_reaper_with(retained, &FALLBACK_TEST_RUNNING, observe_fallback_test_group, |_| { - panic!("an already-running fallback must not spawn another thread"); - }) + FALLBACK_TEST_REPORTED.store(0, Ordering::SeqCst); + start_failed_handoff_reaper_with( + retained, + &FALLBACK_TEST_RUNNING, + observe_fallback_test_group, + report_fallback_test_error, + |_| { + panic!("an already-running fallback must not spawn another thread"); + }, + ) .expect("an already-running fallback accepts more work"); FALLBACK_TEST_RUNNING.store(false, Ordering::Release); retained.lock().expect("fallback ownership mutex is not poisoned").push(2); - let error = start_failed_handoff_reaper_with(retained, &FALLBACK_TEST_RUNNING, observe_fallback_test_group, |job| { - drop(job); - Err(io::Error::other("injected fallback thread failure")) - }) + let error = start_failed_handoff_reaper_with( + retained, + &FALLBACK_TEST_RUNNING, + observe_fallback_test_group, + report_fallback_test_error, + |job| { + drop(job); + Err(io::Error::other("injected fallback thread failure")) + }, + ) .expect_err("fallback thread failure is reported"); assert!(error.to_string().contains("injected fallback thread failure")); assert!(!FALLBACK_TEST_RUNNING.load(Ordering::Acquire)); assert_eq!(retained.lock().expect("fallback ownership mutex is not poisoned").len(), 1); - start_failed_handoff_reaper_with(retained, &FALLBACK_TEST_RUNNING, observe_fallback_test_group, |job| { - thread::Builder::new().spawn(job).map(drop) - }) + retained.lock().expect("fallback ownership mutex is not poisoned").push(3); + start_failed_handoff_reaper_with( + retained, + &FALLBACK_TEST_RUNNING, + observe_fallback_test_group, + report_fallback_test_error, + |job| thread::Builder::new().spawn(job).map(drop), + ) .expect("fallback polling thread starts"); let deadline = Instant::now() + Duration::from_secs(1); while FALLBACK_TEST_RUNNING.load(Ordering::Acquire) && Instant::now() < deadline { @@ -1588,6 +1690,7 @@ mod tests { "fallback polling thread did not finish" ); assert_eq!(FALLBACK_TEST_COLLECTED.load(Ordering::SeqCst), 1); + assert_eq!(FALLBACK_TEST_REPORTED.load(Ordering::SeqCst), 1); assert!(retained.lock().expect("fallback ownership mutex is not poisoned").is_empty()); } From 23bd6eef732a131434af671525e46d5ccd9a2a99 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 22 Sep 2026 20:17:03 +0200 Subject: [PATCH 30/37] test(cargo-each): pin reaper retention decisions Exercise the completed, pending, interrupted, and terminal observation truth table directly so an always-retain mutation fails immediately. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 28 +++++++++++++++++++++++++--- 1 file changed, 25 insertions(+), 3 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index a98540c5f..4d45fb870 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -1084,9 +1084,10 @@ mod tests { SnapshotSource, TemporarySnapshot, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, add_infrastructure_failure, combine_captured_output, display_duration, effective_worker_count, emit_buffered_to, execute_parallel, exit_byte, failed_handoffs, failure_stops_launching, finish_capture, handoff_group, handoff_group_with_fallback, panic_description, - parallel_failure_exit_code, poll_process_exit, poll_reaper, record_emitted_failure, run_captured, run_captured_with, run_streamed, - run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_group, spawn_worker, start_failed_handoff_reaper_with, - terminate_group_bounded, terminate_group_with, wait_for_process, wait_for_worker, with_cleanup_failure, with_reaper_handoff, + parallel_failure_exit_code, poll_process_exit, poll_reaper, record_emitted_failure, retain_after_reaper_observation, run_captured, + run_captured_with, run_streamed, run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_group, spawn_worker, + start_failed_handoff_reaper_with, terminate_group_bounded, terminate_group_with, wait_for_process, wait_for_worker, + with_cleanup_failure, with_reaper_handoff, }; fn invocation(argv: &[&str]) -> Invocation { @@ -1601,6 +1602,27 @@ mod tests { assert_eq!(reports.load(Ordering::SeqCst), 1, "each failing group is reported once"); } + #[test] + fn reaper_observation_retention_distinguishes_transient_and_terminal_states() { + let mut reported = Vec::new(); + assert!(!retain_after_reaper_observation(Ok(Some(successful_status())), &mut |error| { + reported.push(error.kind()); + })); + assert!(retain_after_reaper_observation(Ok(None), &mut |error| { + reported.push(error.kind()); + })); + assert!(retain_after_reaper_observation( + Err(io::Error::new(io::ErrorKind::Interrupted, "interrupted")), + &mut |error| { + reported.push(error.kind()); + }, + )); + assert!(!retain_after_reaper_observation(Err(io::Error::other("terminal")), &mut |error| { + reported.push(error.kind()); + },)); + assert_eq!(reported, [io::ErrorKind::Other]); + } + #[test] fn failed_reaper_handoff_recovers_ownership_before_returning() { let (connected_sender, connected_receiver) = mpsc::channel(); From fecc8466939eaf1c66c2dc1a90ea00d1d7b2edb6 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 22 Sep 2026 20:34:54 +0200 Subject: [PATCH 31/37] test(cargo-each): bound reaper completion wait Use a timed completion channel before joining the fake reaper so an always-retain mutation fails promptly instead of timing out the test binary. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 4d45fb870..ace4c61f4 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -1564,6 +1564,7 @@ mod tests { sender.send(terminal).expect("reaper receiver is connected"); drop(sender); + let (done_sender, done_receiver) = mpsc::channel(); let worker = thread::spawn({ let reports = Arc::clone(&reports); move || { @@ -1586,6 +1587,7 @@ mod tests { reports.fetch_add(1, Ordering::SeqCst); }, ); + let _receiver_gone = done_sender.send(()); } }); let deadline = Instant::now() + Duration::from_secs(1); @@ -1597,6 +1599,9 @@ mod tests { 1, "the ready group must be collected while another group remains pending" ); + done_receiver + .recv_timeout(Duration::from_secs(1)) + .expect("the finite fake reaper must stop after the sender disconnects"); worker.join().expect("the finite fake reaper exits"); assert_eq!(collected.load(Ordering::SeqCst), 2); assert_eq!(reports.load(Ordering::SeqCst), 1, "each failing group is reported once"); From 4b8f9484a3306b310ba2f36419d3f5ee2fabaf1e Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Tue, 22 Sep 2026 22:22:01 +0200 Subject: [PATCH 32/37] fix(cargo-each): clean up streamed wait failures Launch untimed streamed commands in command groups and route leader observation errors through bounded termination and the local reaper while preserving inherited standard streams. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/README.md | 9 +++---- crates/cargo-each/docs/design/README.md | 7 +++--- crates/cargo-each/src/main.rs | 9 +++---- crates/cargo-each/src/run.rs | 32 +++++++++++++++---------- 4 files changed, 33 insertions(+), 24 deletions(-) diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index e04aeb99c..869665637 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -141,10 +141,11 @@ outcomes instead of blocking the scheduler. Worker launch failures retain output already collected at earlier plan indices. Parallel work runs in plan-contiguous waves capped by the effective worker count; each completed wave is emitted and dropped before the next wave starts, bounding retained -temporary-file storage. Without `--timeout`, parallel commands are launched -in a job or process group, but cargo-each observes only the leader and does -not kill background descendants. Every genuinely parallel invocation -redirects stdout and stderr directly to separate unique temporary files. +temporary-file storage. Every command is launched in a job or process group. +Without `--timeout`, cargo-each observes only the leader and does not kill +background descendants; a post-spawn observation failure still receives +bounded group cleanup. Every genuinely parallel invocation redirects stdout +and stderr directly to separate unique temporary files. Child writers and parent readers are separately reopened so parent seeks cannot move descendant write positions. Cargo-each records each file’s current length when the leader completes (or after timeout cleanup), then diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index c40a40efa..8695e2ace 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -335,9 +335,10 @@ no-op. count. A wave is fully observed, emitted, and dropped before the next wave starts, so the number of retained invocation captures is also capped by the effective worker count. - Without `--timeout`, parallel commands are still launched in a Windows job - or Unix process group, but cargo-each observes only the launched leader and - does not kill background descendants. + Every launched command uses a Windows job or Unix process group. Without + `--timeout`, cargo-each observes only the launched leader and does not kill + ordinary background descendants. A post-spawn leader-observation failure + enters the same bounded group termination and reaper path as timeout cleanup. - **Parallel capture uses finite temporary-file snapshots.** Every genuinely parallel invocation redirects stdout and stderr directly to separate unique temporary files before group spawn; no pipe-reader threads are created. The diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 8a72d4792..983d73968 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -131,10 +131,11 @@ //! output already collected at earlier plan indices. Parallel work runs in //! plan-contiguous waves capped by the effective worker count; each completed //! wave is emitted and dropped before the next wave starts, bounding retained -//! temporary-file storage. Without `--timeout`, parallel commands are launched -//! in a job or process group, but cargo-each observes only the leader and does -//! not kill background descendants. Every genuinely parallel invocation -//! redirects stdout and stderr directly to separate unique temporary files. +//! temporary-file storage. Every command is launched in a job or process group. +//! Without `--timeout`, cargo-each observes only the leader and does not kill +//! background descendants; a post-spawn observation failure still receives +//! bounded group cleanup. Every genuinely parallel invocation redirects stdout +//! and stderr directly to separate unique temporary files. //! Child writers and parent readers are separately reopened so parent seeks //! cannot move descendant write positions. Cargo-each records each file's //! current length when the leader completes (or after timeout cleanup), then diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index ace4c61f4..b92adb9a9 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -158,7 +158,7 @@ fn execute_sequential(plan: &Plan, keep_going: bool, timeout: Option, if let Some(timeout) = timeout { run_streamed_with_timeout(invocation, timeout, reaper) } else { - run_streamed(invocation) + run_streamed(invocation, reaper) } }) } @@ -375,15 +375,8 @@ fn panic_description(payload: &(dyn std::any::Any + Send)) -> &str { } } -fn run_streamed(invocation: &Invocation) -> InvocationResult { - let (program, mut command) = match command_for(invocation) { - Ok(command) => command, - Err(message) => return InvocationResult::Infrastructure(message), - }; - match command.status() { - Ok(status) => InvocationResult::Exited(status), - Err(error) => InvocationResult::Infrastructure(format!("failed to spawn `{program}`: {error}")), - } +fn run_streamed(invocation: &Invocation, reaper: &GroupReaper) -> InvocationResult { + run_streamed_group_with(invocation, None, reaper, spawn_group) } fn run_streamed_with_timeout(invocation: &Invocation, timeout: Duration, reaper: &GroupReaper) -> InvocationResult { @@ -395,6 +388,15 @@ fn run_streamed_with_timeout_with( timeout: Duration, reaper: &GroupReaper, spawn: impl FnOnce(Command) -> Result, +) -> InvocationResult { + run_streamed_group_with(invocation, Some(timeout), reaper, spawn) +} + +fn run_streamed_group_with( + invocation: &Invocation, + timeout: Option, + reaper: &GroupReaper, + spawn: impl FnOnce(Command) -> Result, ) -> InvocationResult { let (program, command) = match command_for(invocation) { Ok(command) => command, @@ -408,7 +410,7 @@ fn run_streamed_with_timeout_with( }; wait_for_process( tree, - Some(timeout), + timeout, |process| process.inner().try_wait(), |process| terminate_group_bounded(process, reaper), "observe child process leader", @@ -1799,7 +1801,9 @@ mod tests { fn direct_runners_report_empty_and_unspawnable_commands() { let reaper = test_reaper(); let empty = invocation(&[]); - assert!(matches!(run_streamed(&empty), InvocationResult::Infrastructure(message) if message.contains("empty argument vector"))); + assert!( + matches!(run_streamed(&empty, &reaper), InvocationResult::Infrastructure(message) if message.contains("empty argument vector")) + ); assert!(infrastructure_message(run_captured(&empty, None, &reaper)).contains("empty argument vector")); assert!(matches!( run_streamed_with_timeout(&empty, Duration::from_secs(1), &reaper), @@ -1807,7 +1811,9 @@ mod tests { )); let missing = invocation(&["__cargo_each_missing_program_for_unit_test__"]); - assert!(matches!(run_streamed(&missing), InvocationResult::Infrastructure(message) if message.contains("failed to spawn"))); + assert!( + matches!(run_streamed(&missing, &reaper), InvocationResult::Infrastructure(message) if message.contains("failed to spawn")) + ); assert!(infrastructure_message(run_captured(&missing, None, &reaper)).contains("failed to spawn")); let injected = run_streamed_with_timeout_with(&invocation(&["rustc", "--version"]), Duration::from_secs(1), &reaper, |_| { From 4847961fbb8a6a1d03c97e664b468af004f906cd Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 23 Sep 2026 00:15:39 +0200 Subject: [PATCH 33/37] fix(cargo-each): preserve sequential terminal semantics Keep untimed effective-one commands as ordinary children while separating spawn from wait, applying bounded direct-child cleanup, and prestarting a child-handle reaper for failed waits. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/README.md | 12 +- crates/cargo-each/docs/design/README.md | 11 +- crates/cargo-each/src/main.rs | 12 +- crates/cargo-each/src/run.rs | 208 +++++++++++++++++++----- 4 files changed, 187 insertions(+), 56 deletions(-) diff --git a/crates/cargo-each/README.md b/crates/cargo-each/README.md index 869665637..743cf2852 100644 --- a/crates/cargo-each/README.md +++ b/crates/cargo-each/README.md @@ -141,11 +141,13 @@ outcomes instead of blocking the scheduler. Worker launch failures retain output already collected at earlier plan indices. Parallel work runs in plan-contiguous waves capped by the effective worker count; each completed wave is emitted and dropped before the next wave starts, bounding retained -temporary-file storage. Every command is launched in a job or process group. -Without `--timeout`, cargo-each observes only the leader and does not kill -background descendants; a post-spawn observation failure still receives -bounded group cleanup. Every genuinely parallel invocation redirects stdout -and stderr directly to separate unique temporary files. +temporary-file storage. Untimed effective-one execution uses an ordinary +child, preserving terminal foreground behavior and Ctrl-C delivery; a +post-spawn wait failure gets bounded child cleanup and reaper ownership. +Timed and genuinely parallel commands use a job or process group. Without +`--timeout`, cargo-each observes only the leader and does not kill background +descendants. Every genuinely parallel invocation redirects stdout and +stderr directly to separate unique temporary files. Child writers and parent readers are separately reopened so parent seeks cannot move descendant write positions. Cargo-each records each file’s current length when the leader completes (or after timeout cleanup), then diff --git a/crates/cargo-each/docs/design/README.md b/crates/cargo-each/docs/design/README.md index 8695e2ace..47092ad9f 100644 --- a/crates/cargo-each/docs/design/README.md +++ b/crates/cargo-each/docs/design/README.md @@ -335,10 +335,13 @@ no-op. count. A wave is fully observed, emitted, and dropped before the next wave starts, so the number of retained invocation captures is also capped by the effective worker count. - Every launched command uses a Windows job or Unix process group. Without - `--timeout`, cargo-each observes only the launched leader and does not kill - ordinary background descendants. A post-spawn leader-observation failure - enters the same bounded group termination and reaper path as timeout cleanup. + Untimed effective-one execution uses an ordinary child so inherited terminal + streams, foreground-group behavior, and Ctrl-C delivery match direct command + execution. It spawns and waits separately; a post-spawn observation failure + receives bounded direct-child termination and transfers an unreaped handle + to the local polling reaper. Timed and genuinely parallel commands use a + Windows job or Unix process group. Without `--timeout`, cargo-each observes + only the launched leader and does not kill ordinary background descendants. - **Parallel capture uses finite temporary-file snapshots.** Every genuinely parallel invocation redirects stdout and stderr directly to separate unique temporary files before group spawn; no pipe-reader threads are created. The diff --git a/crates/cargo-each/src/main.rs b/crates/cargo-each/src/main.rs index 983d73968..6babcb115 100644 --- a/crates/cargo-each/src/main.rs +++ b/crates/cargo-each/src/main.rs @@ -131,11 +131,13 @@ //! output already collected at earlier plan indices. Parallel work runs in //! plan-contiguous waves capped by the effective worker count; each completed //! wave is emitted and dropped before the next wave starts, bounding retained -//! temporary-file storage. Every command is launched in a job or process group. -//! Without `--timeout`, cargo-each observes only the leader and does not kill -//! background descendants; a post-spawn observation failure still receives -//! bounded group cleanup. Every genuinely parallel invocation redirects stdout -//! and stderr directly to separate unique temporary files. +//! temporary-file storage. Untimed effective-one execution uses an ordinary +//! child, preserving terminal foreground behavior and Ctrl-C delivery; a +//! post-spawn wait failure gets bounded child cleanup and reaper ownership. +//! Timed and genuinely parallel commands use a job or process group. Without +//! `--timeout`, cargo-each observes only the leader and does not kill background +//! descendants. Every genuinely parallel invocation redirects stdout and +//! stderr directly to separate unique temporary files. //! Child writers and parent readers are separately reopened so parent seeks //! cannot move descendant write positions. Cargo-each records each file's //! current length when the leader completes (or after timeout cleanup), then diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index b92adb9a9..3d4d9c654 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -8,7 +8,7 @@ use std::collections::BTreeSet; use std::io::{self, Read as _, Seek as _, SeekFrom, Write as _}; use std::num::NonZeroUsize; use std::panic::{self, UnwindSafe}; -use std::process::{Command, ExitCode, ExitStatus, Stdio}; +use std::process::{Child, Command, ExitCode, ExitStatus, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Mutex, OnceLock, mpsc}; use std::time::{Duration, Instant}; @@ -140,7 +140,7 @@ fn parse_target_kinds(kinds: &[String]) -> Result, AppError } fn execute(plan: &Plan, keep_going: bool, jobs: NonZeroUsize, timeout: Option) -> Result { - let reaper = GroupReaper::start().into_app_err("failed to start cargo-each process-group reaper")?; + let reaper = ProcessReaper::start().into_app_err("failed to start cargo-each process reaper")?; let worker_count = effective_worker_count(jobs, plan.invocations.len()); if worker_count.get() == 1 { Ok(execute_sequential(plan, keep_going, timeout, &reaper)) @@ -153,7 +153,7 @@ fn effective_worker_count(requested: NonZeroUsize, plan_size: usize) -> NonZeroU NonZeroUsize::new(requested.get().min(plan_size)).expect("Plan::is_empty is checked before execute, so the execution plan is nonempty") } -fn execute_sequential(plan: &Plan, keep_going: bool, timeout: Option, reaper: &GroupReaper) -> ExitCode { +fn execute_sequential(plan: &Plan, keep_going: bool, timeout: Option, reaper: &ProcessReaper) -> ExitCode { execute_sequential_with(plan, keep_going, timeout, |invocation, timeout| { if let Some(timeout) = timeout { run_streamed_with_timeout(invocation, timeout, reaper) @@ -205,7 +205,7 @@ fn execute_parallel( keep_going: bool, worker_count: NonZeroUsize, timeout: Option, - reaper: &GroupReaper, + reaper: &ProcessReaper, ) -> Result { let invocations = plan.invocations.clone(); let mut workers = Vec::with_capacity(worker_count.get()); @@ -306,7 +306,7 @@ fn failure_stops_launching(keep_going: bool, failed: bool) -> bool { matches!((keep_going, failed), (false, true)) } -fn spawn_worker(index: usize, invocation: Invocation, timeout: Option, reaper: GroupReaper) -> io::Result { +fn spawn_worker(index: usize, invocation: Invocation, timeout: Option, reaper: ProcessReaper) -> io::Result { #[cfg(test)] if invocation .argv @@ -375,18 +375,43 @@ fn panic_description(payload: &(dyn std::any::Any + Send)) -> &str { } } -fn run_streamed(invocation: &Invocation, reaper: &GroupReaper) -> InvocationResult { - run_streamed_group_with(invocation, None, reaper, spawn_group) +fn run_streamed(invocation: &Invocation, reaper: &ProcessReaper) -> InvocationResult { + run_streamed_with(invocation, reaper, spawn_child) } -fn run_streamed_with_timeout(invocation: &Invocation, timeout: Duration, reaper: &GroupReaper) -> InvocationResult { +fn run_streamed_with( + invocation: &Invocation, + reaper: &ProcessReaper, + spawn: impl FnOnce(Command) -> Result, +) -> InvocationResult { + let (program, command) = match command_for(invocation) { + Ok(command) => command, + Err(message) => return InvocationResult::Infrastructure(message), + }; + let child = match spawn(command) { + Ok(child) => child, + Err(error) => { + return InvocationResult::Infrastructure(format!("failed to spawn `{program}`: {error}")); + } + }; + wait_for_process( + child, + None, + Child::try_wait, + |child| terminate_child_bounded(child, reaper), + "observe child process", + ) + .result +} + +fn run_streamed_with_timeout(invocation: &Invocation, timeout: Duration, reaper: &ProcessReaper) -> InvocationResult { run_streamed_with_timeout_with(invocation, timeout, reaper, spawn_group) } fn run_streamed_with_timeout_with( invocation: &Invocation, timeout: Duration, - reaper: &GroupReaper, + reaper: &ProcessReaper, spawn: impl FnOnce(Command) -> Result, ) -> InvocationResult { run_streamed_group_with(invocation, Some(timeout), reaper, spawn) @@ -395,7 +420,7 @@ fn run_streamed_with_timeout_with( fn run_streamed_group_with( invocation: &Invocation, timeout: Option, - reaper: &GroupReaper, + reaper: &ProcessReaper, spawn: impl FnOnce(Command) -> Result, ) -> InvocationResult { let (program, command) = match command_for(invocation) { @@ -418,7 +443,7 @@ fn run_streamed_group_with( .result } -fn run_captured(invocation: &Invocation, timeout: Option, reaper: &GroupReaper) -> BufferedOutcome { +fn run_captured(invocation: &Invocation, timeout: Option, reaper: &ProcessReaper) -> BufferedOutcome { #[cfg(test)] assert!( invocation.argv.first().is_none_or(|program| program != WORKER_PANIC_TEST_PROGRAM), @@ -431,7 +456,7 @@ fn run_captured(invocation: &Invocation, timeout: Option, reaper: &Gro fn run_captured_with( invocation: &Invocation, timeout: Option, - reaper: &GroupReaper, + reaper: &ProcessReaper, mut capture: impl FnMut(&'static str) -> io::Result<(Box, Stdio)>, spawner: impl FnOnce(Command) -> Result, ) -> BufferedOutcome { @@ -519,6 +544,10 @@ fn spawn_group(mut command: Command) -> Result { command.group_spawn().map_err(|error| error.to_string()) } +fn spawn_child(mut command: Command) -> Result { + command.spawn().map_err(|error| error.to_string()) +} + fn create_output_capture(_stream: &'static str) -> io::Result<(Box, Stdio)> { let temporary = tempfile::NamedTempFile::new()?; let reader = temporary.reopen()?; @@ -585,16 +614,23 @@ fn wait_for_process( } #[cfg_attr(coverage_nightly, coverage(off))] -fn terminate_group_bounded(child: GroupChild, reaper: &GroupReaper) -> io::Result { +fn terminate_group_bounded(child: GroupChild, reaper: &ProcessReaper) -> io::Result { terminate_group_with( child, TERMINATION_GRACE, GroupChild::kill, |child| child.inner().try_wait(), - |child| reaper.handoff(child), + |child| reaper.handoff_group(child), ) } +#[cfg_attr(coverage_nightly, coverage(off))] +fn terminate_child_bounded(child: Child, reaper: &ProcessReaper) -> io::Result { + terminate_group_with(child, TERMINATION_GRACE, Child::kill, Child::try_wait, |child| { + reaper.handoff_child(child) + }) +} + fn terminate_group_with( mut child: T, grace: Duration, @@ -612,10 +648,10 @@ fn terminate_group_with( Ok(None) => { let reaper = detach(child); let message = kill_error.map_or_else( - || format!("process group did not exit within {} ms after termination", grace.as_millis()), + || format!("process boundary did not exit within {} ms after termination", grace.as_millis()), |error| { format!( - "{error}; process group did not exit within {} ms after termination", + "{error}; process boundary did not exit within {} ms after termination", grace.as_millis() ) }, @@ -626,26 +662,28 @@ fn terminate_group_with( let reaper = detach(child); Err(io::Error::new( error.kind(), - with_reaper_handoff(&format!("failed to observe process group after termination: {error}"), &reaper), + with_reaper_handoff(&format!("failed to observe process boundary after termination: {error}"), &reaper), )) } } } #[derive(Debug)] -struct GroupReaper { - sender: mpsc::Sender, +struct ProcessReaper { + group_sender: mpsc::Sender, + child_sender: mpsc::Sender, } -impl Clone for GroupReaper { +impl Clone for ProcessReaper { fn clone(&self) -> Self { Self { - sender: self.sender.clone(), + group_sender: self.group_sender.clone(), + child_sender: self.child_sender.clone(), } } } -impl GroupReaper { +impl ProcessReaper { fn start() -> io::Result { Self::start_with(|job| { thread::Builder::new() @@ -655,17 +693,29 @@ impl GroupReaper { }) } - fn start_with(spawn: impl FnOnce(ReaperJob) -> io::Result<()>) -> io::Result { - let (sender, receiver) = mpsc::channel(); + fn start_with(mut spawn: impl FnMut(ReaperJob) -> io::Result<()>) -> io::Result { + let (group_sender, group_receiver) = mpsc::channel(); + spawn(Box::new(move || { + poll_reaper(&group_receiver, GroupChild::try_wait, report_reaper_failure); + }))?; + let (child_sender, child_receiver) = mpsc::channel(); spawn(Box::new(move || { - poll_reaper(&receiver, GroupChild::try_wait, report_reaper_failure); + poll_reaper(&child_receiver, Child::try_wait, report_reaper_failure); }))?; - Ok(Self { sender }) + Ok(Self { + group_sender, + child_sender, + }) } - fn handoff(&self, child: GroupChild) -> io::Result<()> { + fn handoff_group(&self, child: GroupChild) -> io::Result<()> { let retained = failed_handoffs(); - handoff_group_with_fallback(&self.sender, retained, child, || start_failed_handoff_reaper(retained)) + handoff_group_with_fallback(&self.group_sender, retained, child, || start_failed_handoff_reaper(retained)) + } + + fn handoff_child(&self, child: Child) -> io::Result<()> { + let retained = failed_child_handoffs(); + handoff_group_with_fallback(&self.child_sender, retained, child, || start_failed_child_handoff_reaper(retained)) } } @@ -684,6 +734,21 @@ fn start_failed_handoff_reaper(retained: &'static Mutex>) -> io: }) } +fn failed_child_handoffs() -> &'static Mutex> { + static FAILED_HANDOFFS: OnceLock>> = OnceLock::new(); + FAILED_HANDOFFS.get_or_init(|| Mutex::new(Vec::new())) +} + +fn start_failed_child_handoff_reaper(retained: &'static Mutex>) -> io::Result<()> { + static RUNNING: AtomicBool = AtomicBool::new(false); + start_failed_handoff_reaper_with(retained, &RUNNING, Child::try_wait, report_reaper_failure, |job| { + thread::Builder::new() + .name("cargo-each-fallback-child-reaper".to_owned()) + .spawn(job) + .map(drop) + }) +} + fn start_failed_handoff_reaper_with( retained: &'static Mutex>, running: &'static AtomicBool, @@ -808,8 +873,8 @@ fn retain_after_reaper_observation(observation: io::Result>, fn with_reaper_handoff(message: &str, reaper: &io::Result<()>) -> String { match reaper { - Ok(()) => format!("{message}; the process group was moved to the local polling reaper"), - Err(error) => format!("{message}; failed to hand the process group to the local reaper: {error}"), + Ok(()) => format!("{message}; the process wait handle was moved to the local polling reaper"), + Err(error) => format!("{message}; failed to hand the process wait handle to the local reaper: {error}"), } } @@ -1082,14 +1147,14 @@ mod tests { use std::{io, thread}; use super::{ - BufferedOutcome, CapturedOutput, CapturedStream, GroupReaper, Invocation, InvocationResult, OutputEmitError, Plan, RunningWorker, + BufferedOutcome, CapturedOutput, CapturedStream, Invocation, InvocationResult, OutputEmitError, Plan, ProcessReaper, RunningWorker, SnapshotSource, TemporarySnapshot, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, add_infrastructure_failure, combine_captured_output, display_duration, effective_worker_count, emit_buffered_to, execute_parallel, - exit_byte, failed_handoffs, failure_stops_launching, finish_capture, handoff_group, handoff_group_with_fallback, panic_description, - parallel_failure_exit_code, poll_process_exit, poll_reaper, record_emitted_failure, retain_after_reaper_observation, run_captured, - run_captured_with, run_streamed, run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_group, spawn_worker, - start_failed_handoff_reaper_with, terminate_group_bounded, terminate_group_with, wait_for_process, wait_for_worker, - with_cleanup_failure, with_reaper_handoff, + exit_byte, failed_child_handoffs, failed_handoffs, failure_stops_launching, finish_capture, handoff_group, + handoff_group_with_fallback, panic_description, parallel_failure_exit_code, poll_process_exit, poll_reaper, record_emitted_failure, + retain_after_reaper_observation, run_captured, run_captured_with, run_streamed, run_streamed_with_timeout, + run_streamed_with_timeout_with, spawn_group, spawn_worker, start_failed_handoff_reaper_with, terminate_group_bounded, + terminate_group_with, wait_for_process, wait_for_worker, with_cleanup_failure, with_reaper_handoff, }; fn invocation(argv: &[&str]) -> Invocation { @@ -1175,8 +1240,8 @@ mod tests { } } - fn test_reaper() -> GroupReaper { - GroupReaper::start().expect("the test process can start its reaper") + fn test_reaper() -> ProcessReaper { + ProcessReaper::start().expect("the test process can start its reaper") } struct FakeProcess { @@ -1523,12 +1588,25 @@ mod tests { #[test] fn reaper_startup_failure_is_reported_synchronously() { - let error = GroupReaper::start_with(|job| { + let error = ProcessReaper::start_with(|job| { drop(job); Err(io::Error::other("injected reaper startup failure")) }) .expect_err("startup failure must be returned"); assert!(error.to_string().contains("injected reaper startup failure")); + + let mut starts = 0; + let error = ProcessReaper::start_with(|job| { + starts += 1; + if starts == 1 { + thread::Builder::new().spawn(job).map(drop) + } else { + drop(job); + Err(io::Error::other("injected child-reaper startup failure")) + } + }) + .expect_err("child-reaper startup failure must be returned"); + assert!(error.to_string().contains("injected child-reaper startup failure")); } #[test] @@ -1726,6 +1804,7 @@ mod tests { #[test] fn failed_handoff_storage_is_process_stable() { assert!(std::ptr::eq(failed_handoffs(), failed_handoffs())); + assert!(std::ptr::eq(failed_child_handoffs(), failed_child_handoffs())); } #[test] @@ -1733,12 +1812,18 @@ mod tests { fn disconnected_reaper_handoff_is_eventually_collected() { let (sender, receiver) = mpsc::channel(); drop(receiver); - let reaper = GroupReaper { sender }; + let (child_sender, _child_receiver) = mpsc::channel(); + let reaper = ProcessReaper { + group_sender: sender, + child_sender, + }; let mut command = Command::new("rustc"); let _ = command.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); let child = spawn_group(command).expect("spawn a short-lived process group"); - let error = reaper.handoff(child).expect_err("the disconnected primary reaper is reported"); + let error = reaper + .handoff_group(child) + .expect_err("the disconnected primary reaper is reported"); assert_eq!(error.kind(), io::ErrorKind::BrokenPipe); let deadline = Instant::now() + Duration::from_secs(2); while !failed_handoffs() @@ -1749,6 +1834,7 @@ mod tests { { thread::sleep(Duration::from_millis(10)); } + assert!( failed_handoffs() .lock() @@ -1758,6 +1844,44 @@ mod tests { ); } + #[test] + #[cfg_attr(miri, ignore = "spawns a child process and a fallback reaper thread")] + fn disconnected_child_reaper_handoff_is_eventually_collected() { + let (group_sender, _group_receiver) = mpsc::channel(); + let (child_sender, child_receiver) = mpsc::channel(); + drop(child_receiver); + let reaper = ProcessReaper { + group_sender, + child_sender, + }; + let mut command = Command::new("rustc"); + let child = command + .arg("--version") + .stdout(Stdio::null()) + .stderr(Stdio::null()) + .spawn() + .expect("spawn a short-lived child"); + + let error = reaper.handoff_child(child).expect_err("the disconnected child reaper is reported"); + assert_eq!(error.kind(), io::ErrorKind::BrokenPipe); + let deadline = Instant::now() + Duration::from_secs(2); + while !failed_child_handoffs() + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .is_empty() + && Instant::now() < deadline + { + thread::sleep(Duration::from_millis(10)); + } + assert!( + failed_child_handoffs() + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .is_empty(), + "the fallback child reaper must eventually collect a recovered handoff" + ); + } + #[test] #[cfg_attr(miri, ignore = "spawns process groups")] fn real_group_execution_observes_completion_and_timeout() { @@ -1775,7 +1899,7 @@ mod tests { let _ = quick.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); let mut group = spawn_group(quick).expect("spawn quick process group"); group.inner().wait().expect("quick leader exits"); - reaper.handoff(group).expect("completed group reaches the local reaper"); + reaper.handoff_group(group).expect("completed group reaches the local reaper"); drop(reaper); thread::sleep(Duration::from_millis(100)); } From 73041226947a9b7f485817b5bb4be6e59fee9466 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 23 Sep 2026 00:18:38 +0200 Subject: [PATCH 34/37] test(cargo-each): cover direct-child cleanup Exercise the bounded direct-child terminator and both process-reaper startup/handoff channels; isolate only the unforceable OS wait-error ownership adapter from coverage and mutation. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 32 +++++++++++++++++++++++++++----- 1 file changed, 27 insertions(+), 5 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 3d4d9c654..14806d316 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -394,16 +394,32 @@ fn run_streamed_with( return InvocationResult::Infrastructure(format!("failed to spawn `{program}`: {error}")); } }; + let control = StreamedChild { child, reaper }; wait_for_process( - child, + control, None, - Child::try_wait, - |child| terminate_child_bounded(child, reaper), + observe_streamed_child, + terminate_streamed_child, "observe child process", ) .result } +struct StreamedChild<'a> { + child: Child, + reaper: &'a ProcessReaper, +} + +fn observe_streamed_child(control: &mut StreamedChild<'_>) -> io::Result> { + control.child.try_wait() +} + +#[cfg_attr(coverage_nightly, coverage(off))] +#[mutants::skip] // Thin ownership adapter for an OS wait-error path; terminate_child_bounded has a real-process regression. +fn terminate_streamed_child(control: StreamedChild<'_>) -> io::Result { + terminate_child_bounded(control.child, control.reaper) +} + fn run_streamed_with_timeout(invocation: &Invocation, timeout: Duration, reaper: &ProcessReaper) -> InvocationResult { run_streamed_with_timeout_with(invocation, timeout, reaper, spawn_group) } @@ -1153,8 +1169,8 @@ mod tests { exit_byte, failed_child_handoffs, failed_handoffs, failure_stops_launching, finish_capture, handoff_group, handoff_group_with_fallback, panic_description, parallel_failure_exit_code, poll_process_exit, poll_reaper, record_emitted_failure, retain_after_reaper_observation, run_captured, run_captured_with, run_streamed, run_streamed_with_timeout, - run_streamed_with_timeout_with, spawn_group, spawn_worker, start_failed_handoff_reaper_with, terminate_group_bounded, - terminate_group_with, wait_for_process, wait_for_worker, with_cleanup_failure, with_reaper_handoff, + run_streamed_with_timeout_with, spawn_group, spawn_worker, start_failed_handoff_reaper_with, terminate_child_bounded, + terminate_group_bounded, terminate_group_with, wait_for_process, wait_for_worker, with_cleanup_failure, with_reaper_handoff, }; fn invocation(argv: &[&str]) -> Invocation { @@ -1895,6 +1911,12 @@ mod tests { assert!(!error.success()); assert!(started.elapsed() < Duration::from_secs(2)); + let child = sleeping_test_command().spawn().expect("spawn sleeping direct child"); + let started = Instant::now(); + let status = terminate_child_bounded(child, &reaper).expect("killed direct child is reaped"); + assert!(!status.success()); + assert!(started.elapsed() < Duration::from_secs(2)); + let mut quick = Command::new("rustc"); let _ = quick.arg("--version").stdout(Stdio::null()).stderr(Stdio::null()); let mut group = spawn_group(quick).expect("spawn quick process group"); From baa7561e45246208309277bea0db90158b9c1179 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 23 Sep 2026 00:39:10 +0200 Subject: [PATCH 35/37] test(cargo-each): pin direct-child observation Verify a completed ordinary child is observed as exited so an always-pending observation mutation fails immediately. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 25 ++++++++++++++++++++++--- 1 file changed, 22 insertions(+), 3 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index 14806d316..af9781b23 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -1164,11 +1164,11 @@ mod tests { use super::{ BufferedOutcome, CapturedOutput, CapturedStream, Invocation, InvocationResult, OutputEmitError, Plan, ProcessReaper, RunningWorker, - SnapshotSource, TemporarySnapshot, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, + SnapshotSource, StreamedChild, TemporarySnapshot, TreeOutcome, WORKER_PANIC_TEST_PROGRAM, WORKER_SPAWN_ERROR_TEST_PROGRAM, add_infrastructure_failure, combine_captured_output, display_duration, effective_worker_count, emit_buffered_to, execute_parallel, exit_byte, failed_child_handoffs, failed_handoffs, failure_stops_launching, finish_capture, handoff_group, - handoff_group_with_fallback, panic_description, parallel_failure_exit_code, poll_process_exit, poll_reaper, record_emitted_failure, - retain_after_reaper_observation, run_captured, run_captured_with, run_streamed, run_streamed_with_timeout, + handoff_group_with_fallback, observe_streamed_child, panic_description, parallel_failure_exit_code, poll_process_exit, poll_reaper, + record_emitted_failure, retain_after_reaper_observation, run_captured, run_captured_with, run_streamed, run_streamed_with_timeout, run_streamed_with_timeout_with, spawn_group, spawn_worker, start_failed_handoff_reaper_with, terminate_child_bounded, terminate_group_bounded, terminate_group_with, wait_for_process, wait_for_worker, with_cleanup_failure, with_reaper_handoff, }; @@ -1971,6 +1971,25 @@ mod tests { )); } + #[test] + #[cfg_attr(miri, ignore = "spawns a child process")] + fn streamed_child_observation_returns_a_completed_status() { + let reaper = test_reaper(); + let mut command = Command::new("rustc"); + let mut child = command + .arg("--version") + .stdout(Stdio::null()) + .stderr(Stdio::null()) + .spawn() + .expect("spawn short-lived child"); + let expected = child.wait().expect("wait for short-lived child"); + let mut control = StreamedChild { child, reaper: &reaper }; + assert_eq!( + observe_streamed_child(&mut control).expect("observe completed child"), + Some(expected) + ); + } + #[test] fn capture_setup_failure_is_reported_before_process_spawn() { let reaper = test_reaper(); From 37504b6492ada8bf03128281046429f7a3e59e92 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 23 Sep 2026 02:31:46 +0200 Subject: [PATCH 36/37] chore(spelling): allow untimed Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- .spelling | 1 + 1 file changed, 1 insertion(+) diff --git a/.spelling b/.spelling index 127d8068f..34d3e98df 100644 --- a/.spelling +++ b/.spelling @@ -935,3 +935,4 @@ nonblocking SHA upserts EWMA +Untimed From 1504923395084ee24c82b6a61a6f6a171d78d228 Mon Sep 17 00:00:00 2001 From: "Martin Kolinek (from Dev Box)" Date: Wed, 23 Sep 2026 04:03:05 +0200 Subject: [PATCH 37/37] fix(cargo-each): generalize reaper diagnostics Use wait-handle wording shared by process-group and ordinary-child reapers. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: a9fc919b-99f7-4134-aad1-2116321b4e0c --- crates/cargo-each/src/run.rs | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/crates/cargo-each/src/run.rs b/crates/cargo-each/src/run.rs index af9781b23..c53ea84ce 100644 --- a/crates/cargo-each/src/run.rs +++ b/crates/cargo-each/src/run.rs @@ -811,7 +811,7 @@ fn handoff_group(sender: &mpsc::Sender, retained: &Mutex>, child: T retained.lock().unwrap_or_else(std::sync::PoisonError::into_inner).push(error.0); Err(io::Error::new( io::ErrorKind::BrokenPipe, - "process-group reaper channel disconnected; a persistent fallback retained the wait handle", + "process reaper channel disconnected; a persistent fallback retained the wait handle", )) } } @@ -829,7 +829,7 @@ fn handoff_group_with_fallback( { return Err(io::Error::new( io::ErrorKind::BrokenPipe, - format!("process-group reaper channel disconnected; the fallback retained the wait handle but failed to start: {error}"), + format!("process reaper channel disconnected; the fallback retained the wait handle but failed to start: {error}"), )); } handoff @@ -844,7 +844,7 @@ fn report_reaper_failure(error: &io::Error) { .spawn(move || { let _ = writeln!( io::stderr().lock(), - "cargo each: process-group reaper failed to observe a retained group: {message}" + "cargo each: process reaper failed to observe a retained wait handle: {message}" ); }); }