Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
189 changes: 189 additions & 0 deletions .github/workflows/aws-runner-e2e.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,189 @@
name: AWS runner end to end

on:
workflow_dispatch:
inputs:
scenario:
description: Lifecycle scenario to verify
required: true
type: choice
default: success
options:
- success
- failure
- maximum-duration

permissions:
contents: read
id-token: write

jobs:
start-runner:
runs-on: ubuntu-latest
outputs:
label: ${{ steps.start.outputs.label }}
microvm-id: ${{ steps.start.outputs.microvm-id }}
region: ${{ steps.start.outputs.region }}
runner-id: ${{ steps.start.outputs.runner-id }}
steps:
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
- uses: aws-actions/configure-aws-credentials@254c19bd240aabef8777f48595e9d2d7b972184b # v6.2.1
with:
role-to-assume: ${{ vars.MICROVM_LAUNCH_ROLE_ARN }}
aws-region: ${{ vars.MICROVM_AWS_REGION }}
- uses: ./
id: start
with:
mode: start
github-token: ${{ secrets.RUNNER_E2E_TOKEN }}
image-id: ${{ vars.MICROVM_RUNNER_IMAGE_ARN }}
image-version: ${{ vars.MICROVM_RUNNER_IMAGE_VERSION }}
execution-role-arn: ${{ vars.MICROVM_EXECUTION_ROLE_ARN }}
cloudwatch-log-group: ${{ vars.MICROVM_RUNTIME_LOG_GROUP }}
runner-labels: lambda-microvm,docker,e2e
maximum-duration-seconds:
${{ inputs.scenario == 'maximum-duration' && '60' || '900' }}

target:
if: inputs.scenario != 'maximum-duration'
needs: start-runner
runs-on: ${{ needs.start-runner.outputs.label }}
timeout-minutes: 15
services:
redis:
image: public.ecr.aws/docker/library/redis:7.4.2-alpine
ports:
- 6379:6379
options: >-
--health-cmd "redis-cli ping" --health-interval 2s --health-timeout 2s
--health-retries 20
steps:
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
- name: Verify host, service container, and egress
run: |
set -euo pipefail
test "$(uname -m)" = "aarch64"
docker info
docker buildx version
docker compose version
aws sts get-caller-identity >/dev/null
docker run --rm --network host \
public.ecr.aws/docker/library/busybox:1.37.0 \
sh -ec '
nc -z 127.0.0.1 6379
nslookup github.com
wget --spider https://github.com
'

- name: Build and run an ARM64 image with Buildx
run: |
set -euo pipefail
build_dir="$(mktemp -d)"
trap 'rm -rf "${build_dir}"' EXIT
cat >"${build_dir}/Dockerfile" <<'EOF'
FROM public.ecr.aws/docker/library/busybox:1.37.0@sha256:9532d8c39891ca2ecde4d30d7710e01fb739c87a8b9299685c63704296b16028
RUN printf 'buildx-ok\n' >/result
CMD ["cat", "/result"]
EOF
docker buildx build \
--platform linux/arm64 \
--load \
--tag lambda-microvm-buildx:e2e \
"${build_dir}"
test "$(docker run --rm lambda-microvm-buildx:e2e)" = "buildx-ok"

- name: Verify Compose bridge DNS and TCP
run: |
set -euo pipefail
compose_dir="$(mktemp -d)"
trap 'docker compose -f "${compose_dir}/compose.yml" down --volumes; rm -rf "${compose_dir}"' EXIT
cat >"${compose_dir}/compose.yml" <<'EOF'
services:
redis:
image: public.ecr.aws/docker/library/redis:7.4.2-alpine
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 2s
timeout: 2s
retries: 20
probe:
image: public.ecr.aws/docker/library/busybox:1.37.0
depends_on:
redis:
condition: service_healthy
command: ["sh", "-ec", "nc -z redis 6379 && nslookup github.com"]
EOF
docker compose -f "${compose_dir}/compose.yml" up \
--abort-on-container-exit \
--exit-code-from probe

- name: Exercise failing-job cleanup
if: inputs.scenario == 'failure'
run: exit 1

duration-audit:
if: inputs.scenario == 'maximum-duration'
needs: start-runner
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6
with:
node-version: "24"
cache: npm
- run: npm ci
- uses: aws-actions/configure-aws-credentials@254c19bd240aabef8777f48595e9d2d7b972184b # v6.2.1
with:
role-to-assume: ${{ vars.MICROVM_LAUNCH_ROLE_ARN }}
aws-region: ${{ needs.start-runner.outputs.region }}
- name: Verify platform-enforced termination
env:
MICROVM_ID: ${{ needs.start-runner.outputs.microvm-id }}
run: |
sleep 90
node --input-type=module <<'NODE'
import {
GetMicrovmCommand,
LambdaMicrovmsClient,
} from "@aws-sdk/client-lambda-microvms";

const client = new LambdaMicrovmsClient({
region: process.env.AWS_REGION,
});
try {
const result = await client.send(
new GetMicrovmCommand({
microvmIdentifier: process.env.MICROVM_ID,
}),
);
if (result.state !== "TERMINATED") {
throw new Error(`MicroVM remained in state ${result.state}`);
}
} catch (error) {
if (error?.name !== "ResourceNotFoundException") {
throw error;
}
}
NODE
- name: Remove the unused JIT runner
env:
GH_TOKEN: ${{ secrets.RUNNER_E2E_TOKEN }}
RUNNER_ID: ${{ needs.start-runner.outputs.runner-id }}
run:
gh api --method DELETE
"repos/${GITHUB_REPOSITORY}/actions/runners/${RUNNER_ID}"

stop-runner:
if: always() && needs.start-runner.outputs.microvm-id != ''
needs: [start-runner, target, duration-audit]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
- uses: aws-actions/configure-aws-credentials@254c19bd240aabef8777f48595e9d2d7b972184b # v6.2.1
with:
role-to-assume: ${{ vars.MICROVM_LAUNCH_ROLE_ARN }}
aws-region: ${{ needs.start-runner.outputs.region }}
- uses: ./
with:
mode: stop
microvm-id: ${{ needs.start-runner.outputs.microvm-id }}
70 changes: 68 additions & 2 deletions .github/workflows/aws-runner-target.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,12 @@ name: AWS runner target

on:
workflow_dispatch:
inputs:
fail_target:
description: Fail after all runtime checks to verify cleanup
required: false
type: boolean
default: false

permissions:
contents: read
Expand All @@ -10,14 +16,74 @@ jobs:
target:
runs-on: [self-hosted, lambda-microvm, e2e]
timeout-minutes: 15
services:
redis:
image: public.ecr.aws/docker/library/redis:7.4.2-alpine
ports:
- 6379:6379
options: >-
--health-cmd "redis-cli ping" --health-interval 2s --health-timeout 2s
--health-retries 20
steps:
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
- name: Verify host and Docker
- name: Verify host, service container, and egress
run: |
set -euo pipefail
test "$(uname -m)" = "aarch64"
docker info
docker buildx version
docker compose version
aws sts get-caller-identity >/dev/null
docker run --rm public.ecr.aws/docker/library/busybox:1.37.0 true
docker run --rm --network host \
public.ecr.aws/docker/library/busybox:1.37.0 \
sh -ec '
nc -z 127.0.0.1 6379
nslookup github.com
wget --spider https://github.com
'

- name: Build and run an ARM64 image with Buildx
run: |
set -euo pipefail
build_dir="$(mktemp -d)"
trap 'rm -rf "${build_dir}"' EXIT
cat >"${build_dir}/Dockerfile" <<'EOF'
FROM public.ecr.aws/docker/library/busybox:1.37.0@sha256:9532d8c39891ca2ecde4d30d7710e01fb739c87a8b9299685c63704296b16028
RUN printf 'buildx-ok\n' >/result
CMD ["cat", "/result"]
EOF
docker buildx build \
--platform linux/arm64 \
--load \
--tag lambda-microvm-buildx:e2e \
"${build_dir}"
test "$(docker run --rm lambda-microvm-buildx:e2e)" = "buildx-ok"

- name: Verify Compose bridge DNS and TCP
run: |
set -euo pipefail
compose_dir="$(mktemp -d)"
trap 'docker compose -f "${compose_dir}/compose.yml" down --volumes; rm -rf "${compose_dir}"' EXIT
cat >"${compose_dir}/compose.yml" <<'EOF'
services:
redis:
image: public.ecr.aws/docker/library/redis:7.4.2-alpine
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 2s
timeout: 2s
retries: 20
probe:
image: public.ecr.aws/docker/library/busybox:1.37.0
depends_on:
redis:
condition: service_healthy
command: ["sh", "-ec", "nc -z redis 6379 && nslookup github.com"]
EOF
docker compose -f "${compose_dir}/compose.yml" up \
--abort-on-container-exit \
--exit-code-from probe

- name: Exercise failing-job cleanup
if: inputs.fail_target
run: exit 1
10 changes: 10 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,8 @@ jobs:
node-version: "24"
cache: npm
- run: npm ci
- name: Audit Node dependencies
run: npm audit --audit-level=high
- run: npm run check
- name: Verify committed Action bundle
run: git diff --exit-code -- dist
Expand Down Expand Up @@ -48,6 +50,14 @@ jobs:
load: true
cache-from: type=gha
cache-to: type=gha,mode=max
- name: Scan runner image
uses: anchore/scan-action@e1165082ffb1fe366ebaf02d8526e7c4989ea9d2 # v7.4.0
with:
image: lambda-microvm-github-runner:test
fail-build: true
severity-cutoff: critical
only-fixed: true
output-format: table
- name: Verify immutable image contents
run: |
docker run --rm --platform linux/arm64 \
Expand Down
10 changes: 10 additions & 0 deletions .github/workflows/release.yml
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,8 @@ jobs:
node-version: "24"
cache: npm
- run: npm ci
- name: Audit Node dependencies
run: npm audit --audit-level=high
- run: npm run check
- name: Verify committed Action bundle
run: git diff --exit-code -- dist
Expand All @@ -32,6 +34,14 @@ jobs:
platforms: linux/arm64
tags: lambda-microvm-github-runner:release
load: true
- name: Scan runner image
uses: anchore/scan-action@e1165082ffb1fe366ebaf02d8526e7c4989ea9d2 # v7.4.0
with:
image: lambda-microvm-github-runner:release
fail-build: true
severity-cutoff: critical
only-fixed: true
output-format: table
- run: mkdir -p build
- name: Generate runner image SBOM
uses: anchore/sbom-action@e22c389904149dbc22b58101806040fa8d37a610 # v0
Expand Down
2 changes: 1 addition & 1 deletion dist/index.js

Large diffs are not rendered by default.

2 changes: 1 addition & 1 deletion dist/index.js.map

Large diffs are not rendered by default.

10 changes: 6 additions & 4 deletions docs/operations.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,10 +8,12 @@ documented baseline includes 5 `RunMicrovm` requests per second with burst 5, 10
requests per second with burst 100.

The Action uses bounded full-jitter launch and termination retries, a stable
launch client token, randomized sequential polling, and immediate failure for
capacity exhaustion. For sustained rates above the launch quota, request a quota
increase or shape GitHub workflow concurrency. Do not add an internal queue to
this product.
launch client token, an initial polling spread of up to 5 seconds, randomized
sequential polling at 2.5–5 second intervals, and immediate failure for capacity
exhaustion. The defaults keep a simulated 200 simultaneous starts within the 100
`GetMicrovm` requests-per-second baseline. For sustained rates above the launch
quota, request a quota increase or shape GitHub workflow concurrency. Do not add
an internal queue to this product.

## Logs

Expand Down
23 changes: 18 additions & 5 deletions docs/testing.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,9 +8,10 @@ npm run check
shellcheck scripts/*.sh
scripts/package-runner-image.sh
npm run test:image
npm audit --audit-level=high
```

`npm run check` covers strict TypeScript, 53 Action tests, 16 supervisor tests,
`npm run check` covers strict TypeScript, 57 Action tests, 17 supervisor tests,
and the bundled Action. Supervisor tests also run successfully under the image's
Python 3.9 runtime.

Expand Down Expand Up @@ -40,14 +41,26 @@ Every candidate version must prove in AWS:

## Private repository gate

The manual `.github/workflows/aws-runner-target.yml` job waits for a runner with
the additional `e2e` label. Launch a candidate with that label to exercise
checkout, the ARM64 host, Docker, Buildx, Compose, and a nested container.
The manual `.github/workflows/aws-runner-e2e.yml` workflow exercises the
production three-job pattern through the GitHub OIDC role. Set a temporary
`RUNNER_E2E_TOKEN` repository secret with Administration read/write permission,
then run its `success`, `failure`, and `maximum-duration` scenarios. Remove the
secret after testing. The workflow verifies checkout, the ARM64 host, Docker,
Buildx, Compose, service containers, DNS, egress, failure cleanup, idempotent
stop, and the platform duration backstop.

The lower-level `.github/workflows/aws-runner-target.yml` job remains available
for a runner launched outside GitHub Actions. It waits for a runner with the
additional `e2e` label.

Run successful and failing target jobs, Docker builds, service containers,
Compose, cancellation, startup timeout cleanup, duplicate launch retry, two
concurrent workflows, five concurrent starts, simulated throttling, capacity
failure, and denied self-termination fallback.

Do not tag `v1` until logs have been checked for plaintext secrets and the full
private-repository matrix passes.
private-repository matrix passes. CI and release builds print all image findings
and fail on fixable critical vulnerabilities. High findings remain visible and
must be reviewed when advancing the pinned upstream runner, Docker, Buildx,
Compose, and AWS CLI versions. Every release publishes Action and runner-image
SBOMs with checksums.
4 changes: 2 additions & 2 deletions src/polling.ts
Original file line number Diff line number Diff line change
Expand Up @@ -31,8 +31,8 @@ export class PollingError extends Error {
export async function pollSequentially<TObserved, TResult>(
options: PollingOptions<TObserved, TResult>,
): Promise<TResult> {
const initialDelayMaxMs = options.initialDelayMaxMs ?? 2_000;
const baseIntervalMs = options.baseIntervalMs ?? 2_000;
const initialDelayMaxMs = options.initialDelayMaxMs ?? 5_000;
const baseIntervalMs = options.baseIntervalMs ?? 2_500;
const maxIntervalMs = options.maxIntervalMs ?? 5_000;
const random = options.random ?? Math.random;
const now = options.now ?? Date.now;
Expand Down
Loading
Loading