diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index 65bc5c6..09680ce 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -11,4 +11,5 @@ - [ ] I kept credentials, private keys, tokens, and machine-specific configuration out of this pull request. - [ ] I added or updated tests where behavior changed. - [ ] I updated relevant documentation. +- [ ] For provider or onboarding changes, I followed the [Development and Extension Principles](../docs/development/principles.md) and documented and tested every intentional exception. - [ ] I read and followed the contributing guide and code of conduct. diff --git a/.github/workflows/host-trust-verification.yml b/.github/workflows/host-trust-verification.yml index 12adece..cfb1641 100644 --- a/.github/workflows/host-trust-verification.yml +++ b/.github/workflows/host-trust-verification.yml @@ -19,6 +19,27 @@ concurrency: cancel-in-progress: true jobs: + go-quality: + name: Go race and vet + runs-on: ubuntu-latest + timeout-minutes: 30 + + steps: + - name: Check out repository + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 + + - name: Set up Go + uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6.5.0 + with: + go-version-file: go.mod + cache: true + + - name: Run race tests + run: go test -race ./... + + - name: Run vet + run: go vet ./... + native: name: Native read-only (${{ matrix.os }}) runs-on: ${{ matrix.os }} @@ -40,7 +61,12 @@ jobs: - name: Run Go tests shell: bash - run: go test ./... + run: | + set -euo pipefail + if [[ "${RUNNER_OS}" == "macOS" ]]; then + export TMPDIR="$(cd "${RUNNER_TEMP}" && pwd -P)" + fi + go test ./... - name: Smoke-test native host root collection without trust-store mutations shell: bash @@ -203,10 +229,18 @@ jobs: if: runner.os == 'Windows' run: scripts/test/host-trust-wrapper-smoke.ps1 -ProjectRoot $env:GITHUB_WORKSPACE - - name: Exercise no-Go first-run start lifecycle + - name: Verify no-Go native-controller contract shell: pwsh if: runner.os == 'Windows' - run: scripts/test/no-go-first-run-smoke.ps1 -ProjectRoot $env:GITHUB_WORKSPACE + run: scripts/test/windows-native-controller-contract.ps1 -ProjectRoot $env:GITHUB_WORKSPACE + + - name: Verify Docker Sandboxes plan-only contracts + shell: pwsh + if: runner.os == 'Windows' + run: | + scripts/test/docker-sandboxes-plan-smoke.ps1 + scripts/docker-sandboxes/validate-assets.ps1 -Platform linux/amd64 + scripts/docker-sandboxes/validate-assets.ps1 -Platform linux/arm64 - name: Exercise official wrapper feed lifecycle and singleton lock shell: bash @@ -218,6 +252,16 @@ jobs: if: runner.os != 'Windows' run: bash scripts/test/no-go-first-run-smoke.sh + - name: Verify start command forwarding + shell: bash + if: runner.os != 'Windows' + run: bash scripts/test/start-command-forwarding.sh + + - name: Verify native-controller cache retention + shell: bash + if: runner.os != 'Windows' + run: bash scripts/test/native-controller-cache-retention.sh + - name: Parse Bash host-trust scripts if: runner.os != 'Windows' shell: bash @@ -229,7 +273,9 @@ jobs: scripts/host-trust/host-trust-feed.sh \ scripts/host-trust/wrapper-lib.sh \ scripts/test/host-trust-wrapper-smoke.sh \ + scripts/test/native-controller-cache-retention.sh \ scripts/test/no-go-first-run-smoke.sh \ + scripts/test/start-command-forwarding.sh \ scripts/guest/ubuntu/apply-trusted-ca-runtime.sh \ scripts/guest/ubuntu/check-host-trust-generation.sh @@ -241,10 +287,15 @@ jobs: $files = @( 'start.ps1', 'scripts/run-with-docker.ps1', + 'scripts/build-native-controller.ps1', + 'scripts/docker-sandboxes/build-template.ps1', + 'scripts/docker-sandboxes/load-template.ps1', + 'scripts/docker-sandboxes/validate-assets.ps1', 'scripts/host-trust/host-trust-feed.ps1', 'scripts/host-trust/wrapper-lib.ps1', + 'scripts/test/docker-sandboxes-plan-smoke.ps1', 'scripts/test/host-trust-wrapper-smoke.ps1', - 'scripts/test/no-go-first-run-smoke.ps1' + 'scripts/test/windows-native-controller-contract.ps1' ) foreach ($file in $files) { $tokens = $null diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..51698c5 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,7 @@ +# Repository Agent Guidance + +Before planning or changing provider, startup, configuration, pool, runner-image, or lifecycle behavior, read [Development and Extension Principles](docs/development/principles.md), [Contributing](CONTRIBUTING.md), [Design](docs/development/design.md), and [Adding a Provider](docs/development/adding-provider.md). + +Treat the missing-config `./start` wizard, the no-Go native-controller path, Catthehacker defaults for providers that can consume Docker images, runner-artifact customization, the shared machine-derived pool-prefix generator, the shared `pool.RunnerName` format, runner routing, strict capacity, logging, host trust, registration, replacement, diagnostics, exact cleanup, and no-silent-fallback behavior as product-wide contracts. Compare an extension with the common manager path and at least one established provider instead of validating only its provider-local implementation. + +A provider or onboarding change is incomplete until its wizard and generated configuration, reusable artifact/update path, shared lifecycle behavior, documentation, normal and race tests, wrapper syntax, and relevant live platform evidence are addressed. Keep unvalidated platforms or capabilities explicitly preview-only. Document and test every intentional exception to the shared contracts. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 8945162..0afd3ef 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -11,9 +11,10 @@ Thanks for taking the time to contribute. ## Development Workflow 1. Fork the repository and create a focused branch from `develop`. -2. Keep the change small and document any operational or security behavior that it changes. -3. Run the relevant tests locally. The baseline Go test suite is `go test ./...`. -4. Open a pull request targeting `develop` and complete the pull-request template. +2. Read and preserve the [Development and Extension Principles](docs/development/principles.md). +3. Keep the change small and document any operational or security behavior that it changes. +4. Run the relevant tests locally. The baseline Go test suite is `go test ./...`. +5. Open a pull request targeting `develop` and complete the pull-request template. Fork pull requests run the safe hosted verification workflow. The live EPAR canary is reserved for branches in this repository because it uses a protected environment and disposable privileged containers. @@ -23,5 +24,8 @@ Fork pull requests run the safe hosted verification workflow. The live EPAR cana - Add or update tests when behavior changes. - Keep credentials, private keys, tokens, and machine-specific configuration out of commits. - Update the relevant documentation when a user-visible or operational behavior changes. +- Document and test every intentional platform or security exception. By contributing, you agree to follow the [Code of Conduct](CODE_OF_CONDUCT.md). + +See the [development documentation](docs/development/) for the architecture, provider extension checklist, verification infrastructure, and release process. diff --git a/README.md b/README.md index 7dd66e2..e16b628 100644 --- a/README.md +++ b/README.md @@ -2,169 +2,80 @@ ![Ephemeral Action Runner banner](docs/assets/brand/epar-banner.jpg) -Ephemeral Action Runner (EPAR) keeps a warm pool of disposable GitHub Actions self-hosted runners on your own machine. - -Each runner is made for one job. EPAR starts it, registers it with GitHub, lets one workflow job run, deletes it, and creates a fresh replacement. +Ephemeral Action Runner (EPAR) keeps a warm pool of disposable GitHub Actions self-hosted runners on a machine you control. A runner accepts one job, is removed, and is replaced with a clean runner so ordinary job files, containers, and caches do not become the next job's starting state. ```mermaid -flowchart TB - EPAR["EPAR"] --> Create["Create runner"] - Create --> Ready["Runner ready"] - Ready --> Job["Run one GitHub Actions job"] - Job --> Delete["Delete runner"] - Delete --> Create +flowchart LR + Start["EPAR starts a runner"] --> Ready["Runner is ready"] + Ready --> Job["One GitHub Actions job"] + Job --> Remove["Runner is removed"] + Remove --> Start ``` -## Use Case - -Private repositories often have limited [GitHub-hosted Actions minutes](https://docs.github.com/en/billing/concepts/product-billing/github-actions#free-use-of-github-actions). If you already have a spare Windows, macOS, Linux, or Docker-capable machine, you can use it for feature-branch CI instead of spending those hosted-runner minutes. - -A normal long-lived self-hosted runner can leave dependencies, files, containers, caches, or other job state behind on that machine. EPAR lowers that risk by running each job in a disposable container, WSL distro, or VM, then deleting it and creating a clean replacement. +## Why EPAR -## Why Use EPAR - -- **Warm pool:** keep ready self-hosted runners online after setup. -- **Disposable jobs:** each runner is cleaned up after one job. -- **Great default image:** Docker-DinD and WSL use Catthehacker's full Ubuntu runner image by default. -- **Docker-friendly isolation:** Docker-DinD gives each runner its own private Docker daemon. -- **Simple host use:** run Linux GitHub Actions jobs from a Windows, macOS, Linux, or Docker-capable host. +- Keep private-repository CI ready without maintaining a long-lived runner workspace. +- Protect the host with [Docker Sandboxes](docs/providers/docker-sandboxes.md): each runner gets a dedicated microVM and private Docker daemon, then is removed after one job. +- Run Docker-friendly Linux jobs from a Windows, macOS, Linux, or other Docker-capable host. ## Quick Start -The easiest path is the default **Docker-DinD** mode. It works well for most Linux GitHub Actions jobs, especially Docker and Docker Compose jobs. - -### 1. Install Docker +The normal path is a source archive plus Docker. EPAR's first run opens a guided setup wizard; it checks what the host supports and writes your ignored local configuration. -The default quick start needs a Docker-compatible daemon: +### 1. Install the host tools -- Windows: [Docker Desktop](https://www.docker.com/products/docker-desktop/), or another Docker daemon reachable from PowerShell -- macOS: [Docker Desktop](https://www.docker.com/products/docker-desktop/) or [OrbStack](https://orbstack.dev/) -- Linux: [Docker Engine](https://docs.docker.com/engine/) +- Install and start Docker. +- For stronger isolation, also install [Docker Sandboxes](https://docs.docker.com/ai/sandboxes/) to enable the Docker Sandboxes provider. -### 2. Download EPAR Source - -Open the [EPAR Releases page](https://github.com/solutionforest/ephemeral-action-runner/releases), select the release you want, and download GitHub's automatically generated **Source code (zip)** or **Source code (tar.gz)**. EPAR releases use these source archives only. - -Extract the source archive and open a terminal in the extracted folder. The folder is usually named `ephemeral-action-runner-`. - -```bash -cd path/to/ephemeral-action-runner- -``` +On macOS or Linux, the first Docker Sandboxes runner may trigger operating-system or security-tool prompts for runtime helpers such as `mkfs.ext4`, `mkfs.erofs`, and `containerd-shim-nerdbox-v1`; macOS may say the helper “is an app downloaded from the Internet.” These are used to create the runner's private Docker filesystem, unpack its read-only template filesystem, and launch the sandbox VM. Confirm that each executable belongs to the installed Docker Sandboxes runtime and that any displayed file target is sandbox-owned before approving it. Denying a required helper prevents that sandbox from starting, and EPAR fails closed without registering it; preserve and clean any diagnostic runtime state through EPAR's exact cleanup path. See the [Docker Sandboxes provider guide](docs/providers/docker-sandboxes.md#private-filesystem-and-vm-helper-approval) and [troubleshooting](docs/troubleshooting.md#docker-sandboxes-creation-fails-after-a-runtime-helper-prompt). -### 3. Create A GitHub App +Docker Sandboxes has a host credential-injecting forward proxy. EPAR's Docker Sandboxes template keeps the private Docker daemon and Actions listener on Docker Sandboxes' policy-enforced transparent egress path by default, so a workflow's own `docker login` remains authoritative instead of being replaced by the host `sbx login` identity. Rebuild older templates after upgrading EPAR. See [Docker Hub Credentials and Transparent Egress](docs/providers/docker-sandboxes.md#docker-hub-credentials-and-transparent-egress). -EPAR uses a GitHub App to create short-lived runner registration tokens. +### 2. Download EPAR -Follow [GitHub App Setup](docs/github-app.md), then keep these three values ready: +From the [EPAR releases page](https://github.com/solutionforest/ephemeral-action-runner/releases), download GitHub's **Source code (zip)** or **Source code (tar.gz)** for the release you want. Extract it and open a terminal in the extracted folder. -- GitHub App ID -- GitHub organization name -- private key file path +### 3. Create a GitHub App -### 4. Run EPAR +EPAR uses a GitHub App to obtain short-lived runner registration tokens. Follow [GitHub App Setup](docs/github-app.md), then have the App ID, organization name, and private-key file path ready. -Run EPAR with the default flow: +### 4. Start EPAR ```bash ./start ``` -On Windows, `./start` also works in modern PowerShell. If your shell does not run it, use `.\start.ps1` or `start.cmd`. +In native Windows PowerShell or cmd, use `.\start.ps1` or `start.cmd` if `./start` is not available. The wrapper uses local Go when it works; otherwise it builds a native controller with Docker. If no configuration exists, the interactive wizard asks for the GitHub App, a runner group, and an available provider. The first start can take longer while EPAR prepares the configured runner image or creates the first runner. -That's it. - -#### What Happens - -EPAR initializes `.local/config.yml` for you if it does not exist. Docker-DinD is the default. The wizard asks whether new Docker-DinD runners should inherit the host's trusted TLS roots and defaults to yes. On native Windows, it also offers WSL2 when `wsl.exe --status` confirms default version 2. On macOS, it offers experimental Tart mode when `tart --version` succeeds. Press Enter to keep Docker-DinD. Existing configs do not enable host trust inheritance automatically. You can customize the config afterward; see [Configuration](docs/configuration.md). - -Then EPAR checks the configured runner image, builds or replaces it when needed, and starts the configured number of runners. The default config uses `pool.instances: 1`. - -The first run can take a while because EPAR may need to build the runner image before it starts the pool. Later runs reuse the aligned image unless the config, EPAR scripts, or source image changed. - -Keep EPAR running while you want runners online. Stop with `Ctrl-C`; EPAR cleans up matching local instances and GitHub runner records by default. - -#### Optional: Config Or Runner Count - -To choose a config or runner count: - -```bash -./start --config .local/custom-config.yml --instances 2 -``` +Keep the process open while runners should accept work. Press `Ctrl-C` once to stop, then wait for cleanup to finish before closing the terminal. For detailed commands, config selection, no-Go startup, verification, and cleanup, read [Usage](docs/usage.md). -If `--instances` is omitted, EPAR uses `pool.instances` from the config. +## Choose a provider -#### GitHub Actions Labels +Choose a provider based on your host OS, available prerequisites, and isolation needs. **Docker Sandboxes** is recommended when its capability checks pass as it provides the highest isolation level. EPAR never silently falls back to another provider. -GitHub Actions picks a runner by matching the job's `runs-on` list with the labels registered on each runner. Every self-hosted runner gets the `self-hosted` label, so the simplest workflow can use: +| Provider | Host OS | Prerequisites | Isolation and compatibility | +| --- | --- | --- | --- | +| [Docker Sandboxes](docs/providers/docker-sandboxes.md) | Linux, macOS, Windows | Docker, the `sbx` CLI, and healthy `sbx diagnose --output json` results | Highest isolation level — each runner uses a dedicated microVM with a private Docker daemon. Recommended when capability checks pass. | +| [Docker Container](docs/providers/docker-container.md) | Linux, macOS, Windows | Docker | Standard isolation level — each disposable runner container has a private Docker daemon. | +| [WSL](docs/providers/wsl.md) | Windows | WSL2 and Docker | Standard isolation level — each runner uses a disposable WSL2 Linux environment. | +| [Tart](docs/providers/tart.md) | Apple Silicon macOS | Tart | Experimental — ARM64 Linux VM with limited compatibility for CI jobs that require non-ARM64 Docker images. | -```yaml -runs-on: [self-hosted] -``` +## Route a workflow to EPAR -If you have multiple self-hosted runners and want this job to run on a specific kind of EPAR runner, add one of its extra labels to the list, e.g.: +Every EPAR runner has GitHub's `self-hosted` label. Add one of the provider labels when a repository has several types of runner: ```yaml -runs-on: [self-hosted, epar-docker-dind-catthehacker-ubuntu] +runs-on: [self-hosted, linux, epar-docker-sandboxes] ``` -EPAR also adds an `epar-host-` label by default, so you can see which host registered each runner. You only need to include that label in `runs-on` when you intentionally want a job to target one machine. - -## Other Modes - -Docker-DinD is the default first choice. Other providers are available when they fit your host better: - -| Provider | Use when | -| --- | --- | -| Docker-DinD | You have a Docker-compatible daemon on Windows, macOS, or Linux, and want a private Docker daemon per runner. | -| WSL2 | You are on Windows and want runners as disposable WSL distros. | -| Tart (experimental) | You are on Apple Silicon macOS and want to experiment with native ARM64 Linux VMs. The default Tart image is a basic Ubuntu OS image and does not include the normal GitHub-hosted runner dependency set. | - -WSL2 also defaults to Catthehacker's full Ubuntu runner image, but it converts that Docker image into a WSL rootfs during `image build`. - -Tart is not a ready-made substitute for GitHub's hosted Ubuntu runners. If you need that environment, build and maintain your own bootable Tart runner image by adapting the scripts from [actions/runner-images](https://github.com/actions/runner-images), then configure EPAR to use it. EPAR does not automate that conversion. - -See [Usage](docs/usage.md) for WSL, Tart, source builds, custom configs, and advanced options. - -## FAQ - -### Can EPAR run multiple runners at once? - -Yes. Set `pool.instances` in `.local/config.yml`, or pass `--instances N` for one run. - -### Can one machine run runners for multiple GitHub organizations? - -Yes. Use one config per organization, then start EPAR once per config. Each config should use its own GitHub App values and a distinct `pool.namePrefix`. - -### Does each job get a clean runner? - -Yes. EPAR registers disposable ephemeral runners. After a job finishes, EPAR deletes that runner and creates a replacement. - -### Can jobs use Docker, Docker Compose, and Buildx? - -Yes, with the default Docker-DinD mode. Each runner gets its own private Docker daemon, so job-created containers, networks, and volumes stay inside that disposable runner. - -## Safety +Use labels that describe the environment your job actually needs. In particular, an ARM64 Tart runner is not a replacement for GitHub-hosted `ubuntu-latest` or an x64-only workload. -EPAR is for trusted jobs. It improves cleanup and reduces stale runner state, but it does not make your machine safe for arbitrary untrusted code. +## Security depends on the provider -GitHub also warns against using self-hosted runners with public repositories that can run untrusted pull request workflows. Read GitHub's self-hosted runner guidance before exposing a runner to untrusted users. +Docker Sandboxes places each runner inside a dedicated microVM sandbox and provides EPAR's strongest host-isolation boundary. Docker Container and WSL remain trusted-workflow providers; Tart is VM-isolated but experimental. With every provider, restrict access with [runner groups](docs/runner-groups.md) and expose only the secrets and services each workflow needs. Read [Security](docs/security.md) before choosing a provider. -## More Docs +## Find the right guide -- [Usage](docs/usage.md): setup, image builds, verification, and pool commands. -- [Configuration](docs/configuration.md): config file sections and common edits. -- [GitHub App Setup](docs/github-app.md): required GitHub App permissions and fields. -- [Docker-DinD Provider](docs/providers/docker-dind.md): default Docker runner mode. -- [WSL Provider](docs/providers/wsl.md): Windows WSL2 runners. -- [Tart Provider (experimental)](docs/providers/tart.md): Apple Silicon ARM64 Linux VM runners and Rosetta compatibility limits. -- [Image Build](docs/image-build.md): image internals and customization. -- [Operations](docs/operations.md): logs, cleanup, and troubleshooting. -- [Troubleshooting](docs/troubleshooting.md): symptom-first diagnostics by host and provider. -- [Support](SUPPORT.md): where to start, what diagnostic information to collect, and where to ask for help. -- [Windows Startup](docs/advanced/windows-startup.md): start EPAR after Windows login. -- [macOS Startup](docs/advanced/macos-startup.md): start EPAR after macOS login. -- [Running EPAR Without Installing Go](docs/advanced/no-go-install.md): run from source with no local Go install. -- [Security](docs/security.md): trust boundaries, secret handling, and private vulnerability reporting. -- [Contributing](CONTRIBUTING.md): how to propose and validate changes. -- [Code of Conduct](CODE_OF_CONDUCT.md): community expectations and reporting concerns. -- [Level 1 Core Runner Verification](docs/core-runner-verification.md): trusted live CI setup, canary behavior, and cleanup. +- **Start and configure:** [Documentation hub](docs/README.md), [Usage](docs/usage.md), [Configuration](docs/configuration.md), and [GitHub App setup](docs/github-app.md). +- **Run and maintain:** [Operations](docs/operations.md), [Troubleshooting](docs/troubleshooting.md), [Logging](docs/logging.md), and [Storage](docs/storage.md). +- **Get help or contribute:** [Support](SUPPORT.md), [Contributing](CONTRIBUTING.md), and [Security reporting](docs/security.md). diff --git a/SUPPORT.md b/SUPPORT.md index 72f7890..78c8a0c 100644 --- a/SUPPORT.md +++ b/SUPPORT.md @@ -1,6 +1,6 @@ # Support -Start with the [troubleshooting guide](docs/troubleshooting.md). It provides symptom-first diagnostics for Docker-DinD, WSL, Tart, host trust, image builds, storage, and cross-architecture containers. +Start with the [troubleshooting guide](docs/troubleshooting.md). It provides symptom-first diagnostics for Docker Container, WSL, Tart, host trust, image builds, storage, and cross-architecture containers. Before asking for help, search the repository's existing [issues](https://github.com/solutionforest/ephemeral-action-runner/issues) and collect: diff --git a/cmd/ephemeral-action-runner/init.go b/cmd/ephemeral-action-runner/init.go index f47271f..1f352c3 100644 --- a/cmd/ephemeral-action-runner/init.go +++ b/cmd/ephemeral-action-runner/init.go @@ -6,21 +6,33 @@ import ( "context" "crypto/rand" "encoding/hex" + "encoding/json" "errors" "flag" "fmt" "io" + "math" "os" "os/exec" "path/filepath" "runtime" + "sort" "strconv" "strings" "time" "unicode/utf16" "github.com/solutionforest/ephemeral-action-runner/internal/config" + gh "github.com/solutionforest/ephemeral-action-runner/internal/github" "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + imageartifact "github.com/solutionforest/ephemeral-action-runner/internal/image" + "github.com/solutionforest/ephemeral-action-runner/internal/logging" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes" + sandboxcapacity "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/capacity" + sandboxpolicy "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/policy" + sandboxpromotion "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/promotion" + providerregistry "github.com/solutionforest/ephemeral-action-runner/internal/provider/registry" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" ) var dockerAvailable = func(ctx context.Context) error { @@ -41,6 +53,146 @@ var initTartVersion = tartVersion var initResolveHostTrust = hosttrust.Resolve +var initSandboxPromotionPlatform = sandboxpromotion.CurrentPlatform + +var initSandboxPromotionLookup = sandboxpromotion.Lookup + +var initDockerSandboxesPreflight = func(ctx context.Context, record sandboxpromotion.Record, projectRoot string) sandboxpromotion.PreflightResult { + return sandboxpromotion.LocalPreflight(ctx, record, projectRoot, os.Getenv("EPAR_CONTROLLER_IN_DOCKER") != "1", sourceRevision) +} + +var initDockerSandboxesLookPath = exec.LookPath + +var initDockerSandboxesStartDaemon = func(ctx context.Context, binary string) error { + return dockersandboxes.New(binary).StartDaemon(ctx) +} + +var initDockerSandboxesDiagnose = func(ctx context.Context, binary string) (dockersandboxes.HostReadiness, error) { + return dockersandboxes.New(binary).VerifyHostReadiness(ctx) +} + +var initDockerSandboxesReadiness = prepareInitDockerSandboxesReadiness + +func prepareInitDockerSandboxesReadiness(ctx context.Context) (dockersandboxes.HostReadiness, error) { + binary := "sbx" + var daemonStartErr error + if installed, err := initDockerSandboxesLookPath(binary); err == nil { + binary = installed + daemonStartErr = initDockerSandboxesStartDaemon(ctx, binary) + } + readiness, readinessErr := initDockerSandboxesDiagnose(ctx, binary) + if readinessErr != nil && daemonStartErr != nil { + return dockersandboxes.HostReadiness{}, fmt.Errorf("automatic 'sbx daemon start --detach' failed: %v; diagnostics also failed: %w", daemonStartErr, readinessErr) + } + return readiness, readinessErr +} + +type initDockerSandboxesTemplate struct { + Reference string + Digest string + CacheID string + Platform string + Size int64 + Label string + SourceChannel string +} + +type dockerSandboxesSourceLock struct { + SchemaVersion int `json:"schemaVersion"` + Profiles map[string]dockerSandboxesLockProfile `json:"profiles"` +} + +type dockerSandboxesLockProfile struct { + ObservedTagReference string `json:"observedTagReference"` + Platforms map[string]dockerSandboxesLockProfilePlatform `json:"platforms"` +} + +type dockerSandboxesLockProfilePlatform struct { + TemplateTag string `json:"templateTag"` +} + +type dockerSandboxesActiveProfile struct { + Name string + ObservedTag string + TemplateReference string + DisplayLabel string +} + +type initDockerSandboxesDiscovery struct { + Templates []initDockerSandboxesTemplate + PolicyFingerprint string +} + +type initDockerSandboxesRootMeasurement struct { + PeakBytes int64 + Evidence string +} + +type initDockerSandboxesCapacityResult struct { + StorageRoot string + AvailableBytes uint64 + TotalBytes uint64 + Reservation uint64 + HostWatermark uint64 + RequiredBytes uint64 + DeficitBytes uint64 + CapacityStatus storage.CapacityStatus +} + +type initDockerSandboxesProfile struct { + Provider string + HostPlatform sandboxpromotion.Platform + GuestPlatform string + SourceImage string + CustomScripts []string + PolicyFingerprint string + RootDisk string + DockerDisk string +} + +type initImageUpdatePolicy struct { + Frequency string + Time string +} + +var initDiscoverDockerSandboxes = discoverDockerSandboxes + +var initDockerSandboxesRootMeasurementFor = dockerSandboxesRootMeasurement + +var initDockerSandboxesCapacityCheck = checkInitDockerSandboxesCapacity + +var initResolveDockerSandboxesSource = imageartifact.ResolveCatthehackerSource + +var initDockerSandboxesPolicyFingerprint = func(ctx context.Context) (string, error) { + adapter := dockersandboxes.New("") + if err := adapter.VerifyAdmission(ctx); err != nil { + return "", fmt.Errorf("Docker Sandboxes admission check failed: %w", err) + } + rules, err := adapter.ReadGlobalNetworkPolicy(ctx) + if err != nil { + return "", fmt.Errorf("read Docker Sandboxes global network policy: %w", err) + } + return sandboxpolicy.Fingerprint(rules) +} + +var initEnsureDockerSandboxesTemplate = func(ctx context.Context, projectRoot, configPath string) error { + manager, err := newImageProvisioningManager(configPath, projectRoot) + if err != nil { + return err + } + defer manager.Close() + return manager.EnsureImage(ctx) +} + +type initRunnerGroupClient interface { + ListRunnerGroups(context.Context) ([]gh.RunnerGroup, error) + ListRunnerGroupRepositories(context.Context, int64) ([]gh.RunnerGroupRepository, error) +} + +var newInitRunnerGroupClient = func(cfg config.GitHubConfig) initRunnerGroupClient { + return gh.New(cfg) +} + func detectedInitHostTrustOS() string { if hostOS := strings.TrimSpace(os.Getenv("EPAR_CONTROLLER_HOST_OS")); hostOS != "" { return hostOS @@ -49,12 +201,15 @@ func detectedInitHostTrustOS() string { } type initOptions struct { + Context context.Context ProjectRoot string ConfigPath string Force bool SkipDockerCheck bool SkipHostTrustCheck bool + EmbeddedInStart bool In io.Reader + Reader *bufio.Reader Out io.Writer } @@ -80,6 +235,7 @@ func runInit(args []string) error { } return runInitWithOptions(initOptions{ + Context: interruptContext(), ProjectRoot: projectRoot, ConfigPath: configPath, Force: *force, @@ -91,6 +247,9 @@ func runInit(args []string) error { } func runInitWithOptions(opts initOptions) error { + if opts.Context == nil { + opts.Context = context.Background() + } if opts.In == nil { opts.In = os.Stdin } @@ -114,196 +273,1367 @@ func runInitWithOptions(opts initOptions) error { fmt.Fprintln(opts.Out, "See README.md and docs/github-app.md for the GitHub App steps.") fmt.Fprintln(opts.Out, "") - reader := bufio.NewReader(opts.In) - appID, err := promptRequiredInt64(opts.Out, reader, "GitHub App ID") + reader := opts.Reader + if reader == nil { + reader = bufio.NewReader(opts.In) + } + appID, err := promptRequiredInt64(opts.Out, reader, "GitHub App ID") + if err != nil { + return err + } + organization, err := promptRequired(opts.Out, reader, "GitHub organization") + if err != nil { + return err + } + privateKeyPath, err := promptRequired(opts.Out, reader, "GitHub App private key path") + if err != nil { + return err + } + githubConfig := config.GitHubConfig{ + AppID: appID, + Organization: organization, + PrivateKeyPath: resolveInitPrivateKeyPath(opts.ProjectRoot, privateKeyPath), + APIBaseURL: "https://api.github.com", + WebBaseURL: "https://github.com", + } + runnerGroup, err := promptRunnerGroup(opts.Context, opts.Out, reader, newInitRunnerGroupClient(githubConfig)) + if err != nil { + return err + } + providerType, _, selectedProfile, err := promptInitProvider(opts.Context, opts.ProjectRoot, opts.Out, reader, opts.SkipDockerCheck) + if err != nil { + return err + } + defaultPrefix, err := generatedPoolNamePrefix() + if err != nil { + return err + } + fmt.Fprintln(opts.Out, "") + fmt.Fprintln(opts.Out, "Pool name prefix must be unique for this machine/config within the GitHub organization.") + fmt.Fprintln(opts.Out, "EPAR cleanup deletes GitHub runner records matching this prefix.") + poolNamePrefix, err := promptPoolNamePrefix(opts.Out, reader, defaultPrefix) + if err != nil { + return err + } + + hostTrustMode := config.HostTrustModeDisabled + hostTrustScopes := []string{config.HostTrustScopeSystem} + if providerType == "docker-container" || providerType == "docker-sandboxes" { + fmt.Fprintln(opts.Out, "Runners need this host's trusted TLS roots to access services that this machine trusts.") + enabled, promptErr := promptYesNo(opts.Out, reader, "Inherit this host's trusted TLS roots into disposable runners?", true) + if promptErr != nil { + return promptErr + } + if enabled { + hostTrustMode = config.HostTrustModeOverlay + hostTrustScopes = hostTrustScopesForOS(initHostTrustOS) + deferred := os.Getenv("EPAR_HOST_TRUST_INIT_DEFERRED") == "1" + if !opts.SkipHostTrustCheck && !deferred { + preflightCtx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + _, collectErr := initResolveHostTrust(preflightCtx, hosttrust.Options{ + Mode: hostTrustMode, + Scopes: hostTrustScopes, + ControllerHostOS: initHostTrustOS, + }) + cancel() + if collectErr != nil { + return fmt.Errorf("collect host trusted TLS roots before writing config: %w", collectErr) + } + } + } + } + updatePolicy, err := promptImageUpdatePolicy(opts.Out, reader) + if err != nil { + return err + } + + profile := selectedProfile + if providerType != "tart" && profile == nil { + return fmt.Errorf("%s image selection did not produce a provisioning profile", providerType) + } + var content string + switch providerType { + case "docker-container": + content = defaultDockerContainerConfig(appID, organization, privateKeyPath, poolNamePrefix, hostTrustMode, hostTrustScopes, runnerGroup, *profile, updatePolicy) + case "wsl": + content = defaultWSLConfig(appID, organization, privateKeyPath, poolNamePrefix, runnerGroup, *profile, updatePolicy) + case "tart": + content = defaultTartConfig(appID, organization, privateKeyPath, poolNamePrefix, runnerGroup, updatePolicy) + case "docker-sandboxes": + guestPlatform, runnerArchitectureLabel, platformErr := dockerSandboxesPlatform(profile.HostPlatform) + if platformErr != nil { + return platformErr + } + content = defaultDockerSandboxesConfig(appID, organization, privateKeyPath, poolNamePrefix, hostTrustMode, hostTrustScopes, runnerGroup, *profile, updatePolicy, guestPlatform, runnerArchitectureLabel) + default: + return fmt.Errorf("unsupported provider.type %q", providerType) + } + if err := os.MkdirAll(filepath.Dir(opts.ConfigPath), 0755); err != nil { + return err + } + if err := logging.WritePrivateFileAtomic(opts.ConfigPath, []byte(content)); err != nil { + return err + } + + fmt.Fprintf(opts.Out, "\nCreated %s\n", opts.ConfigPath) + if opts.EmbeddedInStart { + fmt.Fprintln(opts.Out, "Initialization succeeded. Startup will now provision the selected runner artifact and apply storage admission before side effects.") + return nil + } + fmt.Fprintf(opts.Out, ` +Next: + %s start + +Manual/advanced: + %s image build --replace + %s pool verify --instances 2 --register-only --cleanup + %s pool up --instances 2 +`, binaryName, binaryName, binaryName, binaryName) + return nil +} + +type initRunnerGroupSelection struct { + Group gh.RunnerGroup + Policy config.RunnerGroupSecurityConfig +} + +func promptRunnerGroup(ctx context.Context, out io.Writer, reader *bufio.Reader, client initRunnerGroupClient) (initRunnerGroupSelection, error) { + var groups []gh.RunnerGroup + var repositories map[int64][]gh.RunnerGroupRepository + showBlockedGroups := false + showGroupDetails := false + for { + if groups == nil { + var err error + groups, repositories, err = loadInitRunnerGroups(ctx, client) + if err != nil { + return initRunnerGroupSelection{}, fmt.Errorf("load GitHub runner groups: %w", err) + } + } + fmt.Fprintln(out, "") + fmt.Fprintln(out, "GitHub runner group:") + fmt.Fprintln(out, " Choose which repositories can use these runners.") + fmt.Fprintln(out, " For better security, use a custom group for selected trusted repositories, with public access disabled.") + fmt.Fprintln(out, " See docs/runner-groups.md for details.") + if showGroupDetails { + fmt.Fprintln(out, "") + fmt.Fprintln(out, " Repository access meanings:") + fmt.Fprintln(out, " Selected repositories: Only repositories added to the group can use its runners.") + fmt.Fprintln(out, " All private repositories: All current and future private repositories can use its runners.") + fmt.Fprintln(out, " All repositories: All repositories allowed by the public-repository setting can use its runners.") + } + visibleGroups := filterRunnerGroupsForWizard(groups, repositories, showBlockedGroups) + if len(visibleGroups) == 0 { + fmt.Fprintln(out, "") + fmt.Fprintln(out, " No selectable runner groups found. Show blocked groups to review them.") + } + for i, group := range visibleGroups { + printRunnerGroupChoice(out, i+1, group, repositories[group.ID], showGroupDetails) + } + fmt.Fprintln(out, " R. Refresh runner groups") + if showBlockedGroups { + fmt.Fprintln(out, " B. Hide blocked runner groups") + } else { + fmt.Fprintln(out, " B. Show blocked runner groups") + } + if showGroupDetails { + fmt.Fprintln(out, " D. Hide runner group details") + } else { + fmt.Fprintln(out, " D. Show runner group details") + } + fmt.Fprintln(out, " Q. Quit without writing a config") + + choice, err := promptRequired(out, reader, "Runner group choice") + if err != nil { + return initRunnerGroupSelection{}, err + } + switch strings.ToLower(choice) { + case "r", "refresh": + groups = nil + repositories = nil + continue + case "b", "blocked": + showBlockedGroups = !showBlockedGroups + continue + case "d", "details": + showGroupDetails = !showGroupDetails + continue + case "q", "quit": + return initRunnerGroupSelection{}, fmt.Errorf("runner-group selection cancelled; no config was written") + } + index, parseErr := strconv.Atoi(choice) + if parseErr != nil || index < 1 || index > len(visibleGroups) { + fmt.Fprintln(out, "Choose a runner group number, R to refresh, B for blocked groups, D for details, or Q to quit.") + continue + } + group := visibleGroups[index-1] + selectedRepositories := repositories[group.ID] + _, publicCount := repositoryPrivacyCounts(selectedRepositories) + if runnerGroupVisibilityRank(group.Visibility) == 3 { + fmt.Fprintln(out, "") + fmt.Fprintln(out, "*** SECURITY BLOCK: GITHUB RETURNED AN UNKNOWN REPOSITORY-ACCESS POLICY ***") + fmt.Fprintf(out, "Runner group %q uses repository access %q, which this EPAR version cannot evaluate safely.\n", group.Name, group.Visibility) + fmt.Fprintln(out, "RECOMMENDED ACTION: Do not select this group. Review its policy in GitHub, update EPAR if support is available, and choose Refresh; otherwise choose another documented group.") + refresh, err := promptBackRefreshQuit(out, reader) + if err != nil { + return initRunnerGroupSelection{}, err + } + if refresh { + groups = nil + repositories = nil + } + continue + } + if group.AllowsPublicRepositories || publicCount > 0 { + fmt.Fprintln(out, "") + fmt.Fprintln(out, "*** SECURITY BLOCK: THIS RUNNER GROUP IS NOT ALLOWED BY EPAR'S SAFE DEFAULTS ***") + fmt.Fprintf(out, "Runner group %q permits public repository access. A public repository or fork-triggered workflow may run untrusted code on a self-hosted runner and reach the runner host or connected services.\n", group.Name) + fmt.Fprintln(out, "RECOMMENDED ACTION: Do not use this group for a normal EPAR deployment. Follow docs/runner-groups.md to create a dedicated non-default group, allow only explicitly selected trusted repositories, and disable public repository access. Then return here and choose that group.") + fmt.Fprintln(out, "If you intentionally operate a separately reviewed public-project deployment, finish initialization with a secure group first and document any manual policy override afterward.") + refresh, err := promptBackRefreshQuit(out, reader) + if err != nil { + return initRunnerGroupSelection{}, err + } + if refresh { + groups = nil + repositories = nil + } + continue + } + + if group.Default { + fmt.Fprintln(out, "") + fmt.Fprintln(out, "*** SECURITY REMINDER: DEFAULT RUNNER GROUP ***") + fmt.Fprintln(out, "The Default runner group is fine for trying EPAR.") + fmt.Fprintln(out, "For regular use, a custom group limited to selected trusted repositories offers better security.") + continueSelection, err := promptContinueOrBack(out, reader, "Continue with Default runner group") + if err != nil { + return initRunnerGroupSelection{}, err + } + if !continueSelection { + continue + } + } else { + warnings := runnerGroupSelectionWarnings(group) + if len(warnings) > 0 { + fmt.Fprintln(out, "") + notRecommended := runnerGroupDoesNotMeetRecommendedPolicy(group) + if notRecommended { + fmt.Fprintln(out, "*** SECURITY WARNING: THIS RUNNER GROUP IS NOT RECOMMENDED ***") + } else { + fmt.Fprintln(out, "*** SECURITY ADVISORY: ENTERPRISE-MANAGED RUNNER GROUP ***") + } + fmt.Fprintf(out, "Runner group %q requires explicit review:\n", group.Name) + for _, warning := range warnings { + fmt.Fprintf(out, " - %s\n", warning) + } + continueLabel := "Continue after confirming the enterprise-managed policy" + if notRecommended { + fmt.Fprintln(out, "RECOMMENDED ACTION: Choose Back. Follow docs/runner-groups.md to create a dedicated non-default group with Selected repositories and public repository access disabled, then select that safer group.") + fmt.Fprintln(out, "Continuing will deliberately relax the generated policy to match this broader group. Future repositories may gain access without another EPAR configuration change.") + continueLabel = "Continue anyway and generate a relaxed policy" + } + continueSelection, err := promptContinueOrBack(out, reader, continueLabel) + if err != nil { + return initRunnerGroupSelection{}, err + } + if !continueSelection { + continue + } + } + } + return initRunnerGroupSelection{ + Group: group, + Policy: config.RunnerGroupSecurityConfig{ + Enforcement: config.RunnerGroupEnforcementEnforce, + RequireExplicitGroup: true, + RequireNonDefaultGroup: !group.Default, + RequiredRepositoryAccess: group.Visibility, + RequirePublicRepositoriesDisabled: true, + }, + }, nil + } +} + +func loadInitRunnerGroups(ctx context.Context, client initRunnerGroupClient) ([]gh.RunnerGroup, map[int64][]gh.RunnerGroupRepository, error) { + groups, err := client.ListRunnerGroups(ctx) + if err != nil { + return nil, nil, err + } + if len(groups) == 0 { + return nil, nil, fmt.Errorf("GitHub returned no runner groups for the organization") + } + groups = sortRunnerGroupsForWizard(groups) + repositories := make(map[int64][]gh.RunnerGroupRepository) + for _, group := range groups { + if group.Visibility != config.RunnerGroupRepositoryAccessSelected { + continue + } + selected, err := client.ListRunnerGroupRepositories(ctx, group.ID) + if err != nil { + return nil, nil, err + } + repositories[group.ID] = selected + } + return groups, repositories, nil +} + +func sortRunnerGroupsForWizard(groups []gh.RunnerGroup) []gh.RunnerGroup { + ordered := append([]gh.RunnerGroup(nil), groups...) + sort.SliceStable(ordered, func(i, j int) bool { + left, right := ordered[i], ordered[j] + if left.Default != right.Default { + return left.Default + } + if left.AllowsPublicRepositories != right.AllowsPublicRepositories { + return !left.AllowsPublicRepositories + } + if leftRank, rightRank := runnerGroupVisibilityRank(left.Visibility), runnerGroupVisibilityRank(right.Visibility); leftRank != rightRank { + return leftRank < rightRank + } + if left.Inherited != right.Inherited { + return !left.Inherited + } + return strings.ToLower(left.Name) < strings.ToLower(right.Name) + }) + return ordered +} + +func filterRunnerGroupsForWizard(groups []gh.RunnerGroup, repositories map[int64][]gh.RunnerGroupRepository, showBlocked bool) []gh.RunnerGroup { + if showBlocked { + return groups + } + visible := make([]gh.RunnerGroup, 0, len(groups)) + for _, group := range groups { + if runnerGroupBlockedByWizard(group, repositories[group.ID]) { + continue + } + visible = append(visible, group) + } + return visible +} + +func runnerGroupBlockedByWizard(group gh.RunnerGroup, repositories []gh.RunnerGroupRepository) bool { + _, publicCount := repositoryPrivacyCounts(repositories) + return runnerGroupVisibilityRank(group.Visibility) == 3 || group.AllowsPublicRepositories || publicCount > 0 +} + +func runnerGroupVisibilityRank(visibility string) int { + switch visibility { + case config.RunnerGroupRepositoryAccessSelected: + return 0 + case config.RunnerGroupRepositoryAccessPrivate: + return 1 + case config.RunnerGroupRepositoryAccessAll: + return 2 + default: + return 3 + } +} + +func printRunnerGroupChoice(out io.Writer, number int, group gh.RunnerGroup, repositories []gh.RunnerGroupRepository, showDetails bool) { + privateCount, publicCount := repositoryPrivacyCounts(repositories) + fmt.Fprintf(out, "\n %d. %s\n", number, group.Name) + if showDetails { + switch group.Visibility { + case config.RunnerGroupRepositoryAccessSelected: + fmt.Fprintf(out, " Repository access: Selected repositories — only the %d private and %d public repositories explicitly selected in GitHub can use this group.\n", privateCount, publicCount) + case config.RunnerGroupRepositoryAccessPrivate: + fmt.Fprintln(out, " Repository access: All private repositories — every current and future private repository in the organization can use this group.") + case config.RunnerGroupRepositoryAccessAll: + fmt.Fprintln(out, " Repository access: All repositories — every current and future repository permitted by the public-repository setting can use this group.") + default: + fmt.Fprintf(out, " Repository access: Unknown GitHub value %q — do not select this group until its policy can be understood.\n", group.Visibility) + } + if group.AllowsPublicRepositories || publicCount > 0 { + fmt.Fprintln(out, " Public repositories: ALLOWED — public or fork-triggered workflows may reach self-hosted runners.") + } else { + fmt.Fprintln(out, " Public repositories: Disabled — public repositories cannot use this group.") + } + groupTypes := []string{"organization-managed", "non-default"} + if group.Default { + groupTypes = []string{"GitHub default group"} + } + if group.Inherited { + groupTypes = append(groupTypes, "inherited from the enterprise") + } + fmt.Fprintf(out, " Group type: %s.\n", strings.Join(groupTypes, ", ")) + } + switch { + case runnerGroupVisibilityRank(group.Visibility) == 3: + fmt.Fprintln(out, " Assessment: BLOCKED BY WIZARD — repository access cannot be evaluated safely.") + case group.AllowsPublicRepositories || publicCount > 0: + fmt.Fprintln(out, " Assessment: BLOCKED BY WIZARD — does not satisfy the public-repository safety requirement.") + case group.Default: + fmt.Fprintln(out, " Assessment: It is fine for first-time tasting of EPAR, but generally recommend to create and use custom runner group for better security.") + case runnerGroupDoesNotMeetRecommendedPolicy(group): + fmt.Fprintln(out, " Assessment: NOT RECOMMENDED — requires an explicit warning and a relaxed generated policy.") + case group.Inherited: + fmt.Fprintln(out, " Assessment: REVIEW REQUIRED — access is restrictive, but policy changes are controlled at enterprise level.") + default: + fmt.Fprintln(out, " Assessment: RECOMMENDED — matches EPAR's strict generated policy.") + } +} + +func repositoryPrivacyCounts(repositories []gh.RunnerGroupRepository) (privateCount, publicCount int) { + for _, repository := range repositories { + if repository.Private { + privateCount++ + } else { + publicCount++ + } + } + return privateCount, publicCount +} + +func runnerGroupSelectionWarnings(group gh.RunnerGroup) []string { + var warnings []string + if group.Default { + warnings = append(warnings, "This is GitHub's default runner group. New or unintended repositories may gain access as organization policy changes.") + } + switch group.Visibility { + case config.RunnerGroupRepositoryAccessPrivate: + warnings = append(warnings, "This group is available to every private repository in the organization, including repositories created later.") + case config.RunnerGroupRepositoryAccessAll: + warnings = append(warnings, "This group is available to every repository allowed by its public-repository setting, including repositories created later.") + } + if group.Inherited { + warnings = append(warnings, "This group is inherited from the enterprise. Its policy must be reviewed and changed at enterprise level.") + } + return warnings +} + +func runnerGroupDoesNotMeetRecommendedPolicy(group gh.RunnerGroup) bool { + return group.Default || group.Visibility != config.RunnerGroupRepositoryAccessSelected +} + +func promptContinueOrBack(out io.Writer, reader *bufio.Reader, continueLabel string) (bool, error) { + fmt.Fprintf(out, " 1. %s\n", continueLabel) + fmt.Fprintln(out, " 2. Back to group selection") + for { + choice, err := promptRequired(out, reader, "Choice") + if err != nil { + return false, err + } + switch strings.ToLower(choice) { + case "1", "continue": + return true, nil + case "2", "back": + return false, nil + default: + fmt.Fprintln(out, "Choose 1 to continue or 2 to go back.") + } + } +} + +func promptBackRefreshQuit(out io.Writer, reader *bufio.Reader) (bool, error) { + fmt.Fprintln(out, " 1. Back to group selection") + fmt.Fprintln(out, " 2. Refresh runner groups") + fmt.Fprintln(out, " 3. Quit without writing a config") + for { + choice, err := promptRequired(out, reader, "Choice") + if err != nil { + return false, err + } + switch strings.ToLower(choice) { + case "1", "back": + return false, nil + case "2", "refresh": + return true, nil + case "3", "quit": + return false, fmt.Errorf("runner-group selection cancelled; no config was written") + default: + fmt.Fprintln(out, "Choose 1 to go back, 2 to refresh, or 3 to quit.") + } + } +} + +func resolveInitPrivateKeyPath(projectRoot, path string) string { + if path == "~" || strings.HasPrefix(path, "~/") { + if home, err := os.UserHomeDir(); err == nil { + if path == "~" { + return home + } + return filepath.Join(home, path[2:]) + } + } + return config.ProjectPath(projectRoot, path) +} + +func promptRequired(out io.Writer, reader *bufio.Reader, label string) (string, error) { + for { + fmt.Fprintf(out, "%s: ", label) + value, err := reader.ReadString('\n') + if err != nil && !errors.Is(err, io.EOF) { + return "", err + } + value = strings.TrimSpace(value) + if strings.ContainsAny(value, "\r\n") { + return "", fmt.Errorf("%s must be one line", label) + } + if value != "" { + return value, nil + } + if errors.Is(err, io.EOF) { + return "", fmt.Errorf("%s is required", label) + } + fmt.Fprintf(out, "%s is required.\n", label) + } +} + +func promptRequiredInt64(out io.Writer, reader *bufio.Reader, label string) (int64, error) { + for { + value, err := promptRequired(out, reader, label) + if err != nil { + return 0, err + } + parsed, parseErr := strconv.ParseInt(value, 10, 64) + if parseErr == nil && parsed > 0 { + return parsed, nil + } + fmt.Fprintf(out, "%s must be a positive number.\n", label) + } +} + +func promptPoolNamePrefix(out io.Writer, reader *bufio.Reader, defaultValue string) (string, error) { + for { + value, hitEOF, err := promptDefault(out, reader, "Pool name prefix", defaultValue) + if err != nil { + return "", err + } + if err := config.ValidatePrefix(value); err != nil { + fmt.Fprintf(out, "Pool name prefix is invalid: %v\n", err) + if hitEOF { + return "", err + } + continue + } + return value, nil + } +} + +type initProviderOption struct { + Number string + Type string + Label string + Available bool + Status string + Default bool + Aliases []string +} + +type initProviderPrerequisites struct { + DockerAvailable bool + DockerStatus string + DockerSandboxesAvailable bool + DockerSandboxesStatus string + WSLAvailable bool + WSLStatus string + TartAvailable bool + TartStatus string +} + +func promptInitProvider(ctx context.Context, projectRoot string, out io.Writer, reader *bufio.Reader, skipDockerCheck bool) (string, sandboxpromotion.Record, *initDockerSandboxesProfile, error) { + hostPlatform := initSandboxPromotionPlatform() + record, promoted := initSandboxPromotionLookup(hostPlatform) + for { + providerType, _, refresh, err := promptInitProviderChoice(ctx, projectRoot, hostPlatform, record, promoted, out, reader, skipDockerCheck) + if err != nil { + return "", sandboxpromotion.Record{}, nil, err + } + if refresh { + fmt.Fprintln(out, "Refreshing provider prerequisites...") + continue + } + if providerType == "tart" { + return providerType, sandboxpromotion.Record{}, nil, nil + } + if providerType == "docker-container" || providerType == "docker-sandboxes" || providerType == "wsl" { + profile, accepted, profileErr := promptDockerImageProfile(ctx, projectRoot, providerType, hostPlatform, out, reader) + if profileErr != nil { + return "", sandboxpromotion.Record{}, nil, profileErr + } + if !accepted { + return "", sandboxpromotion.Record{}, nil, fmt.Errorf("%s image setup did not complete; no config was written", providerType) + } + return providerType, sandboxpromotion.Record{}, profile, nil + } + return "", sandboxpromotion.Record{}, nil, fmt.Errorf("provider %q has no registered image onboarding flow", providerType) + } +} + +func promptInitProviderChoice(ctx context.Context, projectRoot string, hostPlatform sandboxpromotion.Platform, record sandboxpromotion.Record, promoted bool, out io.Writer, reader *bufio.Reader, skipDockerCheck bool) (string, bool, bool, error) { + prerequisites := detectInitProviderPrerequisites(ctx, hostPlatform, skipDockerCheck) + operationalDefault := !promoted && prerequisites.DockerSandboxesAvailable + promotionPassed := false + var promotionFailures []sandboxpromotion.Failure + if promoted { + var preflight sandboxpromotion.PreflightResult + if os.Getenv(sandboxpromotion.DisableEnvironment) == "1" { + preflight.Failures = append(preflight.Failures, sandboxpromotion.Failure{ + Gate: "operator kill switch", + Detail: sandboxpromotion.DisableEnvironment + "=1 disables Docker Sandboxes admission and automatic selection", + Resolution: "Unset the kill switch only after the Docker Sandboxes issue is resolved, or explicitly choose another provider.", + }) + } else if err := sandboxpromotion.Validate(record); err != nil { + preflight.Failures = append(preflight.Failures, sandboxpromotion.Failure{ + Gate: "promotion record", + Detail: err.Error(), + Resolution: "Explicitly choose Docker Container or another provider and report the invalid embedded promotion record.", + }) + } else if prerequisites.DockerSandboxesAvailable { + preflightContext, cancel := context.WithTimeout(ctx, 45*time.Second) + preflight = initDockerSandboxesPreflight(preflightContext, record, projectRoot) + cancel() + } + promotionPassed = preflight.Passed() && prerequisites.DockerSandboxesAvailable + promotionFailures = preflight.Failures + fmt.Fprintln(out, "") + fmt.Fprintln(out, "Docker Sandboxes automatic-default preflight:") + if promotionPassed { + fmt.Fprintln(out, " PASS: the exact promoted platform, host, sbx, template, policy, and resource gates passed.") + } else { + for _, failure := range promotionFailures { + fmt.Fprintf(out, " FAIL [%s]: %s\n", failure.Gate, failure.Detail) + fmt.Fprintf(out, " Action: %s\n", failure.Resolution) + } + if len(promotionFailures) == 0 { + fmt.Fprintf(out, " FAIL [prerequisites]: %s\n", prerequisites.DockerSandboxesStatus) + } + prerequisites.DockerSandboxesAvailable = false + prerequisites.DockerSandboxesStatus = "UNAVAILABLE — promoted admission did not pass; review the failures above" + } + } + + defaultProvider := "docker-container" + if promotionPassed || operationalDefault { + defaultProvider = "docker-sandboxes" + } else if promoted || !prerequisites.DockerAvailable { + defaultProvider = "" + } + providerType, refresh, err := promptProviderOptions(out, reader, prerequisites, promoted, promotionPassed, operationalDefault, defaultProvider) + if err != nil { + return "", false, false, err + } + return providerType, promotionPassed, refresh, nil +} + +func promptDockerSandboxesProfile(ctx context.Context, projectRoot string, hostPlatform sandboxpromotion.Platform, out io.Writer, reader *bufio.Reader) (*initDockerSandboxesProfile, bool, error) { + return promptDockerImageProfile(ctx, projectRoot, "docker-sandboxes", hostPlatform, out, reader) +} + +func promptImageUpdatePolicy(out io.Writer, reader *bufio.Reader) (initImageUpdatePolicy, error) { + var frequency string + for frequency == "" { + fmt.Fprintln(out, "") + fmt.Fprintln(out, "Automatic image and Actions runner updates:") + fmt.Fprintln(out, " 1. Weekly (default)") + fmt.Fprintln(out, " 2. Daily") + fmt.Fprintln(out, " 3. Every two weeks") + fmt.Fprintln(out, " 4. Monthly") + fmt.Fprintln(out, " 5. Manual — check only on demand") + fmt.Fprintln(out, " Command: ./start image update") + choice, hitEOF, err := promptDefault(out, reader, "Update frequency", "1") + if err != nil { + return initImageUpdatePolicy{}, err + } + switch strings.ToLower(strings.TrimSpace(choice)) { + case "1", "weekly": + frequency = config.ImageUpdateFrequencyWeekly + case "2", "daily": + frequency = config.ImageUpdateFrequencyDaily + case "3", "biweekly", "every two weeks": + frequency = config.ImageUpdateFrequencyBiweekly + case "4", "monthly": + frequency = config.ImageUpdateFrequencyMonthly + case "5", "manual": + return initImageUpdatePolicy{Frequency: config.ImageUpdateFrequencyManual, Time: config.DefaultImageUpdateTime}, nil + default: + fmt.Fprintln(out, " Choose 1–5 or enter daily, weekly, biweekly, monthly, or manual.") + if hitEOF { + return initImageUpdatePolicy{}, fmt.Errorf("invalid image update frequency %q", choice) + } + } + } + for { + updateTime, _, promptErr := promptDefault(out, reader, "Local update time (24-hour HH:MM)", config.DefaultImageUpdateTime) + if promptErr != nil { + return initImageUpdatePolicy{}, promptErr + } + policy := initImageUpdatePolicy{Frequency: frequency, Time: updateTime} + image := config.Default().Image + image.UpdateFrequency = policy.Frequency + image.UpdateTime = policy.Time + if validationErr := config.ValidateImageUpdatePolicy(image); validationErr != nil { + fmt.Fprintf(out, " %v\n", validationErr) + continue + } + return policy, nil + } +} + +func promptDockerImageProfile(ctx context.Context, projectRoot, providerType string, hostPlatform sandboxpromotion.Platform, out io.Writer, reader *bufio.Reader) (*initDockerSandboxesProfile, bool, error) { + guestPlatform, err := initDockerGuestPlatform(providerType, hostPlatform) + if err != nil { + return nil, false, fmt.Errorf("%s image setup is unavailable: %w", providerType, err) + } + descriptor, found := providerregistry.DescriptorFor(providerType) + if !found || !descriptor.GuidedArtifacts || len(descriptor.WizardImageProfiles) == 0 { + return nil, false, fmt.Errorf("%s has no registered guided image onboarding contribution", providerType) + } + fmt.Fprintln(out, "") + fmt.Fprintf(out, "%s image setup:\n", descriptor.DisplayName) + fmt.Fprintln(out, " Choose the desired Catthehacker Ubuntu image. EPAR will create the configuration now and provision or update the reusable runner artifact during startup.") + fmt.Fprintln(out, "") + fmt.Fprintln(out, "Runner base image:") + for index, profile := range descriptor.WizardImageProfiles { + defaultLabel := "" + if index == 0 { + defaultLabel = " (default)" + } + fmt.Fprintf(out, " %d. %s — %s%s\n", index+1, profile.Name, profile.Tag, defaultLabel) + } + customChoice := strconv.Itoa(len(descriptor.WizardImageProfiles) + 1) + fmt.Fprintf(out, " %s. Another catthehacker/ubuntu tag, such as go-24.04\n", customChoice) + fmt.Fprintln(out, " Image catalog: https://github.com/catthehacker/docker_images#images-available") + + var source imageartifact.ResolvedDockerSource + for { + choice, hitEOF, promptErr := promptDefault(out, reader, "Runner base image", "1") + if promptErr != nil { + return nil, false, promptErr + } + normalizedChoice := strings.ToLower(choice) + input := "" + for index, profile := range descriptor.WizardImageProfiles { + if normalizedChoice == strconv.Itoa(index+1) || normalizedChoice == profile.Name { + input = profile.Name + break + } + } + if normalizedChoice == customChoice { + input, promptErr = promptRequired(out, reader, "catthehacker/ubuntu tag") + if promptErr != nil { + return nil, false, promptErr + } + } + if input == "" { + fmt.Fprintf(out, " Choose a built-in image from 1 to %d, or %s for another catthehacker/ubuntu tag.\n", len(descriptor.WizardImageProfiles), customChoice) + if hitEOF { + return nil, false, fmt.Errorf("invalid runner base image %q", choice) + } + continue + } + resolveContext, cancel := context.WithTimeout(ctx, 90*time.Second) + source, err = initResolveDockerSandboxesSource(resolveContext, input, guestPlatform) + cancel() + if err == nil { + break + } + fmt.Fprintf(out, " That image cannot be used for %s: %v\n", guestPlatform, err) + fmt.Fprintln(out, " Choose an existing ghcr.io/catthehacker/ubuntu tag that publishes this platform.") + if hitEOF { + return nil, false, fmt.Errorf("resolve runner source image: %w", err) + } + } + + var customScripts []string + addScripts, err := promptYesNo(out, reader, "Run custom install scripts while building the runner artifact?", false) + if err != nil { + return nil, false, err + } + if addScripts { + fmt.Fprintln(out, " Scripts run as root during the image build. Do not put secrets in scripts or build inputs.") + for { + script, promptErr := promptOptional(out, reader, "Custom install script path") + if promptErr != nil { + return nil, false, promptErr + } + if script == "" { + break + } + normalized, validationErr := validateInitCustomInstallScript(projectRoot, script) + if validationErr != nil { + fmt.Fprintf(out, " Invalid custom install script: %v\n", validationErr) + continue + } + customScripts = append(customScripts, normalized) + another, promptErr := promptYesNo(out, reader, "Add another custom install script?", false) + if promptErr != nil { + return nil, false, promptErr + } + if !another { + break + } + } + } + + policyFingerprint := "" + if providerType == "docker-sandboxes" { + policyFingerprint, err = initDockerSandboxesPolicyFingerprint(ctx) + if err != nil { + return nil, false, err + } + } + sourceEstimate, err := imageartifact.EstimateSourceSize(source.CompressedLayerBytes, 0) + if err != nil { + return nil, false, err + } + const dockerDisk = config.DockerSandboxesDefaultDockerDisk + dockerDiskBytes, _ := config.ParseByteSize(dockerDisk) + artifactPlan, err := imageartifact.PlanArtifactStorage(providerType, sourceEstimate, false, uint64(dockerDiskBytes)) + if err != nil { + return nil, false, err + } + var availableText = "unknown" + if capacity, probeErr := storage.ProbeFilesystemCapacity(projectRoot, time.Now()); probeErr == nil && capacity.Known { + availableText = formatInitUintByteCount(capacity.AvailableBytes) + } + fmt.Fprintln(out, "") + fmt.Fprintln(out, "Runner artifact estimate:") + fmt.Fprintf(out, " Source: %s\n", source.Reference) + fmt.Fprintf(out, " Platform: %s\n", source.Platform) + if len(customScripts) == 0 { + fmt.Fprintln(out, " Custom install scripts: none") + } else { + fmt.Fprintf(out, " Custom install scripts: %s\n", strings.Join(customScripts, ", ")) + } + fmt.Fprintf(out, " Estimated download: %s compressed layers\n", formatInitUintByteCount(source.CompressedLayerBytes)) + fmt.Fprintf(out, " Estimated expanded source: %s\n", formatInitUintByteCount(sourceEstimate.ExpandedBytes)) + fmt.Fprintf(out, " Estimated incremental physical peak: %s\n", formatInitUintByteCount(artifactPlan.EstimatedIncrementalPeak)) + fmt.Fprintf(out, " Available physical space: %s\n", availableText) + fmt.Fprintln(out, " Fixed free-space reserve: 1GiB") + if providerType == "docker-sandboxes" { + fmt.Fprintf(out, " Automatic sandbox root limit: %s (sparse logical maximum)\n", formatInitUintByteCount(artifactPlan.LogicalRootMaximumBytes)) + fmt.Fprintf(out, " Inner Docker limit: %s (independent sparse logical maximum)\n", formatInitUintByteCount(artifactPlan.LogicalDockerMaximumBytes)) + } + confirmed, err := promptYesNo(out, reader, "Create this configuration?", true) + if err != nil { + return nil, false, err + } + if !confirmed { + return nil, false, nil + } + return &initDockerSandboxesProfile{ + Provider: providerType, + HostPlatform: hostPlatform, + GuestPlatform: guestPlatform, + SourceImage: source.Reference, + CustomScripts: customScripts, + PolicyFingerprint: policyFingerprint, + RootDisk: config.DockerSandboxesAutomaticRootDisk, + DockerDisk: dockerDisk, + }, true, nil +} + +func initDockerGuestPlatform(providerType string, hostPlatform sandboxpromotion.Platform) (string, error) { + if providerType == "docker-sandboxes" { + platform, _, err := dockerSandboxesPlatform(hostPlatform) + return platform, err + } + _, architecture, found := strings.Cut(string(hostPlatform), "/") + if !found { + return "", fmt.Errorf("unsupported controller platform %q", hostPlatform) + } + switch architecture { + case "amd64", "arm64": + return "linux/" + architecture, nil + default: + return "", fmt.Errorf("unsupported Docker image architecture %q", architecture) + } +} + +func validateInitCustomInstallScript(projectRoot, configured string) (string, error) { + value := strings.TrimSpace(configured) + if value == "" || filepath.IsAbs(value) || filepath.VolumeName(value) != "" { + return "", fmt.Errorf("path must be project-relative") + } + path := config.ProjectPath(projectRoot, value) + relative, err := filepath.Rel(projectRoot, path) + if err != nil || relative == ".." || strings.HasPrefix(relative, ".."+string(filepath.Separator)) { + return "", fmt.Errorf("path must remain under the project root") + } + info, err := os.Lstat(path) + if err != nil { + return "", err + } + if !info.Mode().IsRegular() { + return "", fmt.Errorf("path must name a regular file") + } + return filepath.ToSlash(filepath.Clean(relative)), nil +} + +func checkInitDockerSandboxesCapacity(rootDisk, dockerDisk, minHostFreeSpace uint64) (initDockerSandboxesCapacityResult, error) { + _ = rootDisk + _ = dockerDisk + storageRoot, err := sandboxcapacity.DockerSandboxesStorageRoot() + if err != nil { + return initDockerSandboxesCapacityResult{}, err + } + probePath := storageRoot + for { + if _, statErr := os.Lstat(probePath); statErr == nil { + break + } else if !os.IsNotExist(statErr) { + return initDockerSandboxesCapacityResult{}, fmt.Errorf("inspect %s: %w", probePath, statErr) + } + parent := filepath.Dir(probePath) + if parent == probePath { + return initDockerSandboxesCapacityResult{}, fmt.Errorf("find existing filesystem ancestor for %s", storageRoot) + } + probePath = parent + } + capacity, err := storage.ProbeFilesystemCapacity(probePath, time.Now()) + if err != nil { + return initDockerSandboxesCapacityResult{}, fmt.Errorf("probe %s for %s: %w", probePath, storageRoot, err) + } + reservation := uint64(0) + hostWatermark, err := sandboxcapacity.HostWatermark(minHostFreeSpace, capacity.TotalBytes) + if err != nil { + return initDockerSandboxesCapacityResult{}, err + } + check, err := storage.EvaluateCapacity(storage.Surface{ + ID: "docker-sandboxes-backing", + Provider: "docker-sandboxes", + Kind: storage.SurfaceSandboxCache, + Location: storageRoot, + Capacity: capacity, + }, storage.Requirement{ + ID: "docker-sandboxes-instance-create", + Provider: "docker-sandboxes", + SurfaceID: "docker-sandboxes-backing", + PeakBytes: reservation, + MinimumFreeBytes: hostWatermark, + }) + if err != nil { + return initDockerSandboxesCapacityResult{}, err + } + return initDockerSandboxesCapacityResult{ + StorageRoot: storageRoot, + AvailableBytes: capacity.AvailableBytes, + TotalBytes: capacity.TotalBytes, + Reservation: reservation, + HostWatermark: hostWatermark, + RequiredBytes: check.RequiredAvailableBytes, + DeficitBytes: check.DeficitBytes, + CapacityStatus: check.Status, + }, nil +} + +func promptInitByteSize(out io.Writer, reader *bufio.Reader, label, defaultValue string, minimum int64) (string, error) { + for { + value, hitEOF, err := promptDefault(out, reader, label, defaultValue) + if err != nil { + return "", err + } + parsed, parseErr := config.ParseByteSize(value) + if parseErr == nil && parsed >= minimum { + return value, nil + } + if parseErr != nil { + fmt.Fprintf(out, "%s is invalid: %v\n", label, parseErr) + } else { + fmt.Fprintf(out, "%s must be at least %s.\n", label, formatInitByteCount(minimum)) + } + if hitEOF { + return "", fmt.Errorf("invalid %s %q", strings.ToLower(label), value) + } + } +} + +func formatInitByteCount(value int64) string { + const ( + gib = int64(1 << 30) + mib = int64(1 << 20) + ) + if value >= gib { + if value%gib == 0 { + return strconv.FormatInt(value/gib, 10) + "GiB" + } + return strconv.FormatFloat(float64(value)/float64(gib), 'f', 2, 64) + "GiB" + } + if value >= mib { + if value%mib == 0 { + return strconv.FormatInt(value/mib, 10) + "MiB" + } + return strconv.FormatFloat(float64(value)/float64(mib), 'f', 2, 64) + "MiB" + } + return strconv.FormatInt(value, 10) + "B" +} + +func formatInitUintByteCount(value uint64) string { + if value <= math.MaxInt64 { + return formatInitByteCount(int64(value)) + } + return strconv.FormatFloat(float64(value)/float64(uint64(1)<<30), 'f', 2, 64) + "GiB" +} + +func dockerSandboxesRootMeasurement(hostPlatform sandboxpromotion.Platform, template initDockerSandboxesTemplate) (initDockerSandboxesRootMeasurement, bool) { + const fullTemplateDigest = "sha256:00303a3e249a1baf8b0585d20273af408c27182dcfc827a98aa25ffe66b1f67f" + if hostPlatform != sandboxpromotion.WindowsAMD64 || template.Digest != fullTemplateDigest || template.Platform != "linux/amd64" { + return initDockerSandboxesRootMeasurement{}, false + } + return initDockerSandboxesRootMeasurement{ + PeakBytes: 324_780_032, + Evidence: "exact full-template Buildx and Compose validation probe on Docker Sandboxes", + }, true +} + +func discoverDockerSandboxes(ctx context.Context, projectRoot, guestPlatform string) (initDockerSandboxesDiscovery, error) { + profiles, err := readDockerSandboxesActiveProfiles(projectRoot, guestPlatform) if err != nil { - return err + return initDockerSandboxesDiscovery{}, err } - organization, err := promptRequired(opts.Out, reader, "GitHub organization") - if err != nil { - return err + adapter := dockersandboxes.New("") + if err := adapter.VerifyAdmission(ctx); err != nil { + return initDockerSandboxesDiscovery{}, fmt.Errorf("Docker Sandboxes admission check failed: %w", err) } - privateKeyPath, err := promptRequired(opts.Out, reader, "GitHub App private key path") + cached, err := adapter.CachedTemplates(ctx) if err != nil { - return err + return initDockerSandboxesDiscovery{}, fmt.Errorf("read Docker Sandboxes template inventory: %w", err) } - providerType := "docker-dind" - if wsl2Available() { - providerType, err = promptProviderType(opts.Out, reader, "wsl") - if err != nil { - return err - } - } else if tartAvailable() { - providerType, err = promptProviderType(opts.Out, reader, "tart") - if err != nil { - return err + cachedByReference := make(map[string]dockersandboxes.CachedTemplate, len(cached)) + for _, template := range cached { + reference, canonicalErr := canonicalDockerSandboxesTemplateReference(template.Reference) + if canonicalErr != nil { + continue } + cachedByReference[reference] = template } - if !opts.SkipDockerCheck && providerType != "tart" { - ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) - dockerErr := dockerAvailable(ctx) - cancel() - if dockerErr != nil { - return fmt.Errorf("Docker is required for the selected %s setup. Install Docker Desktop, Docker Engine, or a compatible Docker host, then rerun %s init", providerDisplayName(providerType), binaryName) + var templates []initDockerSandboxesTemplate + for _, profile := range profiles { + template, found := cachedByReference[profile.TemplateReference] + if !found { + continue } + templates = append(templates, initDockerSandboxesTemplate{ + Reference: template.Reference, + Digest: "", + CacheID: template.CacheID, + Platform: guestPlatform, + Size: template.SizeBytes, + Label: profile.DisplayLabel, + SourceChannel: profile.ObservedTag, + }) } - defaultPrefix, err := generatedPoolNamePrefix() + rules, err := adapter.ReadGlobalNetworkPolicy(ctx) if err != nil { - return err + return initDockerSandboxesDiscovery{}, fmt.Errorf("read Docker Sandboxes global policy: %w", err) } - fmt.Fprintln(opts.Out, "") - fmt.Fprintln(opts.Out, "Pool name prefix must be unique for this machine/config within the GitHub organization.") - fmt.Fprintln(opts.Out, "EPAR cleanup deletes GitHub runner records matching this prefix.") - poolNamePrefix, err := promptPoolNamePrefix(opts.Out, reader, defaultPrefix) + fingerprint, err := sandboxpolicy.Fingerprint(rules) if err != nil { - return err + return initDockerSandboxesDiscovery{}, fmt.Errorf("fingerprint Docker Sandboxes global policy: %w", err) } + if err := sandboxpolicy.VerifyBaseline(fingerprint, "epar-preview-policy-probe", rules); err != nil { + return initDockerSandboxesDiscovery{}, fmt.Errorf("verify Docker Sandboxes global policy: %w", err) + } + return initDockerSandboxesDiscovery{Templates: templates, PolicyFingerprint: fingerprint}, nil +} - hostTrustMode := config.HostTrustModeDisabled - hostTrustScopes := []string{config.HostTrustScopeSystem} - if providerType == "docker-dind" { - enabled, promptErr := promptYesNo(opts.Out, reader, "Inherit this host's trusted TLS roots into disposable runners?", true) - if promptErr != nil { - return promptErr +func readDockerSandboxesActiveProfiles(projectRoot, guestPlatform string) ([]dockerSandboxesActiveProfile, error) { + if projectRoot == "" { + return nil, errors.New("read Docker Sandboxes source lock: project root is empty") + } + lockPath := filepath.Join(projectRoot, "templates", "docker-sandboxes", "sources.lock.json") + contents, err := os.ReadFile(lockPath) + if err != nil { + return nil, fmt.Errorf("read Docker Sandboxes source lock %q: %w", lockPath, err) + } + var lock dockerSandboxesSourceLock + if err := json.Unmarshal(contents, &lock); err != nil { + return nil, fmt.Errorf("parse Docker Sandboxes source lock %q: %w", lockPath, err) + } + if lock.SchemaVersion != 2 { + return nil, fmt.Errorf("read Docker Sandboxes source lock %q: unsupported schemaVersion %d", lockPath, lock.SchemaVersion) + } + if len(lock.Profiles) == 0 { + return nil, fmt.Errorf("read Docker Sandboxes source lock %q: no active profiles", lockPath) + } + profileOrder := []string{"full", "act-22.04"} + expectedSourceChannels := map[string]string{ + "full": "ghcr.io/catthehacker/ubuntu:full-latest", + "act-22.04": "ghcr.io/catthehacker/ubuntu:act-22.04", + } + profiles := make([]dockerSandboxesActiveProfile, 0, len(profileOrder)) + for _, name := range profileOrder { + profile, found := lock.Profiles[name] + if !found { + return nil, fmt.Errorf("read Docker Sandboxes source lock %q: active profile %q is missing", lockPath, name) } - if enabled { - hostTrustMode = config.HostTrustModeOverlay - hostTrustScopes = hostTrustScopesForOS(initHostTrustOS) - deferred := os.Getenv("EPAR_HOST_TRUST_INIT_DEFERRED") == "1" - if !opts.SkipHostTrustCheck && !deferred { - preflightCtx, cancel := context.WithTimeout(context.Background(), 30*time.Second) - _, collectErr := initResolveHostTrust(preflightCtx, hosttrust.Options{ - Mode: hostTrustMode, - Scopes: hostTrustScopes, - ControllerHostOS: initHostTrustOS, - }) - cancel() - if collectErr != nil { - return fmt.Errorf("collect host trusted TLS roots before writing config: %w", collectErr) - } - } + platform, found := profile.Platforms[guestPlatform] + if !found || platform.TemplateTag == "" { + return nil, fmt.Errorf("read Docker Sandboxes source lock %q: active profile %q has no templateTag for %s", lockPath, name, guestPlatform) + } + reference, err := canonicalDockerSandboxesTemplateReference(platform.TemplateTag) + if err != nil { + return nil, fmt.Errorf("read Docker Sandboxes source lock %q: active profile %q has invalid templateTag: %w", lockPath, name, err) + } + if profile.ObservedTagReference != expectedSourceChannels[name] { + return nil, fmt.Errorf("read Docker Sandboxes source lock %q: active profile %q has unexpected observedTagReference %q", lockPath, name, profile.ObservedTagReference) + } + label := "Catthehacker Ubuntu Act 22.04" + if name == "full" { + label = "Catthehacker Ubuntu Full" } + profiles = append(profiles, dockerSandboxesActiveProfile{ + Name: name, + ObservedTag: profile.ObservedTagReference, + TemplateReference: reference, + DisplayLabel: label, + }) } + if len(lock.Profiles) != len(profiles) { + return nil, fmt.Errorf("read Docker Sandboxes source lock %q: active profiles must be exactly full and act-22.04", lockPath) + } + profiles[0].DisplayLabel += " (recommended)" + profiles[1].DisplayLabel += " (current lean profile)" + return profiles, nil +} - content := defaultDockerDindConfig(appID, organization, privateKeyPath, poolNamePrefix, hostTrustMode, hostTrustScopes) - switch providerType { - case "wsl": - content = defaultWSLConfig(appID, organization, privateKeyPath, poolNamePrefix) - case "tart": - content = defaultTartConfig(appID, organization, privateKeyPath, poolNamePrefix) +func canonicalDockerSandboxesTemplateReference(reference string) (string, error) { + if strings.HasPrefix(reference, "docker.io/library/") { + reference = strings.TrimPrefix(reference, "docker.io/library/") } - if err := os.MkdirAll(filepath.Dir(opts.ConfigPath), 0755); err != nil { - return err + if strings.ContainsAny(reference, "@/ \t\r\n") || !strings.HasPrefix(reference, "epar-docker-sandboxes-catthehacker-") || strings.Count(reference, ":") != 1 { + return "", fmt.Errorf("must be an EPAR repository:tag reference, got %q", reference) } - if err := os.WriteFile(opts.ConfigPath, []byte(content), 0600); err != nil { - return err + parts := strings.SplitN(reference, ":", 2) + if parts[0] == "" || parts[1] == "" { + return "", fmt.Errorf("must include a repository and tag, got %q", reference) } + return "docker.io/library/" + reference, nil +} - fmt.Fprintf(opts.Out, ` -Created %s +func detectInitProviderPrerequisites(ctx context.Context, hostPlatform sandboxpromotion.Platform, skipDockerCheck bool) initProviderPrerequisites { + result := initProviderPrerequisites{} + if skipDockerCheck { + result.DockerAvailable = true + result.DockerStatus = "AVAILABLE — Docker check skipped by --skip-docker-check" + } else { + dockerContext, cancel := context.WithTimeout(ctx, 10*time.Second) + err := dockerAvailable(dockerContext) + cancel() + if err == nil { + result.DockerAvailable = true + result.DockerStatus = "READY — Docker CLI and daemon are available" + } else { + result.DockerStatus = fmt.Sprintf("UNAVAILABLE — Docker CLI or daemon check failed: %v", err) + } + } -Next: - %s start + if _, _, err := dockerSandboxesPlatform(hostPlatform); err != nil { + result.DockerSandboxesStatus = fmt.Sprintf("UNAVAILABLE — %v", err) + } else if os.Getenv(sandboxpromotion.DisableEnvironment) == "1" { + result.DockerSandboxesStatus = "UNAVAILABLE — " + sandboxpromotion.DisableEnvironment + "=1 disables admission" + } else if !result.DockerAvailable { + result.DockerSandboxesStatus = "UNAVAILABLE — Docker CLI and daemon are required" + } else { + readinessContext, cancel := context.WithTimeout(ctx, 30*time.Second) + readiness, err := initDockerSandboxesReadiness(readinessContext) + cancel() + if err == nil { + result.DockerSandboxesAvailable = true + result.DockerSandboxesStatus = fmt.Sprintf("READY — Docker and sbx diagnostics passed (%d pass, %d warn, %d fail, %d skip)", readiness.ChecksPassed, readiness.ChecksWarned, readiness.ChecksFailed, readiness.ChecksSkipped) + } else { + result.DockerSandboxesStatus = fmt.Sprintf("UNAVAILABLE — sbx readiness failed: %v", err) + if !strings.Contains(err.Error(), "sbx diagnose --output json") { + result.DockerSandboxesStatus += ". Run 'sbx diagnose --output json' and review the hints for each failed check." + } + } + } -Manual/advanced: - %s image build --replace - %s pool verify --instances 2 --register-only --cleanup - %s pool up --instances 2 -`, opts.ConfigPath, binaryName, binaryName, binaryName, binaryName) - return nil + switch { + case initGOOS != "windows": + result.WSLStatus = "UNAVAILABLE — native Windows, WSL2, and Docker are required" + case !result.DockerAvailable: + result.WSLStatus = "UNAVAILABLE — Docker CLI and daemon are required" + case !wsl2Available(): + result.WSLStatus = "UNAVAILABLE — wsl.exe must report Default Version: 2" + default: + result.WSLAvailable = true + result.WSLStatus = "READY — Docker and WSL2 are available" + } + + switch { + case initGOOS != "darwin": + result.TartStatus = "UNAVAILABLE — native macOS and tart are required" + case !tartAvailable(): + result.TartStatus = "UNAVAILABLE — tart --version failed" + default: + result.TartAvailable = true + result.TartStatus = "READY — tart is available" + } + return result } -func promptRequired(out io.Writer, reader *bufio.Reader, label string) (string, error) { - for { - fmt.Fprintf(out, "%s: ", label) - value, err := reader.ReadString('\n') - if err != nil && !errors.Is(err, io.EOF) { - return "", err - } - value = strings.TrimSpace(value) - if strings.ContainsAny(value, "\r\n") { - return "", fmt.Errorf("%s must be one line", label) - } - if value != "" { - return value, nil +func promptProviderOptions(out io.Writer, reader *bufio.Reader, prerequisites initProviderPrerequisites, promoted, promotionPassed, operationalDefault bool, defaultProvider string) (string, bool, error) { + options := make([]initProviderOption, 0, len(providerregistry.Descriptors())) + for _, descriptor := range providerregistry.Descriptors() { + option := initProviderOption{ + Number: descriptor.WizardNumber, + Type: descriptor.Type, + Label: descriptor.WizardLabel, + Aliases: append([]string(nil), descriptor.WizardAliases...), } - if errors.Is(err, io.EOF) { - return "", fmt.Errorf("%s is required", label) + switch descriptor.Type { + case "docker-container": + option.Available = prerequisites.DockerAvailable + option.Status = prerequisites.DockerStatus + case "docker-sandboxes": + option.Available = prerequisites.DockerSandboxesAvailable && (!promoted || promotionPassed) + option.Status = prerequisites.DockerSandboxesStatus + if operationalDefault { + option.Label = "Docker Sandboxes — recommended" + } else if promoted { + option.Label = "Docker Sandboxes (independently certified for this exact platform)" + } + case "wsl": + option.Available = prerequisites.WSLAvailable + option.Status = prerequisites.WSLStatus + case "tart": + option.Available = prerequisites.TartAvailable + option.Status = prerequisites.TartStatus + default: + return "", false, fmt.Errorf("registered provider %q has no prerequisite contribution", descriptor.Type) } - fmt.Fprintf(out, "%s is required.\n", label) + options = append(options, option) } -} + if err := validateWizardProviderOptions(options); err != nil { + return "", false, err + } + defaultNumber := prioritizeDefaultProviderOption(options, defaultProvider) -func promptRequiredInt64(out io.Writer, reader *bufio.Reader, label string) (int64, error) { + fmt.Fprintln(out, "") + if defaultNumber == "" { + fmt.Fprintln(out, "Runner provider (explicit choice required):") + } else { + fmt.Fprintln(out, "Runner provider:") + } + for _, option := range options { + defaultLabel := "" + if option.Default { + defaultLabel = " (default)" + } + fmt.Fprintf(out, " %s. %s%s\n", option.Number, option.Label, defaultLabel) + fmt.Fprintf(out, " Prerequisites: %s\n", option.Status) + } + fmt.Fprintln(out, " R. Refresh provider prerequisites") for { - value, err := promptRequired(out, reader, label) + var value string + var hitEOF bool + var err error + if defaultNumber == "" { + fmt.Fprint(out, "Runner provider: ") + value, err = reader.ReadString('\n') + if err != nil && !errors.Is(err, io.EOF) { + return "", false, err + } + hitEOF = errors.Is(err, io.EOF) + if hitEOF { + err = nil + } + value = strings.TrimSpace(value) + } else { + value, hitEOF, err = promptDefault(out, reader, "Runner provider", defaultNumber) + } if err != nil { - return 0, err + return "", false, err } - parsed, parseErr := strconv.ParseInt(value, 10, 64) - if parseErr == nil && parsed > 0 { - return parsed, nil + normalized := strings.ToLower(value) + if normalized == "r" || normalized == "refresh" { + return "", true, nil + } + var selected *initProviderOption + for index := range options { + option := &options[index] + if normalized == option.Number || normalized == option.Type { + selected = option + break + } + for _, alias := range option.Aliases { + if normalized == alias { + selected = option + break + } + } + if selected != nil { + break + } + } + if selected != nil && selected.Available { + return selected.Type, false, nil + } + if selected != nil { + fmt.Fprintf(out, "%s is unavailable: %s\n", selected.Label, selected.Status) + } else { + fmt.Fprintln(out, "Choose an available provider number or name shown above, or R to refresh.") + } + if hitEOF { + if selected != nil { + return "", false, fmt.Errorf("runner provider %q is unavailable: %s", value, selected.Status) + } + return "", false, fmt.Errorf("invalid runner provider %q", value) } - fmt.Fprintf(out, "%s must be a positive number.\n", label) } } -func promptPoolNamePrefix(out io.Writer, reader *bufio.Reader, defaultValue string) (string, error) { - for { - value, hitEOF, err := promptDefault(out, reader, "Pool name prefix", defaultValue) - if err != nil { - return "", err - } - if err := config.ValidatePrefix(value); err != nil { - fmt.Fprintf(out, "Pool name prefix is invalid: %v\n", err) - if hitEOF { - return "", err - } - continue +func prioritizeDefaultProviderOption(options []initProviderOption, defaultProvider string) string { + defaultIndex := -1 + for index := range options { + options[index].Default = false + if options[index].Type == defaultProvider && options[index].Available { + defaultIndex = index } - return value, nil } + if defaultIndex > 0 { + selected := options[defaultIndex] + copy(options[1:defaultIndex+1], options[:defaultIndex]) + options[0] = selected + defaultIndex = 0 + } + for index := range options { + options[index].Number = strconv.Itoa(index + 1) + } + if defaultIndex < 0 { + return "" + } + options[defaultIndex].Default = true + return options[defaultIndex].Number } -func promptProviderType(out io.Writer, reader *bufio.Reader, alternative string) (string, error) { - alternativeLabel := providerDisplayName(alternative) - fmt.Fprintln(out, "") - fmt.Fprintln(out, "Runner provider:") - fmt.Fprintln(out, " 1. Docker-DinD (default)") - fmt.Fprintf(out, " 2. %s\n", alternativeLabel) - for { - value, hitEOF, err := promptDefault(out, reader, "Runner provider", "1") - if err != nil { - return "", err +func validateWizardProviderOptions(options []initProviderOption) error { + registered := make(map[string]struct{}, len(options)) + for _, option := range options { + descriptor, found := providerregistry.DescriptorFor(option.Type) + if !found { + return fmt.Errorf("wizard provider %q has no registry entry", option.Type) } - switch strings.ToLower(value) { - case "1", "docker", "docker-dind": - return "docker-dind", nil - case "2", alternative: - return alternative, nil - case "wsl2": - if alternative == "wsl" { - return alternative, nil - } - default: - // Continue below so aliases that belong to an unavailable provider are rejected. + if !descriptor.WizardSupported { + return fmt.Errorf("wizard provider %q is not registered for onboarding", option.Type) } - fmt.Fprintf(out, "Runner provider must be 1 (Docker-DinD) or 2 (%s).\n", alternativeLabel) - if hitEOF { - return "", fmt.Errorf("invalid runner provider %q", value) + if option.Number != descriptor.WizardNumber || option.Label == "" || len(option.Aliases) == 0 { + return fmt.Errorf("wizard provider %q does not use its complete registry contribution", option.Type) + } + if _, duplicate := registered[option.Type]; duplicate { + return fmt.Errorf("wizard provider %q is duplicated", option.Type) + } + registered[option.Type] = struct{}{} + } + for _, descriptor := range providerregistry.Descriptors() { + if descriptor.WizardSupported { + if _, found := registered[descriptor.Type]; !found { + return fmt.Errorf("registered provider %q has no ./start wizard option", descriptor.Type) + } } } + return nil } func providerDisplayName(providerType string) string { - switch providerType { - case "wsl": - return "WSL2" - case "tart": - return "Tart (experimental)" - default: - return "Docker-DinD" + if descriptor, found := providerregistry.DescriptorFor(providerType); found { + return descriptor.DisplayName } + return providerType } func promptDefault(out io.Writer, reader *bufio.Reader, label string, defaultValue string) (string, bool, error) { @@ -323,6 +1653,19 @@ func promptDefault(out io.Writer, reader *bufio.Reader, label string, defaultVal return value, hitEOF, nil } +func promptOptional(out io.Writer, reader *bufio.Reader, label string) (string, error) { + fmt.Fprintf(out, "%s (press Enter for none): ", label) + value, err := reader.ReadString('\n') + if err != nil && !errors.Is(err, io.EOF) { + return "", err + } + value = strings.TrimSpace(value) + if strings.ContainsAny(value, "\r\n") { + return "", fmt.Errorf("%s must be one line", label) + } + return value, nil +} + func promptYesNo(out io.Writer, reader *bufio.Reader, label string, defaultYes bool) (bool, error) { defaultValue := "Y" if !defaultYes { @@ -499,7 +1842,7 @@ func (b *boundedBuffer) Write(p []byte) (int, error) { return b.Buffer.Write(p) } -func defaultDockerDindConfig(appID int64, organization, privateKeyPath string, poolNamePrefix, hostTrustMode string, hostTrustScopes []string) string { +func defaultDockerContainerConfig(appID int64, organization, privateKeyPath string, poolNamePrefix, hostTrustMode string, hostTrustScopes []string, runnerGroup initRunnerGroupSelection, profile initDockerSandboxesProfile, updatePolicy initImageUpdatePolicy) string { return fmt.Sprintf(`github: appId: %d organization: %s @@ -509,14 +1852,18 @@ func defaultDockerDindConfig(appID int64, organization, privateKeyPath string, p image: sourceType: docker-image - sourceImage: ghcr.io/catthehacker/ubuntu:full-latest - outputImage: epar-docker-dind-catthehacker-ubuntu + sourceImage: %s + sourcePlatform: %s + outputImage: epar-docker-container-catthehacker-ubuntu upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: %s + updateTime: "%s" hostTrustMode: %s hostTrustScopes: [%s] customInstallScripts: +%s pool: instances: 1 @@ -525,6 +1872,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -546,13 +1902,22 @@ logging: retentionIntervalMinutes: 60 runner: - labels: [self-hosted, linux, epar-docker-dind-catthehacker-ubuntu] + group: %s + labels: [self-hosted, linux, epar-docker-container-catthehacker-ubuntu] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: %s + requireExplicitGroup: %t + requireNonDefaultGroup: %t + requiredRepositoryAccess: %s + requirePublicRepositoriesDisabled: %t + provider: - type: docker-dind - sourceImage: epar-docker-dind-catthehacker-ubuntu + type: docker-container + sourceImage: epar-docker-container-catthehacker-ubuntu network: default docker: @@ -563,10 +1928,135 @@ timeouts: bootSeconds: 180 githubOnlineSeconds: 180 commandSeconds: 900 -`, appID, organization, privateKeyPath, hostTrustMode, strings.Join(hostTrustScopes, ", "), poolNamePrefix) +`, appID, organization, privateKeyPath, profile.SourceImage, profile.GuestPlatform, updatePolicy.Frequency, updatePolicy.Time, hostTrustMode, strings.Join(hostTrustScopes, ", "), renderInitCustomInstallScripts(profile.CustomScripts), poolNamePrefix, strconv.Quote(runnerGroup.Group.Name), runnerGroup.Policy.Enforcement, runnerGroup.Policy.RequireExplicitGroup, runnerGroup.Policy.RequireNonDefaultGroup, runnerGroup.Policy.RequiredRepositoryAccess, runnerGroup.Policy.RequirePublicRepositoriesDisabled) +} + +func promotedDockerSandboxesPlatform(record sandboxpromotion.Record) (string, string, error) { + return dockerSandboxesPlatform(record.Platform) +} + +func dockerSandboxesPlatform(platform sandboxpromotion.Platform) (string, string, error) { + hostOS, hostArch, found := strings.Cut(string(platform), "/") + if !found || hostOS == "" || hostArch == "" || strings.Contains(hostArch, "/") { + return "", "", fmt.Errorf("unsupported Docker Sandboxes controller platform %q", platform) + } + guestPlatform, err := config.DockerSandboxesGuestPlatform(hostOS, hostArch) + if err != nil { + return "", "", err + } + switch guestPlatform { + case "linux/amd64": + return guestPlatform, "X64", nil + case "linux/arm64": + return guestPlatform, "ARM64", nil + default: + return "", "", fmt.Errorf("Docker Sandboxes guest platform %q has no GitHub runner architecture label", guestPlatform) + } +} + +func defaultDockerSandboxesConfig(appID int64, organization, privateKeyPath string, poolNamePrefix, hostTrustMode string, hostTrustScopes []string, runnerGroup initRunnerGroupSelection, profile initDockerSandboxesProfile, updatePolicy initImageUpdatePolicy, guestPlatform, runnerArchitectureLabel string) string { + return fmt.Sprintf(`github: + appId: %d + organization: %s + privateKeyPath: %s + apiBaseUrl: https://api.github.com + webBaseUrl: https://github.com + +image: + sourceType: docker-image + sourceImage: %s + sourcePlatform: %s + runnerVersion: latest + updateFrequency: %s + updateTime: "%s" + customInstallScripts: +%s + hostTrustMode: %s + hostTrustScopes: [%s] + +pool: + instances: 1 + namePrefix: %s + replacementRetryInitialSeconds: 15 + replacementRetryMaxSeconds: 1800 + replacementRetryMultiplier: 2 + replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + +logging: + directory: work/logs + managerSinks: [console] + managerConsoleFormat: text + managerConsoleTextFormat: "{time} [{level}] {message}" + managerFileFormat: json + transcriptSinks: [file] + transcriptConsoleFormat: text + maxFileSizeMiB: 100 + maxBackups: 3 + compressBackups: true + retentionEnabled: true + retentionMaxTotalMiB: 1024 + managerMaxAgeDays: 14 + instanceMaxAgeDays: 14 + buildMaxAgeDays: 14 + errorMaxAgeDays: 30 + benchmarkMaxAgeDays: 90 + retentionIntervalMinutes: 60 + +runner: + group: %s + labels: [self-hosted, linux, %s, epar-docker-sandboxes] + includeHostLabel: true + ephemeral: true + +security: + runnerGroup: + enforcement: %s + requireExplicitGroup: %t + requireNonDefaultGroup: %t + requiredRepositoryAccess: %s + requirePublicRepositoriesDisabled: %t + +provider: + type: docker-sandboxes + platform: %s + +dockerSandboxes: + policyGeneration: %s + networkBaseline: open + stagingRoot: .local/docker-sandboxes-staging + cpus: 4 + memory: 8GiB + rootDisk: %s + dockerDisk: %s + maxConcurrentCreates: 2 + +timeouts: + bootSeconds: 180 + githubOnlineSeconds: 180 + commandSeconds: 900 +`, appID, organization, privateKeyPath, profile.SourceImage, guestPlatform, updatePolicy.Frequency, updatePolicy.Time, renderInitCustomInstallScripts(profile.CustomScripts), hostTrustMode, strings.Join(hostTrustScopes, ", "), poolNamePrefix, strconv.Quote(runnerGroup.Group.Name), runnerArchitectureLabel, runnerGroup.Policy.Enforcement, runnerGroup.Policy.RequireExplicitGroup, runnerGroup.Policy.RequireNonDefaultGroup, runnerGroup.Policy.RequiredRepositoryAccess, runnerGroup.Policy.RequirePublicRepositoriesDisabled, guestPlatform, profile.PolicyFingerprint, profile.RootDisk, profile.DockerDisk) +} + +func renderInitCustomInstallScripts(paths []string) string { + if len(paths) == 0 { + return " # - examples/custom-install/install-extra-apt-tools.sh" + } + var lines []string + for _, path := range paths { + lines = append(lines, " - "+strconv.Quote(path)) + } + return strings.Join(lines, "\n") } -func defaultWSLConfig(appID int64, organization, privateKeyPath string, poolNamePrefix string) string { +func defaultWSLConfig(appID int64, organization, privateKeyPath string, poolNamePrefix string, runnerGroup initRunnerGroupSelection, profile initDockerSandboxesProfile, updatePolicy initImageUpdatePolicy) string { return fmt.Sprintf(`github: appId: %d organization: %s @@ -576,14 +2066,16 @@ func defaultWSLConfig(appID int64, organization, privateKeyPath string, poolName image: sourceType: docker-image - sourceImage: ghcr.io/catthehacker/ubuntu:full-latest - sourcePlatform: linux/amd64 + sourceImage: %s + sourcePlatform: %s outputImage: work/images/epar-wsl-catthehacker-ubuntu.tar upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: %s + updateTime: "%s" customInstallScripts: - # - examples/custom-install/install-extra-apt-tools.sh +%s pool: instances: 1 @@ -593,6 +2085,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -614,10 +2115,19 @@ logging: retentionIntervalMinutes: 60 runner: + group: %s labels: [self-hosted, linux, X64, epar-wsl-catthehacker-ubuntu] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: %s + requireExplicitGroup: %t + requireNonDefaultGroup: %t + requiredRepositoryAccess: %s + requirePublicRepositoriesDisabled: %t + provider: type: wsl sourceImage: work/images/epar-wsl-catthehacker-ubuntu.tar @@ -632,10 +2142,10 @@ timeouts: bootSeconds: 180 githubOnlineSeconds: 180 commandSeconds: 900 -`, appID, organization, privateKeyPath, poolNamePrefix) +`, appID, organization, privateKeyPath, profile.SourceImage, profile.GuestPlatform, updatePolicy.Frequency, updatePolicy.Time, renderInitCustomInstallScripts(profile.CustomScripts), poolNamePrefix, strconv.Quote(runnerGroup.Group.Name), runnerGroup.Policy.Enforcement, runnerGroup.Policy.RequireExplicitGroup, runnerGroup.Policy.RequireNonDefaultGroup, runnerGroup.Policy.RequiredRepositoryAccess, runnerGroup.Policy.RequirePublicRepositoriesDisabled) } -func defaultTartConfig(appID int64, organization, privateKeyPath string, poolNamePrefix string) string { +func defaultTartConfig(appID int64, organization, privateKeyPath string, poolNamePrefix string, runnerGroup initRunnerGroupSelection, updatePolicy initImageUpdatePolicy) string { return fmt.Sprintf(`# Experimental: this default is a basic Ubuntu ARM64 Tart VM, not a GitHub-hosted runner image. # It does not include the broad dependency set from https://github.com/actions/runner-images. github: @@ -651,6 +2161,8 @@ image: upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: %s + updateTime: "%s" customInstallScripts: # - examples/custom-install/install-extra-apt-tools.sh @@ -662,6 +2174,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -683,10 +2204,19 @@ logging: retentionIntervalMinutes: 60 runner: + group: %s labels: [self-hosted, linux, ARM64, epar-tart-ubuntu-24.04-base] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: %s + requireExplicitGroup: %t + requireNonDefaultGroup: %t + requiredRepositoryAccess: %s + requirePublicRepositoriesDisabled: %t + provider: type: tart sourceImage: epar-ubuntu-24-arm64 @@ -700,7 +2230,7 @@ timeouts: bootSeconds: 180 githubOnlineSeconds: 180 commandSeconds: 900 -`, appID, organization, privateKeyPath, poolNamePrefix) +`, appID, organization, privateKeyPath, updatePolicy.Frequency, updatePolicy.Time, poolNamePrefix, strconv.Quote(runnerGroup.Group.Name), runnerGroup.Policy.Enforcement, runnerGroup.Policy.RequireExplicitGroup, runnerGroup.Policy.RequireNonDefaultGroup, runnerGroup.Policy.RequiredRepositoryAccess, runnerGroup.Policy.RequirePublicRepositoriesDisabled) } var stdinIsInteractive = func() bool { diff --git a/cmd/ephemeral-action-runner/init_test.go b/cmd/ephemeral-action-runner/init_test.go index 641f65d..c0c689f 100644 --- a/cmd/ephemeral-action-runner/init_test.go +++ b/cmd/ephemeral-action-runner/init_test.go @@ -1,24 +1,91 @@ package main import ( + "bufio" "bytes" "context" "errors" + "io" "os" + "os/exec" "path/filepath" + "reflect" "runtime" "slices" "strings" "testing" + "time" "unicode/utf16" "github.com/solutionforest/ephemeral-action-runner/internal/config" + gh "github.com/solutionforest/ephemeral-action-runner/internal/github" "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + imageartifact "github.com/solutionforest/ephemeral-action-runner/internal/image" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes" + sandboxpromotion "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/promotion" + providerregistry "github.com/solutionforest/ephemeral-action-runner/internal/provider/registry" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" ) -func TestInitCreatesDefaultDockerDindConfig(t *testing.T) { +func TestWizardCoversEveryRegisteredProvider(t *testing.T) { + options := make([]initProviderOption, 0, len(providerregistry.Descriptors())) + for _, descriptor := range providerregistry.Descriptors() { + options = append(options, initProviderOption{ + Number: descriptor.WizardNumber, + Type: descriptor.Type, + Label: descriptor.WizardLabel, + Aliases: descriptor.WizardAliases, + }) + } + if err := validateWizardProviderOptions(options); err != nil { + t.Fatal(err) + } +} + +func TestProviderWizardPutsAvailableDefaultFirst(t *testing.T) { + options := []initProviderOption{ + {Number: "1", Type: "docker-container", Available: true}, + {Number: "2", Type: "docker-sandboxes", Available: true}, + {Number: "3", Type: "wsl", Available: true}, + } + + if got := prioritizeDefaultProviderOption(options, "docker-sandboxes"); got != "1" { + t.Fatalf("default number = %q, want 1", got) + } + if options[0].Type != "docker-sandboxes" || !options[0].Default { + t.Fatalf("first option = %+v, want Docker Sandboxes default", options[0]) + } + if options[1].Type != "docker-container" || options[1].Number != "2" || options[1].Default { + t.Fatalf("second option = %+v, want non-default Docker Container", options[1]) + } + if options[2].Type != "wsl" || options[2].Number != "3" { + t.Fatalf("third option = %+v, want WSL", options[2]) + } +} + +func TestProviderWizardDoesNotPromoteUnavailableDefault(t *testing.T) { + options := []initProviderOption{ + {Number: "1", Type: "docker-container", Available: true}, + {Number: "2", Type: "docker-sandboxes", Available: false}, + } + + if got := prioritizeDefaultProviderOption(options, "docker-sandboxes"); got != "" { + t.Fatalf("default number = %q, want none", got) + } + if options[0].Type != "docker-container" || options[0].Number != "1" || options[0].Default { + t.Fatalf("first option = %+v, want unchanged non-default Docker Container", options[0]) + } +} + +func TestInitCreatesDefaultDockerContainerConfig(t *testing.T) { stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) stubNoWSL2(t) + oldPreflight := initDockerSandboxesPreflight + initDockerSandboxesPreflight = func(context.Context, sandboxpromotion.Record, string) sandboxpromotion.PreflightResult { + t.Fatal("empty promotion table must not run Docker Sandboxes preflight") + return sandboxpromotion.PreflightResult{} + } + t.Cleanup(func() { initDockerSandboxesPreflight = oldPreflight }) dir := t.TempDir() path := filepath.Join(dir, ".local", "config.yml") @@ -29,7 +96,7 @@ func TestInitCreatesDefaultDockerDindConfig(t *testing.T) { ConfigPath: path, SkipDockerCheck: true, SkipHostTrustCheck: true, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n\n"), Out: &out, }); err != nil { t.Fatal(err) @@ -42,13 +109,20 @@ func TestInitCreatesDefaultDockerDindConfig(t *testing.T) { if cfg.GitHub.AppID != 123456 || cfg.GitHub.Organization != "solutionforest" || cfg.GitHub.PrivateKeyPath != ".local/github-app.pem" { t.Fatalf("unexpected GitHub config: %+v", cfg.GitHub) } - if got, want := cfg.Provider.Type, "docker-dind"; got != want { + if cfg.Runner.Group != "restricted group" { + t.Fatalf("runner.group = %q, want selected group", cfg.Runner.Group) + } + policy := cfg.Security.RunnerGroup + if policy.Enforcement != config.RunnerGroupEnforcementEnforce || !policy.RequireExplicitGroup || !policy.RequireNonDefaultGroup || policy.RequiredRepositoryAccess != config.RunnerGroupRepositoryAccessSelected || !policy.RequirePublicRepositoriesDisabled { + t.Fatalf("unexpected generated runner-group policy: %+v", policy) + } + if got, want := cfg.Provider.Type, "docker-container"; got != want { t.Fatalf("provider.type = %q, want %q", got, want) } if got, want := cfg.Image.SourceImage, "ghcr.io/catthehacker/ubuntu:full-latest"; got != want { t.Fatalf("image.sourceImage = %q, want %q", got, want) } - if got, want := cfg.Image.OutputImage, "epar-docker-dind-catthehacker-ubuntu"; got != want { + if got, want := cfg.Image.OutputImage, "epar-docker-container-catthehacker-ubuntu"; got != want { t.Fatalf("image.outputImage = %q, want %q", got, want) } if got, want := cfg.Image.HostTrustMode, config.HostTrustModeOverlay; got != want { @@ -89,7 +163,10 @@ func TestInitCreatesDefaultDockerDindConfig(t *testing.T) { if !strings.Contains(string(configText), "replacementRetryInitialSeconds: 15\n replacementRetryMaxSeconds: 1800\n replacementRetryMultiplier: 2\n replacementRetryJitterPercent: 20\n") { t.Fatalf("generated config did not include replacement retry settings:\n%s", configText) } - if got := strings.Join(cfg.Runner.Labels, ","); !strings.Contains(got, "epar-docker-dind-catthehacker-ubuntu") { + if !strings.Contains(string(configText), "storage:\n minimumFree: 1GiB\n gracePeriod: 168h\n keepPrevious: 0\n automaticHousekeeping: conservative\n buildCacheLimit: 20GiB\n goCacheLimit: 10GiB\n") { + t.Fatalf("generated config did not include bounded storage settings:\n%s", configText) + } + if got := strings.Join(cfg.Runner.Labels, ","); !strings.Contains(got, "epar-docker-container-catthehacker-ubuntu") { t.Fatalf("runner labels = %q", got) } if !strings.Contains(out.String(), "start") || !strings.Contains(out.String(), "pool up --instances 2") { @@ -98,6 +175,370 @@ func TestInitCreatesDefaultDockerDindConfig(t *testing.T) { if !strings.Contains(out.String(), "Pool name prefix (press Enter to use build-box-01-a4f9c2):") { t.Fatalf("init output did not explain default prefix acceptance:\n%s", out.String()) } + hostTrustExplanation := "Runners need this host's trusted TLS roots to access services that this machine trusts." + hostTrustPrompt := "Inherit this host's trusted TLS roots into disposable runners?" + if explanationIndex, promptIndex := strings.Index(out.String(), hostTrustExplanation), strings.Index(out.String(), hostTrustPrompt); explanationIndex < 0 || promptIndex < 0 || explanationIndex > promptIndex { + t.Fatalf("init output did not explain host trust before prompting:\n%s", out.String()) + } + for _, want := range []string{"1. Docker Container", "private daemon (default)", "2. Docker Sandboxes — recommended when ready"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("init output did not preserve explicit Docker Container default and capability-driven Docker Sandboxes labeling %q:\n%s", want, out.String()) + } + } + for _, want := range []string{"D. Show runner group details", "B. Show blocked runner groups", "Assessment: RECOMMENDED"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("init output did not include concise runner-group choice %q:\n%s", want, out.String()) + } + } + for _, hidden := range []string{"Repository access meanings:", "Repository access:", "Public repositories:", "Group type:"} { + if strings.Contains(out.String(), hidden) { + t.Fatalf("init output included runner-group details before D was selected %q:\n%s", hidden, out.String()) + } + } +} + +func TestInitRunnerGroupDetailsAreOptIn(t *testing.T) { + client := &fakeInitRunnerGroupClient{ + groups: []gh.RunnerGroup{ + {ID: 1, Name: "restricted", Visibility: config.RunnerGroupRepositoryAccessSelected}, + {ID: 2, Name: "blocked-public", Visibility: config.RunnerGroupRepositoryAccessSelected, AllowsPublicRepositories: true}, + }, + repositories: map[int64][]gh.RunnerGroupRepository{ + 1: {{FullName: "example/private", Private: true}}, + 2: {{FullName: "example/public", Private: false}}, + }, + } + var out bytes.Buffer + selection, err := promptRunnerGroup(context.Background(), &out, bufio.NewReader(strings.NewReader("d\n1\n")), client) + if err != nil { + t.Fatal(err) + } + if selection.Group.Name != "restricted" { + t.Fatalf("selected group = %q, want restricted", selection.Group.Name) + } + for _, want := range []string{"Repository access meanings:", "Repository access: Selected repositories", "Public repositories: Disabled", "Group type: organization-managed, non-default", "D. Hide runner group details"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("runner-group details did not include %q after D was selected:\n%s", want, out.String()) + } + } + if strings.Contains(out.String(), "blocked-public") { + t.Fatalf("blocked group became visible when only details were requested:\n%s", out.String()) + } +} + +func TestInitRunnerGroupBlockedChoicesAreHiddenByDefault(t *testing.T) { + client := &fakeInitRunnerGroupClient{ + groups: []gh.RunnerGroup{ + {ID: 1, Name: "restricted", Visibility: config.RunnerGroupRepositoryAccessSelected}, + {ID: 2, Name: "blocked-public", Visibility: config.RunnerGroupRepositoryAccessSelected, AllowsPublicRepositories: true}, + {ID: 3, Name: "blocked-unknown", Visibility: "future-value"}, + }, + repositories: map[int64][]gh.RunnerGroupRepository{ + 1: {{FullName: "example/private", Private: true}}, + 2: {{FullName: "example/public", Private: false}}, + }, + } + var out bytes.Buffer + selection, err := promptRunnerGroup(context.Background(), &out, bufio.NewReader(strings.NewReader("1\n")), client) + if err != nil { + t.Fatal(err) + } + if selection.Group.Name != "restricted" { + t.Fatalf("selected group = %q, want first visible restricted group", selection.Group.Name) + } + for _, hidden := range []string{"blocked-public", "blocked-unknown"} { + if strings.Contains(out.String(), hidden) { + t.Fatalf("default runner-group list included blocked group %q:\n%s", hidden, out.String()) + } + } + if !strings.Contains(out.String(), "B. Show blocked runner groups") { + t.Fatalf("default runner-group list did not offer blocked groups on request:\n%s", out.String()) + } +} + +func TestInitAllowsDefaultGroupWithReminderAndWritesMatchingPolicy(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + setInitRunnerGroupClient(t, &fakeInitRunnerGroupClient{groups: []gh.RunnerGroup{{ID: 1, Name: "Default", Visibility: config.RunnerGroupRepositoryAccessAll, Default: true}}}) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123\nexample\nkey.pem\n1\n1\n\nn\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + policy := cfg.Security.RunnerGroup + if cfg.Runner.Group != "Default" || policy.RequireNonDefaultGroup || policy.RequiredRepositoryAccess != config.RunnerGroupRepositoryAccessAll || !policy.RequirePublicRepositoriesDisabled { + t.Fatalf("unexpected default-group config: group=%q policy=%+v", cfg.Runner.Group, policy) + } + for _, want := range []string{"*** SECURITY REMINDER: DEFAULT RUNNER GROUP ***", "The Default runner group is fine for trying EPAR.", "For regular use, a custom group limited to selected trusted repositories offers better security.", "Continue with Default runner group"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("default-group reminder did not include %q:\n%s", want, out.String()) + } + } + for _, unwanted := range []string{"SECURITY WARNING", "NOT RECOMMENDED", "requires explicit review", "RECOMMENDED ACTION", "Continue anyway", "relaxed policy"} { + if strings.Contains(out.String(), unwanted) { + t.Fatalf("default-group reminder included alarming wording %q:\n%s", unwanted, out.String()) + } + } + if !strings.Contains(out.String(), "Assessment: It is fine for first-time tasting of EPAR, but generally recommend to create and use custom runner group for better security.") { + t.Fatalf("default-group first-use assessment missing:\n%s", out.String()) + } +} + +func TestInitCanBackFromBroadGroupAndChooseRestrictedGroup(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + setInitRunnerGroupClient(t, &fakeInitRunnerGroupClient{ + groups: []gh.RunnerGroup{ + {ID: 1, Name: "broad", Visibility: config.RunnerGroupRepositoryAccessPrivate}, + {ID: 2, Name: "restricted", Visibility: config.RunnerGroupRepositoryAccessSelected}, + }, + repositories: map[int64][]gh.RunnerGroupRepository{2: {{FullName: "example/private", Private: true}}}, + }) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123\nexample\nkey.pem\n2\n2\n1\n\nn\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if cfg.Runner.Group != "restricted" || cfg.Security.RunnerGroup.RequiredRepositoryAccess != config.RunnerGroupRepositoryAccessSelected { + t.Fatalf("unexpected selection after back: group=%q policy=%+v", cfg.Runner.Group, cfg.Security.RunnerGroup) + } + if !strings.Contains(out.String(), "*** SECURITY WARNING: THIS RUNNER GROUP IS NOT RECOMMENDED ***") || !strings.Contains(out.String(), "Continue anyway and generate a relaxed policy") { + t.Fatalf("non-default broad group no longer showed its security warning:\n%s", out.String()) + } +} + +func TestInitRejectsPublicGroupAndAllowsAnotherSelection(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + client := &fakeInitRunnerGroupClient{ + groups: []gh.RunnerGroup{ + {ID: 1, Name: "public", Visibility: config.RunnerGroupRepositoryAccessSelected, AllowsPublicRepositories: true}, + {ID: 2, Name: "restricted", Visibility: config.RunnerGroupRepositoryAccessSelected}, + }, + repositories: map[int64][]gh.RunnerGroupRepository{ + 1: {{FullName: "example/public", Private: false}}, + 2: {{FullName: "example/private", Private: true}}, + }, + } + setInitRunnerGroupClient(t, client) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123\nexample\nkey.pem\nb\n2\n1\n1\n\nn\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if cfg.Runner.Group != "restricted" { + t.Fatalf("runner.group = %q, want restricted", cfg.Runner.Group) + } + if !strings.Contains(out.String(), "*** SECURITY BLOCK: THIS RUNNER GROUP IS NOT ALLOWED BY EPAR'S SAFE DEFAULTS ***") || !strings.Contains(out.String(), "public repository or fork-triggered workflow") || !strings.Contains(out.String(), "RECOMMENDED ACTION: Do not use this group") || !strings.Contains(out.String(), "docs/runner-groups.md") { + t.Fatalf("public-repository warning missing:\n%s", out.String()) + } + if client.groupCalls != 1 { + t.Fatalf("ListRunnerGroups calls = %d, want 1 when choosing Back", client.groupCalls) + } +} + +func TestInitBlocksUnknownRunnerGroupRepositoryAccess(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + setInitRunnerGroupClient(t, &fakeInitRunnerGroupClient{groups: []gh.RunnerGroup{{ID: 1, Name: "future-policy", Visibility: "future-value"}}}) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123\nexample\nkey.pem\nb\n1\n3\n"), + Out: &out, + }) + if err == nil || !strings.Contains(err.Error(), "selection cancelled") { + t.Fatalf("init error = %v, want cancellation after unknown policy block", err) + } + if !strings.Contains(out.String(), "*** SECURITY BLOCK: GITHUB RETURNED AN UNKNOWN REPOSITORY-ACCESS POLICY ***") { + t.Fatalf("unknown access policy was not blocked clearly:\n%s", out.String()) + } + if _, statErr := os.Stat(path); !errors.Is(statErr, os.ErrNotExist) { + t.Fatalf("config exists after unknown runner-group policy: %v", statErr) + } +} + +func TestSortRunnerGroupsForWizardPutsDefaultFirst(t *testing.T) { + groups := []gh.RunnerGroup{ + {ID: 1, Name: "Default", Visibility: config.RunnerGroupRepositoryAccessAll, Default: true}, + {ID: 2, Name: "public-selected", Visibility: config.RunnerGroupRepositoryAccessSelected, AllowsPublicRepositories: true}, + {ID: 3, Name: "all-private-repositories", Visibility: config.RunnerGroupRepositoryAccessPrivate}, + {ID: 4, Name: "recommended", Visibility: config.RunnerGroupRepositoryAccessSelected}, + {ID: 5, Name: "inherited-recommended", Visibility: config.RunnerGroupRepositoryAccessSelected, Inherited: true}, + } + ordered := sortRunnerGroupsForWizard(groups) + got := make([]string, len(ordered)) + for i, group := range ordered { + got[i] = group.Name + } + want := []string{"Default", "recommended", "inherited-recommended", "all-private-repositories", "public-selected"} + if !slices.Equal(got, want) { + t.Fatalf("runner-group order = %#v, want %#v", got, want) + } + if groups[0].Name != "Default" { + t.Fatalf("sort mutated API response order: %#v", groups) + } +} + +func TestInitRefreshesRunnerGroupsOnlyWhenRequested(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + client := &fakeInitRunnerGroupClient{ + groupResponses: [][]gh.RunnerGroup{ + {{ID: 1, Name: "before-refresh", Visibility: config.RunnerGroupRepositoryAccessSelected}}, + {{ID: 2, Name: "after-refresh", Visibility: config.RunnerGroupRepositoryAccessSelected}}, + }, + repositories: map[int64][]gh.RunnerGroupRepository{ + 1: {{FullName: "example/private", Private: true}}, + 2: {{FullName: "example/private", Private: true}}, + }, + } + setInitRunnerGroupClient(t, client) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123\nexample\nkey.pem\nr\n1\n\n"), + Out: &bytes.Buffer{}, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if cfg.Runner.Group != "after-refresh" || client.groupCalls != 2 { + t.Fatalf("refresh result: group=%q ListRunnerGroups calls=%d", cfg.Runner.Group, client.groupCalls) + } +} + +func TestInitAllowsInheritedGroupWithAdvisory(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + setInitRunnerGroupClient(t, &fakeInitRunnerGroupClient{ + groups: []gh.RunnerGroup{{ID: 1, Name: "enterprise-restricted", Visibility: config.RunnerGroupRepositoryAccessSelected, Inherited: true}}, + repositories: map[int64][]gh.RunnerGroupRepository{1: {{FullName: "example/private", Private: true}}}, + }) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123\nexample\nkey.pem\n1\n1\n\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if cfg.Runner.Group != "enterprise-restricted" || !cfg.Security.RunnerGroup.RequireNonDefaultGroup || !strings.Contains(out.String(), "enterprise level") { + t.Fatalf("inherited selection missing advisory or strict policy: group=%q policy=%+v output=%s", cfg.Runner.Group, cfg.Security.RunnerGroup, out.String()) + } +} + +func TestInitRunnerGroupAPIFailureDoesNotWriteConfig(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + setInitRunnerGroupClient(t, &fakeInitRunnerGroupClient{err: errors.New("permission denied")}) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123\nexample\nkey.pem\n"), + Out: &bytes.Buffer{}, + }) + if err == nil || !strings.Contains(err.Error(), "load GitHub runner groups") || !strings.Contains(err.Error(), "permission denied") { + t.Fatalf("init error = %v, want runner-group API failure", err) + } + if _, statErr := os.Stat(path); !errors.Is(statErr, os.ErrNotExist) { + t.Fatalf("config exists after runner-group API failure: %v", statErr) + } +} + +func TestInitRunnerGroupAPIFailureDoesNotOverwriteExistingConfig(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + setInitRunnerGroupClient(t, &fakeInitRunnerGroupClient{err: errors.New("permission denied")}) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + if err := os.MkdirAll(filepath.Dir(path), 0755); err != nil { + t.Fatal(err) + } + const original = "existing config must remain\n" + if err := os.WriteFile(path, []byte(original), 0600); err != nil { + t.Fatal(err) + } + err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + Force: true, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123\nexample\nkey.pem\n"), + Out: &bytes.Buffer{}, + }) + if err == nil || !strings.Contains(err.Error(), "load GitHub runner groups") { + t.Fatalf("init error = %v, want runner-group API failure", err) + } + contents, readErr := os.ReadFile(path) + if readErr != nil { + t.Fatal(readErr) + } + if string(contents) != original { + t.Fatalf("existing config changed after API failure: %q", contents) + } } func TestDetectedInitHostTrustOSUsesWrapperHost(t *testing.T) { @@ -117,7 +558,7 @@ func TestInitCanDisableHostTrustOverlay(t *testing.T) { ConfigPath: path, SkipDockerCheck: true, SkipHostTrustCheck: true, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n\nn\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n\n\nn\n\n\nn\n"), Out: &bytes.Buffer{}, }); err != nil { t.Fatal(err) @@ -145,7 +586,7 @@ func TestInitDoesNotWriteEnabledConfigWhenHostTrustPreflightFails(t *testing.T) ProjectRoot: dir, ConfigPath: path, SkipDockerCheck: true, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n\n\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n\n\n"), Out: &bytes.Buffer{}, }) if err == nil || !strings.Contains(err.Error(), "collector unavailable") { @@ -167,7 +608,7 @@ func TestInitAcceptsCustomPoolNamePrefix(t *testing.T) { ConfigPath: path, SkipDockerCheck: true, SkipHostTrustCheck: true, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\ncustom-prefix\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n\n\nn\n\ncustom-prefix\n\n"), Out: &bytes.Buffer{}, }); err != nil { t.Fatal(err) @@ -193,7 +634,7 @@ func TestInitRepromptsInvalidPoolNamePrefix(t *testing.T) { ConfigPath: path, SkipDockerCheck: true, SkipHostTrustCheck: true, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n-bad\nfixed-prefix\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n\n\nn\n\n-bad\nfixed-prefix\n\n"), Out: &out, }); err != nil { t.Fatal(err) @@ -278,6 +719,20 @@ func TestGeneratedPoolNamePrefixFallsBackWhenHostnameSanitizesEmpty(t *testing.T } func TestInitRefusesExistingConfig(t *testing.T) { + oldPlatform := initSandboxPromotionPlatform + oldLookup := initSandboxPromotionLookup + initSandboxPromotionPlatform = func() sandboxpromotion.Platform { + t.Fatal("existing config must be rejected before promotion lookup") + return "" + } + initSandboxPromotionLookup = func(sandboxpromotion.Platform) (sandboxpromotion.Record, bool) { + t.Fatal("existing config must be rejected before promotion lookup") + return sandboxpromotion.Record{}, false + } + t.Cleanup(func() { + initSandboxPromotionPlatform = oldPlatform + initSandboxPromotionLookup = oldLookup + }) dir := t.TempDir() path := filepath.Join(dir, ".local", "config.yml") if err := os.MkdirAll(filepath.Dir(path), 0755); err != nil { @@ -301,6 +756,7 @@ func TestInitRefusesExistingConfig(t *testing.T) { } func TestInitChecksDockerByDefault(t *testing.T) { + stubInitRunnerGroupClient(t) oldDockerAvailable := dockerAvailable t.Cleanup(func() { dockerAvailable = oldDockerAvailable @@ -309,14 +765,18 @@ func TestInitChecksDockerByDefault(t *testing.T) { return errors.New("docker unavailable") } + var out bytes.Buffer err := runInitWithOptions(initOptions{ ProjectRoot: t.TempDir(), ConfigPath: filepath.Join(t.TempDir(), ".local", "config.yml"), - In: strings.NewReader("123\norg\nkey.pem\n"), - Out: &bytes.Buffer{}, + In: strings.NewReader("123\norg\nkey.pem\n1\n1"), + Out: &out, }) - if err == nil || !strings.Contains(err.Error(), "Docker is required") { - t.Fatalf("error = %v, want Docker requirement", err) + if err == nil || !strings.Contains(err.Error(), `runner provider "1" is unavailable`) { + t.Fatalf("error = %v, want unavailable Docker Container selection", err) + } + if !strings.Contains(out.String(), "Docker CLI or daemon check failed: docker unavailable") { + t.Fatalf("output did not report Docker prerequisite status:\n%s", out.String()) } } @@ -332,7 +792,7 @@ func TestInitOffersWSL2ConfigWhenAvailable(t *testing.T) { ConfigPath: path, SkipDockerCheck: true, SkipHostTrustCheck: true, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n2\n\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n3\n\n"), Out: &out, }); err != nil { t.Fatal(err) @@ -351,12 +811,13 @@ func TestInitOffersWSL2ConfigWhenAvailable(t *testing.T) { "organization: your-org", "organization: solutionforest", "privateKeyPath: ~/.config/ephemeral-action-runner/github-app.pem", "privateKeyPath: .local/github-app.pem", "namePrefix: CHANGE-ME-unique-machine-prefix", "namePrefix: build-box-01-a4f9c2", + "group: your-runner-group", "group: \"restricted group\"", ).Replace(string(want)) wantText = strings.ReplaceAll(wantText, "\r\n", "\n") if string(got) != wantText { t.Fatalf("WSL config did not match configs/wsl.example.yml:\nwant:\n%s\ngot:\n%s", wantText, got) } - if !strings.Contains(out.String(), "2. WSL2") { + if !strings.Contains(out.String(), "2. Docker Sandboxes — recommended when ready") || !strings.Contains(out.String(), "3. WSL2") { t.Fatalf("init output did not offer WSL2:\n%s", out.String()) } providerPrompt := strings.Index(out.String(), "Runner provider:") @@ -366,7 +827,7 @@ func TestInitOffersWSL2ConfigWhenAvailable(t *testing.T) { } } -func TestInitWSL2ChoiceDefaultsToDockerDindAndRepromptsInvalidValues(t *testing.T) { +func TestInitWSL2ChoiceDefaultsToDockerContainerAndRepromptsInvalidValues(t *testing.T) { stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) stubWSL2Available(t) @@ -378,7 +839,7 @@ func TestInitWSL2ChoiceDefaultsToDockerDindAndRepromptsInvalidValues(t *testing. ConfigPath: path, SkipDockerCheck: true, SkipHostTrustCheck: true, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\ninvalid\n\n\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\ninvalid\n\n\n"), Out: &out, }); err != nil { t.Fatal(err) @@ -387,23 +848,28 @@ func TestInitWSL2ChoiceDefaultsToDockerDindAndRepromptsInvalidValues(t *testing. if err != nil { t.Fatal(err) } - if !strings.Contains(string(configBytes), "type: docker-dind") { - t.Fatalf("config did not use the default Docker-DinD provider:\n%s", configBytes) + if !strings.Contains(string(configBytes), "type: docker-container") { + t.Fatalf("config did not use the default Docker Container provider:\n%s", configBytes) } - if !strings.Contains(out.String(), "Runner provider must be 1 (Docker-DinD) or 2 (WSL2).") { + if !strings.Contains(out.String(), "Choose an available provider number or name shown above, or R to refresh.") { t.Fatalf("init output did not explain invalid provider input:\n%s", out.String()) } + if strings.Contains(out.String(), "Start runners now?") { + t.Fatalf("standalone init unexpectedly asked whether to start runners:\n%s", out.String()) + } } -func TestInitOffersTartConfigWhenAvailable(t *testing.T) { +func TestInitDockerSandboxesGeneratesDesiredImageConfigAndProvisionsTemplate(t *testing.T) { stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) - stubTartAvailable(t) - oldDockerAvailable := dockerAvailable - dockerAvailable = func(context.Context) error { - t.Fatal("Docker availability should not be checked for Tart") - return nil - } - t.Cleanup(func() { dockerAvailable = oldDockerAvailable }) + stubNoWSL2(t) + policyFingerprint := "sha256:" + strings.Repeat("b", 64) + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + Templates: []initDockerSandboxesTemplate{ + {Reference: "docker.io/library/epar-docker-sandboxes-catthehacker-full:20260723-r2-amd64", Digest: "sha256:" + strings.Repeat("a", 64), CacheID: strings.Repeat("a", 12), Platform: "linux/amd64", Size: 8 << 30, Label: "Catthehacker Ubuntu Full (recommended)", SourceChannel: "ghcr.io/catthehacker/ubuntu:full-latest"}, + {Reference: "docker.io/library/epar-docker-sandboxes-catthehacker-act-22.04:20260723-r4-amd64", Digest: "sha256:" + strings.Repeat("c", 64), CacheID: strings.Repeat("c", 12), Platform: "linux/amd64", Size: 4 << 30, Label: "Catthehacker Ubuntu Act 22.04 (current lean profile)", SourceChannel: "ghcr.io/catthehacker/ubuntu:act-22.04"}, + }, + PolicyFingerprint: policyFingerprint, + }, nil) dir := t.TempDir() path := filepath.Join(dir, ".local", "config.yml") @@ -411,32 +877,1173 @@ func TestInitOffersTartConfigWhenAvailable(t *testing.T) { if err := runInitWithOptions(initOptions{ ProjectRoot: dir, ConfigPath: path, + SkipDockerCheck: true, SkipHostTrustCheck: true, - In: strings.NewReader("654321\nexample\n.local/github-app.pem\n2\n\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n1\n2\nn\nnot-a-size\n30GiB\n\n\nn\n"), Out: &out, }); err != nil { t.Fatal(err) } - got, err := os.ReadFile(path) + cfg, err := config.Load(path) if err != nil { t.Fatal(err) } - want, err := os.ReadFile(filepath.Join("..", "..", "configs", "tart.example.yml")) + if got, want := cfg.Provider.Type, "docker-sandboxes"; got != want { + t.Fatalf("provider.type = %q, want %q", got, want) + } + if got, want := cfg.Provider.Platform, "linux/amd64"; got != want { + t.Fatalf("provider.platform = %q, want %q", got, want) + } + if got, want := cfg.Image.SourceImage, "ghcr.io/catthehacker/ubuntu:act-latest"; got != want { + t.Fatalf("image.sourceImage = %q, want %q", got, want) + } + configContent, err := os.ReadFile(path) if err != nil { t.Fatal(err) } - wantText := strings.NewReplacer( - "appId: 123456", "appId: 654321", + if strings.Contains(string(configContent), "templateDigest:") || strings.Contains(string(configContent), "\n template:") { + t.Fatalf("generated config retained local template identities:\n%s", configContent) + } + if got, want := cfg.DockerSandboxes.PolicyGeneration, policyFingerprint; got != want { + t.Fatalf("dockerSandboxes.policyGeneration = %q, want %q", got, want) + } + if got, want := cfg.DockerSandboxes.NetworkBaseline, config.DockerSandboxesNetworkBaselineOpen; got != want { + t.Fatalf("dockerSandboxes.networkBaseline = %q, want %q", got, want) + } + for key, values := range map[string]struct{ got, want string }{ + "rootDisk": {cfg.DockerSandboxes.RootDisk, "auto"}, + "dockerDisk": {cfg.DockerSandboxes.DockerDisk, "50GiB"}, + } { + if values.got != values.want { + t.Fatalf("dockerSandboxes.%s = %q, want %q", key, values.got, values.want) + } + } + for _, want := range []string{"Docker Sandboxes image setup:", "Runner base image:", "1. full — full-latest (default)", "2. act — act-latest", "Image catalog:", "Runner artifact estimate:", "Source: ghcr.io/catthehacker/ubuntu:act-latest", "Automatic sandbox root limit:"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("init output omitted %q:\n%s", want, out.String()) + } + } + for _, removed := range []string{"informational; configuration creation is not blocked", "Estimate confidence:", "Expected duration:"} { + if strings.Contains(out.String(), removed) { + t.Fatalf("init output retained removed estimate text %q:\n%s", removed, out.String()) + } + } +} + +func TestInitDockerSandboxesWritesConfigBeforeOrdinaryProvisioning(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + Templates: []initDockerSandboxesTemplate{{Platform: "linux/amd64", Size: 4 << 30}}, + PolicyFingerprint: "sha256:" + strings.Repeat("b", 64), + }, nil) + initEnsureDockerSandboxesTemplate = func(context.Context, string, string) error { + return errors.New("simulated import readback failure") + } + + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n1\n\nn\n\n\n\nn\n"), + Out: io.Discard, + }) + if err != nil { + t.Fatalf("runInitWithOptions() error = %v", err) + } + if _, statErr := os.Stat(path); statErr != nil { + t.Fatalf("configuration was not created before ordinary provisioning: %v", statErr) + } +} + +func TestDockerSandboxesWizardResolvesEveryBuiltInImageChoice(t *testing.T) { + for _, test := range []struct { + choice string + want string + }{ + {choice: "", want: "ghcr.io/catthehacker/ubuntu:full-latest"}, + {choice: "1", want: "ghcr.io/catthehacker/ubuntu:full-latest"}, + {choice: "2", want: "ghcr.io/catthehacker/ubuntu:act-latest"}, + {choice: "3", want: "ghcr.io/catthehacker/ubuntu:dotnet-latest"}, + {choice: "4", want: "ghcr.io/catthehacker/ubuntu:js-latest"}, + } { + t.Run(test.want, func(t *testing.T) { + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + PolicyFingerprint: "sha256:" + strings.Repeat("b", 64), + }, nil) + input := test.choice + "\nn\n\n\n" + profile, accepted, err := promptDockerSandboxesProfile(context.Background(), t.TempDir(), sandboxpromotion.WindowsAMD64, io.Discard, bufio.NewReader(strings.NewReader(input))) + if err != nil { + t.Fatal(err) + } + if !accepted || profile == nil || profile.SourceImage != test.want { + t.Fatalf("wizard profile = %+v, accepted=%t, want source %q", profile, accepted, test.want) + } + }) + } +} + +func TestSharedDockerImageWizardCoversDockerContainerSandboxesAndWSL(t *testing.T) { + for _, providerType := range []string{"docker-container", "docker-sandboxes", "wsl"} { + t.Run(providerType, func(t *testing.T) { + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + PolicyFingerprint: "sha256:" + strings.Repeat("b", 64), + }, nil) + var out bytes.Buffer + profile, accepted, err := promptDockerImageProfile(context.Background(), t.TempDir(), providerType, sandboxpromotion.WindowsAMD64, &out, bufio.NewReader(strings.NewReader("\nn\n\n"))) + if err != nil { + t.Fatal(err) + } + if !accepted || profile == nil || profile.Provider != providerType || profile.SourceImage != "ghcr.io/catthehacker/ubuntu:full-latest" { + t.Fatalf("shared wizard profile = %+v, accepted=%t", profile, accepted) + } + for _, want := range []string{"1. full — full-latest (default)", "2. act — act-latest", "3. dotnet — dotnet-latest", "4. js — js-latest", "Runner artifact estimate:"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("%s wizard output omitted %q:\n%s", providerType, want, out.String()) + } + } + }) + } +} + +func TestImageUpdatePolicyWizardDefaultsAndOrdering(t *testing.T) { + var out bytes.Buffer + policy, err := promptImageUpdatePolicy(&out, bufio.NewReader(strings.NewReader("\n\n"))) + if err != nil { + t.Fatal(err) + } + if policy.Frequency != config.ImageUpdateFrequencyWeekly || policy.Time != config.DefaultImageUpdateTime { + t.Fatalf("default policy = %+v", policy) + } + text := out.String() + choices := []string{"1. Weekly (default)", "2. Daily", "3. Every two weeks", "4. Monthly", "5. Manual"} + last := -1 + for _, choice := range choices { + index := strings.Index(text, choice) + if index <= last { + t.Fatalf("wizard choices are missing or out of order: %q\n%s", choice, text) + } + last = index + } + if strings.Contains(strings.ToLower(text), "monthly") && strings.Contains(strings.ToLower(text), "warning") { + t.Fatalf("monthly choice should not carry a warning:\n%s", text) + } +} + +func TestImageUpdatePolicyWizardManualSkipsTimePrompt(t *testing.T) { + var out bytes.Buffer + policy, err := promptImageUpdatePolicy(&out, bufio.NewReader(strings.NewReader("5\n"))) + if err != nil { + t.Fatal(err) + } + if policy.Frequency != config.ImageUpdateFrequencyManual { + t.Fatalf("manual policy = %+v", policy) + } + if strings.Contains(out.String(), "Local update time") { + t.Fatalf("manual policy unexpectedly prompted for a time:\n%s", out.String()) + } + if !strings.Contains(out.String(), "5. Manual — check only on demand\n Command: ./start image update") { + t.Fatalf("manual policy omitted its neutral trigger explanation:\n%s", out.String()) + } +} + +func TestImageUpdatePolicyWizardRepromptsInvalidChoiceAndTime(t *testing.T) { + var out bytes.Buffer + policy, err := promptImageUpdatePolicy(&out, bufio.NewReader(strings.NewReader("invalid\n2\n7am\n06:30\n"))) + if err != nil { + t.Fatal(err) + } + if policy.Frequency != config.ImageUpdateFrequencyDaily || policy.Time != "06:30" { + t.Fatalf("reprompted policy = %+v", policy) + } + if !strings.Contains(out.String(), "Choose 1") || !strings.Contains(out.String(), "24-hour HH:MM") { + t.Fatalf("wizard did not explain invalid input:\n%s", out.String()) + } +} + +func TestDockerSandboxesWizardCollectsCustomTagAndInstallScript(t *testing.T) { + stubInitDockerSandboxesSetup(t, sandboxpromotion.DarwinARM64, initDockerSandboxesDiscovery{ + PolicyFingerprint: "sha256:" + strings.Repeat("b", 64), + }, nil) + projectRoot := t.TempDir() + scriptPath := filepath.Join(projectRoot, "scripts", "install-extra.sh") + if err := os.MkdirAll(filepath.Dir(scriptPath), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(scriptPath, []byte("#!/usr/bin/env bash\nset -euo pipefail\n"), 0o700); err != nil { + t.Fatal(err) + } + var out bytes.Buffer + profile, accepted, err := promptDockerSandboxesProfile(context.Background(), projectRoot, sandboxpromotion.DarwinARM64, &out, bufio.NewReader(strings.NewReader("5\ngo-24.04\ny\nscripts/install-extra.sh\nn\n\n\n"))) + if err != nil { + t.Fatal(err) + } + if !accepted || profile == nil || profile.SourceImage != "ghcr.io/catthehacker/ubuntu:go-24.04" || profile.HostPlatform != sandboxpromotion.DarwinARM64 { + t.Fatalf("wizard profile = %+v, accepted=%t", profile, accepted) + } + if len(profile.CustomScripts) != 1 || profile.CustomScripts[0] != "scripts/install-extra.sh" { + t.Fatalf("custom scripts = %#v", profile.CustomScripts) + } + for _, want := range []string{"Scripts run as root", "Do not put secrets", "Platform: linux/arm64", "Custom install scripts: scripts/install-extra.sh"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("wizard output omitted %q:\n%s", want, out.String()) + } + } +} + +func TestDockerSandboxesWizardAllowsEmptyCustomInstallScriptChoice(t *testing.T) { + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + PolicyFingerprint: "sha256:" + strings.Repeat("b", 64), + }, nil) + var out bytes.Buffer + profile, accepted, err := promptDockerSandboxesProfile(context.Background(), t.TempDir(), sandboxpromotion.WindowsAMD64, &out, bufio.NewReader(strings.NewReader("1\ny\n\n\n"))) + if err != nil { + t.Fatal(err) + } + if !accepted || profile == nil { + t.Fatalf("wizard profile = %+v, accepted=%t", profile, accepted) + } + if len(profile.CustomScripts) != 0 { + t.Fatalf("custom scripts = %#v, want none", profile.CustomScripts) + } + for _, want := range []string{"Custom install script path (press Enter for none):", "Custom install scripts: none"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("wizard output omitted %q:\n%s", want, out.String()) + } + } + if strings.Contains(out.String(), "Custom install script path is required") { + t.Fatalf("empty custom install script path was treated as required:\n%s", out.String()) + } +} + +func TestDockerSandboxesWizardRepromptsAfterUnresolvableTag(t *testing.T) { + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + PolicyFingerprint: "sha256:" + strings.Repeat("b", 64), + }, nil) + resolver := initResolveDockerSandboxesSource + initResolveDockerSandboxesSource = func(ctx context.Context, input, platform string) (imageartifact.ResolvedDockerSource, error) { + if input == "missing" { + return imageartifact.ResolvedDockerSource{}, errors.New("tag does not exist") + } + return resolver(ctx, input, platform) + } + var out bytes.Buffer + profile, accepted, err := promptDockerSandboxesProfile(context.Background(), t.TempDir(), sandboxpromotion.WindowsAMD64, &out, bufio.NewReader(strings.NewReader("5\nmissing\n4\nn\n\n\n"))) + if err != nil { + t.Fatal(err) + } + if !accepted || profile == nil || profile.SourceImage != "ghcr.io/catthehacker/ubuntu:js-latest" { + t.Fatalf("wizard profile = %+v, accepted=%t", profile, accepted) + } + if !strings.Contains(out.String(), "That image cannot be used for linux/amd64: tag does not exist") { + t.Fatalf("wizard did not explain the failed tag resolution:\n%s", out.String()) + } +} + +func TestDockerSandboxesWizardConfirmationRefusalReturnsNoProfile(t *testing.T) { + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + PolicyFingerprint: "sha256:" + strings.Repeat("b", 64), + }, nil) + profile, accepted, err := promptDockerSandboxesProfile(context.Background(), t.TempDir(), sandboxpromotion.WindowsAMD64, io.Discard, bufio.NewReader(strings.NewReader("1\nn\nn\n"))) + if err != nil { + t.Fatal(err) + } + if profile != nil || accepted { + t.Fatalf("confirmation refusal returned profile=%+v accepted=%t", profile, accepted) + } +} + +func TestPrepareDockerSandboxesReadinessStartsInstalledDaemonBeforeDiagnostics(t *testing.T) { + oldLookPath := initDockerSandboxesLookPath + oldStartDaemon := initDockerSandboxesStartDaemon + oldDiagnose := initDockerSandboxesDiagnose + t.Cleanup(func() { + initDockerSandboxesLookPath = oldLookPath + initDockerSandboxesStartDaemon = oldStartDaemon + initDockerSandboxesDiagnose = oldDiagnose + }) + + const binary = `C:\Program Files\Docker\Docker\resources\bin\sbx.exe` + var operations []string + initDockerSandboxesLookPath = func(name string) (string, error) { + if name != "sbx" { + t.Fatalf("looked up %q, want sbx", name) + } + return binary, nil + } + initDockerSandboxesStartDaemon = func(_ context.Context, actual string) error { + if actual != binary { + t.Fatalf("daemon binary = %q, want %q", actual, binary) + } + operations = append(operations, "start") + return nil + } + initDockerSandboxesDiagnose = func(_ context.Context, actual string) (dockersandboxes.HostReadiness, error) { + if actual != binary { + t.Fatalf("diagnostic binary = %q, want %q", actual, binary) + } + operations = append(operations, "diagnose") + return dockersandboxes.HostReadiness{ChecksPassed: 8}, nil + } + + readiness, err := prepareInitDockerSandboxesReadiness(context.Background()) + if err != nil { + t.Fatal(err) + } + if readiness.ChecksPassed != 8 || !reflect.DeepEqual(operations, []string{"start", "diagnose"}) { + t.Fatalf("readiness = %+v, operations = %v", readiness, operations) + } +} + +func TestPrepareDockerSandboxesReadinessSkipsDaemonStartWhenSBXIsMissing(t *testing.T) { + oldLookPath := initDockerSandboxesLookPath + oldStartDaemon := initDockerSandboxesStartDaemon + oldDiagnose := initDockerSandboxesDiagnose + t.Cleanup(func() { + initDockerSandboxesLookPath = oldLookPath + initDockerSandboxesStartDaemon = oldStartDaemon + initDockerSandboxesDiagnose = oldDiagnose + }) + + initDockerSandboxesLookPath = func(string) (string, error) { + return "", exec.ErrNotFound + } + initDockerSandboxesStartDaemon = func(context.Context, string) error { + t.Fatal("daemon start was attempted without an installed sbx executable") + return nil + } + initDockerSandboxesDiagnose = func(_ context.Context, binary string) (dockersandboxes.HostReadiness, error) { + if binary != "sbx" { + t.Fatalf("diagnostic binary = %q, want sbx", binary) + } + return dockersandboxes.HostReadiness{}, exec.ErrNotFound + } + + if _, err := prepareInitDockerSandboxesReadiness(context.Background()); !errors.Is(err, exec.ErrNotFound) { + t.Fatalf("readiness error = %v, want executable-not-found error", err) + } +} + +func TestPrepareDockerSandboxesReadinessUsesSuccessfulDiagnosticsAfterStartWarning(t *testing.T) { + oldLookPath := initDockerSandboxesLookPath + oldStartDaemon := initDockerSandboxesStartDaemon + oldDiagnose := initDockerSandboxesDiagnose + t.Cleanup(func() { + initDockerSandboxesLookPath = oldLookPath + initDockerSandboxesStartDaemon = oldStartDaemon + initDockerSandboxesDiagnose = oldDiagnose + }) + + initDockerSandboxesLookPath = func(string) (string, error) { return "sbx-test", nil } + initDockerSandboxesStartDaemon = func(context.Context, string) error { return errors.New("daemon already running") } + initDockerSandboxesDiagnose = func(context.Context, string) (dockersandboxes.HostReadiness, error) { + return dockersandboxes.HostReadiness{ChecksPassed: 8, ChecksWarned: 1}, nil + } + + readiness, err := prepareInitDockerSandboxesReadiness(context.Background()) + if err != nil { + t.Fatalf("healthy diagnostics were rejected after a daemon-start warning: %v", err) + } + if readiness.ChecksPassed != 8 || readiness.ChecksWarned != 1 { + t.Fatalf("readiness = %+v", readiness) + } +} + +func TestPrepareDockerSandboxesReadinessReportsStartAndDiagnosticFailures(t *testing.T) { + oldLookPath := initDockerSandboxesLookPath + oldStartDaemon := initDockerSandboxesStartDaemon + oldDiagnose := initDockerSandboxesDiagnose + t.Cleanup(func() { + initDockerSandboxesLookPath = oldLookPath + initDockerSandboxesStartDaemon = oldStartDaemon + initDockerSandboxesDiagnose = oldDiagnose + }) + + initDockerSandboxesLookPath = func(string) (string, error) { return "sbx-test", nil } + initDockerSandboxesStartDaemon = func(context.Context, string) error { return errors.New("daemon startup failed") } + initDockerSandboxesDiagnose = func(context.Context, string) (dockersandboxes.HostReadiness, error) { + return dockersandboxes.HostReadiness{}, errors.New("daemon diagnostic failed") + } + + _, err := prepareInitDockerSandboxesReadiness(context.Background()) + if err == nil || !strings.Contains(err.Error(), "sbx daemon start --detach") || !strings.Contains(err.Error(), "daemon startup failed") || !strings.Contains(err.Error(), "daemon diagnostic failed") { + t.Fatalf("combined readiness error = %v", err) + } +} + +func TestDockerSandboxesPrerequisitesIgnoreStorageCapacity(t *testing.T) { + stubNoWSL2(t) + oldReadiness := initDockerSandboxesReadiness + oldCapacityCheck := initDockerSandboxesCapacityCheck + initDockerSandboxesReadiness = func(context.Context) (dockersandboxes.HostReadiness, error) { + return dockersandboxes.HostReadiness{ChecksPassed: 8, ChecksWarned: 1}, nil + } + initDockerSandboxesCapacityCheck = func(rootDisk, dockerDisk, minHostFreeSpace uint64) (initDockerSandboxesCapacityResult, error) { + t.Fatal("provider prerequisite detection must not perform storage admission") + return initDockerSandboxesCapacityResult{}, nil + } + t.Cleanup(func() { + initDockerSandboxesReadiness = oldReadiness + initDockerSandboxesCapacityCheck = oldCapacityCheck + }) + + got := detectInitProviderPrerequisites(context.Background(), sandboxpromotion.WindowsAMD64, true) + if !got.DockerSandboxesAvailable { + t.Fatalf("Docker Sandboxes was unavailable despite passing tooling diagnostics: %s", got.DockerSandboxesStatus) + } +} + +func TestDockerSandboxesPrerequisitesExplainHowToInspectFailedDiagnosticHints(t *testing.T) { + stubNoWSL2(t) + oldReadiness := initDockerSandboxesReadiness + oldCapacityCheck := initDockerSandboxesCapacityCheck + initDockerSandboxesReadiness = func(context.Context) (dockersandboxes.HostReadiness, error) { + return dockersandboxes.HostReadiness{}, errors.New("docker sandboxes diagnostics reported 1 failed check") + } + initDockerSandboxesCapacityCheck = func(uint64, uint64, uint64) (initDockerSandboxesCapacityResult, error) { + t.Fatal("capacity check must not run after failed diagnostics") + return initDockerSandboxesCapacityResult{}, nil + } + t.Cleanup(func() { + initDockerSandboxesReadiness = oldReadiness + initDockerSandboxesCapacityCheck = oldCapacityCheck + }) + + got := detectInitProviderPrerequisites(context.Background(), sandboxpromotion.WindowsAMD64, true) + if got.DockerSandboxesAvailable { + t.Fatal("Docker Sandboxes was available despite failed diagnostics") + } + for _, want := range []string{"1 failed check", "sbx diagnose --output json", "hints for each failed check"} { + if !strings.Contains(got.DockerSandboxesStatus, want) { + t.Fatalf("Docker Sandboxes status omitted %q: %s", want, got.DockerSandboxesStatus) + } + } +} + +func TestInitProviderRefreshRechecksAvailabilityAndRedrawsMenu(t *testing.T) { + record := validInitPromotionRecord() + stubInitSandboxPromotion(t, record, sandboxpromotion.PreflightResult{}) + oldReadiness := initDockerSandboxesReadiness + readinessCalls := 0 + initDockerSandboxesReadiness = func(context.Context) (dockersandboxes.HostReadiness, error) { + readinessCalls++ + if readinessCalls == 1 { + return dockersandboxes.HostReadiness{}, errors.New("docker sandboxes diagnostics reported 1 failed check") + } + return dockersandboxes.HostReadiness{ChecksPassed: 8, ChecksWarned: 1}, nil + } + t.Cleanup(func() { + initDockerSandboxesReadiness = oldReadiness + }) + + var out bytes.Buffer + providerType, selectedRecord, profile, err := promptInitProvider(context.Background(), t.TempDir(), &out, bufio.NewReader(strings.NewReader("2\nr\n1\n")), true) + if err != nil { + t.Fatal(err) + } + if providerType != "docker-sandboxes" || selectedRecord.Template != "" || profile == nil || profile.SourceImage != "ghcr.io/catthehacker/ubuntu:full-latest" { + t.Fatalf("refreshed selection = provider %q, record template %q, profile %+v", providerType, selectedRecord.Template, profile) + } + if readinessCalls != 2 { + t.Fatalf("Docker Sandboxes readiness calls = %d, want 2", readinessCalls) + } + for _, want := range []string{ + "R. Refresh provider prerequisites", + "Docker Sandboxes (independently certified for this exact platform) is unavailable", + "Refreshing provider prerequisites...", + "PASS: the exact promoted platform, host, sbx, template, policy, and resource gates passed.", + } { + if !strings.Contains(out.String(), want) { + t.Fatalf("provider refresh output omitted %q:\n%s", want, out.String()) + } + } + if got := strings.Count(out.String(), "R. Refresh provider prerequisites"); got != 2 { + t.Fatalf("provider menu render count = %d, want 2:\n%s", got, out.String()) + } +} + +func TestDockerSandboxesProfileShowsEstimateWithoutCapacityAdmission(t *testing.T) { + policyFingerprint := "sha256:" + strings.Repeat("b", 64) + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + Templates: []initDockerSandboxesTemplate{{ + Reference: "docker.io/library/epar-docker-sandboxes-catthehacker-act-22.04:test-amd64", + Digest: "sha256:" + strings.Repeat("c", 64), + CacheID: strings.Repeat("c", 12), + Platform: "linux/amd64", + Size: 4 << 30, + Label: "Catthehacker Ubuntu Act 22.04", + SourceChannel: "ghcr.io/catthehacker/ubuntu:act-22.04", + }}, + PolicyFingerprint: policyFingerprint, + }, nil) + oldCapacityCheck := initDockerSandboxesCapacityCheck + initDockerSandboxesCapacityCheck = func(rootDisk, dockerDisk, minHostFreeSpace uint64) (initDockerSandboxesCapacityResult, error) { + t.Fatal("image onboarding must not perform storage admission") + return initDockerSandboxesCapacityResult{}, nil + } + t.Cleanup(func() { initDockerSandboxesCapacityCheck = oldCapacityCheck }) + + var out bytes.Buffer + profile, accepted, err := promptDockerSandboxesProfile(context.Background(), t.TempDir(), sandboxpromotion.WindowsAMD64, &out, bufio.NewReader(strings.NewReader("1\nn\n\n"))) + if err != nil { + t.Fatalf("informational estimate error = %v", err) + } + if profile == nil || !accepted { + t.Fatalf("informational estimate returned profile=%+v accepted=%t", profile, accepted) + } + for _, want := range []string{"Runner artifact estimate:", "Available physical space:", "Fixed free-space reserve: 1GiB", "sparse logical maximum"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("informational estimate output omitted %q:\n%s", want, out.String()) + } + } +} + +func TestInitCapabilityReadyDockerSandboxesIsDefaultWithoutPreviewAcknowledgement(t *testing.T) { + for _, test := range []struct { + name string + hostPlatform sandboxpromotion.Platform + guestPlatform string + }{ + {name: "windows amd64", hostPlatform: sandboxpromotion.WindowsAMD64, guestPlatform: "linux/amd64"}, + {name: "linux amd64", hostPlatform: sandboxpromotion.LinuxAMD64, guestPlatform: "linux/amd64"}, + {name: "darwin arm64", hostPlatform: sandboxpromotion.DarwinARM64, guestPlatform: "linux/arm64"}, + {name: "future os amd64", hostPlatform: sandboxpromotion.Platform("futureos/amd64"), guestPlatform: "linux/amd64"}, + } { + t.Run(test.name, func(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + policyFingerprint := "sha256:" + strings.Repeat("b", 64) + stubInitDockerSandboxesSetup(t, test.hostPlatform, initDockerSandboxesDiscovery{ + Templates: []initDockerSandboxesTemplate{{ + Reference: "docker.io/library/epar-docker-sandboxes-catthehacker-full:capability-default", + Digest: "sha256:" + strings.Repeat("a", 64), + CacheID: strings.Repeat("a", 12), + Platform: test.guestPlatform, + Size: 18_730_706_190, + Label: "Catthehacker Ubuntu Full (recommended)", + SourceChannel: "ghcr.io/catthehacker/ubuntu:full-latest", + }}, + PolicyFingerprint: policyFingerprint, + }, nil) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n\n\n\n\n\n\nn\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if got, want := cfg.Provider.Type, "docker-sandboxes"; got != want { + t.Fatalf("provider.type = %q, want %q", got, want) + } + if got, want := cfg.Provider.Platform, test.guestPlatform; got != want { + t.Fatalf("provider.platform = %q, want %q", got, want) + } + for _, want := range []string{ + "1. Docker Sandboxes — recommended (default)", + "2. Docker Container — private daemon", + "Docker Sandboxes — recommended (default)", + "Runner provider (press Enter to use 1):", + "Docker Sandboxes image setup:", + "Runner base image:", + "full — full-latest (default)", + "Runner artifact estimate:", + } { + if !strings.Contains(out.String(), want) { + t.Fatalf("capability-default output omitted %q:\n%s", want, out.String()) + } + } + for _, forbidden := range []string{"Continue with explicit preview setup?", "Docker Sandboxes preview:"} { + if strings.Contains(out.String(), forbidden) { + t.Fatalf("capability-default output retained preview interaction %q:\n%s", forbidden, out.String()) + } + } + }) + } +} + +func TestReadDockerSandboxesActiveProfilesUsesOnlyCurrentLockedTemplates(t *testing.T) { + projectRoot := t.TempDir() + lockDirectory := filepath.Join(projectRoot, "templates", "docker-sandboxes") + if err := os.MkdirAll(lockDirectory, 0755); err != nil { + t.Fatal(err) + } + const lock = `{ + "schemaVersion": 2, + "profiles": { + "full": { + "observedTagReference": "ghcr.io/catthehacker/ubuntu:full-latest", + "platforms": {"linux/amd64": {"templateTag": "epar-docker-sandboxes-catthehacker-full:20260723-r2-amd64"}} + }, + "act-22.04": { + "observedTagReference": "ghcr.io/catthehacker/ubuntu:act-22.04", + "platforms": {"linux/amd64": {"templateTag": "epar-docker-sandboxes-catthehacker-act-22.04:20260723-r4-amd64"}} + } + }, + "supersededRecords": { + "linux/amd64": { + "full": {"templateTag": "epar-docker-sandboxes-catthehacker-full:20260723-r1-amd64"}, + "act-22.04": {"templateTag": "epar-docker-sandboxes-catthehacker-act-22.04:20260723-r3-amd64"} + } + } +}` + if err := os.WriteFile(filepath.Join(lockDirectory, "sources.lock.json"), []byte(lock), 0644); err != nil { + t.Fatal(err) + } + + profiles, err := readDockerSandboxesActiveProfiles(projectRoot, "linux/amd64") + if err != nil { + t.Fatal(err) + } + if got, want := len(profiles), 2; got != want { + t.Fatalf("active profile count = %d, want %d", got, want) + } + for index, want := range []struct { + name, channel, reference, label string + }{ + {"full", "ghcr.io/catthehacker/ubuntu:full-latest", "docker.io/library/epar-docker-sandboxes-catthehacker-full:20260723-r2-amd64", "Catthehacker Ubuntu Full (recommended)"}, + {"act-22.04", "ghcr.io/catthehacker/ubuntu:act-22.04", "docker.io/library/epar-docker-sandboxes-catthehacker-act-22.04:20260723-r4-amd64", "Catthehacker Ubuntu Act 22.04 (current lean profile)"}, + } { + got := profiles[index] + if got.Name != want.name || got.ObservedTag != want.channel || got.TemplateReference != want.reference || got.DisplayLabel != want.label { + t.Fatalf("active profile %d = %#v, want name=%q channel=%q reference=%q label=%q", index, got, want.name, want.channel, want.reference, want.label) + } + if strings.Contains(got.TemplateReference, "-r1-") || strings.Contains(got.TemplateReference, "-r3-") { + t.Fatalf("historical template leaked into active profiles: %#v", got) + } + } +} + +func TestReadDockerSandboxesActiveProfilesRejectsUnexpectedSourceChannel(t *testing.T) { + projectRoot := t.TempDir() + lockDirectory := filepath.Join(projectRoot, "templates", "docker-sandboxes") + if err := os.MkdirAll(lockDirectory, 0755); err != nil { + t.Fatal(err) + } + const lock = `{"schemaVersion":2,"profiles":{"full":{"observedTagReference":"ghcr.io/catthehacker/ubuntu:full-latest\nnot-a-channel","platforms":{"linux/amd64":{"templateTag":"epar-docker-sandboxes-catthehacker-full:20260723-r2-amd64"}}},"act-22.04":{"observedTagReference":"ghcr.io/catthehacker/ubuntu:act-22.04","platforms":{"linux/amd64":{"templateTag":"epar-docker-sandboxes-catthehacker-act-22.04:20260723-r4-amd64"}}}}}` + if err := os.WriteFile(filepath.Join(lockDirectory, "sources.lock.json"), []byte(lock), 0644); err != nil { + t.Fatal(err) + } + + if _, err := readDockerSandboxesActiveProfiles(projectRoot, "linux/amd64"); err == nil || !strings.Contains(err.Error(), "unexpected observedTagReference") { + t.Fatalf("unexpected source channel was accepted: %v", err) + } +} + +func TestInitDockerSandboxesUsesGuidedRootDiskDefault(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + policyFingerprint := "sha256:" + strings.Repeat("b", 64) + template := initDockerSandboxesTemplate{ + Reference: "docker.io/library/epar-docker-sandboxes-catthehacker-full:measured", + Digest: "sha256:" + strings.Repeat("a", 64), + CacheID: strings.Repeat("a", 12), + Platform: "linux/amd64", + Size: 18_730_706_190, + } + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + Templates: []initDockerSandboxesTemplate{ + {Reference: "docker.io/library/epar-docker-sandboxes-catthehacker-full:unmeasured-newer", Digest: "sha256:" + strings.Repeat("f", 64), CacheID: strings.Repeat("f", 12), Platform: "linux/amd64", Size: 19 << 30}, + template, + }, + PolicyFingerprint: policyFingerprint, + }, nil) + oldMeasurement := initDockerSandboxesRootMeasurementFor + initDockerSandboxesRootMeasurementFor = func(host sandboxpromotion.Platform, actual initDockerSandboxesTemplate) (initDockerSandboxesRootMeasurement, bool) { + if host != sandboxpromotion.WindowsAMD64 { + t.Fatalf("measurement host = %q, want %q", host, sandboxpromotion.WindowsAMD64) + } + if actual.Digest != template.Digest { + return initDockerSandboxesRootMeasurement{}, false + } + return initDockerSandboxesRootMeasurement{ + PeakBytes: 324_780_032, + Evidence: "test workload", + }, true + } + t.Cleanup(func() { + initDockerSandboxesRootMeasurementFor = oldMeasurement + }) + + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n1\n\nn\n\n\n\nn\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if got, want := cfg.DockerSandboxes.RootDisk, "auto"; got != want { + t.Fatalf("dockerSandboxes.rootDisk = %q, want %q", got, want) + } + if got, want := cfg.Image.SourceImage, "ghcr.io/catthehacker/ubuntu:full-latest"; got != want { + t.Fatalf("image.sourceImage = %q, want guided default %q", got, want) + } + for _, want := range []string{ + "1. full — full-latest (default)", + "Automatic sandbox root limit:", + "Estimated download:", + } { + if !strings.Contains(out.String(), want) { + t.Fatalf("init output omitted %q:\n%s", want, out.String()) + } + } +} + +func TestFormatInitByteCountUsesReadableBinaryUnits(t *testing.T) { + for _, test := range []struct { + value int64 + want string + }{ + {value: 18_730_706_190, want: "17.44GiB"}, + {value: 324_780_032, want: "309.73MiB"}, + {value: 20 << 30, want: "20GiB"}, + {value: 512, want: "512B"}, + } { + if got := formatInitByteCount(test.value); got != test.want { + t.Fatalf("formatInitByteCount(%d) = %q, want %q", test.value, got, test.want) + } + } +} + +func TestDockerSandboxesRootMeasurementIsBoundToExactTemplateIdentity(t *testing.T) { + const digest = "sha256:00303a3e249a1baf8b0585d20273af408c27182dcfc827a98aa25ffe66b1f67f" + measurement, ok := dockerSandboxesRootMeasurement(sandboxpromotion.WindowsAMD64, initDockerSandboxesTemplate{Digest: digest, Platform: "linux/amd64"}) + if !ok { + t.Fatal("exact measured template identity did not return capacity evidence") + } + if got, want := measurement.PeakBytes, int64(324_780_032); got != want { + t.Fatalf("measurement peak = %d, want %d", got, want) + } + for _, test := range []struct { + host sandboxpromotion.Platform + template initDockerSandboxesTemplate + }{ + {host: sandboxpromotion.WindowsAMD64, template: initDockerSandboxesTemplate{Digest: "sha256:" + strings.Repeat("f", 64), Platform: "linux/amd64"}}, + {host: sandboxpromotion.WindowsAMD64, template: initDockerSandboxesTemplate{Digest: digest, Platform: "linux/arm64"}}, + {host: sandboxpromotion.LinuxAMD64, template: initDockerSandboxesTemplate{Digest: digest, Platform: "linux/amd64"}}, + } { + if _, ok := dockerSandboxesRootMeasurement(test.host, test.template); ok { + t.Fatalf("unexpected measurement for host %q digest %q platform %q", test.host, test.template.Digest, test.template.Platform) + } + } +} + +func TestInitDockerSandboxesDiscoveryRetryKeepsProviderSelectionAndWritesVerifiedConfig(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + policyFingerprint := "sha256:" + strings.Repeat("b", 64) + discovery := initDockerSandboxesDiscovery{ + Templates: []initDockerSandboxesTemplate{{ + Reference: "docker.io/library/epar-docker-sandboxes-catthehacker-act-22.04:20260723-r4-amd64", + Digest: "sha256:" + strings.Repeat("c", 64), + CacheID: strings.Repeat("c", 12), + Platform: "linux/amd64", + Size: 4 << 30, + Label: "Catthehacker Ubuntu Act 22.04 (current lean profile)", + SourceChannel: "ghcr.io/catthehacker/ubuntu:act-22.04", + }}, + PolicyFingerprint: policyFingerprint, + } + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, discovery, nil) + originalDiscovery := initDiscoverDockerSandboxes + discoveryCalls := 0 + initDiscoverDockerSandboxes = func(ctx context.Context, projectRoot, guestPlatform string) (initDockerSandboxesDiscovery, error) { + discoveryCalls++ + if discoveryCalls == 1 { + return initDockerSandboxesDiscovery{}, errors.New("template cache unavailable") + } + return originalDiscovery(ctx, projectRoot, guestPlatform) + } + + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n1\n\n\nn\n\n\n\nn\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if got, want := cfg.Provider.Type, "docker-sandboxes"; got != want { + t.Fatalf("provider.type = %q, want %q after retained-selection retry", got, want) + } + if got := strings.Count(out.String(), "Runner provider:"); got != 1 { + t.Fatalf("Runner provider prompts = %d, want 1 after Docker Sandboxes discovery retry:\n%s", got, out.String()) + } + if got := strings.Count(out.String(), "Continue with explicit preview setup?"); got != 0 { + t.Fatalf("preview acknowledgements = %d, want 0 after capability-driven default:\n%s", got, out.String()) + } + if discoveryCalls != 3 { + t.Fatalf("Docker Sandboxes discovery calls = %d, want 3 including policy readback", discoveryCalls) + } + if !strings.Contains(out.String(), "That image cannot be used for linux/amd64: template cache unavailable") || !strings.Contains(out.String(), "Choose an existing ghcr.io/catthehacker/ubuntu tag") { + t.Fatalf("init output did not explain Docker Sandboxes source recovery:\n%s", out.String()) + } +} + +func TestInitDockerSandboxesDiscoveryRetryDeclinedExitsWithoutRepeatingProviderOrWritingConfig(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{}, errors.New("template cache unavailable")) + + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n1\nn\n"), + Out: &out, + }) + if err == nil || !strings.Contains(err.Error(), "resolve runner source image") { + t.Fatalf("retry-declined error = %v", err) + } + if got := strings.Count(out.String(), "Runner provider:"); got != 1 { + t.Fatalf("Runner provider prompts = %d, want 1:\n%s", got, out.String()) + } + if _, statErr := os.Stat(path); !os.IsNotExist(statErr) { + t.Fatalf("retry-declined setup wrote a config: %v", statErr) + } +} + +func TestInitDockerSandboxesKillSwitchDoesNotInvokeDiscovery(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + oldPlatform := initSandboxPromotionPlatform + oldLookup := initSandboxPromotionLookup + oldDiscovery := initDiscoverDockerSandboxes + initSandboxPromotionPlatform = func() sandboxpromotion.Platform { return sandboxpromotion.WindowsAMD64 } + initSandboxPromotionLookup = func(sandboxpromotion.Platform) (sandboxpromotion.Record, bool) { + return sandboxpromotion.Record{}, false + } + initDiscoverDockerSandboxes = func(context.Context, string, string) (initDockerSandboxesDiscovery, error) { + t.Fatal("kill switch must prevent Docker Sandboxes discovery") + return initDockerSandboxesDiscovery{}, nil + } + t.Cleanup(func() { + initSandboxPromotionPlatform = oldPlatform + initSandboxPromotionLookup = oldLookup + initDiscoverDockerSandboxes = oldDiscovery + }) + t.Setenv(sandboxpromotion.DisableEnvironment, "1") + + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n2"), + Out: &out, + }) + if err == nil || !strings.Contains(err.Error(), "EPAR_DISABLE_DOCKER_SANDBOXES") { + t.Fatalf("kill-switch setup error = %v", err) + } + if _, statErr := os.Stat(path); !os.IsNotExist(statErr) { + t.Fatalf("kill-switch setup wrote a config: %v", statErr) + } + if !strings.Contains(out.String(), "EPAR_DISABLE_DOCKER_SANDBOXES=1 disables admission") { + t.Fatalf("init output did not explain preview kill switch:\n%s", out.String()) + } +} + +func TestInitPromotedDockerSandboxesDefaultsOnlyAfterPassingPreflight(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + record := validInitPromotionRecord() + stubInitSandboxPromotion(t, record, sandboxpromotion.PreflightResult{}) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n\n\nn\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if got, want := cfg.Provider.Type, "docker-sandboxes"; got != want { + t.Fatalf("provider.type = %q, want %q", got, want) + } + if got, want := cfg.Provider.Platform, "linux/amd64"; got != want { + t.Fatalf("provider.platform = %q, want %q", got, want) + } + if !slices.Contains(cfg.Runner.Labels, "X64") { + t.Fatalf("runner.labels = %q, want the mapped X64 guest architecture", cfg.Runner.Labels) + } + if got, want := cfg.Image.SourceImage, "ghcr.io/catthehacker/ubuntu:full-latest"; got != want { + t.Fatalf("image.sourceImage = %q, want %q", got, want) + } + if got, want := cfg.DockerSandboxes.PolicyGeneration, record.PolicyFingerprint; got != want { + t.Fatalf("dockerSandboxes.policyGeneration = %q, want %q", got, want) + } + for key, values := range map[string]struct { + got string + want string + }{ + "rootDisk": {cfg.DockerSandboxes.RootDisk, "auto"}, + "dockerDisk": {cfg.DockerSandboxes.DockerDisk, "50GiB"}, + } { + if values.got != values.want { + t.Fatalf("dockerSandboxes.%s = %q, want %q", key, values.got, values.want) + } + } + configText, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + for _, required := range []string{"type: docker-sandboxes", "dockerSandboxes:", "epar-docker-sandboxes", "policyGeneration: " + record.PolicyFingerprint} { + if !strings.Contains(string(configText), required) { + t.Fatalf("generated Docker Sandboxes config omitted %q:\n%s", required, configText) + } + } + for _, forbidden := range []string{"dockerSandbox:", "\n type: docker-sandbox\n", "epar-docker-sandbox]"} { + if strings.Contains(string(configText), forbidden) { + t.Fatalf("generated config used singular Docker Sandbox key %q:\n%s", forbidden, configText) + } + } + if !strings.Contains(out.String(), "PASS: the exact promoted platform") || !strings.Contains(out.String(), "Docker Sandboxes (independently certified for this exact platform) (default)") { + t.Fatalf("init output did not explain the promoted default:\n%s", out.String()) + } +} + +func TestPromotedDockerSandboxesPlatformUsesSharedHostGuestMapping(t *testing.T) { + tests := []struct { + host sandboxpromotion.Platform + wantGuest string + wantRunnerLabel string + wantUnsupported bool + }{ + {host: sandboxpromotion.WindowsAMD64, wantGuest: "linux/amd64", wantRunnerLabel: "X64"}, + {host: sandboxpromotion.LinuxAMD64, wantGuest: "linux/amd64", wantRunnerLabel: "X64"}, + {host: sandboxpromotion.DarwinARM64, wantGuest: "linux/arm64", wantRunnerLabel: "ARM64"}, + {host: sandboxpromotion.Platform("linux/arm64"), wantGuest: "linux/arm64", wantRunnerLabel: "ARM64"}, + {host: sandboxpromotion.Platform("windows/arm64"), wantGuest: "linux/arm64", wantRunnerLabel: "ARM64"}, + {host: sandboxpromotion.Platform("darwin/amd64"), wantGuest: "linux/amd64", wantRunnerLabel: "X64"}, + {host: sandboxpromotion.Platform("futureos/amd64"), wantGuest: "linux/amd64", wantRunnerLabel: "X64"}, + {host: sandboxpromotion.Platform("futureos/386"), wantUnsupported: true}, + } + for _, test := range tests { + t.Run(string(test.host), func(t *testing.T) { + guest, runnerLabel, err := promotedDockerSandboxesPlatform(sandboxpromotion.Record{Platform: test.host}) + if test.wantUnsupported { + if err == nil { + t.Fatalf("promotedDockerSandboxesPlatform(%q) unexpectedly succeeded", test.host) + } + return + } + if err != nil { + t.Fatal(err) + } + if guest != test.wantGuest || runnerLabel != test.wantRunnerLabel { + t.Fatalf("promotedDockerSandboxesPlatform(%q) = (%q, %q), want (%q, %q)", test.host, guest, runnerLabel, test.wantGuest, test.wantRunnerLabel) + } + }) + } +} + +func TestInitPromotionGateFailuresRequireExplicitProviderAndExplainAction(t *testing.T) { + record := validInitPromotionRecord() + tests := []struct { + name string + gate string + detail string + disable bool + }{ + {name: "kill switch", gate: "operator kill switch", detail: sandboxpromotion.DisableEnvironment, disable: true}, + {name: "native controller", gate: "native controller", detail: "native controller is unavailable"}, + {name: "daemon", gate: "daemon health", detail: "daemon is not running"}, + {name: "virtualization", gate: "virtualization", detail: "virtualization is unavailable"}, + {name: "template", gate: "promoted template", detail: "template identity differs"}, + {name: "policy", gate: "promoted policy", detail: "policy fingerprint differs"}, + {name: "resource", gate: "resource availability", detail: "insufficient free space"}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + preflight := sandboxpromotion.PreflightResult{Failures: []sandboxpromotion.Failure{{ + Gate: test.gate, + Detail: test.detail, + Resolution: "take the exact corrective action and rerun setup", + }}} + stubInitSandboxPromotion(t, record, preflight) + if test.disable { + t.Setenv(sandboxpromotion.DisableEnvironment, "1") + initDockerSandboxesPreflight = func(context.Context, sandboxpromotion.Record, string) sandboxpromotion.PreflightResult { + t.Fatal("kill switch must stop automatic-default preflight before admission checks") + return sandboxpromotion.PreflightResult{} + } + } else { + t.Setenv(sandboxpromotion.DisableEnvironment, "") + } + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\ndocker-sandboxes\n1\n\nn\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if cfg.Provider.Type != "docker-container" { + t.Fatalf("provider.type = %q, want explicit Docker Container selection", cfg.Provider.Type) + } + if !strings.Contains(out.String(), "FAIL ["+test.gate+"]") || !strings.Contains(out.String(), test.detail) || !strings.Contains(out.String(), "Action:") { + t.Fatalf("init output omitted actionable %s failure:\n%s", test.gate, out.String()) + } + if !strings.Contains(out.String(), "explicit choice required") || !strings.Contains(out.String(), "Docker Sandboxes (independently certified for this exact platform) is unavailable") { + t.Fatalf("init output did not reject explicit unavailable Docker Sandboxes selection:\n%s", out.String()) + } + }) + } +} + +func TestInitFailedPromotionAllowsExplicitWSLSelection(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubWSL2Available(t) + record := validInitPromotionRecord() + stubInitSandboxPromotion(t, record, sandboxpromotion.PreflightResult{Failures: []sandboxpromotion.Failure{{ + Gate: "authentication", Detail: "authentication is not valid", Resolution: "run sbx login", + }}}) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n3\n\n"), + Out: &bytes.Buffer{}, + }); err != nil { + t.Fatal(err) + } + cfg, err := config.Load(path) + if err != nil { + t.Fatal(err) + } + if cfg.Provider.Type != "wsl" { + t.Fatalf("provider.type = %q, want explicit WSL selection", cfg.Provider.Type) + } +} + +func TestInitFailedPromotionDoesNotSilentlyFallBackOnEOF(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + record := validInitPromotionRecord() + stubInitSandboxPromotion(t, record, sandboxpromotion.PreflightResult{Failures: []sandboxpromotion.Failure{{ + Gate: "authentication", Detail: "authentication is not valid", Resolution: "run sbx login", + }}}) + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipDockerCheck: true, + SkipHostTrustCheck: true, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n"), + Out: &bytes.Buffer{}, + }) + if err == nil || !strings.Contains(err.Error(), "invalid runner provider") { + t.Fatalf("error = %v, want explicit provider requirement", err) + } + if _, statErr := os.Stat(path); !errors.Is(statErr, os.ErrNotExist) { + t.Fatalf("config was written after failed explicit selection: %v", statErr) + } +} + +func TestInitOffersTartConfigWhenAvailable(t *testing.T) { + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubTartAvailable(t) + oldDockerAvailable := dockerAvailable + dockerAvailable = func(context.Context) error { + return errors.New("Docker is unavailable on this Mac") + } + t.Cleanup(func() { dockerAvailable = oldDockerAvailable }) + + dir := t.TempDir() + path := filepath.Join(dir, ".local", "config.yml") + var out bytes.Buffer + if err := runInitWithOptions(initOptions{ + ProjectRoot: dir, + ConfigPath: path, + SkipHostTrustCheck: true, + In: strings.NewReader("654321\nexample\n.local/github-app.pem\n1\n4\n\n"), + Out: &out, + }); err != nil { + t.Fatal(err) + } + + got, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + want, err := os.ReadFile(filepath.Join("..", "..", "configs", "tart.example.yml")) + if err != nil { + t.Fatal(err) + } + wantText := strings.NewReplacer( + "appId: 123456", "appId: 654321", "organization: your-org", "organization: example", "privateKeyPath: ~/.config/ephemeral-action-runner/github-app.pem", "privateKeyPath: .local/github-app.pem", "namePrefix: CHANGE-ME-unique-machine-prefix", "namePrefix: build-box-01-a4f9c2", + "group: your-runner-group", "group: \"restricted group\"", ).Replace(string(want)) wantText = strings.ReplaceAll(wantText, "\r\n", "\n") if string(got) != wantText { t.Fatalf("Tart config did not match configs/tart.example.yml:\nwant:\n%s\ngot:\n%s", wantText, got) } - if !strings.Contains(out.String(), "2. Tart (experimental)") { + if !strings.Contains(out.String(), "2. Docker Sandboxes — recommended when ready") || !strings.Contains(out.String(), "3. WSL2") || !strings.Contains(out.String(), "4. Tart (experimental)") || !strings.Contains(out.String(), "Docker CLI or daemon check failed: Docker is unavailable on this Mac") { t.Fatalf("init output did not offer Tart:\n%s", out.String()) } } @@ -529,14 +2136,76 @@ func stubInitHostAndRandom(t *testing.T, hostname string, random []byte) { t.Helper() oldHostname := initHostname oldRandomRead := initRandomRead + oldResolver := initResolveDockerSandboxesSource initHostname = func() (string, error) { return hostname, nil } initRandomRead = fixedRandomRead(random) + initResolveDockerSandboxesSource = func(_ context.Context, input, platform string) (imageartifact.ResolvedDockerSource, error) { + reference, err := imageartifact.NormalizeCatthehackerSource(input) + if err != nil { + return imageartifact.ResolvedDockerSource{}, err + } + return imageartifact.ResolvedDockerSource{ + Reference: reference, + ImmutableReference: "ghcr.io/catthehacker/ubuntu@sha256:" + strings.Repeat("a", 64), + IndexDigest: "sha256:" + strings.Repeat("a", 64), + PlatformDigest: "sha256:" + strings.Repeat("b", 64), + Platform: platform, + CompressedLayerBytes: 8 << 30, + }, nil + } + stubInitRunnerGroupClient(t) t.Cleanup(func() { initHostname = oldHostname initRandomRead = oldRandomRead + initResolveDockerSandboxesSource = oldResolver }) } +type fakeInitRunnerGroupClient struct { + groups []gh.RunnerGroup + groupResponses [][]gh.RunnerGroup + repositories map[int64][]gh.RunnerGroupRepository + err error + groupCalls int +} + +func (f *fakeInitRunnerGroupClient) ListRunnerGroups(context.Context) ([]gh.RunnerGroup, error) { + f.groupCalls++ + if len(f.groupResponses) > 0 { + index := f.groupCalls - 1 + if index >= len(f.groupResponses) { + index = len(f.groupResponses) - 1 + } + return append([]gh.RunnerGroup(nil), f.groupResponses[index]...), f.err + } + return append([]gh.RunnerGroup(nil), f.groups...), f.err +} + +func (f *fakeInitRunnerGroupClient) ListRunnerGroupRepositories(_ context.Context, groupID int64) ([]gh.RunnerGroupRepository, error) { + return append([]gh.RunnerGroupRepository(nil), f.repositories[groupID]...), f.err +} + +func stubInitRunnerGroupClient(t *testing.T) { + t.Helper() + oldFactory := newInitRunnerGroupClient + newInitRunnerGroupClient = func(config.GitHubConfig) initRunnerGroupClient { + return &fakeInitRunnerGroupClient{ + groups: []gh.RunnerGroup{{ID: 1, Name: "restricted group", Visibility: config.RunnerGroupRepositoryAccessSelected}}, + repositories: map[int64][]gh.RunnerGroupRepository{ + 1: {{ID: 1, FullName: "example/private", Private: true}}, + }, + } + } + t.Cleanup(func() { newInitRunnerGroupClient = oldFactory }) +} + +func setInitRunnerGroupClient(t *testing.T, client initRunnerGroupClient) { + t.Helper() + oldFactory := newInitRunnerGroupClient + newInitRunnerGroupClient = func(config.GitHubConfig) initRunnerGroupClient { return client } + t.Cleanup(func() { newInitRunnerGroupClient = oldFactory }) +} + func fixedRandomRead(random []byte) func([]byte) (int, error) { return func(data []byte) (int, error) { copy(data, random) @@ -558,6 +2227,7 @@ func utf16LE(text string, includeBOM bool) []byte { func stubNoWSL2(t *testing.T) { t.Helper() + stubDockerSandboxesUnavailable(t) oldGOOS := initGOOS oldWSLStatus := initWSLStatus initGOOS = "linux" @@ -573,15 +2243,32 @@ func stubNoWSL2(t *testing.T) { func stubWSL2Available(t *testing.T) { t.Helper() + stubDockerSandboxesUnavailable(t) oldGOOS := initGOOS oldWSLStatus := initWSLStatus + oldPlatform := initSandboxPromotionPlatform initGOOS = "windows" initWSLStatus = func(context.Context) ([]byte, error) { return []byte("Default Distribution: Ubuntu\nDefault Version: 2\n"), nil } + initSandboxPromotionPlatform = func() sandboxpromotion.Platform { + return sandboxpromotion.WindowsAMD64 + } t.Cleanup(func() { initGOOS = oldGOOS initWSLStatus = oldWSLStatus + initSandboxPromotionPlatform = oldPlatform + }) +} + +func stubDockerSandboxesUnavailable(t *testing.T) { + t.Helper() + oldReadiness := initDockerSandboxesReadiness + initDockerSandboxesReadiness = func(context.Context) (dockersandboxes.HostReadiness, error) { + return dockersandboxes.HostReadiness{}, errors.New("Docker Sandboxes unavailable in provider-neutral wizard test") + } + t.Cleanup(func() { + initDockerSandboxesReadiness = oldReadiness }) } @@ -596,3 +2283,187 @@ func stubTartAvailable(t *testing.T) { initTartVersion = oldTartVersion }) } + +func stubInitDockerSandboxesSetup(t *testing.T, platform sandboxpromotion.Platform, discovery initDockerSandboxesDiscovery, discoveryErr error) { + t.Helper() + oldPlatform := initSandboxPromotionPlatform + oldLookup := initSandboxPromotionLookup + oldDiscovery := initDiscoverDockerSandboxes + oldReadiness := initDockerSandboxesReadiness + oldCapacityCheck := initDockerSandboxesCapacityCheck + oldResolver := initResolveDockerSandboxesSource + oldPolicyFingerprint := initDockerSandboxesPolicyFingerprint + oldEnsureTemplate := initEnsureDockerSandboxesTemplate + initSandboxPromotionPlatform = func() sandboxpromotion.Platform { return platform } + initSandboxPromotionLookup = func(actual sandboxpromotion.Platform) (sandboxpromotion.Record, bool) { + if actual != platform { + t.Fatalf("promotion lookup platform = %q, want %q", actual, platform) + } + return sandboxpromotion.Record{}, false + } + initDiscoverDockerSandboxes = func(_ context.Context, projectRoot, guestPlatform string) (initDockerSandboxesDiscovery, error) { + if projectRoot == "" { + t.Fatal("Docker Sandboxes discovery project root is empty") + } + expectedGuestPlatform, _, err := dockerSandboxesPlatform(platform) + if err != nil { + t.Fatalf("derive Docker Sandboxes discovery guest platform: %v", err) + } + if guestPlatform != expectedGuestPlatform { + t.Fatalf("Docker Sandboxes discovery guest platform = %q, want %q", guestPlatform, expectedGuestPlatform) + } + return discovery, discoveryErr + } + initDockerSandboxesReadiness = func(context.Context) (dockersandboxes.HostReadiness, error) { + return dockersandboxes.HostReadiness{ChecksPassed: 8, ChecksWarned: 1}, nil + } + initResolveDockerSandboxesSource = func(ctx context.Context, input, guestPlatform string) (imageartifact.ResolvedDockerSource, error) { + current, err := initDiscoverDockerSandboxes(ctx, "test-project", guestPlatform) + if err != nil { + return imageartifact.ResolvedDockerSource{}, err + } + reference, err := imageartifact.NormalizeCatthehackerSource(input) + if err != nil { + return imageartifact.ResolvedDockerSource{}, err + } + compressed := uint64(8 << 30) + if len(current.Templates) > 0 && current.Templates[0].Size > 0 { + compressed = uint64(current.Templates[0].Size) + } + return imageartifact.ResolvedDockerSource{ + Reference: reference, + ImmutableReference: "ghcr.io/catthehacker/ubuntu@sha256:" + strings.Repeat("a", 64), + IndexDigest: "sha256:" + strings.Repeat("a", 64), + PlatformDigest: "sha256:" + strings.Repeat("b", 64), + Platform: guestPlatform, + CompressedLayerBytes: compressed, + }, nil + } + initDockerSandboxesPolicyFingerprint = func(ctx context.Context) (string, error) { + current, err := initDiscoverDockerSandboxes(ctx, "test-project", expectedDockerSandboxesGuestPlatform(t, platform)) + return current.PolicyFingerprint, err + } + initEnsureDockerSandboxesTemplate = func(context.Context, string, string) error { return nil } + initDockerSandboxesCapacityCheck = func(rootDisk, dockerDisk, minHostFreeSpace uint64) (initDockerSandboxesCapacityResult, error) { + return initDockerSandboxesCapacityResult{ + StorageRoot: `C:\stub\DockerSandboxes`, + AvailableBytes: 1 << 40, + TotalBytes: 2 << 40, + Reservation: rootDisk + dockerDisk, + HostWatermark: minHostFreeSpace, + RequiredBytes: rootDisk + dockerDisk + minHostFreeSpace, + CapacityStatus: storage.CapacityReady, + }, nil + } + t.Cleanup(func() { + initSandboxPromotionPlatform = oldPlatform + initSandboxPromotionLookup = oldLookup + initDiscoverDockerSandboxes = oldDiscovery + initDockerSandboxesReadiness = oldReadiness + initDockerSandboxesCapacityCheck = oldCapacityCheck + initResolveDockerSandboxesSource = oldResolver + initDockerSandboxesPolicyFingerprint = oldPolicyFingerprint + initEnsureDockerSandboxesTemplate = oldEnsureTemplate + }) +} + +func expectedDockerSandboxesGuestPlatform(t *testing.T, platform sandboxpromotion.Platform) string { + t.Helper() + guestPlatform, _, err := dockerSandboxesPlatform(platform) + if err != nil { + t.Fatal(err) + } + return guestPlatform +} + +func stubInitSandboxPromotion(t *testing.T, record sandboxpromotion.Record, result sandboxpromotion.PreflightResult) { + t.Helper() + oldPlatform := initSandboxPromotionPlatform + oldLookup := initSandboxPromotionLookup + oldPreflight := initDockerSandboxesPreflight + oldReadiness := initDockerSandboxesReadiness + oldCapacityCheck := initDockerSandboxesCapacityCheck + oldResolver := initResolveDockerSandboxesSource + oldPolicyFingerprint := initDockerSandboxesPolicyFingerprint + oldEnsureTemplate := initEnsureDockerSandboxesTemplate + initSandboxPromotionPlatform = func() sandboxpromotion.Platform { return record.Platform } + initSandboxPromotionLookup = func(platform sandboxpromotion.Platform) (sandboxpromotion.Record, bool) { + if platform != record.Platform { + t.Fatalf("promotion lookup platform = %q, want %q", platform, record.Platform) + } + return record, true + } + initDockerSandboxesPreflight = func(context.Context, sandboxpromotion.Record, string) sandboxpromotion.PreflightResult { + return result + } + initDockerSandboxesReadiness = func(context.Context) (dockersandboxes.HostReadiness, error) { + return dockersandboxes.HostReadiness{ChecksPassed: 8}, nil + } + initResolveDockerSandboxesSource = func(_ context.Context, input, guestPlatform string) (imageartifact.ResolvedDockerSource, error) { + reference, err := imageartifact.NormalizeCatthehackerSource(input) + if err != nil { + return imageartifact.ResolvedDockerSource{}, err + } + return imageartifact.ResolvedDockerSource{ + Reference: reference, + ImmutableReference: "ghcr.io/catthehacker/ubuntu@" + record.TemplateDigest, + IndexDigest: record.TemplateDigest, + PlatformDigest: record.TemplateDigest, + Platform: guestPlatform, + CompressedLayerBytes: 8 << 30, + }, nil + } + initDockerSandboxesPolicyFingerprint = func(context.Context) (string, error) { return record.PolicyFingerprint, nil } + initEnsureDockerSandboxesTemplate = func(context.Context, string, string) error { return nil } + initDockerSandboxesCapacityCheck = func(rootDisk, dockerDisk, minHostFreeSpace uint64) (initDockerSandboxesCapacityResult, error) { + return initDockerSandboxesCapacityResult{ + StorageRoot: `C:\stub\DockerSandboxes`, + AvailableBytes: 1 << 40, + TotalBytes: 2 << 40, + Reservation: rootDisk + dockerDisk, + HostWatermark: minHostFreeSpace, + RequiredBytes: rootDisk + dockerDisk + minHostFreeSpace, + CapacityStatus: storage.CapacityReady, + }, nil + } + t.Cleanup(func() { + initSandboxPromotionPlatform = oldPlatform + initSandboxPromotionLookup = oldLookup + initDockerSandboxesPreflight = oldPreflight + initDockerSandboxesReadiness = oldReadiness + initDockerSandboxesCapacityCheck = oldCapacityCheck + initResolveDockerSandboxesSource = oldResolver + initDockerSandboxesPolicyFingerprint = oldPolicyFingerprint + initEnsureDockerSandboxesTemplate = oldEnsureTemplate + }) +} + +func validInitPromotionRecord() sandboxpromotion.Record { + digest := func(character string) string { return "sha256:" + strings.Repeat(character, 64) } + return sandboxpromotion.Record{ + Platform: sandboxpromotion.WindowsAMD64, + EPARRevision: digest("1"), + Template: "epar-docker-sandboxes-catthehacker-full:promoted", + TemplateDigest: digest("a"), + TemplateCacheID: strings.Repeat("a", 12), + TemplateMetadataDigest: digest("b"), + TemplateArchiveDigest: digest("b"), + PolicyFingerprint: digest("b"), + EvidenceDigest: digest("c"), + SBOMDigest: digest("d"), + ProvenanceDigest: digest("e"), + SoftwareInventoryDigest: digest("f"), + VerifiedAt: time.Date(2026, 7, 23, 0, 0, 0, 0, time.UTC), + Verifier: "independent-test-verifier", + Gates: sandboxpromotion.GateResults{Local: true, Functional: true, Recovery: true, Security: true, Policy: true, Cleanup: true, SecretScanning: true, ConcurrentProvisioning: true, IndependentSecurityReview: true}, + RootDiskBytes: 120 << 30, + DockerDiskBytes: 100 << 30, + MinHostFreeSpaceBytes: 50 << 30, + ReliabilityJobs: 25, + ReliabilityDuration: 2 * time.Hour, + CachedCreateP95: 30 * time.Second, + QueueToOnlineP95: 90 * time.Second, + ForceRemoveP95: 60 * time.Second, + BuildxComposeSlowdownPct: 10, + } +} diff --git a/cmd/ephemeral-action-runner/main.go b/cmd/ephemeral-action-runner/main.go index 7c5c680..fa42c90 100644 --- a/cmd/ephemeral-action-runner/main.go +++ b/cmd/ephemeral-action-runner/main.go @@ -8,21 +8,24 @@ import ( "os/signal" "path/filepath" "strings" + "sync" "syscall" "time" "github.com/solutionforest/ephemeral-action-runner/internal/config" gh "github.com/solutionforest/ephemeral-action-runner/internal/github" + "github.com/solutionforest/ephemeral-action-runner/internal/invocation" "github.com/solutionforest/ephemeral-action-runner/internal/logging" "github.com/solutionforest/ephemeral-action-runner/internal/pool" "github.com/solutionforest/ephemeral-action-runner/internal/provider" - dockerdindprovider "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockerdind" - tartprovider "github.com/solutionforest/ephemeral-action-runner/internal/provider/tart" - wslprovider "github.com/solutionforest/ephemeral-action-runner/internal/provider/wsl" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/registry" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" ) const binaryName = "ephemeral-action-runner" +var imageUpdateDefaultNotice sync.Once + func main() { if err := run(os.Args[1:]); err != nil { fmt.Fprintln(os.Stderr, binaryName+":", err) @@ -113,6 +116,8 @@ func run(args []string) error { return runStatus(args[1:]) case "logs": return runLogs(args[1:]) + case "storage": + return runStorage(args[1:]) case "version": printVersion(os.Stdout) return nil @@ -218,12 +223,40 @@ func retentionPolicy(cfg config.LoggingConfig) logging.RetentionPolicy { func runImage(args []string) error { if len(args) == 0 { - return fmt.Errorf("image requires subcommand: update-upstream or build") + return fmt.Errorf("image requires subcommand: update, update-upstream, build, or refresh-scripts") } switch args[0] { + case "update": + fs := flag.NewFlagSet("image update", flag.ExitOnError) + common := addCommonFlags(fs) + allowInsufficientStorage := fs.Bool("allow-insufficient-storage", false, "continue this invocation after storage-only admission warnings") + if err := fs.Parse(args[1:]); err != nil { + return err + } + m, err := newManager(*common.configPath, *common.projectRoot, *common.dryRun, false) + if err != nil { + return err + } + defer m.Close() + m.ConfigureStorageAdmissionOverride(*allowInsufficientStorage, invocation.Command(append([]string{"image", "update"}, appendStorageOverride(args[1:])...)...)) + ctx := interruptContext() + poolControllerLock, err := m.AcquirePoolControllerLock() + if err != nil { + return err + } + defer poolControllerLock.Close() + hostTrustControllerLock, err := m.AcquireHostTrustControllerLock() + if err != nil { + return err + } + if hostTrustControllerLock != nil { + defer hostTrustControllerLock.Close() + } + return m.UpdateImage(ctx) case "update-upstream": fs := flag.NewFlagSet("image update-upstream", flag.ExitOnError) common := addCommonFlags(fs) + allowInsufficientStorage := fs.Bool("allow-insufficient-storage", false, "continue this invocation after storage-only admission warnings") if err := fs.Parse(args[1:]); err != nil { return err } @@ -232,6 +265,10 @@ func runImage(args []string) error { return err } defer m.Close() + m.ConfigureStorageAdmissionOverride(*allowInsufficientStorage, invocation.Command(append([]string{"image", "update-upstream"}, appendStorageOverride(args[1:])...)...)) + if err := rejectDockerSandboxesImageCommand(m, "image update-upstream"); err != nil { + return err + } controllerLock, err := m.AcquirePoolControllerLock() if err != nil { return err @@ -244,6 +281,7 @@ func runImage(args []string) error { replace := fs.Bool("replace", false, "delete an existing output image before building") update := fs.Bool("update-upstream", false, "refresh runner-images before building") skipUpstream := fs.Bool("skip-upstream-check", false, "skip checking the runner-images checkout") + allowInsufficientStorage := fs.Bool("allow-insufficient-storage", false, "continue this invocation after storage-only admission warnings") if err := fs.Parse(args[1:]); err != nil { return err } @@ -252,6 +290,7 @@ func runImage(args []string) error { return err } defer m.Close() + m.ConfigureStorageAdmissionOverride(*allowInsufficientStorage, invocation.Command(append([]string{"image", "build"}, appendStorageOverride(args[1:])...)...)) ctx := interruptContext() poolControllerLock, err := m.AcquirePoolControllerLock() if err != nil { @@ -282,6 +321,9 @@ func runImage(args []string) error { return err } defer m.Close() + if err := rejectDockerSandboxesImageCommand(m, "image refresh-scripts"); err != nil { + return err + } controllerLock, err := m.AcquirePoolControllerLock() if err != nil { return err @@ -303,7 +345,8 @@ func runPool(args []string) error { common := addCommonFlags(fs) instances := fs.Int("instances", 0, "number of concurrent instances to verify; overrides pool.instances") registerOnly := fs.Bool("register-only", false, "register runners and verify online/idle without dispatching a job") - cleanup := fs.Bool("cleanup", false, "clean up prefixed instances and GitHub runners after verification") + cleanup := fs.Bool("cleanup", false, "clean up verification resources; legacy providers use the configured pool prefix, while Docker Sandboxes uses exact owned records") + allowInsufficientStorage := fs.Bool("allow-insufficient-storage", false, "continue this invocation after storage-only admission warnings") if err := fs.Parse(args[1:]); err != nil { return err } @@ -315,6 +358,7 @@ func runPool(args []string) error { return err } defer m.Close() + m.ConfigureStorageAdmissionOverride(*allowInsufficientStorage, invocation.Command(append([]string{"pool", "verify"}, appendStorageOverride(args[1:])...)...)) return m.Verify(interruptContext(), pool.VerifyOptions{Instances: *instances, RegisterOnly: *registerOnly, Cleanup: *cleanup}) case "up": fs := flag.NewFlagSet("pool up", flag.ExitOnError) @@ -324,6 +368,7 @@ func runPool(args []string) error { keepOnExit := fs.Bool("keep-on-exit", false, "leave prefixed instances and GitHub runners running when interrupted") replaceCompleted := fs.Bool("replace-completed", true, "replace an instance when its ephemeral runner exits after a job") monitorInterval := fs.Duration("monitor-interval", 15*time.Second, "interval for runner liveness checks") + allowInsufficientStorage := fs.Bool("allow-insufficient-storage", false, "continue this invocation after storage-only admission warnings") if err := fs.Parse(args[1:]); err != nil { return err } @@ -334,6 +379,7 @@ func runPool(args []string) error { if err != nil { return err } + m.ConfigureStorageAdmissionOverride(*allowInsufficientStorage, invocation.Command(append([]string{"pool", "up"}, appendStorageOverride(args[1:])...)...)) return m.RunPool(interruptContext(), pool.RunOptions{ Instances: *instances, Register: *register, @@ -352,6 +398,7 @@ func runCleanup(args []string) error { fs := flag.NewFlagSet("cleanup", flag.ExitOnError) common := addCommonFlags(fs) noGitHub := fs.Bool("no-github", false, "skip GitHub runner deletion") + acknowledgeFailedDiagnostics := fs.Bool("acknowledge-failed-diagnostics", false, "allow exact cleanup of retained sandboxes after failed diagnostics evidence has been reviewed") if err := fs.Parse(args); err != nil { return err } @@ -360,6 +407,7 @@ func runCleanup(args []string) error { return err } defer m.Close() + m.AcknowledgeFailedDiagnostics = *acknowledgeFailedDiagnostics return m.Cleanup(context.Background()) } @@ -409,6 +457,14 @@ func flagPassed(fs *flag.FlagSet, name string) bool { } func newManager(configPath, projectRoot string, dryRun bool, githubEnabled bool) (*pool.Manager, error) { + return newManagerWithLifecycleState(configPath, projectRoot, dryRun, githubEnabled, true) +} + +func newImageProvisioningManager(configPath, projectRoot string) (*pool.Manager, error) { + return newManagerWithLifecycleState(configPath, projectRoot, false, false, false) +} + +func newManagerWithLifecycleState(configPath, projectRoot string, dryRun bool, githubEnabled bool, openLifecycleState bool) (*pool.Manager, error) { projectRoot, err := filepath.Abs(projectRoot) if err != nil { return nil, err @@ -428,6 +484,18 @@ func newManager(configPath, projectRoot string, dryRun bool, githubEnabled bool) if err := config.Validate(cfg); err != nil { return nil, err } + if !dryRun { + if err := importNativeBootstrapAcquisition(projectRoot, resolvedConfigPath, time.Now().UTC()); err != nil { + return nil, err + } + } + providerRuntime, err := registry.New(cfg, projectRoot, dryRun) + if err != nil { + return nil, err + } + if providerRuntime.Lifecycle == nil || providerRuntime.Storage == nil { + return nil, fmt.Errorf("provider %q registry entry is missing required lifecycle or storage behavior", cfg.Provider.Type) + } var client pool.GitHubClient if githubEnabled && !dryRun { if err := config.ValidateGitHub(cfg); err != nil { @@ -435,10 +503,6 @@ func newManager(configPath, projectRoot string, dryRun bool, githubEnabled bool) } client = gh.New(cfg.GitHub) } - provider, err := newProvider(cfg, projectRoot, dryRun) - if err != nil { - return nil, err - } runtime, err := logging.NewRuntime(logging.Options{ Directory: config.ProjectPath(projectRoot, cfg.Logging.Directory), ManagerSinks: loggingSinks(cfg.Logging.ManagerSinks), @@ -458,13 +522,18 @@ func newManager(configPath, projectRoot string, dryRun bool, githubEnabled bool) return nil, err } manager := &pool.Manager{ - Config: cfg, - Provider: provider, - GitHub: client, - ProjectRoot: projectRoot, - ConfigPath: resolvedConfigPath, - DryRun: dryRun, - Logging: runtime, + Config: cfg, + Provider: providerRuntime.Legacy, + Lifecycle: providerRuntime.Lifecycle, + PolicyManager: providerRuntime.PolicyManager, + Storage: providerRuntime.Storage, + LifecycleStateEnabled: !dryRun && openLifecycleState, + GitHub: client, + ProjectRoot: projectRoot, + ConfigPath: resolvedConfigPath, + DryRun: dryRun, + Logging: runtime, + AutomaticImageLifecycle: true, } if cfg.Logging.RetentionEnabled { report, pruneErr := manager.PruneLogs(false) @@ -479,6 +548,74 @@ func newManager(configPath, projectRoot string, dryRun bool, githubEnabled bool) return manager, nil } +func preflightControllerStorage(projectRoot string, cfg config.Config, contributions ...provider.StorageContribution) error { + minimumFree, err := config.EffectiveMinimumFreeBytes(cfg) + if err != nil { + return err + } + if len(contributions) != 0 && contributions[0] != nil { + snapshot, err := contributions[0].StorageSnapshot(context.Background(), provider.StorageRequest{ + Operation: "controller-bootstrap", + Now: time.Now(), + MinimumFreeBytes: minimumFree, + }) + if err != nil { + return fmt.Errorf("provider storage surface cannot be measured before controller bootstrap: %w\n\nInspect storage with:\n %s", err, invocation.Command("storage", "status", "--provider", cfg.Provider.Type)) + } + surfaces := make(map[string]storage.Surface, len(snapshot.Surfaces)) + for _, surface := range snapshot.Surfaces { + surfaces[surface.ID] = surface + } + for _, requirement := range snapshot.Requirements { + surface, found := surfaces[requirement.SurfaceID] + if !found { + return fmt.Errorf("controller storage requirement %q references unknown surface %q", requirement.ID, requirement.SurfaceID) + } + check, err := storage.EvaluateCapacity(surface, requirement) + if err != nil { + return err + } + if check.Status != storage.CapacityReady { + return storage.CapacityAdmissionError("initialize the EPAR controller", surface, requirement, check, invocation.Command("storage", "prune", "--provider", cfg.Provider.Type)) + } + } + return nil + } + capacity, err := storage.ProbeFilesystemCapacity(projectRoot, time.Now()) + if err != nil { + return fmt.Errorf("controller storage surface %q cannot be measured before bootstrap: %w\n\nInspect storage with:\n %s", projectRoot, err, invocation.Command("storage", "status", "--provider", cfg.Provider.Type)) + } + check, err := storage.EvaluateCapacity(storage.Surface{ + ID: "project", + Provider: cfg.Provider.Type, + Kind: storage.SurfaceHostFilesystem, + Location: projectRoot, + Capacity: capacity, + }, storage.Requirement{ + ID: "controller-bootstrap", + Provider: cfg.Provider.Type, + SurfaceID: "project", + MinimumFreeBytes: minimumFree, + }) + if err != nil { + return fmt.Errorf("evaluate controller storage capacity: %w", err) + } + if check.Status != storage.CapacityReady { + return storage.CapacityAdmissionError("initialize the EPAR controller", storage.Surface{ + Location: projectRoot, + Capacity: check.Capacity, + }, check.Requirement, check, invocation.Command("storage", "prune", "--provider", cfg.Provider.Type)) + } + return nil +} + +func rejectDockerSandboxesImageCommand(manager *pool.Manager, command string) error { + if manager.Config.Provider.Type != "docker-sandboxes" { + return nil + } + return fmt.Errorf("%s is not applicable to docker-sandboxes; edit image.sourceImage or image.customInstallScripts, then run %s", command, invocation.Command("image", "build")) +} + func loggingSinks(values []string) logging.Sinks { var sinks logging.Sinks for _, value := range values { @@ -494,6 +631,12 @@ func loggingSinks(values []string) logging.Sinks { func printConfigWarnings(cfg config.Config) { for _, warning := range cfg.Warnings() { + if strings.Contains(warning, "image update policy is not configured") { + imageUpdateDefaultNotice.Do(func() { + fmt.Fprintln(os.Stderr, "warning:", warning) + }) + continue + } fmt.Fprintln(os.Stderr, "warning:", warning) } } @@ -522,25 +665,6 @@ func resolveConfigPath(projectRoot, explicit string) (string, error) { return "", nil } -func newProvider(cfg config.Config, projectRoot string, dryRun bool) (provider.Provider, error) { - switch cfg.Provider.Type { - case "tart": - return tartprovider.New("", dryRun), nil - case "wsl": - return wslprovider.New("", config.ProjectPath(projectRoot, cfg.Provider.InstallRoot), projectRoot, dryRun), nil - case "docker-dind": - hostGateway := config.DockerConfigNeedsHostGateway(cfg.Docker) - environment := map[string]string{ - "HTTP_PROXY": cfg.Docker.HTTPProxy, - "HTTPS_PROXY": cfg.Docker.HTTPSProxy, - "NO_PROXY": cfg.Docker.NoProxy, - } - return dockerdindprovider.NewWithOptions("", cfg.Provider.Platform, hostGateway, environment, dryRun), nil - default: - return nil, provider.UnsupportedTypeError(cfg.Provider.Type) - } -} - func interruptContext() context.Context { ctx, cancel := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) go func() { @@ -557,6 +681,7 @@ Commands: ephemeral-action-runner ephemeral-action-runner start [--instances N] [--config .local/config.yml] ephemeral-action-runner init + ephemeral-action-runner image update [--config .local/config.yml] ephemeral-action-runner image update-upstream [--config .local/config.yml] ephemeral-action-runner image build [--replace] [--update-upstream] ephemeral-action-runner image refresh-scripts @@ -568,6 +693,8 @@ Commands: ephemeral-action-runner logs path ephemeral-action-runner logs list ephemeral-action-runner logs prune [--dry-run] + ephemeral-action-runner storage status [--provider NAME] [--json] + ephemeral-action-runner storage prune [--provider NAME] [--json] [--execute] ephemeral-action-runner version `) } diff --git a/cmd/ephemeral-action-runner/provider_test.go b/cmd/ephemeral-action-runner/provider_test.go index 7ab3206..1e21637 100644 --- a/cmd/ephemeral-action-runner/provider_test.go +++ b/cmd/ephemeral-action-runner/provider_test.go @@ -1,38 +1,17 @@ package main import ( + "strings" "testing" "github.com/solutionforest/ephemeral-action-runner/internal/config" - dockerdindprovider "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockerdind" + "github.com/solutionforest/ephemeral-action-runner/internal/pool" ) -func TestNewProviderWiresDockerDaemonProxy(t *testing.T) { - cfg := config.Default() - cfg.Provider.Type = "docker-dind" - cfg.Provider.Platform = "linux/amd64" - cfg.Docker.HTTPProxy = "http://host.docker.internal:3128" - cfg.Docker.HTTPSProxy = "http://host.docker.internal:3128" - cfg.Docker.NoProxy = "localhost,127.0.0.1" - - created, err := newProvider(cfg, t.TempDir(), false) - if err != nil { - t.Fatal(err) - } - dind, ok := created.(*dockerdindprovider.Provider) - if !ok { - t.Fatalf("newProvider() type = %T, want Docker-DinD provider", created) - } - if !dind.HostGateway { - t.Fatal("host.docker.internal proxy did not enable host gateway") - } - for key, want := range map[string]string{ - "HTTP_PROXY": cfg.Docker.HTTPProxy, - "HTTPS_PROXY": cfg.Docker.HTTPSProxy, - "NO_PROXY": cfg.Docker.NoProxy, - } { - if got := dind.Environment[key]; got != want { - t.Errorf("provider environment %s = %q, want %q", key, got, want) - } +func TestDockerSandboxesSourceMaintenanceCommandsAreRejectedClearly(t *testing.T) { + manager := &pool.Manager{Config: config.Config{Provider: config.ProviderConfig{Type: "docker-sandboxes"}}} + err := rejectDockerSandboxesImageCommand(manager, "image update-upstream") + if err == nil || !strings.Contains(err.Error(), "not applicable") || !strings.Contains(err.Error(), "image build") { + t.Fatalf("image command rejection = %v", err) } } diff --git a/cmd/ephemeral-action-runner/start.go b/cmd/ephemeral-action-runner/start.go index a22efe8..9a98905 100644 --- a/cmd/ephemeral-action-runner/start.go +++ b/cmd/ephemeral-action-runner/start.go @@ -1,6 +1,7 @@ package main import ( + "bufio" "context" "errors" "flag" @@ -11,10 +12,12 @@ import ( "time" "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/invocation" "github.com/solutionforest/ephemeral-action-runner/internal/pool" ) type starterManager interface { + PreflightRunnerGroup(context.Context) error EnsureImage(context.Context) error RunPool(context.Context, pool.RunOptions) error } @@ -36,6 +39,10 @@ type closingStarterManager interface { Close() error } +type storageAdmissionConfiguringStarterManager interface { + ConfigureStorageAdmissionOverride(bool, string) +} + type starterManagerFactory func(configPath, projectRoot string, dryRun bool, githubEnabled bool) (starterManager, error) var newStarterManager starterManagerFactory = func(configPath, projectRoot string, dryRun bool, githubEnabled bool) (starterManager, error) { @@ -43,18 +50,20 @@ var newStarterManager starterManagerFactory = func(configPath, projectRoot strin } type startOptions struct { - Context context.Context - ProjectRoot string - ConfigPath string - DryRun bool - Instances int - Register bool - KeepOnExit bool - ReplaceCompleted bool - MonitorInterval time.Duration - In io.Reader - Out io.Writer - ManagerFactory starterManagerFactory + Context context.Context + ProjectRoot string + ConfigPath string + DryRun bool + Instances int + Register bool + KeepOnExit bool + ReplaceCompleted bool + MonitorInterval time.Duration + AllowInsufficientStorage bool + StorageOverrideCommand string + In io.Reader + Out io.Writer + ManagerFactory starterManagerFactory } func runStart(args []string) error { @@ -65,6 +74,7 @@ func runStart(args []string) error { keepOnExit := fs.Bool("keep-on-exit", false, "leave prefixed instances and GitHub runners running when interrupted") replaceCompleted := fs.Bool("replace-completed", true, "replace an instance when its ephemeral runner exits after a job") monitorInterval := fs.Duration("monitor-interval", 15*time.Second, "interval for runner liveness checks") + allowInsufficientStorage := fs.Bool("allow-insufficient-storage", false, "continue this invocation after storage-only admission warnings") if err := fs.Parse(args); err != nil { return err } @@ -72,18 +82,20 @@ func runStart(args []string) error { return fmt.Errorf("--instances must be 1 or greater") } return runStartWithOptions(startOptions{ - Context: interruptContext(), - ProjectRoot: *common.projectRoot, - ConfigPath: *common.configPath, - DryRun: *common.dryRun, - Instances: *instances, - Register: *register, - KeepOnExit: *keepOnExit, - ReplaceCompleted: *replaceCompleted, - MonitorInterval: *monitorInterval, - In: os.Stdin, - Out: os.Stdout, - ManagerFactory: newStarterManager, + Context: interruptContext(), + ProjectRoot: *common.projectRoot, + ConfigPath: *common.configPath, + DryRun: *common.dryRun, + Instances: *instances, + Register: *register, + KeepOnExit: *keepOnExit, + ReplaceCompleted: *replaceCompleted, + MonitorInterval: *monitorInterval, + AllowInsufficientStorage: *allowInsufficientStorage, + StorageOverrideCommand: matchingStartCommand(appendStorageOverride(args)), + In: os.Stdin, + Out: os.Stdout, + ManagerFactory: newStarterManager, }) } @@ -107,7 +119,8 @@ func runStartWithOptions(opts startOptions) (err error) { if err != nil { return err } - configPath, err := ensureConfigForStart(startOptions{ + configPath, startNow, err := ensureConfigForStart(startOptions{ + Context: opts.Context, ProjectRoot: projectRoot, ConfigPath: opts.ConfigPath, In: opts.In, @@ -116,6 +129,9 @@ func runStartWithOptions(opts startOptions) (err error) { if err != nil { return err } + if !startNow { + return nil + } manager, err := opts.ManagerFactory(configPath, projectRoot, opts.DryRun, opts.Register) if err != nil { return err @@ -123,6 +139,13 @@ func runStartWithOptions(opts startOptions) (err error) { if closingManager, ok := manager.(closingStarterManager); ok { defer closingManager.Close() } + if configuringManager, ok := manager.(storageAdmissionConfiguringStarterManager); ok { + overrideCommand := opts.StorageOverrideCommand + if overrideCommand == "" { + overrideCommand = matchingStartCommand([]string{"--allow-insufficient-storage"}) + } + configuringManager.ConfigureStorageAdmissionOverride(opts.AllowInsufficientStorage, overrideCommand) + } if timingManager, ok := manager.(startupTimingStarterManager); ok { if _, err := timingManager.StartStartupTiming(); err != nil { return fmt.Errorf("start startup timing log: %w", err) @@ -131,6 +154,12 @@ func runStartWithOptions(opts startOptions) (err error) { timingManager.FinishStartupTiming(err) }() } + if opts.Register { + fmt.Fprintf(opts.Out, "Checking GitHub runner-group security policy for %s\n", configPath) + if err = manager.PreflightRunnerGroup(opts.Context); err != nil { + return err + } + } poolLockHeld := false if lockingManager, ok := manager.(poolLockingStarterManager); ok { controllerLock, err := lockingManager.AcquirePoolControllerLock() @@ -153,14 +182,18 @@ func runStartWithOptions(opts startOptions) (err error) { hostTrustLockHeld = true } } - fmt.Fprintf(opts.Out, "Ensuring runner image is current for %s\n", configPath) + fmt.Fprintf(opts.Out, "Ensuring the runner image or sandbox template is current for %s\n", configPath) if err = manager.EnsureImage(opts.Context); err != nil { return err } + stopGuidance := "Press Ctrl-C once to stop, then wait for cleanup to finish before closing this window." + if opts.KeepOnExit { + stopGuidance = "Press Ctrl-C once to stop; --keep-on-exit will leave owned runner resources running." + } if opts.Instances > 0 { - fmt.Fprintf(opts.Out, "Starting EPAR pool with %d instance(s). Press Ctrl-C to stop; cleanup is enabled by default.\n", opts.Instances) + fmt.Fprintf(opts.Out, "Starting EPAR pool with %d instance(s). %s\n", opts.Instances, stopGuidance) } else { - fmt.Fprintf(opts.Out, "Starting EPAR pool using pool.instances from config. Press Ctrl-C to stop; cleanup is enabled by default.\n") + fmt.Fprintf(opts.Out, "Starting EPAR pool using pool.instances from config. %s\n", stopGuidance) } err = manager.RunPool(opts.Context, pool.RunOptions{ Instances: opts.Instances, @@ -174,31 +207,61 @@ func runStartWithOptions(opts startOptions) (err error) { return err } -func ensureConfigForStart(opts startOptions) (string, error) { +func appendStorageOverride(args []string) []string { + result := append([]string(nil), args...) + for _, arg := range result { + if arg == "--allow-insufficient-storage" || arg == "--allow-insufficient-storage=true" { + return result + } + } + return append(result, "--allow-insufficient-storage") +} + +func matchingStartCommand(args []string) string { + if os.Getenv(invocation.Environment) == "start" { + return invocation.Command(args...) + } + return invocation.Command(append([]string{"start"}, args...)...) +} + +func ensureConfigForStart(opts startOptions) (string, bool, error) { path, exists, err := resolveStartConfigPath(opts.ProjectRoot, opts.ConfigPath) if err != nil { - return "", err + return "", false, err } if exists { - return path, nil + return path, true, nil } if path == "" { path = filepath.Join(opts.ProjectRoot, ".local", "config.yml") } if !stdinIsInteractive() { - return "", fmt.Errorf("no EPAR config found; run %s init from the EPAR directory, or pass --config after creating a config. See README.md and docs/github-app.md for GitHub App setup", binaryName) + return "", false, fmt.Errorf("no EPAR config found; run %s init from the EPAR directory, or pass --config after creating a config. See README.md and docs/github-app.md for GitHub App setup", binaryName) } fmt.Fprintf(opts.Out, "No EPAR config found. Starting first-run setup.\n\n") + reader := bufio.NewReader(opts.In) if err := runInitWithOptions(initOptions{ - ProjectRoot: projectRootOrCwd(opts.ProjectRoot), - ConfigPath: path, - In: opts.In, - Out: opts.Out, + Context: opts.Context, + ProjectRoot: projectRootOrCwd(opts.ProjectRoot), + ConfigPath: path, + EmbeddedInStart: true, + In: opts.In, + Reader: reader, + Out: opts.Out, }); err != nil { - return "", err + return "", false, err + } + fmt.Fprintln(opts.Out, "") + startNow, err := promptYesNo(opts.Out, reader, fmt.Sprintf("Start runners now? Choose No to exit and review %s", path), true) + if err != nil { + return "", false, err + } + if !startNow { + fmt.Fprintf(opts.Out, "\nConfig saved at %s. Exiting before runner startup.\nReview the config, then run %s when ready.\n", path, invocation.Command()) + return path, false, nil } fmt.Fprintf(opts.Out, "\nContinuing with %s\n", path) - return path, nil + return path, true, nil } func resolveStartConfigPath(projectRoot, explicit string) (string, bool, error) { diff --git a/cmd/ephemeral-action-runner/start_test.go b/cmd/ephemeral-action-runner/start_test.go index 3bc1d56..58463c8 100644 --- a/cmd/ephemeral-action-runner/start_test.go +++ b/cmd/ephemeral-action-runner/start_test.go @@ -3,6 +3,7 @@ package main import ( "bytes" "context" + "errors" "os" "path/filepath" "strings" @@ -11,6 +12,7 @@ import ( "github.com/solutionforest/ephemeral-action-runner/internal/config" "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" "github.com/solutionforest/ephemeral-action-runner/internal/pool" + sandboxpromotion "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/promotion" ) func TestNoArgsRoutesToStart(t *testing.T) { @@ -55,12 +57,13 @@ func TestStartPropagatesConfigAndInstances(t *testing.T) { } fake := &fakeStarterManager{} var gotPath string + var out bytes.Buffer err := runStartWithOptions(startOptions{ Context: context.Background(), ProjectRoot: dir, ConfigPath: "custom.yml", Instances: 3, - Out: &bytes.Buffer{}, + Out: &out, ManagerFactory: func(path, _ string, _ bool, _ bool) (starterManager, error) { gotPath = path return fake, nil @@ -75,11 +78,51 @@ func TestStartPropagatesConfigAndInstances(t *testing.T) { if fake.runOptions.Instances != 3 { t.Fatalf("instances = %d, want 3", fake.runOptions.Instances) } + if !strings.Contains(out.String(), "Press Ctrl-C once to stop, then wait for cleanup to finish before closing this window.") { + t.Fatalf("start guidance = %q", out.String()) + } + if strings.Contains(out.String(), "Start runners now?") { + t.Fatalf("existing config unexpectedly triggered the new-config start prompt:\n%s", out.String()) + } +} + +func TestStartConfiguresOneInvocationStorageOverride(t *testing.T) { + dir := t.TempDir() + configPath := filepath.Join(dir, "config.yml") + if err := os.WriteFile(configPath, []byte("config"), 0600); err != nil { + t.Fatal(err) + } + fake := &fakeStarterManager{} + err := runStartWithOptions(startOptions{ + Context: context.Background(), + ProjectRoot: dir, + ConfigPath: configPath, + AllowInsufficientStorage: true, + StorageOverrideCommand: "./start --allow-insufficient-storage", + Out: &bytes.Buffer{}, + ManagerFactory: func(string, string, bool, bool) (starterManager, error) { + return fake, nil + }, + }) + if err != nil { + t.Fatal(err) + } + if !fake.allowStorage || fake.overrideHint != "./start --allow-insufficient-storage" { + t.Fatalf("storage override = allow %t hint %q", fake.allowStorage, fake.overrideHint) + } +} + +func TestMatchingStartCommandPreservesWrapperEntryPoint(t *testing.T) { + t.Setenv("EPAR_INVOCATION", "start") + if got, want := matchingStartCommand([]string{"--allow-insufficient-storage"}), "./start --allow-insufficient-storage"; got != want { + t.Fatalf("matchingStartCommand() = %q, want %q", got, want) + } } func TestStartInteractiveMissingConfigRunsInitAndContinues(t *testing.T) { dir := t.TempDir() stubNoWSL2(t) + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) oldInteractive := stdinIsInteractive oldDocker := dockerAvailable oldResolveHostTrust := initResolveHostTrust @@ -99,7 +142,7 @@ func TestStartInteractiveMissingConfigRunsInitAndContinues(t *testing.T) { err := runStartWithOptions(startOptions{ Context: context.Background(), ProjectRoot: dir, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n\n\nn\n\n\n\n\n"), Out: &out, ManagerFactory: func(path, _ string, _ bool, _ bool) (starterManager, error) { if path != filepath.Join(dir, ".local", "config.yml") { @@ -114,14 +157,62 @@ func TestStartInteractiveMissingConfigRunsInitAndContinues(t *testing.T) { if _, err := os.Stat(filepath.Join(dir, ".local", "config.yml")); err != nil { t.Fatalf("config was not created: %v", err) } - if !strings.Contains(out.String(), "Continuing with") { - t.Fatalf("output missing continuation message:\n%s", out.String()) + for _, want := range []string{"Start runners now?", "Continuing with"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("output missing %q:\n%s", want, out.String()) + } } if fake.ensureCalls != 1 || fake.runCalls != 1 { t.Fatalf("ensure/run calls = %d/%d, want 1/1", fake.ensureCalls, fake.runCalls) } } +func TestStartInteractiveMissingConfigCanExitToReview(t *testing.T) { + dir := t.TempDir() + stubNoWSL2(t) + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + oldInteractive := stdinIsInteractive + oldDocker := dockerAvailable + oldResolveHostTrust := initResolveHostTrust + t.Cleanup(func() { + stdinIsInteractive = oldInteractive + dockerAvailable = oldDocker + initResolveHostTrust = oldResolveHostTrust + }) + stdinIsInteractive = func() bool { return true } + dockerAvailable = func(context.Context) error { return nil } + initResolveHostTrust = func(context.Context, hosttrust.Options) (hosttrust.Snapshot, error) { + return hosttrust.Snapshot{}, nil + } + + var out bytes.Buffer + err := runStartWithOptions(startOptions{ + Context: context.Background(), + ProjectRoot: dir, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n1\n\nn\n\n\nn\n\n\nn\n"), + Out: &out, + ManagerFactory: func(string, string, bool, bool) (starterManager, error) { + t.Fatal("manager factory should not run after choosing to review the new config") + return nil, nil + }, + }) + if err != nil { + t.Fatal(err) + } + configPath := filepath.Join(dir, ".local", "config.yml") + if _, err := os.Stat(configPath); err != nil { + t.Fatalf("config was not created: %v", err) + } + for _, want := range []string{"Start runners now?", "Config saved at " + configPath, "Exiting before runner startup", "Review the config"} { + if !strings.Contains(out.String(), want) { + t.Fatalf("review exit output omitted %q:\n%s", want, out.String()) + } + } + if strings.Contains(out.String(), "Continuing with") { + t.Fatalf("review exit unexpectedly continued startup:\n%s", out.String()) + } +} + func TestStartInteractiveMissingConfigCanSelectWSL2(t *testing.T) { dir := t.TempDir() stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) @@ -140,7 +231,7 @@ func TestStartInteractiveMissingConfigCanSelectWSL2(t *testing.T) { err := runStartWithOptions(startOptions{ Context: context.Background(), ProjectRoot: dir, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n2\n\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n3\n\n"), Out: &out, ManagerFactory: func(path, _ string, _ bool, _ bool) (starterManager, error) { if path != filepath.Join(dir, ".local", "config.yml") { @@ -167,6 +258,70 @@ func TestStartInteractiveMissingConfigCanSelectWSL2(t *testing.T) { } } +func TestStartInteractiveMissingConfigCanSelectDockerSandboxes(t *testing.T) { + dir := t.TempDir() + stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) + stubNoWSL2(t) + policyFingerprint := "sha256:" + strings.Repeat("b", 64) + stubInitDockerSandboxesSetup(t, sandboxpromotion.WindowsAMD64, initDockerSandboxesDiscovery{ + Templates: []initDockerSandboxesTemplate{{ + Reference: "docker.io/library/epar-docker-sandboxes-catthehacker-full:preview", + Digest: "sha256:" + strings.Repeat("a", 64), + CacheID: strings.Repeat("a", 12), + Platform: "linux/amd64", + Size: 8 << 30, + }}, + PolicyFingerprint: policyFingerprint, + }, nil) + oldInteractive := stdinIsInteractive + oldDocker := dockerAvailable + oldResolveHostTrust := initResolveHostTrust + t.Cleanup(func() { + stdinIsInteractive = oldInteractive + dockerAvailable = oldDocker + initResolveHostTrust = oldResolveHostTrust + }) + stdinIsInteractive = func() bool { return true } + dockerAvailable = func(context.Context) error { return nil } + initResolveHostTrust = func(context.Context, hosttrust.Options) (hosttrust.Snapshot, error) { + return hosttrust.Snapshot{}, nil + } + + fake := &fakeStarterManager{} + var out bytes.Buffer + err := runStartWithOptions(startOptions{ + Context: context.Background(), + ProjectRoot: dir, + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n1\n\nn\n\n\n\n\n\n"), + Out: &out, + ManagerFactory: func(path, _ string, _ bool, _ bool) (starterManager, error) { + if path != filepath.Join(dir, ".local", "config.yml") { + t.Fatalf("config path = %q", path) + } + return fake, nil + }, + }) + if err != nil { + t.Fatal(err) + } + cfg, err := config.Load(filepath.Join(dir, ".local", "config.yml")) + if err != nil { + t.Fatal(err) + } + if got, want := cfg.Provider.Type, "docker-sandboxes"; got != want { + t.Fatalf("provider.type = %q, want %q", got, want) + } + if got, want := cfg.DockerSandboxes.PolicyGeneration, policyFingerprint; got != want { + t.Fatalf("dockerSandboxes.policyGeneration = %q, want %q", got, want) + } + if !strings.Contains(out.String(), "Docker Sandboxes — recommended") || !strings.Contains(out.String(), "Continuing with") { + t.Fatalf("output did not include capability-ready Docker Sandboxes selection and start continuation:\n%s", out.String()) + } + if fake.ensureCalls != 1 || fake.runCalls != 1 { + t.Fatalf("ensure/run calls = %d/%d, want 1/1", fake.ensureCalls, fake.runCalls) + } +} + func TestStartInteractiveMissingConfigCanSelectTartWithoutDocker(t *testing.T) { dir := t.TempDir() stubInitHostAndRandom(t, "Build Box 01", []byte{0xa4, 0xf9, 0xc2}) @@ -179,8 +334,7 @@ func TestStartInteractiveMissingConfigCanSelectTartWithoutDocker(t *testing.T) { }) stdinIsInteractive = func() bool { return true } dockerAvailable = func(context.Context) error { - t.Fatal("Docker availability should not be checked for Tart") - return nil + return errors.New("Docker is unavailable on this Mac") } fake := &fakeStarterManager{} @@ -188,7 +342,7 @@ func TestStartInteractiveMissingConfigCanSelectTartWithoutDocker(t *testing.T) { err := runStartWithOptions(startOptions{ Context: context.Background(), ProjectRoot: dir, - In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n2\n\n"), + In: strings.NewReader("123456\nsolutionforest\n.local/github-app.pem\n1\n4\n\n"), Out: &out, ManagerFactory: func(path, _ string, _ bool, _ bool) (starterManager, error) { if path != filepath.Join(dir, ".local", "config.yml") { @@ -244,18 +398,81 @@ func TestStartRejectsNonPositiveInstancesOverride(t *testing.T) { } } +func TestStartPreflightsBeforeImageAndPool(t *testing.T) { + dir := t.TempDir() + configPath := filepath.Join(dir, "config.yml") + if err := os.WriteFile(configPath, []byte("config"), 0600); err != nil { + t.Fatal(err) + } + fake := &fakeStarterManager{} + err := runStartWithOptions(startOptions{ + Context: context.Background(), + ProjectRoot: dir, + ConfigPath: configPath, + Register: true, + Out: &bytes.Buffer{}, + ManagerFactory: func(string, string, bool, bool) (starterManager, error) { + return fake, nil + }, + }) + if err != nil { + t.Fatal(err) + } + if got := strings.Join(fake.calls, ","); got != "preflight,image,pool" { + t.Fatalf("call order = %q, want preflight,image,pool", got) + } + + fake = &fakeStarterManager{preflightErr: errors.New("unsafe group")} + err = runStartWithOptions(startOptions{ + Context: context.Background(), + ProjectRoot: dir, + ConfigPath: configPath, + Register: true, + AllowInsufficientStorage: true, + Out: &bytes.Buffer{}, + ManagerFactory: func(string, string, bool, bool) (starterManager, error) { + return fake, nil + }, + }) + if err == nil || !strings.Contains(err.Error(), "unsafe group") { + t.Fatalf("start error = %v, want preflight failure", err) + } + if got := strings.Join(fake.calls, ","); got != "preflight" { + t.Fatalf("call order after rejection = %q, want preflight only", got) + } + if !fake.allowStorage { + t.Fatal("storage override was not configured before the non-storage safety check") + } +} + type fakeStarterManager struct { - ensureCalls int - runCalls int - runOptions pool.RunOptions + preflightErr error + ensureCalls int + runCalls int + runOptions pool.RunOptions + calls []string + allowStorage bool + overrideHint string +} + +func (m *fakeStarterManager) ConfigureStorageAdmissionOverride(allow bool, command string) { + m.allowStorage = allow + m.overrideHint = command +} + +func (m *fakeStarterManager) PreflightRunnerGroup(context.Context) error { + m.calls = append(m.calls, "preflight") + return m.preflightErr } func (m *fakeStarterManager) EnsureImage(context.Context) error { + m.calls = append(m.calls, "image") m.ensureCalls++ return nil } func (m *fakeStarterManager) RunPool(_ context.Context, opts pool.RunOptions) error { + m.calls = append(m.calls, "pool") m.runCalls++ m.runOptions = opts return nil diff --git a/cmd/ephemeral-action-runner/storage.go b/cmd/ephemeral-action-runner/storage.go new file mode 100644 index 0000000..d809efc --- /dev/null +++ b/cmd/ephemeral-action-runner/storage.go @@ -0,0 +1,456 @@ +package main + +import ( + "context" + "encoding/json" + "errors" + "flag" + "fmt" + "os" + "path/filepath" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + artifactimage "github.com/solutionforest/ephemeral-action-runner/internal/image" + "github.com/solutionforest/ephemeral-action-runner/internal/invocation" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + providerregistry "github.com/solutionforest/ephemeral-action-runner/internal/provider/registry" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" + "github.com/solutionforest/ephemeral-action-runner/internal/storage/inventory" +) + +type storageCommandReport struct { + Inventory storageInventorySummary `json:"inventory"` + Plan storage.Plan `json:"plan"` + Execution *storage.ExecutionReport `json:"execution,omitempty"` + Legacy bool `json:"legacy,omitempty"` +} + +type storageInventorySummary struct { + ProjectRoot string `json:"projectRoot"` + Provider string `json:"provider,omitempty"` + Warnings []string `json:"warnings,omitempty"` +} + +func runStorage(args []string) error { + if len(args) == 0 { + return fmt.Errorf("storage requires subcommand: status or prune") + } + subcommand := args[0] + if subcommand == "effective-go-cache-limit" { + return runEffectiveGoCacheLimit(args[1:]) + } + if subcommand != "status" && subcommand != "prune" { + return fmt.Errorf("unknown storage subcommand %q", subcommand) + } + fs := flag.NewFlagSet("storage "+subcommand, flag.ContinueOnError) + cwd, _ := os.Getwd() + projectRootFlag := fs.String("project-root", cwd, "project root containing EPAR state and artifacts") + configPathFlag := fs.String("config", "", "config file path; defaults to EPAR_CONFIG or .local/config.yml when present") + providerFlag := fs.String("provider", "", "limit provider-specific inventory") + jsonOutput := fs.Bool("json", false, "write the complete storage report as JSON") + execute := fs.Bool("execute", false, "execute the exact policy-selected prune plan") + legacy := fs.Bool("legacy", false, "include prefix-era EPAR resources in an operator-approved exact preview") + approvedPlan := fs.String("plan", "", "approved legacy preview plan hash") + if err := fs.Parse(args[1:]); err != nil { + return err + } + if fs.NArg() != 0 { + return fmt.Errorf("storage %s does not accept positional arguments", subcommand) + } + if subcommand == "status" && *execute { + return fmt.Errorf("storage status does not support --execute") + } + if subcommand == "status" && (*legacy || strings.TrimSpace(*approvedPlan) != "") { + return fmt.Errorf("storage status does not support --legacy or --plan") + } + if !*legacy && strings.TrimSpace(*approvedPlan) != "" { + return fmt.Errorf("--plan is valid only with storage prune --legacy --execute") + } + if *legacy && *execute && strings.TrimSpace(*approvedPlan) == "" { + return fmt.Errorf("legacy cleanup requires the exact preview hash: storage prune --legacy --execute --plan ") + } + if *providerFlag != "" { + if _, found := providerregistry.DescriptorFor(*providerFlag); !found { + return provider.UnsupportedTypeError(*providerFlag) + } + } + + projectRoot, cfg, configPath, configTime, err := loadStorageConfig(*projectRootFlag, *configPathFlag) + if err != nil { + return err + } + now := time.Now().UTC() + if configPath != "" { + if err := importNativeBootstrapAcquisition(projectRoot, configPath, now); err != nil { + return err + } + } + var selections []inventory.TemplateSelection + activeTemplateRootDisk := "" + legacyTemplateReceiptProtected := false + staleTemplateReceiptWarning := "" + if cfg.Provider.Type == "docker-sandboxes" { + artifact, metadataSHA256, activatedAt, receiptErr := artifactimage.LoadDockerSandboxesReceiptForConfig(projectRoot, configPath) + if errors.Is(receiptErr, os.ErrNotExist) { + artifact, metadataSHA256, activatedAt, receiptErr = artifactimage.LoadDockerSandboxesReceipt(projectRoot) + legacyTemplateReceiptProtected = receiptErr == nil + } + if receiptErr == nil { + activeTemplateRootDisk = artifact.RootDisk + selections = append(selections, inventory.TemplateSelection{ + Platform: artifact.Platform, + Tag: artifact.Reference, + TemplateDigest: artifact.Digest, + MetadataSHA256: metadataSHA256, + ActivatedAt: activatedAt, + }) + } else if !errors.Is(receiptErr, os.ErrNotExist) { + staleTemplateReceiptWarning = fmt.Sprintf("The unpublished Docker Sandboxes receipt is stale and is not treated as active ownership evidence: %v. Normal startup will rebuild and replace it after exact Sandbox-cache readback.", receiptErr) + } + } + currentExecutable, _ := os.Executable() + configuredFiles := configuredStorageFiles(cfg, projectRoot, configTime) + snapshot, err := inventory.Collect(inventory.Options{ + ProjectRoot: projectRoot, + Provider: *providerFlag, + Now: now, + LogsRoot: config.ProjectPath(projectRoot, cfg.Logging.Directory), + NativeRoot: filepath.Join(projectRoot, ".local", "bin"), + TemplateRoot: filepath.Join(projectRoot, "work", "template-builds", "docker-sandboxes"), + CurrentExecutable: currentExecutable, + ConfiguredTemplates: selections, + ConfiguredFiles: configuredFiles, + }) + if err != nil { + return err + } + if legacyTemplateReceiptProtected { + snapshot.Warnings = append(snapshot.Warnings, "The exact template in the retired shared Docker Sandboxes receipt remains protected for legacy cleanup; normal startup still requires regeneration into per-config state.") + } + if staleTemplateReceiptWarning != "" { + snapshot.Warnings = append(snapshot.Warnings, staleTemplateReceiptWarning) + } + collectExternalStorage(&snapshot, *providerFlag, configPath) + protectConfiguredSandboxTemplates(&snapshot, selections) + catalogValue, catalogErr := addCatalogStorage(&snapshot, *providerFlag, now) + if catalogErr != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("EPAR host resource catalog is unavailable; catalog-owned resources remain report-only: %v", catalogErr)) + } + if *legacy { + selectLegacyStorage(&snapshot, catalogValue, now) + snapshot.Warnings = append(snapshot.Warnings, "Legacy preview is limited to resources visible on this host. Unregistered old EPAR checkouts and their intended references cannot be inferred.") + } + if cfg.Storage.AutomaticHousekeeping == config.StorageHousekeepingDisabled { + for index := range snapshot.Artifacts { + snapshot.Artifacts[index].Protections = append(snapshot.Artifacts[index].Protections, storage.Protection{ + Kind: storage.ProtectionOperator, + Detail: "storage.automaticHousekeeping is disabled", + }) + } + } + policy, minimumFree, err := storagePolicyFromConfig(cfg.Storage) + if err != nil { + return err + } + storageProvider := cfg.Provider.Type + if *providerFlag != "" { + storageProvider = *providerFlag + } + storageConfig := cfg + storageConfig.Provider.Type = storageProvider + effectiveMinimumFree, err := config.EffectiveMinimumFreeBytes(storageConfig) + if err != nil { + return err + } + if effectiveMinimumFree > minimumFree { + minimumFree = effectiveMinimumFree + } + requirements := []storage.Requirement{{ + ID: "controller-bootstrap", + Provider: *providerFlag, + SurfaceID: inventory.ProjectSurfaceID, + MinimumFreeBytes: minimumFree, + }} + providerRuntime, runtimeErr := providerregistry.New(storageConfig, projectRoot, true) + if runtimeErr != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Provider storage surfaces are unavailable: %v", runtimeErr)) + } else { + providerSnapshot, snapshotErr := providerRuntime.Storage.StorageSnapshot(context.Background(), provider.StorageRequest{ + Operation: "storage-status", + Now: now, + MinimumFreeBytes: minimumFree, + }) + if snapshotErr != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Provider storage surfaces are unavailable: %v", snapshotErr)) + } else { + snapshot.Surfaces = append(snapshot.Surfaces, providerSnapshot.Surfaces...) + snapshot.Artifacts = append(snapshot.Artifacts, providerSnapshot.Artifacts...) + requirements = append(requirements, providerSnapshot.Requirements...) + } + } + if storageProvider == "docker-sandboxes" { + rootDisk := cfg.DockerSandboxes.RootDisk + if rootDisk == config.DockerSandboxesAutomaticRootDisk { + rootDisk = activeTemplateRootDisk + } + appendLogicalSurface := func(id, configured string) { + if configured == "" && id == "docker-sandboxes-root-logical" { + snapshot.Surfaces = append(snapshot.Surfaces, storage.Surface{ + ID: id, + Provider: "docker-sandboxes", + Kind: storage.SurfaceExternal, + Classification: "logical", + Sparse: true, + Confidence: "pending-artifact-resolution", + Advisory: true, + Capacity: storage.Capacity{ObservedAt: now}, + }) + return + } + parsed, parseErr := config.ParseByteSize(configured) + if parseErr != nil || parsed <= 0 { + return + } + snapshot.Surfaces = append(snapshot.Surfaces, storage.Surface{ + ID: id, + Provider: "docker-sandboxes", + Kind: storage.SurfaceExternal, + Classification: "logical", + Sparse: true, + VirtualMaximumBytes: uint64(parsed), + Confidence: "configured-logical-limit", + Advisory: true, + Capacity: storage.Capacity{ObservedAt: now}, + }) + } + appendLogicalSurface("docker-sandboxes-root-logical", rootDisk) + appendLogicalSurface("docker-sandboxes-inner-docker-logical", cfg.DockerSandboxes.DockerDisk) + } + plan, err := storage.Preview(snapshot.PreviewRequest(policy, requirements)) + if err != nil { + return err + } + report := storageCommandReport{ + Inventory: storageInventorySummary{ + ProjectRoot: projectRoot, + Provider: *providerFlag, + Warnings: snapshot.Warnings, + }, + Plan: plan, + Legacy: *legacy, + } + + if subcommand == "prune" && *execute { + for _, decision := range plan.Decisions { + if decision.Action == storage.ActionRemove { + fmt.Fprintf(os.Stderr, "storage prune exact identity=%s kind=%s target=%s bytes=%d\n", decision.Artifact.Target.Identity, decision.Artifact.Target.Kind, decision.Artifact.Target.Locator, decision.Artifact.SizeBytes) + } + } + if plan.RemovalCount > 0 { + executor, err := newHostStorageExecutor(projectRoot) + if err != nil { + return err + } + planApproval := plan.Hash + if *legacy { + planApproval = strings.TrimSpace(*approvedPlan) + } + execution, err := storage.Execute(context.Background(), plan, planApproval, executor) + report.Execution = &execution + if catalogErr == nil { + if updateErr := removeExecutedCatalogEntries(execution, time.Now().UTC()); updateErr != nil { + report.Inventory.Warnings = append(report.Inventory.Warnings, fmt.Sprintf("exact removals completed but the host catalog could not be compacted: %v", updateErr)) + } + } + if err != nil { + if *jsonOutput { + _ = writeStorageJSON(report) + } + return err + } + } else { + execution := storage.ExecutionReport{PlanHash: plan.Hash} + report.Execution = &execution + } + } + if *jsonOutput { + return writeStorageJSON(report) + } + printStorageReport(subcommand, report) + if configPath == "" { + fmt.Fprintln(os.Stdout, "Configuration: defaults (no config file found)") + } else { + fmt.Fprintln(os.Stdout, "Configuration:", configPath) + } + return nil +} + +func runEffectiveGoCacheLimit(args []string) error { + fs := flag.NewFlagSet("storage effective-go-cache-limit", flag.ContinueOnError) + cwd, _ := os.Getwd() + projectRootFlag := fs.String("project-root", cwd, "project root containing EPAR configuration") + configPathFlag := fs.String("config", "", "config file path") + if err := fs.Parse(args); err != nil { + return err + } + if fs.NArg() != 0 { + return fmt.Errorf("storage effective-go-cache-limit does not accept positional arguments") + } + projectRoot, cfg, configPath, _, err := loadStorageConfig(*projectRootFlag, *configPathFlag) + if err != nil { + return err + } + if configPath != "" { + if err := importNativeBootstrapAcquisition(projectRoot, configPath, time.Now().UTC()); err != nil { + return err + } + } + if err := config.ValidateStorage(cfg.Storage); err != nil { + return err + } + limit, err := config.ParseByteSize(cfg.Storage.GoCacheLimit) + if err != nil { + return err + } + fmt.Fprintln(os.Stdout, uint64(limit)) + return nil +} + +func configuredStorageFiles(cfg config.Config, projectRoot string, configuredAt time.Time) []inventory.ConfiguredFile { + if cfg.Provider.Type != "wsl" || strings.TrimSpace(cfg.Image.OutputImage) == "" { + return nil + } + output := config.ProjectPath(projectRoot, cfg.Image.OutputImage) + files := []inventory.ConfiguredFile{ + {Provider: "wsl", Role: "reusable-image", Path: output, Kind: storage.ArtifactProviderImage, Current: true, ConfiguredAt: configuredAt, ProtectionKind: storage.ProtectionConfiguration, ProtectionDetail: "current reusable WSL image"}, + {Provider: "wsl", Role: "image-manifest", Path: artifactimage.WSLImageManifestPath(output), Kind: storage.ArtifactOther, Current: true, ConfiguredAt: configuredAt, ProtectionKind: storage.ProtectionCertification, ProtectionDetail: "current WSL image manifest"}, + } + if cfg.Image.SourceType == "docker-image" { + rootfs := artifactimage.WSLSourceRootfsPath(output) + files = append(files, + inventory.ConfiguredFile{Provider: "wsl", Role: "source-rootfs-cache", Path: rootfs, Kind: storage.ArtifactProviderCache, Current: true, ConfiguredAt: configuredAt, ProtectionKind: storage.ProtectionLock, ProtectionDetail: "current reusable WSL source rootfs cache"}, + inventory.ConfiguredFile{Provider: "wsl", Role: "source-cache-manifest", Path: artifactimage.SourceCacheManifestPath(rootfs), Kind: storage.ArtifactOther, Current: true, ConfiguredAt: configuredAt, ProtectionKind: storage.ProtectionCertification, ProtectionDetail: "current WSL source cache manifest"}, + ) + } + return files +} + +func loadStorageConfig(projectRoot, explicitConfig string) (string, config.Config, string, time.Time, error) { + absoluteRoot, err := filepath.Abs(projectRoot) + if err != nil { + return "", config.Config{}, "", time.Time{}, err + } + cfg := config.Default() + configPath, err := resolveConfigPath(absoluteRoot, explicitConfig) + if err != nil { + return "", config.Config{}, "", time.Time{}, err + } + var configTime time.Time + if configPath != "" { + cfg, err = config.Load(configPath) + if err != nil { + return "", config.Config{}, "", time.Time{}, err + } + if err := config.ValidateStorage(cfg.Storage); err != nil { + return "", config.Config{}, "", time.Time{}, err + } + if info, statErr := os.Stat(configPath); statErr == nil { + configTime = info.ModTime().UTC() + } + } + return absoluteRoot, cfg, configPath, configTime, nil +} + +func storagePolicyFromConfig(cfg config.StorageConfig) (storage.Policy, uint64, error) { + grace, err := time.ParseDuration(cfg.GracePeriod) + if err != nil { + return storage.Policy{}, 0, err + } + minimumFree, err := config.ParseByteSize(cfg.MinimumFree) + if err != nil { + return storage.Policy{}, 0, err + } + buildLimit, err := config.ParseByteSize(cfg.BuildCacheLimit) + if err != nil { + return storage.Policy{}, 0, err + } + goLimit, err := config.ParseByteSize(cfg.GoCacheLimit) + if err != nil { + return storage.Policy{}, 0, err + } + return storage.Policy{ + GracePeriod: grace, + KeepPrevious: cfg.KeepPrevious, + Budgets: []storage.Budget{ + {Kind: storage.ArtifactBuildKitCache, MaxBytes: uint64(buildLimit)}, + {Kind: storage.ArtifactGoCache, MaxBytes: uint64(goLimit)}, + }, + }, uint64(minimumFree), nil +} + +func writeStorageJSON(report storageCommandReport) error { + encoder := json.NewEncoder(os.Stdout) + encoder.SetIndent("", " ") + return encoder.Encode(report) +} + +func printStorageReport(subcommand string, report storageCommandReport) { + fmt.Fprintf(os.Stdout, "Storage %s plan: %s\n", subcommand, report.Plan.Hash) + for _, surface := range report.Plan.Surfaces { + available := "unknown" + if surface.Capacity.Known { + available = formatStorageBytes(surface.Capacity.AvailableBytes) + } + fmt.Fprintf(os.Stdout, "Surface %s\tprovider=%s\tkind=%s\tclassification=%s\tsparse=%t\tavailable=%s\tallocated=%s\tvirtualMaximum=%s\tconfidence=%s\tauthoritative=%t\tadvisory=%t\tlocation=%s\n", surface.ID, valueOrDash(surface.Provider), surface.Kind, valueOrDash(surface.Classification), surface.Sparse, available, formatStorageOptionalBytes(surface.AllocatedBytes), formatStorageOptionalBytes(surface.VirtualMaximumBytes), valueOrDash(surface.Confidence), surface.AdmissionAuthoritative, surface.Advisory, surface.Location) + } + for _, check := range report.Plan.CapacityChecks { + fmt.Fprintf(os.Stdout, "Capacity %s\tstatus=%s\tavailable=%s\testimated=%s\treserve=%s\trequired=%s\n", check.Requirement.ID, check.Status, formatStorageBytes(check.Capacity.AvailableBytes), formatStorageBytes(check.Requirement.PeakBytes), formatStorageBytes(check.Requirement.MinimumFreeBytes), formatStorageBytes(check.RequiredAvailableBytes)) + } + for _, decision := range report.Plan.Decisions { + fmt.Fprintf(os.Stdout, "Artifact %s\taction=%s\tprovider=%s\tkind=%s\tbytes=%s\tidentity=%s\ttarget=%s\treason=%s\n", decision.Artifact.ID, decision.Action, valueOrDash(decision.Artifact.Provider), decision.Artifact.Kind, formatStorageBytes(decision.Artifact.SizeBytes), valueOrDash(decision.Artifact.Target.Identity), decision.Artifact.Target.Locator, strings.Join(decision.Reasons, ",")) + } + for _, warning := range append(append([]string(nil), report.Inventory.Warnings...), report.Plan.Warnings...) { + fmt.Fprintln(os.Stdout, "Warning:", warning) + } + fmt.Fprintf(os.Stdout, "Summary: removals=%d reclaimable=%s", report.Plan.RemovalCount, formatStorageBytes(report.Plan.ReclaimableBytes)) + if report.Execution != nil { + fmt.Fprintf(os.Stdout, " removed=%d reclaimed=%s", report.Execution.RemovedCount, formatStorageBytes(report.Execution.ReclaimedBytes)) + } + fmt.Fprintln(os.Stdout) + if subcommand == "prune" && report.Execution == nil { + if report.Legacy { + fmt.Fprintf(os.Stdout, "Preview only. Legacy execution requires this exact plan:\n %s\n", invocation.Command("storage", "prune", "--legacy", "--execute", "--plan", report.Plan.Hash)) + } else { + fmt.Fprintln(os.Stdout, "Preview only. Re-run with --execute to apply only the exact identities marked remove.") + } + } +} + +func formatStorageOptionalBytes(value uint64) string { + if value == 0 { + return "-" + } + return formatStorageBytes(value) +} + +func formatStorageBytes(value uint64) string { + const gib = uint64(1 << 30) + const mib = uint64(1 << 20) + switch { + case value >= gib: + return fmt.Sprintf("%.2fGiB", float64(value)/float64(gib)) + case value >= mib: + return fmt.Sprintf("%.2fMiB", float64(value)/float64(mib)) + default: + return fmt.Sprintf("%dB", value) + } +} + +func valueOrDash(value string) string { + if value == "" { + return "-" + } + return value +} diff --git a/cmd/ephemeral-action-runner/storage_bootstrap.go b/cmd/ephemeral-action-runner/storage_bootstrap.go new file mode 100644 index 0000000..962a0b9 --- /dev/null +++ b/cmd/ephemeral-action-runner/storage_bootstrap.go @@ -0,0 +1,163 @@ +package main + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "os" + "os/exec" + "path/filepath" + "strings" + "time" + + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +type nativeBootstrapAcquisition struct { + SchemaVersion int `json:"schemaVersion"` + ProjectRoot string `json:"projectRoot"` + Phase string `json:"phase"` + GoImage string `json:"goImage"` + DevImage string `json:"devImage"` + PreviousGoImageID string `json:"previousGoImageID"` + ResolvedGoImageID string `json:"resolvedGoImageID"` + PreviousDevImageID string `json:"previousDevImageID,omitempty"` + ResolvedDevImageID string `json:"resolvedDevImageID"` + UpdatedAtUTC string `json:"updatedAtUtc,omitempty"` + UpdatedAtUnix int64 `json:"updatedAtUnix,omitempty"` +} + +func importNativeBootstrapAcquisition(projectRoot, configPath string, now time.Time) error { + path := filepath.Join(projectRoot, ".local", "storage", "bootstrap", "native-controller-acquisition.json") + content, err := os.ReadFile(path) + if errors.Is(err, os.ErrNotExist) { + return nil + } + if err != nil { + return fmt.Errorf("read native-controller acquisition journal: %w", err) + } + var record nativeBootstrapAcquisition + if err := json.Unmarshal(content, &record); err != nil { + return fmt.Errorf("decode native-controller acquisition journal: %w", err) + } + if record.SchemaVersion != 1 || record.Phase != "toolchain-built" || record.ResolvedDevImageID == "" { + return fmt.Errorf("native-controller acquisition journal is incomplete at phase %q", record.Phase) + } + backendID, err := currentDockerBackendID() + if err != nil { + return fmt.Errorf("identify Docker backend for native-controller acquisition journal: %w", err) + } + store, err := storagecatalog.Open("") + if err != nil { + return err + } + lockContext, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + backendLock, err := store.AcquireBackendLock(lockContext, backendID) + if err != nil { + return err + } + defer backendLock.Close() + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + configRecord, err := storagecatalog.RegisterConfig(value, projectRoot, configPath, now) + if err != nil { + return err + } + devTag := normalizeDockerTag(record.DevImage) + dev := storagecatalog.Resource{ + BackendID: backendID, InstallationIDs: []string{configRecord.InstallationID}, Kind: "docker-image", Role: "native-toolchain", Locator: devTag, + Identity: record.ResolvedDevImageID, Custody: storagecatalog.CustodyGenerated, State: storagecatalog.StateCurrent, + IntroducedTags: []string{devTag}, CreatedAt: now, LastSeenAt: now, + } + dev.Key = storagecatalog.ResourceKey(dev.BackendID, dev.Kind, dev.Identity) + var existingReferences []storagecatalog.Reference + for _, resource := range value.Resources { + if resource.Key == dev.Key { + existingReferences = append(existingReferences, resource.References...) + dev.CreatedAt = resource.CreatedAt + dev.InstallationIDs = mergeUniqueStrings(dev.InstallationIDs, resource.InstallationIDs) + } + } + dev.References = existingReferences + if err := storagecatalog.UpsertResource(value, dev); err != nil { + return err + } + storagecatalog.ReplaceConfigRoleReferences(value, configRecord.ID, "native-toolchain", map[string]storagecatalog.Reference{ + dev.Key: {}, + }, now) + if record.PreviousDevImageID != "" && record.PreviousDevImageID != record.ResolvedDevImageID { + old := storagecatalog.Resource{ + BackendID: backendID, InstallationIDs: []string{configRecord.InstallationID}, Kind: "docker-image", Role: "native-toolchain", Locator: record.PreviousDevImageID, + Identity: record.PreviousDevImageID, Custody: storagecatalog.CustodyGenerated, State: storagecatalog.StateSuperseded, + CreatedAt: now, LastSeenAt: now, + } + when := now.UTC() + old.SupersededAt = &when + if err := storagecatalog.UpsertResource(value, old); err != nil { + return err + } + } + if record.ResolvedGoImageID != "" && record.ResolvedGoImageID != record.PreviousGoImageID { + source := storagecatalog.Resource{ + BackendID: backendID, InstallationIDs: []string{configRecord.InstallationID}, Kind: "docker-image", Role: "native-toolchain-source", + Locator: normalizeDockerTag(record.GoImage), Identity: record.ResolvedGoImageID, + Custody: storagecatalog.CustodyAcquired, State: storagecatalog.StateSuperseded, + CreatedAt: now, LastSeenAt: now, + } + if record.PreviousGoImageID == "" { + source.IntroducedTags = []string{normalizeDockerTag(record.GoImage)} + } + when := now.UTC() + source.SupersededAt = &when + if err := storagecatalog.UpsertResource(value, source); err != nil { + return err + } + } + return nil + }) + if err != nil { + return err + } + if err := os.Remove(path); err != nil && !errors.Is(err, os.ErrNotExist) { + return fmt.Errorf("retire imported native-controller acquisition journal: %w", err) + } + return nil +} + +func mergeUniqueStrings(values ...[]string) []string { + seen := make(map[string]bool) + var result []string + for _, group := range values { + for _, value := range group { + value = strings.TrimSpace(value) + if value == "" || seen[value] { + continue + } + seen[value] = true + result = append(result, value) + } + } + return result +} + +func normalizeDockerTag(reference string) string { + reference = strings.TrimSpace(reference) + lastSlash := strings.LastIndex(reference, "/") + if strings.LastIndex(reference, ":") <= lastSlash { + return reference + ":latest" + } + return reference +} + +func currentDockerBackendID() (string, error) { + output, err := exec.Command("docker", "info", "--format", "{{.ID}}").CombinedOutput() + if err != nil { + return "", fmt.Errorf("docker info failed: %w: %s", err, strings.TrimSpace(string(output))) + } + id := strings.TrimSpace(string(output)) + if id == "" { + return "", errors.New("Docker Engine returned an empty daemon identity") + } + return "docker:" + id, nil +} diff --git a/cmd/ephemeral-action-runner/storage_catalog.go b/cmd/ephemeral-action-runner/storage_catalog.go new file mode 100644 index 0000000..d9312ac --- /dev/null +++ b/cmd/ephemeral-action-runner/storage_catalog.go @@ -0,0 +1,497 @@ +package main + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "os" + "os/exec" + "path/filepath" + "runtime" + "sort" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes" + tartprovider "github.com/solutionforest/ephemeral-action-runner/internal/provider/tart" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" + "github.com/solutionforest/ephemeral-action-runner/internal/storage/inventory" +) + +const legacyOwnershipEvidence = "operator-selected legacy preview; exact identity readback is still required at execution" + +func addCatalogStorage(snapshot *inventory.Snapshot, providerFilter string, now time.Time) (storagecatalog.Catalog, error) { + store, err := storagecatalog.Open("") + if err != nil { + return storagecatalog.Catalog{}, err + } + value, err := store.Load(now) + if err != nil { + return storagecatalog.Catalog{}, err + } + surfaces := make(map[string]bool, len(snapshot.Surfaces)) + for _, surface := range snapshot.Surfaces { + surfaces[surface.ID] = true + } + for _, resource := range value.Resources { + if !storageProviderMatches(providerFilter, resource.Provider) { + continue + } + artifact, ok := catalogStorageArtifact(value, resource, now) + if !ok { + continue + } + if !surfaces[artifact.SurfaceID] { + snapshot.Surfaces = append(snapshot.Surfaces, storage.Surface{ + ID: artifact.SurfaceID, + Provider: artifact.Provider, + Kind: storage.SurfaceExternal, + Location: resource.BackendID, + Capacity: storage.Capacity{ObservedAt: now}, + }) + surfaces[artifact.SurfaceID] = true + } + mergeCatalogStorageArtifact(snapshot, artifact) + } + return value, nil +} + +func mergeCatalogStorageArtifact(snapshot *inventory.Snapshot, catalogArtifact storage.Artifact) { + artifacts := snapshot.Artifacts[:0] + for _, existing := range snapshot.Artifacts { + if !sameCatalogStorageTarget(existing, catalogArtifact) { + artifacts = append(artifacts, existing) + continue + } + if existing.SizeBytes > catalogArtifact.SizeBytes { + catalogArtifact.SizeBytes = existing.SizeBytes + } + if catalogArtifact.CreatedAt.IsZero() || (!existing.CreatedAt.IsZero() && existing.CreatedAt.Before(catalogArtifact.CreatedAt)) { + catalogArtifact.CreatedAt = existing.CreatedAt + } + if existing.Current { + catalogArtifact.Current = true + for _, protection := range existing.Protections { + if protection.Kind == storage.ProtectionConfiguration { + catalogArtifact.Protections = append(catalogArtifact.Protections, protection) + } + } + } + } + snapshot.Artifacts = append(artifacts, catalogArtifact) +} + +func sameCatalogStorageTarget(left, right storage.Artifact) bool { + if left.Kind != right.Kind || left.Target.Kind != right.Target.Kind || left.Target.Identity == "" || left.Target.Identity != right.Target.Identity { + return false + } + leftLocator := strings.TrimSpace(left.Target.Locator) + rightLocator := strings.TrimSpace(right.Target.Locator) + if left.Target.Kind == storage.TargetSandboxTemplate { + leftLocator = strings.TrimPrefix(strings.ToLower(leftLocator), "docker.io/library/") + rightLocator = strings.TrimPrefix(strings.ToLower(rightLocator), "docker.io/library/") + return leftLocator == rightLocator + } + if left.Target.Kind == storage.TargetFile || left.Target.Kind == storage.TargetDirectory { + leftLocator = filepath.Clean(leftLocator) + rightLocator = filepath.Clean(rightLocator) + if runtime.GOOS == "windows" { + return strings.EqualFold(leftLocator, rightLocator) + } + } + return leftLocator == rightLocator +} + +func catalogStorageArtifact(value storagecatalog.Catalog, resource storagecatalog.Resource, now time.Time) (storage.Artifact, bool) { + ownerID := value.InstallationID + if len(resource.InstallationIDs) != 0 { + ownerID = strings.Join(resource.InstallationIDs, ",") + } + artifact := storage.Artifact{ + ID: "catalog-" + resource.Key, + Provider: resource.Provider, + SurfaceID: catalogSurfaceID(resource), + RetentionGroup: resource.BackendID + "/" + resource.Provider + "/" + resource.Role, + Ownership: storage.Ownership{ + Kind: storage.OwnershipExact, + OwnerID: ownerID, + Evidence: "EPAR host resource catalog " + resource.Key, + }, + CreatedAt: resource.CreatedAt, + LastUsedAt: resource.LastSeenAt, + SupersededAt: resource.SupersededAt, + BackendID: resource.BackendID, + Custody: string(resource.Custody), + LifecycleState: string(resource.State), + CleanupError: resource.CleanupError, + } + for _, reference := range resource.References { + artifact.ConfigRefs = append(artifact.ConfigRefs, reference.ConfigID) + } + sort.Strings(artifact.ConfigRefs) + if resource.LeaseExpiresAt != nil { + artifact.Lease = &storage.Lease{ID: "catalog-resource", ExpiresAt: *resource.LeaseExpiresAt} + } + if len(resource.References) != 0 || resource.State == storagecatalog.StateCurrent { + artifact.Current = true + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionConfiguration, Detail: "referenced by registered EPAR configuration"}) + } + switch resource.Kind { + case "docker-image": + artifact.Kind = storage.ArtifactDockerImage + artifact.Target = storage.Target{Kind: storage.TargetDockerImageTag, Locator: resource.Locator, Identity: resource.Identity, Fingerprint: resource.Fingerprint, Match: storage.MatchExact} + if resource.Custody == storagecatalog.CustodyAcquired && len(resource.IntroducedTags) == 0 { + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionUncertain, Detail: "the Docker tag existed before EPAR acquisition; it remains shared and report-only"}) + } + case "sandbox-template": + artifact.Kind = storage.ArtifactSandboxTemplate + artifact.Target = storage.Target{Kind: storage.TargetSandboxTemplate, Locator: resource.Locator, Identity: resource.Identity, Fingerprint: resource.Fingerprint, Match: storage.MatchExact} + case "provider-image": + target, err := storage.SnapshotFilesystemTarget(resource.Locator) + if err == nil { + artifact.Kind = storage.ArtifactProviderImage + artifact.Target = target + if info, statErr := os.Lstat(target.Locator); statErr == nil { + artifact.SizeBytes = uint64(maxInt64(info.Size(), 0)) + } + } else { + artifact.Kind = storage.ArtifactProviderImage + artifact.Target = storage.Target{Kind: storage.TargetExternal, Locator: resource.Locator, Identity: resource.Identity, Fingerprint: resource.Fingerprint, Match: storage.MatchExact} + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionUncertain, Detail: "catalog filesystem target cannot be read exactly"}) + } + case "template-staging-directory": + target, err := storage.SnapshotFilesystemTarget(resource.Locator) + artifact.Kind = storage.ArtifactOther + if err == nil { + artifact.Target = target + if info, statErr := os.Lstat(target.Locator); statErr == nil { + artifact.SizeBytes = uint64(maxInt64(info.Size(), 0)) + } + } else { + artifact.Target = storage.Target{Kind: storage.TargetExternal, Locator: resource.Locator, Identity: resource.Identity, Fingerprint: resource.Fingerprint, Match: storage.MatchExact} + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionUncertain, Detail: "catalog template staging directory cannot be read exactly"}) + } + case "tart-image": + artifact.Kind = storage.ArtifactProviderImage + artifact.Target = storage.Target{Kind: storage.TargetExternal, Locator: "tart-image:" + resource.Locator, Identity: resource.Identity, Fingerprint: resource.Fingerprint, Match: storage.MatchExact} + default: + return storage.Artifact{}, false + } + if artifact.SupersededAt == nil && (resource.State == storagecatalog.StateSuperseded || resource.State == storagecatalog.StateCleanupPending) { + supersededAt := now.UTC() + artifact.SupersededAt = &supersededAt + } + return artifact, true +} + +func catalogSurfaceID(resource storagecatalog.Resource) string { + switch resource.Kind { + case "docker-image": + return "docker-engine" + case "sandbox-template": + return "docker-sandboxes-template-cache" + case "template-staging-directory": + return inventory.ProjectSurfaceID + case "tart-image": + return "tart-images" + default: + return "catalog-" + resource.Key[:12] + } +} + +func selectLegacyStorage(snapshot *inventory.Snapshot, catalogValue storagecatalog.Catalog, now time.Time) { + known := make(map[string]bool) + for _, resource := range catalogValue.Resources { + known[string(resource.Kind)+"\x00"+resource.Identity] = true + known[string(resource.Kind)+"\x00"+resource.Locator] = true + } + for index := range snapshot.Artifacts { + artifact := &snapshot.Artifacts[index] + if artifact.Ownership.Kind != storage.OwnershipUnknown || !legacyEPARArtifact(*artifact) { + continue + } + catalogKind := "" + switch artifact.Kind { + case storage.ArtifactDockerImage: + catalogKind = "docker-image" + case storage.ArtifactSandboxTemplate: + catalogKind = "sandbox-template" + case storage.ArtifactDockerVolume: + catalogKind = "docker-volume" + default: + continue + } + if known[catalogKind+"\x00"+artifact.Target.Identity] || known[catalogKind+"\x00"+artifact.Target.Locator] { + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionConfiguration, Detail: "exact identity is present in the host resource catalog"}) + continue + } + artifact.Ownership = storage.Ownership{Kind: storage.OwnershipExact, OwnerID: "legacy-preview", Evidence: legacyOwnershipEvidence} + artifact.LifecycleState = string(storagecatalog.StateSuperseded) + artifact.RetentionGroup = "legacy/" + string(artifact.Kind) + supersededAt := now.UTC() + artifact.SupersededAt = &supersededAt + artifact.Protections = removeUncertainProtections(artifact.Protections) + switch artifact.Kind { + case storage.ArtifactDockerImage: + if blockers, err := dockerImageContainerBlockers(artifact.Target.Identity); err != nil { + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionUncertain, Detail: "container references could not be checked: " + err.Error()}) + } else if len(blockers) != 0 { + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionActive, Detail: "Docker containers reference this image: " + strings.Join(blockers, ",")}) + } + case storage.ArtifactSandboxTemplate: + if active, err := activeSandboxCount(); err != nil { + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionUncertain, Detail: "active Sandboxes could not be checked: " + err.Error()}) + } else if active != 0 { + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionActive, Detail: fmt.Sprintf("%d live Docker Sandbox instance(s) protect template cleanup", active)}) + } + case storage.ArtifactDockerVolume: + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionUncertain, Detail: "prefix-era volumes require stronger role evidence than a name"}) + } + } +} + +func legacyEPARArtifact(artifact storage.Artifact) bool { + value := strings.ToLower(artifact.Target.Locator) + if artifact.Kind == storage.ArtifactSandboxTemplate && (value == "docker/sandbox-templates:shell-docker" || value == "docker.io/docker/sandbox-templates:shell-docker") { + return false + } + switch artifact.Kind { + case storage.ArtifactDockerImage: + repository := value + if cut := strings.LastIndex(repository, ":"); cut > strings.LastIndex(repository, "/") { + repository = repository[:cut] + } + repository = strings.TrimPrefix(repository, "docker.io/library/") + return strings.HasPrefix(repository, "epar-") + case storage.ArtifactSandboxTemplate: + return strings.HasPrefix(strings.TrimPrefix(value, "docker.io/library/"), "epar-") + case storage.ArtifactDockerVolume: + return strings.HasPrefix(value, "epar-") + default: + return false + } +} + +func removeUncertainProtections(values []storage.Protection) []storage.Protection { + out := values[:0] + for _, value := range values { + if value.Kind != storage.ProtectionUncertain { + out = append(out, value) + } + } + return out +} + +func dockerImageContainerBlockers(identity string) ([]string, error) { + ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) + defer cancel() + output, err := exec.CommandContext(ctx, "docker", "ps", "-a", "--filter", "ancestor="+identity, "--format", "{{.ID}}").Output() + if err != nil { + return nil, err + } + return strings.Fields(string(output)), nil +} + +func activeSandboxCount() (int, error) { + ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) + defer cancel() + items, err := dockersandboxes.New("").Inventory(ctx) + return len(items), err +} + +type hostStorageExecutor struct { + filesystem storage.ExactExecutor + sandboxes provider.TemplateArtifactCleaner +} + +func newHostStorageExecutor(projectRoot string) (*hostStorageExecutor, error) { + filesystem, err := storage.NewFilesystemExecutor( + filepath.Join(projectRoot, ".local", "bin"), + filepath.Join(projectRoot, "work", "template-builds", "docker-sandboxes"), + ) + if err != nil { + return nil, err + } + return &hostStorageExecutor{filesystem: filesystem, sandboxes: dockersandboxes.New("")}, nil +} + +func (e *hostStorageExecutor) ObserveExact(ctx context.Context, target storage.Target) (storage.Observation, error) { + switch target.Kind { + case storage.TargetFile, storage.TargetDirectory: + return e.filesystem.ObserveExact(ctx, target) + case storage.TargetDockerImageTag: + return observeDockerImageTarget(ctx, target) + case storage.TargetSandboxTemplate: + templates, err := dockersandboxes.New("").CachedTemplates(ctx) + if err != nil { + return storage.Observation{}, err + } + for _, item := range templates { + if item.CacheID == target.Identity && item.Reference == target.Locator { + return storage.Observation{Exists: true, Target: target}, nil + } + } + return storage.Observation{Exists: false, Target: target}, nil + case storage.TargetExternal: + name, ok := strings.CutPrefix(target.Locator, "tart-image:") + if !ok || strings.TrimSpace(name) == "" { + return storage.Observation{}, fmt.Errorf("unsupported exact external storage target %q", target.Locator) + } + instances, err := tartprovider.New("", false).List(ctx) + if err != nil { + return storage.Observation{}, err + } + for _, instance := range instances { + if instance.Name != name { + continue + } + observed := target + observed.Identity = instance.ProviderID + return storage.Observation{Exists: true, Target: observed}, nil + } + return storage.Observation{Exists: false, Target: target}, nil + default: + return storage.Observation{}, fmt.Errorf("exact storage executor does not support %s", target.Kind) + } +} + +func (e *hostStorageExecutor) RemoveExact(ctx context.Context, removal storage.Removal) error { + switch removal.Target.Kind { + case storage.TargetFile, storage.TargetDirectory: + return e.filesystem.RemoveExact(ctx, removal) + case storage.TargetDockerImageTag: + blockers, err := dockerImageContainerBlockers(removal.Target.Identity) + if err != nil { + return err + } + if len(blockers) != 0 { + return fmt.Errorf("Docker containers still reference image %s", removal.Target.Identity) + } + observed, err := observeDockerImageTarget(ctx, removal.Target) + if err != nil { + return err + } + if !observed.Exists { + return nil + } + return runExactStorageCommand(ctx, "docker", "image", "rm", removal.Target.Locator) + case storage.TargetSandboxTemplate: + if count, err := activeSandboxCount(); err != nil { + return err + } else if count != 0 { + return fmt.Errorf("%d live Docker Sandbox instance(s) protect template cleanup", count) + } + return e.sandboxes.RemoveTemplate(ctx, provider.TemplateArtifact{Reference: removal.Target.Locator, CacheID: removal.Target.Identity, Digest: removal.Target.Fingerprint}) + case storage.TargetExternal: + name, ok := strings.CutPrefix(removal.Target.Locator, "tart-image:") + if !ok || strings.TrimSpace(name) == "" { + return fmt.Errorf("unsupported exact external storage target %q", removal.Target.Locator) + } + tart := tartprovider.New("", false) + instances, err := tart.List(ctx) + if err != nil { + return err + } + var exact *provider.Instance + for index := range instances { + instance := instances[index] + if instance.Source == name && instance.Name != name { + return fmt.Errorf("Tart instance %q still references image %q", instance.Name, name) + } + if instance.Name == name { + if instance.ProviderID != removal.Target.Identity { + return errors.New("Tart image identity changed") + } + copy := instance + exact = © + } + } + if exact == nil { + return nil + } + if strings.EqualFold(exact.State, "running") { + return errors.New("Tart image is running") + } + return tart.Delete(ctx, exact.Name) + default: + return fmt.Errorf("exact storage executor does not support %s", removal.Target.Kind) + } +} + +func observeDockerImageTarget(ctx context.Context, target storage.Target) (storage.Observation, error) { + command := exec.CommandContext(ctx, "docker", "image", "inspect", "--format", "{{json .}}", target.Locator) + output, err := command.CombinedOutput() + if err != nil { + var exitErr *exec.ExitError + message := strings.ToLower(string(output)) + if errors.As(err, &exitErr) && (strings.Contains(message, "no such image") || strings.Contains(message, "not found")) { + return storage.Observation{Exists: false, Target: target}, nil + } + return storage.Observation{}, fmt.Errorf("docker image inspect %s failed: %w: %s", target.Locator, err, strings.TrimSpace(string(output))) + } + var record struct { + ID string `json:"Id"` + } + if err := json.Unmarshal(output, &record); err != nil { + return storage.Observation{}, err + } + if record.ID != target.Identity { + drifted := target + drifted.Identity = record.ID + return storage.Observation{Exists: true, Target: drifted}, nil + } + return storage.Observation{Exists: true, Target: target}, nil +} + +func runExactStorageCommand(ctx context.Context, name string, args ...string) error { + command := exec.CommandContext(ctx, name, args...) + output, err := command.CombinedOutput() + if err != nil { + return fmt.Errorf("%s %s failed: %w: %s", name, strings.Join(args, " "), err, strings.TrimSpace(string(output))) + } + return nil +} + +func removeExecutedCatalogEntries(report storage.ExecutionReport, now time.Time) error { + if report.RemovedCount == 0 { + return nil + } + removed := make(map[string]bool) + for _, entry := range report.Entries { + if entry.Status == storage.ExecutionRemoved { + removed[string(entry.Removal.Target.Kind)+"\x00"+entry.Removal.Target.Identity] = true + } + } + store, err := storagecatalog.Open("") + if err != nil { + return err + } + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + resources := value.Resources[:0] + for _, resource := range value.Resources { + targetKind := storage.TargetExternal + switch resource.Kind { + case "docker-image": + targetKind = storage.TargetDockerImageTag + case "sandbox-template": + targetKind = storage.TargetSandboxTemplate + case "provider-image": + if runtime.GOOS == "windows" || filepath.IsAbs(resource.Locator) { + targetKind = storage.TargetFile + } + case "template-staging-directory": + targetKind = storage.TargetDirectory + } + if removed[string(targetKind)+"\x00"+resource.Identity] && len(resource.References) == 0 { + continue + } + resources = append(resources, resource) + } + value.Resources = resources + return nil + }) + return err +} diff --git a/cmd/ephemeral-action-runner/storage_catalog_test.go b/cmd/ephemeral-action-runner/storage_catalog_test.go new file mode 100644 index 0000000..445255f --- /dev/null +++ b/cmd/ephemeral-action-runner/storage_catalog_test.go @@ -0,0 +1,174 @@ +package main + +import ( + "strings" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" + "github.com/solutionforest/ephemeral-action-runner/internal/storage/inventory" +) + +func TestStorageLegacyExecutionRequiresPreviewPlanHash(t *testing.T) { + err := runStorage([]string{"prune", "--legacy", "--execute"}) + if err == nil || !strings.Contains(err.Error(), "--plan ") { + t.Fatalf("runStorage() error = %v, want legacy plan-hash requirement", err) + } + if err := runStorage([]string{"status", "--legacy"}); err == nil || !strings.Contains(err.Error(), "does not support") { + t.Fatalf("storage status --legacy error = %v, want rejection", err) + } +} + +func TestLegacyArtifactSelectionNeverIncludesShellDockerOrUnknownVolumes(t *testing.T) { + now := time.Now().UTC() + snapshot := inventory.Snapshot{CollectedAt: now, Artifacts: []storage.Artifact{ + { + ID: "template", Kind: storage.ArtifactSandboxTemplate, + Target: storage.Target{Kind: storage.TargetSandboxTemplate, Locator: "docker.io/docker/sandbox-templates:shell-docker", Identity: "39cf20eca861", Match: storage.MatchExact}, + Ownership: storage.Ownership{Kind: storage.OwnershipUnknown}, + }, + { + ID: "volume", Kind: storage.ArtifactDockerVolume, + Target: storage.Target{Kind: storage.TargetDockerVolume, Locator: "epar-old-cache", Identity: "epar-old-cache", Match: storage.MatchExact}, + Ownership: storage.Ownership{Kind: storage.OwnershipUnknown}, + Protections: []storage.Protection{{Kind: storage.ProtectionUncertain, Detail: "prefix only"}}, + }, + }} + selectLegacyStorage(&snapshot, storagecatalog.Catalog{}, now) + if snapshot.Artifacts[0].Ownership.Kind != storage.OwnershipUnknown { + t.Fatal("shell-docker became a legacy cleanup candidate") + } + if snapshot.Artifacts[1].Ownership.Kind != storage.OwnershipExact { + t.Fatal("legacy volume was not represented in the approved exact preview") + } + if len(snapshot.Artifacts[1].Protections) == 0 { + t.Fatal("prefix-era volume lost its fail-closed protection") + } +} + +func TestConfiguredSandboxTemplateProtectionMatchesFullReceiptDigest(t *testing.T) { + digest := "sha256:" + strings.Repeat("a", 64) + snapshot := inventory.Snapshot{Artifacts: []storage.Artifact{{ + Kind: storage.ArtifactSandboxTemplate, + Target: storage.Target{Kind: storage.TargetSandboxTemplate, Locator: "docker.io/library/epar-docker-sandboxes-test:current", Identity: strings.Repeat("a", 12), Match: storage.MatchExact}, + }}} + + protectConfiguredSandboxTemplates(&snapshot, []inventory.TemplateSelection{{ + Tag: "epar-docker-sandboxes-test:current", + TemplateDigest: digest, + }}) + + if !snapshot.Artifacts[0].Current { + t.Fatal("configured Sandbox template was not marked current") + } + if !hasStorageProtection(snapshot.Artifacts[0].Protections, storage.ProtectionConfiguration) { + t.Fatalf("configured Sandbox template protections = %#v", snapshot.Artifacts[0].Protections) + } +} + +func hasStorageProtection(values []storage.Protection, want storage.ProtectionKind) bool { + for _, value := range values { + if value.Kind == want { + return true + } + } + return false +} + +func TestCatalogArtifactExposesReferencesCustodyAndCleanupState(t *testing.T) { + now := time.Now().UTC() + supersededAt := now.Add(-time.Minute) + value := storagecatalog.Catalog{InstallationID: "installation"} + resource := storagecatalog.Resource{ + Key: "resource", BackendID: "docker:one", Kind: "docker-image", Provider: "docker-container", + Role: "runtime-image", Locator: "epar-image:old", Identity: "sha256:old", + Custody: storagecatalog.CustodyGenerated, State: storagecatalog.StateCleanupPending, + References: []storagecatalog.Reference{{ConfigID: "config-one"}}, SupersededAt: &supersededAt, + CleanupError: "in use", + } + artifact, ok := catalogStorageArtifact(value, resource, now) + if !ok { + t.Fatal("catalog Docker image was not exposed") + } + if artifact.BackendID != "docker:one" || artifact.Custody != "generated" || artifact.LifecycleState != "cleanup-pending" || artifact.CleanupError != "in use" || len(artifact.ConfigRefs) != 1 { + t.Fatalf("catalog artifact omitted lifecycle evidence: %#v", artifact) + } +} + +func TestCatalogSandboxTemplateReplacesMatchingExternalInventory(t *testing.T) { + cacheID := strings.Repeat("a", 12) + snapshot := inventory.Snapshot{Artifacts: []storage.Artifact{{ + ID: "external", + Provider: "docker-sandboxes", + SurfaceID: "docker-sandboxes-template-cache", + Kind: storage.ArtifactSandboxTemplate, + Target: storage.Target{ + Kind: storage.TargetSandboxTemplate, Locator: "epar-docker-sandboxes-test:current", Identity: cacheID, Match: storage.MatchExact, + }, + Ownership: storage.Ownership{Kind: storage.OwnershipUnknown}, + SizeBytes: 1234, + Protections: []storage.Protection{{ + Kind: storage.ProtectionUncertain, Detail: "external inventory", + }}, + }}} + catalogArtifact := storage.Artifact{ + ID: "catalog", + Provider: "docker-sandboxes", + SurfaceID: "docker-sandboxes-template-cache", + Kind: storage.ArtifactSandboxTemplate, + Target: storage.Target{ + Kind: storage.TargetSandboxTemplate, Locator: "docker.io/library/epar-docker-sandboxes-test:current", Identity: cacheID, Match: storage.MatchExact, + }, + Ownership: storage.Ownership{Kind: storage.OwnershipExact}, + Current: true, + } + + mergeCatalogStorageArtifact(&snapshot, catalogArtifact) + + if len(snapshot.Artifacts) != 1 { + t.Fatalf("merged artifacts = %#v, want one authoritative catalog artifact", snapshot.Artifacts) + } + if snapshot.Artifacts[0].ID != "catalog" || snapshot.Artifacts[0].Ownership.Kind != storage.OwnershipExact || snapshot.Artifacts[0].SizeBytes != 1234 { + t.Fatalf("merged catalog artifact = %#v", snapshot.Artifacts[0]) + } + if hasStorageProtection(snapshot.Artifacts[0].Protections, storage.ProtectionUncertain) { + t.Fatalf("merged catalog artifact retained external uncertainty: %#v", snapshot.Artifacts[0].Protections) + } +} + +func TestCatalogPreexistingAcquiredDockerTagRemainsReportOnly(t *testing.T) { + now := time.Now().UTC() + resource := storagecatalog.Resource{ + Key: "source", BackendID: "docker:one", Kind: "docker-image", Role: "build-source", + Locator: "golang:latest", Identity: "sha256:source", Custody: storagecatalog.CustodyAcquired, + State: storagecatalog.StateSuperseded, SupersededAt: &now, + } + artifact, ok := catalogStorageArtifact(storagecatalog.Catalog{InstallationID: "host"}, resource, now) + if !ok { + t.Fatal("catalog Docker source was not exposed") + } + if !hasStorageProtection(artifact.Protections, storage.ProtectionUncertain) { + t.Fatalf("preexisting acquired Docker tag protections = %#v", artifact.Protections) + } +} + +func TestCatalogTartImageUsesExactExternalIdentity(t *testing.T) { + now := time.Now().UTC() + value := storagecatalog.Catalog{InstallationID: "host", Resources: nil} + resource := storagecatalog.Resource{ + Key: "tart-resource", BackendID: "tart:one", InstallationIDs: []string{"installation"}, + Kind: "tart-image", Provider: "tart", Role: "runtime-image", Locator: "epar-current", + Identity: "tart-mac:02:00:00:00:00:01", Custody: storagecatalog.CustodyGenerated, State: storagecatalog.StateCurrent, + } + artifact, ok := catalogStorageArtifact(value, resource, now) + if !ok { + t.Fatal("catalog Tart image was not exposed") + } + if artifact.Target.Kind != storage.TargetExternal || artifact.Target.Locator != "tart-image:epar-current" || artifact.Target.Identity != resource.Identity { + t.Fatalf("Tart artifact lost exact backend identity: %#v", artifact) + } + if artifact.Ownership.OwnerID != "installation" { + t.Fatalf("Tart artifact owner = %q, want installation identity", artifact.Ownership.OwnerID) + } +} diff --git a/cmd/ephemeral-action-runner/storage_configured_test.go b/cmd/ephemeral-action-runner/storage_configured_test.go new file mode 100644 index 0000000..83b7f69 --- /dev/null +++ b/cmd/ephemeral-action-runner/storage_configured_test.go @@ -0,0 +1,42 @@ +package main + +import ( + "path/filepath" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +func TestConfiguredStorageFilesIncludesCurrentWSLArtifacts(t *testing.T) { + project := t.TempDir() + configuredAt := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + cfg := config.Default() + cfg.Provider.Type = "wsl" + cfg.Image.SourceType = "docker-image" + cfg.Image.OutputImage = filepath.Join("work", "images", "runner.tar") + + files := configuredStorageFiles(cfg, project, configuredAt) + if len(files) != 4 { + t.Fatalf("configuredStorageFiles() returned %d files, want 4: %+v", len(files), files) + } + byRole := make(map[string]storage.ArtifactKind, len(files)) + for _, file := range files { + if file.Provider != "wsl" || !file.Current || file.ConfiguredAt != configuredAt { + t.Fatalf("configured file = %+v", file) + } + byRole[file.Role] = file.Kind + } + if byRole["reusable-image"] != storage.ArtifactProviderImage || byRole["source-rootfs-cache"] != storage.ArtifactProviderCache || byRole["image-manifest"] != storage.ArtifactOther || byRole["source-cache-manifest"] != storage.ArtifactOther { + t.Fatalf("configured roles = %+v", byRole) + } +} + +func TestConfiguredStorageFilesSkipsNonWSLProvider(t *testing.T) { + cfg := config.Default() + cfg.Provider.Type = "docker-container" + if files := configuredStorageFiles(cfg, t.TempDir(), time.Now()); len(files) != 0 { + t.Fatalf("configuredStorageFiles() = %+v, want none", files) + } +} diff --git a/cmd/ephemeral-action-runner/storage_external.go b/cmd/ephemeral-action-runner/storage_external.go new file mode 100644 index 0000000..4e47f99 --- /dev/null +++ b/cmd/ephemeral-action-runner/storage_external.go @@ -0,0 +1,480 @@ +package main + +import ( + "bufio" + "bytes" + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "fmt" + "math" + "os" + "os/exec" + "path/filepath" + "runtime" + "strconv" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + artifactimage "github.com/solutionforest/ephemeral-action-runner/internal/image" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" + "github.com/solutionforest/ephemeral-action-runner/internal/storage/inventory" +) + +func collectExternalStorage(snapshot *inventory.Snapshot, providerFilter, configPath string) { + if providerFilter == "" || providerFilter == "docker-container" || providerFilter == "docker-sandboxes" || providerFilter == "wsl" { + collectDockerStorage(snapshot, providerFilter, configPath) + } + if providerFilter == "" || providerFilter == "docker-sandboxes" { + collectDockerSandboxesStorage(snapshot) + } + if runtime.GOOS == "windows" && (providerFilter == "" || providerFilter == "wsl") { + collectWSLStorage(snapshot) + } + if runtime.GOOS == "darwin" && (providerFilter == "" || providerFilter == "tart") { + collectTartStorage(snapshot) + } +} + +func collectDockerStorage(snapshot *inventory.Snapshot, providerFilter, configPath string) { + const surfaceID = "docker-engine" + snapshot.Surfaces = append(snapshot.Surfaces, storage.Surface{ + ID: surfaceID, + Kind: storage.SurfaceDockerEngine, + Location: "docker-engine", + Capacity: storage.Capacity{ObservedAt: snapshot.CollectedAt}, + }) + ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) + defer cancel() + output, err := exec.CommandContext(ctx, "docker", "image", "ls", "--all", "--no-trunc", "--format", "{{.Repository}}\t{{.Tag}}\t{{.ID}}\t{{.Size}}").Output() + if err != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Docker image inventory is unavailable and remains report-only: %v", err)) + } else { + for _, line := range strings.Split(strings.TrimSpace(string(output)), "\n") { + parts := strings.Split(strings.TrimSpace(line), "\t") + if len(parts) != 4 || !strings.HasPrefix(parts[0], "epar-") { + continue + } + reference := parts[0] + ":" + parts[1] + artifactProvider := dockerImageProvider(parts[0]) + if !storageProviderMatches(providerFilter, artifactProvider) { + continue + } + snapshot.Artifacts = append(snapshot.Artifacts, storage.Artifact{ + ID: externalStorageID("docker-image", reference, parts[2]), + Provider: artifactProvider, + SurfaceID: surfaceID, + Kind: storage.ArtifactDockerImage, + Target: storage.Target{ + Kind: storage.TargetDockerImageTag, + Locator: reference, + Identity: parts[2], + Match: storage.MatchExact, + }, + Ownership: storage.Ownership{Kind: storage.OwnershipUnknown, Evidence: "EPAR prefix is not exact ownership"}, + SizeBytes: parseDockerSize(parts[3]), + Protections: []storage.Protection{{Kind: storage.ProtectionUncertain, Detail: "image lacks persisted EPAR owner metadata; explicit prune only"}}, + }) + } + } + output, err = exec.CommandContext(ctx, "docker", "system", "df", "-v", "--format", "json").Output() + if err != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Docker volume size inventory is unavailable and remains report-only: %v", err)) + collectDockerVolumeNames(snapshot, surfaceID) + } else { + volumes, parseErr := parseDockerDiskUsageVolumes(output) + if parseErr != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Docker volume size inventory is unreadable and remains report-only: %v", parseErr)) + collectDockerVolumeNames(snapshot, surfaceID) + } else { + collectDockerVolumeRecords(snapshot, surfaceID, volumes) + } + } + collectDedicatedBuildxStorage(snapshot, surfaceID, configPath) +} + +type dockerDiskUsageVolume struct { + Name string `json:"Name"` + Size string `json:"Size"` + Labels string `json:"Labels"` +} + +func parseDockerDiskUsageVolumes(output []byte) ([]dockerDiskUsageVolume, error) { + var usage struct { + Volumes []dockerDiskUsageVolume `json:"Volumes"` + } + if err := json.Unmarshal(output, &usage); err != nil { + return nil, err + } + return usage.Volumes, nil +} + +func collectDockerVolumeNames(snapshot *inventory.Snapshot, surfaceID string) { + ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) + defer cancel() + output, err := exec.CommandContext(ctx, "docker", "volume", "ls", "--format", "{{.Name}}").Output() + if err != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Docker volume inventory is unavailable and remains report-only: %v", err)) + return + } + volumes := make([]dockerDiskUsageVolume, 0) + for _, name := range strings.Fields(string(output)) { + volumes = append(volumes, dockerDiskUsageVolume{Name: name}) + } + collectDockerVolumeRecords(snapshot, surfaceID, volumes) +} + +func collectDockerVolumeRecords(snapshot *inventory.Snapshot, surfaceID string, volumes []dockerDiskUsageVolume) { + for _, volume := range volumes { + if !strings.HasPrefix(volume.Name, "epar-") { + continue + } + labels := parseDockerLabels(volume.Labels) + role := labels["io.solutionforest.epar.cache"] + if labels["io.solutionforest.epar.project"] == storageProjectID(snapshot.ProjectRoot) && + labels["io.solutionforest.epar.schema"] == "1" && + sameStorageProjectRoot(labels["io.solutionforest.epar.root"], snapshot.ProjectRoot) && + (role == "gomod" || role == "gobuild") { + snapshot.Artifacts = append(snapshot.Artifacts, storage.Artifact{ + ID: externalStorageID("go-cache-volume", volume.Name, role, labels["io.solutionforest.epar.root"]), + SurfaceID: surfaceID, + Kind: storage.ArtifactGoCache, + Target: storage.Target{Kind: storage.TargetDockerVolume, Locator: volume.Name, Identity: volume.Name, Fingerprint: labels["io.solutionforest.epar.project"] + "\x00" + role + "\x001", Match: storage.MatchExact}, + Ownership: storage.Ownership{Kind: storage.OwnershipExact, OwnerID: labels["io.solutionforest.epar.project"], Evidence: "exact Docker volume labels"}, + SizeBytes: parseDockerSize(volume.Size), + LastUsedAt: snapshot.CollectedAt, + Protections: []storage.Protection{{ + Kind: storage.ProtectionLock, + Detail: "project-scoped Go cache is bounded by the native-controller wrapper", + }}, + }) + continue + } + snapshot.Artifacts = append(snapshot.Artifacts, storage.Artifact{ + ID: externalStorageID("docker-volume", volume.Name), + SurfaceID: surfaceID, + Kind: storage.ArtifactDockerVolume, + Target: storage.Target{Kind: storage.TargetDockerVolume, Locator: volume.Name, Identity: volume.Name, Match: storage.MatchExact}, + Ownership: storage.Ownership{Kind: storage.OwnershipUnknown, Evidence: "name prefix or incomplete labels are not exact ownership"}, + SizeBytes: parseDockerSize(volume.Size), + Protections: []storage.Protection{{Kind: storage.ProtectionUncertain, Detail: "volume lacks exact persisted EPAR ownership labels; explicit operator review only"}}, + }) + } +} + +func parseDockerLabels(value string) map[string]string { + labels := make(map[string]string) + for _, field := range strings.Split(value, ",") { + key, item, found := strings.Cut(strings.TrimSpace(field), "=") + if found && key != "" { + labels[key] = item + } + } + return labels +} + +func storageProjectID(projectRoot string) string { + canonical := filepath.Clean(projectRoot) + if runtime.GOOS == "windows" { + canonical = strings.ToLower(canonical) + } + sum := sha256.Sum256([]byte(canonical)) + return hex.EncodeToString(sum[:6]) +} + +func sameStorageProjectRoot(labelRoot, projectRoot string) bool { + if labelRoot == "" { + return false + } + left, right := filepath.Clean(labelRoot), filepath.Clean(projectRoot) + if runtime.GOOS == "windows" { + return strings.EqualFold(left, right) + } + return left == right +} + +func collectDedicatedBuildxStorage(snapshot *inventory.Snapshot, surfaceID, configPath string) { + metadata, err := artifactimage.LoadBuildxMetadataForConfig(snapshot.ProjectRoot, configPath) + if err != nil { + if !os.IsNotExist(err) { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("EPAR Buildx ownership metadata is invalid; no builder cache is trusted: %v", err)) + } + } else { + collectBuildxMetadataStorage(snapshot, surfaceID, metadata, metadata.ConfigID, metadata.EPARConfigPath, false) + } + legacy, legacyErr := artifactimage.LoadLegacyBuildxMetadata(snapshot.ProjectRoot) + if legacyErr != nil { + if !os.IsNotExist(legacyErr) { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Legacy EPAR Buildx ownership metadata is invalid and was left untouched: %v", legacyErr)) + } + return + } + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Legacy project-scoped EPAR Buildx builder %q is retained for explicit cleanup and is not reused by config-scoped controllers", legacy.Builder)) + collectBuildxMetadataStorage(snapshot, surfaceID, legacy, legacy.ProjectRoot, artifactimage.LegacyBuildxMetadataPath(snapshot.ProjectRoot), true) +} + +func collectBuildxMetadataStorage(snapshot *inventory.Snapshot, surfaceID string, metadata artifactimage.BuildxMetadata, ownerID, evidence string, legacy bool) { + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + output, err := exec.CommandContext(ctx, "docker", "buildx", "du", "--builder", metadata.Builder, "--format", "json").Output() + if err != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("EPAR Buildx cache inventory for %q is unavailable: %v", metadata.Builder, err)) + return + } + total, err := parseBuildxDiskUsage(output) + if err != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("EPAR Buildx cache inventory for %q is unreadable: %v", metadata.Builder, err)) + return + } + cacheLimit, limitErr := config.ParseByteSize(metadata.CacheLimit) + if limitErr != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("EPAR Buildx cache limit for %q is invalid: %v", metadata.Builder, limitErr)) + } + artifact := storage.Artifact{ + ID: externalStorageID("buildx-cache", metadata.Builder, ownerID), + SurfaceID: surfaceID, + Kind: storage.ArtifactBuildKitCache, + Target: storage.Target{Kind: storage.TargetBuildKitRecord, Locator: metadata.Builder, Identity: metadata.Builder, Fingerprint: ownerID + "\x00" + metadata.CacheLimit, Match: storage.MatchExact}, + Ownership: storage.Ownership{Kind: storage.OwnershipExact, OwnerID: ownerID, Evidence: evidence}, + SizeBytes: total, + LastUsedAt: snapshot.CollectedAt, + Protections: []storage.Protection{{Kind: storage.ProtectionLock, Detail: "dedicated BuildKit enforces its configured garbage-collection ceiling"}}, + } + if legacy { + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionOperator, Detail: "legacy project-scoped builder requires explicit operator cleanup"}) + } + snapshot.Artifacts = append(snapshot.Artifacts, artifact) + if limitErr == nil && total > uint64(cacheLimit) { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("EPAR Buildx cache %q currently uses %d bytes above its %d-byte configured ceiling; BuildKit garbage collection is authoritative", metadata.Builder, total, uint64(cacheLimit))) + } +} + +func parseBuildxDiskUsage(output []byte) (uint64, error) { + type record struct { + ID string `json:"ID"` + Size json.RawMessage `json:"Size"` + Total json.RawMessage `json:"Total"` + } + parseBytes := func(raw json.RawMessage) (uint64, bool) { + if len(raw) == 0 || bytes.Equal(raw, []byte("null")) { + return 0, false + } + var numeric uint64 + if err := json.Unmarshal(raw, &numeric); err == nil { + return numeric, true + } + var text string + if err := json.Unmarshal(raw, &text); err == nil { + value := parseDockerSize(text) + return value, value > 0 || strings.TrimSpace(text) == "0B" + } + return 0, false + } + trimmed := bytes.TrimSpace(output) + if len(trimmed) == 0 { + return 0, nil + } + var records []record + if trimmed[0] == '[' { + if err := json.Unmarshal(trimmed, &records); err != nil { + return 0, err + } + } else { + scanner := bufio.NewScanner(bytes.NewReader(trimmed)) + for scanner.Scan() { + var value record + if err := json.Unmarshal(scanner.Bytes(), &value); err != nil { + return 0, err + } + records = append(records, value) + } + if err := scanner.Err(); err != nil { + return 0, err + } + } + var sum uint64 + for _, value := range records { + if total, found := parseBytes(value.Total); found { + return total, nil + } + if value.ID == "" { + continue + } + size, found := parseBytes(value.Size) + if found { + sum += size + } + } + return sum, nil +} + +func parseDockerSize(value string) uint64 { + value = strings.TrimSpace(value) + index := 0 + for index < len(value) && (value[index] == '.' || value[index] >= '0' && value[index] <= '9') { + index++ + } + if index == 0 || index == len(value) { + return 0 + } + number, err := strconv.ParseFloat(value[:index], 64) + if err != nil || number < 0 { + return 0 + } + multiplier := float64(0) + switch strings.ToUpper(strings.TrimSpace(value[index:])) { + case "B": + multiplier = 1 + case "KB": + multiplier = 1e3 + case "MB": + multiplier = 1e6 + case "GB": + multiplier = 1e9 + case "TB": + multiplier = 1e12 + case "KIB": + multiplier = 1 << 10 + case "MIB": + multiplier = 1 << 20 + case "GIB": + multiplier = 1 << 30 + case "TIB": + multiplier = 1 << 40 + default: + return 0 + } + bytes := number * multiplier + if math.IsInf(bytes, 0) || bytes > math.MaxUint64 { + return 0 + } + return uint64(bytes) +} + +func dockerImageProvider(repository string) string { + switch { + case strings.HasPrefix(repository, "epar-docker-sandboxes-"): + return "docker-sandboxes" + case strings.HasPrefix(repository, "epar-docker-container-"): + return "docker-container" + default: + // Development images, retired naming schemes, and cache volumes can be + // shared by more than one provider. Keep them provider-neutral. + return "" + } +} + +func storageProviderMatches(filter, artifactProvider string) bool { + return filter == "" || artifactProvider == "" || filter == artifactProvider +} + +func collectDockerSandboxesStorage(snapshot *inventory.Snapshot) { + const surfaceID = "docker-sandboxes-template-cache" + snapshot.Surfaces = append(snapshot.Surfaces, storage.Surface{ + ID: surfaceID, + Provider: "docker-sandboxes", + Kind: storage.SurfaceSandboxCache, + Location: "docker-sandboxes-containerd", + Capacity: storage.Capacity{ObservedAt: snapshot.CollectedAt}, + }) + ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) + defer cancel() + templates, err := dockersandboxes.New("").CachedTemplates(ctx) + if err != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Docker Sandboxes template inventory is unavailable and remains report-only: %v", err)) + return + } + for _, template := range templates { + snapshot.Artifacts = append(snapshot.Artifacts, storage.Artifact{ + ID: externalStorageID("sandbox-template", template.Reference, template.CacheID), + Provider: "docker-sandboxes", + SurfaceID: surfaceID, + Kind: storage.ArtifactSandboxTemplate, + Target: storage.Target{Kind: storage.TargetSandboxTemplate, Locator: template.Reference, Identity: template.CacheID, Match: storage.MatchExact}, + Ownership: storage.Ownership{Kind: storage.OwnershipUnknown, Evidence: "imported template has no persisted EPAR owner receipt"}, + SizeBytes: uint64(maxInt64(template.SizeBytes, 0)), + Protections: []storage.Protection{{Kind: storage.ProtectionUncertain, Detail: "imported templates require explicit prune execution"}}, + }) + } +} + +func protectConfiguredSandboxTemplates(snapshot *inventory.Snapshot, selections []inventory.TemplateSelection) { + for _, selection := range selections { + digest := strings.TrimPrefix(strings.ToLower(strings.TrimSpace(selection.TemplateDigest)), "sha256:") + if len(digest) != 64 { + continue + } + cacheID := digest[:12] + reference := strings.TrimPrefix(strings.ToLower(strings.TrimSpace(selection.Tag)), "docker.io/library/") + for index := range snapshot.Artifacts { + artifact := &snapshot.Artifacts[index] + if artifact.Kind != storage.ArtifactSandboxTemplate { + continue + } + locator := strings.TrimPrefix(strings.ToLower(strings.TrimSpace(artifact.Target.Locator)), "docker.io/library/") + if locator != reference || strings.ToLower(artifact.Target.Identity) != cacheID { + continue + } + artifact.Current = true + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionConfiguration, Detail: "configured Docker Sandboxes template receipt"}) + } + } +} + +func collectWSLStorage(snapshot *inventory.Snapshot) { + const surfaceID = "wsl-distributions" + snapshot.Surfaces = append(snapshot.Surfaces, storage.Surface{ID: surfaceID, Provider: "wsl", Kind: storage.SurfaceExternal, Location: "wsl", Capacity: storage.Capacity{ObservedAt: snapshot.CollectedAt}}) + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + output, err := exec.CommandContext(ctx, "wsl.exe", "--list", "--quiet").Output() + if err != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("WSL distribution inventory is unavailable and remains report-only: %v", err)) + return + } + for _, name := range strings.Fields(strings.ReplaceAll(string(output), "\x00", "")) { + if !strings.HasPrefix(strings.ToLower(name), "epar-") { + continue + } + snapshot.Artifacts = append(snapshot.Artifacts, externalReportOnlyArtifact("wsl-distribution", "wsl", surfaceID, name)) + } +} + +func collectTartStorage(snapshot *inventory.Snapshot) { + const surfaceID = "tart-images" + snapshot.Surfaces = append(snapshot.Surfaces, storage.Surface{ID: surfaceID, Provider: "tart", Kind: storage.SurfaceExternal, Location: "tart", Capacity: storage.Capacity{ObservedAt: snapshot.CollectedAt}}) + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + output, err := exec.CommandContext(ctx, "tart", "list", "--format", "json").Output() + if err != nil { + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("Tart image inventory is unavailable and remains report-only: %v", err)) + return + } + snapshot.Artifacts = append(snapshot.Artifacts, externalReportOnlyArtifact("tart-inventory", "tart", surfaceID, strings.TrimSpace(string(output)))) +} + +func externalReportOnlyArtifact(kind, providerName, surfaceID, identity string) storage.Artifact { + return storage.Artifact{ + ID: externalStorageID(kind, identity), + Provider: providerName, + SurfaceID: surfaceID, + Kind: storage.ArtifactOther, + Target: storage.Target{Kind: storage.TargetExternal, Locator: kind + ":" + identity, Identity: identity, Match: storage.MatchExact}, + Ownership: storage.Ownership{Kind: storage.OwnershipUnknown}, + Protections: []storage.Protection{{Kind: storage.ProtectionUncertain, Detail: "external provider resource requires explicit operator-reviewed prune"}}, + } +} + +func externalStorageID(parts ...string) string { + sum := sha256.Sum256([]byte(strings.Join(parts, "\x00"))) + return "external-" + hex.EncodeToString(sum[:12]) +} + +func maxInt64(value, minimum int64) int64 { + if value < minimum { + return minimum + } + return value +} diff --git a/cmd/ephemeral-action-runner/storage_external_test.go b/cmd/ephemeral-action-runner/storage_external_test.go new file mode 100644 index 0000000..c513712 --- /dev/null +++ b/cmd/ephemeral-action-runner/storage_external_test.go @@ -0,0 +1,162 @@ +package main + +import ( + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" + "github.com/solutionforest/ephemeral-action-runner/internal/storage/inventory" +) + +func TestDockerImageProvider(t *testing.T) { + tests := map[string]string{ + "epar-docker-sandboxes-catthehacker-full": "docker-sandboxes", + "epar-docker-container-catthehacker-full": "docker-container", + "epar-docker-dind-act": "", + "epar-dev-toolchain": "", + } + for repository, want := range tests { + if got := dockerImageProvider(repository); got != want { + t.Fatalf("dockerImageProvider(%q) = %q, want %q", repository, got, want) + } + } +} + +func TestStorageProviderMatches(t *testing.T) { + tests := []struct { + filter string + provider string + want bool + }{ + {filter: "", provider: "docker-container", want: true}, + {filter: "docker-sandboxes", provider: "", want: true}, + {filter: "docker-sandboxes", provider: "docker-sandboxes", want: true}, + {filter: "docker-sandboxes", provider: "docker-container", want: false}, + } + for _, test := range tests { + if got := storageProviderMatches(test.filter, test.provider); got != test.want { + t.Fatalf("storageProviderMatches(%q, %q) = %t, want %t", test.filter, test.provider, got, test.want) + } + } +} + +func TestParseDockerSize(t *testing.T) { + tests := map[string]uint64{ + "0B": 0, + "972.1MB": 972_100_000, + "17.44GB": 17_440_000_000, + "1.5GiB": 1_610_612_736, + "unknown": 0, + } + for input, want := range tests { + if got := parseDockerSize(input); got != want { + t.Fatalf("parseDockerSize(%q) = %d, want %d", input, got, want) + } + } +} + +func TestParseBuildxDiskUsageArray(t *testing.T) { + got, err := parseBuildxDiskUsage([]byte(`[{"ID":"a","Size":1024},{"ID":"b","Size":"1KiB"}]`)) + if err != nil { + t.Fatal(err) + } + if got != 2048 { + t.Fatalf("parseBuildxDiskUsage() = %d, want 2048", got) + } +} + +func TestParseBuildxDiskUsageNDJSONSummary(t *testing.T) { + got, err := parseBuildxDiskUsage([]byte("{\"ID\":\"a\",\"Size\":1024}\n{\"Total\":4096}\n")) + if err != nil { + t.Fatal(err) + } + if got != 4096 { + t.Fatalf("parseBuildxDiskUsage() = %d, want 4096", got) + } +} + +func TestParseBuildxDiskUsageRejectsInvalidJSON(t *testing.T) { + if _, err := parseBuildxDiskUsage([]byte("{")); err == nil { + t.Fatal("parseBuildxDiskUsage() error = nil, want invalid JSON error") + } +} + +func TestParseDockerDiskUsageVolumes(t *testing.T) { + volumes, err := parseDockerDiskUsageVolumes([]byte(`{"Volumes":[{"Name":"epar-project-gocache","Size":"1.5GiB","Labels":"io.solutionforest.epar.cache=gobuild"}]}`)) + if err != nil { + t.Fatal(err) + } + if len(volumes) != 1 || volumes[0].Name != "epar-project-gocache" || volumes[0].Size != "1.5GiB" { + t.Fatalf("unexpected Docker volume records: %#v", volumes) + } +} + +func TestParseDockerLabels(t *testing.T) { + labels := parseDockerLabels("io.solutionforest.epar.project=abc123,io.solutionforest.epar.cache=gomod,missing") + if labels["io.solutionforest.epar.project"] != "abc123" || labels["io.solutionforest.epar.cache"] != "gomod" { + t.Fatalf("unexpected labels: %#v", labels) + } + if _, found := labels["missing"]; found { + t.Fatalf("malformed label was accepted: %#v", labels) + } +} + +func TestCollectDockerVolumeRecordsTrustsOnlyExactProjectLabels(t *testing.T) { + root := t.TempDir() + projectID := storageProjectID(root) + snapshot := inventory.Snapshot{ProjectRoot: root, CollectedAt: time.Now().UTC()} + collectDockerVolumeRecords(&snapshot, "docker-engine", []dockerDiskUsageVolume{ + { + Name: "epar-" + projectID + "-gocache", + Size: "1GiB", + Labels: "io.solutionforest.epar.project=" + projectID + ",io.solutionforest.epar.cache=gobuild,io.solutionforest.epar.schema=1,io.solutionforest.epar.root=" + root, + }, + {Name: "epar-gocache", Size: "4GiB"}, + }) + if len(snapshot.Artifacts) != 2 { + t.Fatalf("artifact count = %d, want 2", len(snapshot.Artifacts)) + } + if snapshot.Artifacts[0].Kind != storage.ArtifactGoCache || snapshot.Artifacts[0].Ownership.Kind != storage.OwnershipExact { + t.Fatalf("exact labelled cache was not trusted: %#v", snapshot.Artifacts[0]) + } + if snapshot.Artifacts[1].Kind != storage.ArtifactDockerVolume || snapshot.Artifacts[1].Ownership.Kind != storage.OwnershipUnknown { + t.Fatalf("unlabelled cache was trusted: %#v", snapshot.Artifacts[1]) + } +} + +func TestRunEffectiveGoCacheLimitUsesDefaults(t *testing.T) { + root := t.TempDir() + output, err := captureStdout(t, func() error { + return runStorage([]string{"effective-go-cache-limit", "--project-root", root}) + }) + if err != nil { + t.Fatal(err) + } + if output != "10737418240\n" { + t.Fatalf("effective Go cache limit = %q, want 10GiB in bytes", output) + } +} + +func TestNativeWrappersUseExactBoundedGoCaches(t *testing.T) { + for _, name := range []string{"build-native-controller.sh", "build-native-controller.ps1"} { + content, err := os.ReadFile(filepath.Join("..", "..", "scripts", name)) + if err != nil { + t.Fatal(err) + } + text := string(content) + for _, required := range []string{ + "io.solutionforest.epar.project", + "io.solutionforest.epar.cache", + "io.solutionforest.epar.root", + "effective-go-cache-limit", + "EPAR_GO_CACHE_LIMIT_BYTES", + } { + if !strings.Contains(text, required) { + t.Fatalf("%s is missing exact bounded Go cache contract %q", name, required) + } + } + } +} diff --git a/cmd/ephemeral-action-runner/storage_preflight_test.go b/cmd/ephemeral-action-runner/storage_preflight_test.go new file mode 100644 index 0000000..1e063b8 --- /dev/null +++ b/cmd/ephemeral-action-runner/storage_preflight_test.go @@ -0,0 +1,30 @@ +package main + +import ( + "strings" + "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" +) + +func TestPreflightControllerStorageRejectsInsufficientSurface(t *testing.T) { + t.Setenv("EPAR_INVOCATION", "start") + cfg := config.Default() + cfg.Provider.Type = "docker-container" + cfg.Storage.MinimumFree = "9223372036854775807B" + + err := preflightControllerStorage(t.TempDir(), cfg) + if err == nil { + t.Fatal("preflightControllerStorage() error = nil, want insufficient capacity") + } + for _, want := range []string{ + "not enough disk space to initialize the EPAR controller", + "Estimated operation growth: 0 bytes", + "Free-space reserve: 8.00 EiB", + "./start storage prune --provider docker-container", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("preflightControllerStorage() error = %q, want %q", err, want) + } + } +} diff --git a/cmd/ephemeral-action-runner/version.go b/cmd/ephemeral-action-runner/version.go index 28909c3..46ff31f 100644 --- a/cmd/ephemeral-action-runner/version.go +++ b/cmd/ephemeral-action-runner/version.go @@ -10,15 +10,20 @@ var ( version = "dev" commit = "unknown" buildDate = "unknown" + // sourceRevision is populated only by the native-controller build scripts. + // Promotion requires its exact clean-source sha256 identity; go run, + // release builds without equivalent plumbing, and dirty builds fail closed. + sourceRevision = "unknown" ) func versionString() string { return fmt.Sprintf(`%s %s commit: %s buildDate: %s +sourceRevision: %s go: %s platform: %s/%s -`, binaryName, version, commit, buildDate, runtime.Version(), runtime.GOOS, runtime.GOARCH) +`, binaryName, version, commit, buildDate, sourceRevision, runtime.Version(), runtime.GOOS, runtime.GOARCH) } func printVersion(w io.Writer) { diff --git a/cmd/ephemeral-action-runner/version_test.go b/cmd/ephemeral-action-runner/version_test.go index 380bbe5..eb765cd 100644 --- a/cmd/ephemeral-action-runner/version_test.go +++ b/cmd/ephemeral-action-runner/version_test.go @@ -14,6 +14,7 @@ func TestVersionStringDefaults(t *testing.T) { "ephemeral-action-runner dev", "commit: unknown", "buildDate: unknown", + "sourceRevision: unknown", "go: " + runtime.Version(), "platform: " + runtime.GOOS + "/" + runtime.GOARCH, } { @@ -24,20 +25,22 @@ func TestVersionStringDefaults(t *testing.T) { } func TestVersionStringInjectedMetadata(t *testing.T) { - oldVersion, oldCommit, oldBuildDate := version, commit, buildDate + oldVersion, oldCommit, oldBuildDate, oldSourceRevision := version, commit, buildDate, sourceRevision t.Cleanup(func() { - version, commit, buildDate = oldVersion, oldCommit, oldBuildDate + version, commit, buildDate, sourceRevision = oldVersion, oldCommit, oldBuildDate, oldSourceRevision }) version = "v1.2.3-beta.1" commit = "abc1234" buildDate = "2026-07-07T00:00:00Z" + sourceRevision = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef" got := versionString() for _, want := range []string{ "ephemeral-action-runner v1.2.3-beta.1", "commit: abc1234", "buildDate: 2026-07-07T00:00:00Z", + "sourceRevision: sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef", } { if !strings.Contains(got, want) { t.Fatalf("versionString() missing %q in:\n%s", want, got) diff --git a/configs/docker-dind.act.example.yml b/configs/docker-container.act.example.yml similarity index 72% rename from configs/docker-dind.act.example.yml rename to configs/docker-container.act.example.yml index 3afb483..588f5fb 100644 --- a/configs/docker-dind.act.example.yml +++ b/configs/docker-container.act.example.yml @@ -8,10 +8,12 @@ github: image: sourceType: docker-image sourceImage: ghcr.io/catthehacker/ubuntu:act-latest - outputImage: epar-docker-dind-catthehacker-act + outputImage: epar-docker-container-catthehacker-act upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" # Existing configs stay disabled. Set overlay to add the host's trusted root # anchors to Ubuntu runners. Windows/macOS: [system, user]; Linux: [system]. hostTrustMode: disabled @@ -26,6 +28,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -47,13 +58,22 @@ logging: retentionIntervalMinutes: 60 runner: - labels: [self-hosted, linux, epar-docker-dind-catthehacker-act] + group: your-runner-group + labels: [self-hosted, linux, epar-docker-container-catthehacker-act] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true + provider: - type: docker-dind - sourceImage: epar-docker-dind-catthehacker-act + type: docker-container + sourceImage: epar-docker-container-catthehacker-act network: default docker: diff --git a/configs/docker-dind.core.example.yml b/configs/docker-container.core.example.yml similarity index 80% rename from configs/docker-dind.core.example.yml rename to configs/docker-container.core.example.yml index f4fdfbc..6527953 100644 --- a/configs/docker-dind.core.example.yml +++ b/configs/docker-container.core.example.yml @@ -13,6 +13,8 @@ image: upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" # Existing configs stay disabled. Set overlay to add the host's trusted root # anchors to Ubuntu runners. Windows/macOS: [system, user]; Linux: [system]. hostTrustMode: disabled @@ -26,6 +28,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -53,8 +64,16 @@ runner: noDefaultLabels: true ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: false + provider: - type: docker-dind + type: docker-container sourceImage: epar-ci-core network: default platform: linux/amd64 diff --git a/configs/docker-dind.example.yml b/configs/docker-container.example.yml similarity index 71% rename from configs/docker-dind.example.yml rename to configs/docker-container.example.yml index 30c6655..adac62d 100644 --- a/configs/docker-dind.example.yml +++ b/configs/docker-container.example.yml @@ -8,10 +8,12 @@ github: image: sourceType: docker-image sourceImage: ghcr.io/catthehacker/ubuntu:full-latest - outputImage: epar-docker-dind-catthehacker-ubuntu + outputImage: epar-docker-container-catthehacker-ubuntu upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" # Existing configs stay disabled. Set overlay to add the host's trusted root # anchors to Ubuntu runners. Windows/macOS: [system, user]; Linux: [system]. hostTrustMode: disabled @@ -26,6 +28,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -47,13 +58,22 @@ logging: retentionIntervalMinutes: 60 runner: - labels: [self-hosted, linux, epar-docker-dind-catthehacker-ubuntu] + group: your-runner-group + labels: [self-hosted, linux, epar-docker-container-catthehacker-ubuntu] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true + provider: - type: docker-dind - sourceImage: epar-docker-dind-catthehacker-ubuntu + type: docker-container + sourceImage: epar-docker-container-catthehacker-ubuntu network: default docker: diff --git a/configs/docker-dind.web-e2e.example.yml b/configs/docker-container.web-e2e.example.yml similarity index 71% rename from configs/docker-dind.web-e2e.example.yml rename to configs/docker-container.web-e2e.example.yml index 70da56e..af2c374 100644 --- a/configs/docker-dind.web-e2e.example.yml +++ b/configs/docker-container.web-e2e.example.yml @@ -8,10 +8,12 @@ github: image: sourceType: docker-image sourceImage: ghcr.io/catthehacker/ubuntu:act-latest - outputImage: epar-docker-dind-catthehacker-ubuntu-web-e2e + outputImage: epar-docker-container-catthehacker-ubuntu-web-e2e upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" # Existing configs stay disabled. Set overlay to add the host's trusted root # anchors to Ubuntu runners. Windows/macOS: [system, user]; Linux: [system]. hostTrustMode: disabled @@ -27,6 +29,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -48,13 +59,22 @@ logging: retentionIntervalMinutes: 60 runner: - labels: [self-hosted, linux, epar-docker-dind-catthehacker-ubuntu-web-e2e] + group: your-runner-group + labels: [self-hosted, linux, epar-docker-container-catthehacker-ubuntu-web-e2e] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true + provider: - type: docker-dind - sourceImage: epar-docker-dind-catthehacker-ubuntu-web-e2e + type: docker-container + sourceImage: epar-docker-container-catthehacker-ubuntu-web-e2e network: default docker: diff --git a/configs/docker-sandboxes.example.yml b/configs/docker-sandboxes.example.yml new file mode 100644 index 0000000..9e74387 --- /dev/null +++ b/configs/docker-sandboxes.example.yml @@ -0,0 +1,73 @@ +github: + appId: 123456 + organization: your-org + privateKeyPath: ~/.config/ephemeral-action-runner/github-app.pem + apiBaseUrl: https://api.github.com + webBaseUrl: https://github.com + +image: + sourceType: docker-image + sourceImage: ghcr.io/catthehacker/ubuntu:full-latest + sourcePlatform: linux/amd64 + runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" + customInstallScripts: + # - examples/custom-install/install-extra-apt-tools.sh + hostTrustMode: overlay + hostTrustScopes: [system] + +pool: + instances: 1 + namePrefix: change-me-docker-sandboxes + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + +runner: + group: your-runner-group + # On Apple Silicon, replace X64 with ARM64 when provider.platform is linux/arm64. + labels: [self-hosted, linux, X64, epar-docker-sandboxes] + includeHostLabel: true + ephemeral: true + +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true + +provider: + type: docker-sandboxes + # Exact host mapping: Windows/Linux amd64 -> linux/amd64; macOS arm64 -> linux/arm64. + platform: linux/amd64 + +dockerSandboxes: + # The wizard reads and records the exact active host-global policy fingerprint. + policyGeneration: sha256:abcdef0123456789abcdef0123456789abcdef0123456789abcdef0123456789 + networkBaseline: open + additionalAllow: + - api.github.com + - '*.githubusercontent.com:443' + additionalDeny: + - telemetry.example.invalid + stagingRoot: .local/docker-sandboxes-staging + cpus: 4 + memory: 8GiB + # EPAR derives this sparse logical maximum from the selected image and stores the effective size in the active artifact receipt. + rootDisk: auto + # Independent sparse workload capacity for Docker inside the sandbox. + dockerDisk: 50GiB + maxConcurrentCreates: 2 + +timeouts: + bootSeconds: 180 + githubOnlineSeconds: 180 + commandSeconds: 900 diff --git a/configs/tart.example.yml b/configs/tart.example.yml index efe103e..2932a04 100644 --- a/configs/tart.example.yml +++ b/configs/tart.example.yml @@ -13,6 +13,8 @@ image: upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" customInstallScripts: # - examples/custom-install/install-extra-apt-tools.sh @@ -24,6 +26,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -45,10 +56,19 @@ logging: retentionIntervalMinutes: 60 runner: + group: your-runner-group labels: [self-hosted, linux, ARM64, epar-tart-ubuntu-24.04-base] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true + provider: type: tart sourceImage: epar-ubuntu-24-arm64 diff --git a/configs/tart.web-e2e.example.yml b/configs/tart.web-e2e.example.yml index ad7fce8..e21957b 100644 --- a/configs/tart.web-e2e.example.yml +++ b/configs/tart.web-e2e.example.yml @@ -11,6 +11,8 @@ image: upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" customInstallScripts: - scripts/guest/ubuntu/install-web-e2e.sh @@ -22,6 +24,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -43,10 +54,19 @@ logging: retentionIntervalMinutes: 60 runner: + group: your-runner-group labels: [self-hosted, linux, ARM64, epar-tart-ubuntu-24.04-web-e2e, epar-tart-rosetta-amd64] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true + provider: type: tart sourceImage: epar-ubuntu-24-arm64-web-e2e diff --git a/configs/wsl.example.yml b/configs/wsl.example.yml index 2caf870..043b4b5 100644 --- a/configs/wsl.example.yml +++ b/configs/wsl.example.yml @@ -13,6 +13,8 @@ image: upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" customInstallScripts: # - examples/custom-install/install-extra-apt-tools.sh @@ -24,6 +26,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -45,10 +56,19 @@ logging: retentionIntervalMinutes: 60 runner: + group: your-runner-group labels: [self-hosted, linux, X64, epar-wsl-catthehacker-ubuntu] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true + provider: type: wsl sourceImage: work/images/epar-wsl-catthehacker-ubuntu.tar diff --git a/configs/wsl.lean.example.yml b/configs/wsl.lean.example.yml index 6fdd411..2c7a0a8 100644 --- a/configs/wsl.lean.example.yml +++ b/configs/wsl.lean.example.yml @@ -12,6 +12,8 @@ image: upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" customInstallScripts: # - examples/custom-install/install-extra-apt-tools.sh @@ -23,6 +25,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -44,10 +55,19 @@ logging: retentionIntervalMinutes: 60 runner: + group: your-runner-group labels: [self-hosted, linux, X64, epar-wsl-ubuntu-24.04-base] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true + provider: type: wsl sourceImage: work/images/epar-ubuntu-24-wsl.tar diff --git a/configs/wsl.web-e2e.example.yml b/configs/wsl.web-e2e.example.yml index b454272..787d0d1 100644 --- a/configs/wsl.web-e2e.example.yml +++ b/configs/wsl.web-e2e.example.yml @@ -12,6 +12,8 @@ image: upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" customInstallScripts: - scripts/guest/ubuntu/install-web-e2e.sh @@ -23,6 +25,15 @@ pool: replacementRetryMaxSeconds: 1800 replacementRetryMultiplier: 2 replacementRetryJitterPercent: 20 + +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB + logging: directory: work/logs managerSinks: [console] @@ -44,10 +55,19 @@ logging: retentionIntervalMinutes: 60 runner: + group: your-runner-group labels: [self-hosted, linux, X64, epar-wsl-ubuntu-24.04-web-e2e] includeHostLabel: true ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true + provider: type: wsl sourceImage: work/images/epar-ubuntu-24-wsl-web-e2e.tar diff --git a/docs/README.md b/docs/README.md new file mode 100644 index 0000000..19a1148 --- /dev/null +++ b/docs/README.md @@ -0,0 +1,37 @@ +# EPAR documentation + +Use these guides after the short [README quick start](../README.md). Start with the task you need to complete, then open a provider guide only when choosing or changing the runner environment. + +## Start and configure + +- [Usage](usage.md): start, initialize, verify, inspect, clean up, labels, and dry runs. +- [GitHub App setup](github-app.md): create the App EPAR uses for short-lived registration tokens. +- [Runner group security](runner-groups.md): restrict which repositories can route work to your runners. +- [Configuration](configuration.md): edit local configuration and provider defaults. + +## Choose a provider + +- [Docker Container](providers/docker-container.md): disposable containers with a private Docker daemon. +- [Docker Sandboxes](providers/docker-sandboxes.md): dedicated microVM runners and guided template provisioning. +- [WSL](providers/wsl.md): disposable Windows WSL2 runners. +- [Tart](providers/tart.md): experimental Apple Silicon ARM64 Linux VMs. + +## Operate and maintain + +- [Operations](operations.md): supervision, capacity, cleanup, recovery, and maintenance. +- [Troubleshooting](troubleshooting.md): symptom-first diagnostics. +- [Logging](logging.md) and [Storage](storage.md): retention, capacity, and exact cleanup boundaries. +- [Image customization](image-build.md): build layers and custom install scripts. +- [Docker Sandboxes templates](advanced/docker-sandboxes-template.md): build, verify, import, size, and retain exact templates. +- [Cross-architecture containers](advanced/cross-architecture-containers.md): image platforms, emulation, labels, and verification. +- [Docker registry mirrors](advanced/docker-registry-mirrors.md): an optional pull-time optimization. +- [Windows startup](advanced/windows-startup.md), [macOS startup](advanced/macos-startup.md), and [no-Go startup](advanced/no-go-install.md): host-specific launch help. + +## Safety and support + +- [Security](security.md): provider-dependent isolation, secrets, provider caveats, and private vulnerability reporting. +- [Support](../SUPPORT.md): information to collect before opening an issue. + +## Contribute + +Read [Contributing](../CONTRIBUTING.md), then use the [developer documentation](development/README.md) for architecture, extension contracts, provider work, and the live core-runner canary. diff --git a/docs/advanced/cross-architecture-containers.md b/docs/advanced/cross-architecture-containers.md new file mode 100644 index 0000000..7076753 --- /dev/null +++ b/docs/advanced/cross-architecture-containers.md @@ -0,0 +1,74 @@ +# Cross-architecture containers + +Use this guide when a trusted EPAR job must run a container image whose CPU architecture differs from the Linux Docker daemon that executes it. It explains the boundary between image selection, emulation, runner labels, and provider support. + +## Start with evidence + +Determine the runner and daemon architecture, inspect the image manifest, and inspect any Compose override before choosing a workaround: + +```bash +uname -m +docker info --format '{{.OSType}}/{{.Architecture}}' +docker image inspect --format '{{.Os}}/{{.Architecture}}' IMAGE +docker buildx imagetools inspect IMAGE +docker compose config +``` + +An x64 Linux daemon normally runs `linux/amd64` images natively; an ARM64 daemon normally runs `linux/arm64` images natively. A `platform:` value in Compose can override the image's normal selection. Pulling or loading a foreign image proves only that the daemon obtained it, not that it can execute it. + +| Symptom | Meaning | Correct next action | +| --- | --- | --- | +| `no matching manifest for linux/amd64` or `linux/arm64` | The registry does not publish the requested platform. | Choose an available image/platform or publish a multi-platform image. QEMU cannot create a missing manifest. | +| `exec format error` or `cannot execute binary file` | The daemon tried to launch an incompatible executable without a usable handler. | Confirm the selected platform and install/verify an emulator only if the workload supports it. | +| `qemu-x86_64: Could not open '/lib64/ld-linux-x86-64.so.2'` | Translation started but the foreign userspace/loader is missing or incompatible. | Use a compatible image or native runner; binfmt registration alone is insufficient. | +| Exit code `139` | A process segfaulted. | Treat architecture as one hypothesis, not proof; inspect the workload log and run a minimal container test. | +| Docker platform warning | Requested and detected platforms differ. | Verify execution; the warning alone is not a failure. | + +## Set up and verify Linux user-mode emulation + +For a trusted Linux job that intentionally runs a foreign Linux container, configure QEMU/binfmt before the first foreign container. Pin the action and helper image according to the repository's dependency policy: + +```yaml +jobs: + test: + runs-on: [self-hosted, linux, epar-docker-container-catthehacker-ubuntu] + steps: + - name: Set up ARM64 container emulation + uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4 + with: + image: docker.io/tonistiigi/binfmt@sha256:400a4873b838d1b89194d982c45e5fb3cda4593fbfd7e08a02e76b03b21166f0 + platforms: arm64 + + - name: Verify the foreign container + run: docker run --rm --platform linux/arm64 alpine:3.22 uname -m +``` + +The verification should print `aarch64`. Select only the foreign platforms that the workflow needs. QEMU/binfmt translates Linux user-space executables; it does not change the runner CPU, create a foreign VM, make arbitrary host programs compatible, or guarantee workload performance. Emulated builds, databases, browsers, and compute-heavy tools can be slower or unsupported. + +The setup helper is privileged. Use it only in trusted workflows and treat its pinned action/image revisions as reviewed dependencies. Prefer a native matching runner whenever compatibility, performance, or a security boundary matters. + +## Provider and platform scope + +| Execution surface | What to do | +| --- | --- | +| Docker Container | Run the setup action inside the disposable job before Docker or Compose uses a foreign image. No EPAR configuration key enables universal emulation. | +| WSL | Run the setup action inside the WSL runner if its Linux Docker daemon needs a foreign image. An x64 WSL runner does not gain ARM64 execution merely by pulling an ARM64 image. | +| Tart on Apple Silicon | The guest is ARM64. The optional Rosetta path is experimental and not equivalent to QEMU/binfmt. Use a distinct label and validate the exact image/workload. | +| Docker Sandboxes | The pinned template and `provider.platform` determine the guest architecture. Treat unsupported host/template combinations as preview-only and use only independently validated combinations. | +| GitHub-hosted Windows or macOS | These labels do not replace a Linux Docker daemon for container actions or service containers. Use a suitable Linux execution surface. | + +Keep architecture-specific jobs on a distinct `runs-on` label. Do not label an ARM64 runner as `ubuntu-latest`: GitHub's `ubuntu-latest` is a GitHub-managed environment, and x64 assumptions can fail on ARM64. + +## Operational examples + +For an amd64-only service on an ARM64 host, first try a published ARM64 or multi-platform image. If none exists, use a trusted Linux runner with the emulation setup above, then prove the actual service starts and passes its health check. If the service is performance-sensitive or fails under emulation, route it to a native x64 Docker Container, WSL x64, or another native x64 Linux runner instead. + +For an ARM64 image on an x64 Linux runner, follow the same process with `platforms: arm64` and `--platform linux/arm64`. Never treat a successful `docker pull` as the proof; run a container and check both the expected architecture output and the real workload. + +## References + +- [Docker Setup QEMU action](https://github.com/docker/setup-qemu-action) +- [Docker multi-platform build strategies](https://docs.docker.com/build/building/multi-platform/) +- [GitHub-hosted runner labels and limitations](https://docs.github.com/en/actions/reference/runners/github-hosted-runners) +- [GitHub self-hosted runner container requirements](https://docs.github.com/en/actions/reference/runners/self-hosted-runners#requirements-for-self-hosted-runner-machines) + diff --git a/docs/advanced/docker-registry-mirrors.md b/docs/advanced/docker-registry-mirrors.md index e28f056..14cc0f2 100644 --- a/docs/advanced/docker-registry-mirrors.md +++ b/docs/advanced/docker-registry-mirrors.md @@ -22,7 +22,7 @@ Before enabling this config, provide one of these: - a mirror service running on another machine in the same LAN or intranet; - a managed registry cache, such as a cloud registry pull-through cache. -For a local mirror on the EPAR host, Docker Engine, Docker Desktop, or OrbStack is enough to run the mirror container. No extra EPAR package is required. +For a local mirror on the EPAR host, Docker is enough to run the mirror container. No extra EPAR package is required. For an intranet mirror, runners should use the mirror's LAN DNS name or IP address. This is often better for multiple office machines because all EPAR hosts can share one warm cache. @@ -48,7 +48,7 @@ The generated daemon config is equivalent to: } ``` -The same config surface works for Docker-DinD, Tart, and WSL when Docker is installed in the runner instance. +The same config surface works for Docker Container, Tart, and WSL when Docker is installed in the runner instance. ## What EPAR Does Not Run @@ -62,7 +62,7 @@ If the mirror is not running or is not reachable from the runner, Docker falls b ## Local Docker Hub Cache -For local development, a Docker Hub pull-through cache can run on the same host as EPAR. Docker, Docker Desktop, or OrbStack is enough to run the mirror container; no extra EPAR dependency is required. +For local development, a Docker Hub pull-through cache can run on the same host as EPAR. Docker is enough to run the mirror container; no extra EPAR dependency is required. For a quick public-image cache: @@ -107,7 +107,7 @@ docker: - http://host.docker.internal:5050 ``` -For Docker-DinD, EPAR adds Docker's `host.docker.internal:host-gateway` alias when any configured mirror uses `host.docker.internal`. On macOS Docker Desktop and OrbStack this name is usually already available; on Linux Docker Engine the alias helps runner containers reach a host-published mirror. +For Docker Container, EPAR adds Docker's `host.docker.internal:host-gateway` alias when any configured mirror uses `host.docker.internal`. Some host runtimes already provide this name; others support Docker's `host-gateway` token and can use the added alias. If the host runtime supports neither behavior, use a LAN address or DNS name reachable from the runner instead. For Tart and WSL, `host.docker.internal` may not resolve the way it does in Docker containers. Use a LAN address, DNS name, or other route that is reachable from the guest. @@ -117,7 +117,7 @@ The mirror URL must be valid from the runner instance's point of view: | Provider | Same-host mirror URL guidance | | --- | --- | -| Docker-DinD | `http://host.docker.internal:5050` is a good one-machine choice when the local cache publishes host port `5050`. EPAR adds Docker's `host-gateway` alias for this name when needed. | +| Docker Container | `http://host.docker.internal:5050` is a good one-machine choice when the local cache publishes host port `5050`. EPAR adds Docker's `host-gateway` alias for this name when needed. | | Tart | Use an IP address or DNS name reachable from inside the VM, such as the host's LAN IP or an intranet DNS name. `host.docker.internal` is not guaranteed. | | WSL | Use an address reachable from inside the WSL distro. Depending on Windows and WSL networking, this may be the Windows host address, a LAN IP, or an intranet DNS name. `host.docker.internal` is not guaranteed. | @@ -129,7 +129,7 @@ docker: - http://docker-cache.office.example:5000 ``` -For one-machine Docker-DinD development, a host-published local cache is usually enough: +For one-machine Docker Container development, a host-published local cache is usually enough: ```yaml docker: @@ -141,7 +141,7 @@ docker: A mirror cannot bypass registry authorization. -For Docker Hub private images, keep doing `docker login` inside the GitHub Actions job with repository or organization secrets. Host-side `docker login` is not copied into EPAR runners, and EPAR does not bake Docker credentials into images. +For Docker Hub private images, keep doing `docker login` inside the GitHub Actions job with repository or organization secrets. Host-side `docker login` is not copied into EPAR runners, and EPAR does not bake Docker credentials into images. EPAR's current Docker Sandboxes template routes its private Docker daemon transparently so the host forward proxy cannot replace the job's registry authorization during normal operation; see [Docker Hub Credentials and Transparent Egress](../providers/docker-sandboxes.md#docker-hub-credentials-and-transparent-egress). Private pulls can use a mirror in two common ways: @@ -168,7 +168,7 @@ Start a runner instance with mirrors configured: ./bin/ephemeral-action-runner pool verify --instances 1 --cleanup ``` -For Docker-DinD, inspect the inner daemon: +For Docker Container, inspect the inner daemon: ```bash docker exec docker info diff --git a/docs/advanced/docker-sandboxes-template.md b/docs/advanced/docker-sandboxes-template.md new file mode 100644 index 0000000..2bb508a --- /dev/null +++ b/docs/advanced/docker-sandboxes-template.md @@ -0,0 +1,55 @@ +# Docker Sandboxes Template Build And Retention + +Docker Sandboxes requires an EPAR runner template built from the desired Catthehacker image and imported into the local Docker Sandboxes cache. `./start` performs this provisioning during first-run setup, and `./start image build` uses the same implementation. + +## Before You Start + +Use a native Docker server that matches the template platform. An amd64 server builds `linux/amd64`; an ARM64 server builds `linux/arm64`. The pinned build intentionally does not use emulation. Confirm Docker Sandboxes readiness before loading a template: + +```bash +sbx diagnose --output json +``` + +EPAR requires at least one diagnostic pass and no diagnostic failures. Warnings and skipped checks are accepted. When a diagnostic fails, review the failed item and its hint in the JSON output. The lock file at [`templates/docker-sandboxes/sources.lock.json`](../../templates/docker-sandboxes/sources.lock.json) pins build tooling and platform inputs. The selected Catthehacker source tag and Actions runner selector are resolved independently to exact immutable identities when the update schedule is due. + +## Build, Import, And Review + +Use `./start` for first-run provisioning or the shared image command afterward: + +```powershell +./start image build --replace +``` + +EPAR resolves the configured Catthehacker tag for the native platform, checks capacity, and has BuildKit stream one verified Docker-compatible archive directly to disk. EPAR verifies the archive tag, platform, labels, configuration, layers, and digests without loading it into Docker, imports it with `sbx template load`, and reads back the exact Sandbox-cache identity. Build metadata, provenance, an SPDX SBOM descriptor, compatibility evidence, and a software inventory are retained compactly; the active receipt is updated atomically only after every step succeeds. + +The compatibility scripts under `scripts/docker-sandboxes` delegate to this command. They no longer maintain a separate build or load implementation. + +## Configure And Prewarm + +Run `./start` with no configuration. The wizard offers `full-latest`, `act-latest`, `dotnet-latest`, `js-latest`, or another `catthehacker/ubuntu` tag; verifies the tag and native platform; validates optional custom install scripts; displays source, platform, size estimates, reserve, and duration; then saves the desired configuration after one confirmation. Normal startup performs the build and import, and a provisioning failure leaves the configuration available for a retry. + +After configuration, prewarm the selected template outside the job path: + +```powershell +powershell.exe -NoProfile -ExecutionPolicy Bypass -File scripts/build-native-controller.ps1 pool verify --config .local/docker-sandboxes.yml --project-root . --instances 1 --cleanup +``` + +Do not add `--register-only`. This creates, verifies, and exactly removes one unregistered sandbox without requesting a GitHub registration token. The first create can still be slow; later creates reuse the host-level template cache. + +## Capacity + +Use `rootDisk: auto` unless a deployment requires an explicit larger sparse logical maximum. EPAR derives the effective root from the exact expanded source estimate plus a 5 GiB customization allowance and 20 GiB writable headroom, rounded up to the next 10 GiB. `dockerDisk` is an independent sparse workload limit with a 50 GiB default and 1 GiB minimum. Physical admission uses only the estimated incremental host growth plus `storage.minimumFree`; it never reserves a percentage of the backing volume or adds both virtual maxima to host usage. + +The template cache and archive are host-cache measurements, not each sandbox's root-disk baseline. EPAR rechecks backing storage, configured reservations, and uncertain cleanup reservations before every create. A failed capacity admission does not silently choose another provider. + +## Retention And Replacement + +Docker Sandboxes retains loaded templates after individual sandboxes are deleted. At startup and after activating a replacement, EPAR removes a superseded template only when its catalog receipt, configurations, leases, and live-sandbox inventory prove it is unreferenced. It verifies absence after `sbx template rm `. Do not use `sbx reset`: it removes the whole Docker Sandboxes cache. + +The direct build does not create a Docker staging image. Once the imported template is authoritative, EPAR removes the transient archive workspace and retains only compact evidence. A completely verified archive left by an interrupted import can resume directly; partial or malformed archives are never activated. Prefix-era or shared templates remain report-only until `storage prune --legacy` produces an approved exact plan. EPAR never broadly prunes Docker images, BuildKit state, Docker Sandboxes templates, WSL distributions, or Tart images. Docker Desktop VHDX compaction is separate offline host maintenance. + +## Evidence And Certification + +The source lock pins build tooling, Tini, helper inputs, and platform-specific inputs. The default policy checks mutable source and Actions runner selectors weekly at 07:00 local time; `./start image update` checks immediately. EPAR activates a new immutable template only after build, import, and exact readback succeed. + +The current ARM64 path has pinned inputs and code support but no equivalent recorded native real-host lifecycle or independent-certification evidence. Treat it as capability-ready only after local admission and your own workload validation. An independent certification record, when available, must bind the reviewed native-controller source/build, full template identity, cache ID, metadata/archive digests, and reviewed evidence. diff --git a/docs/advanced/macos-startup.md b/docs/advanced/macos-startup.md index f29a858..5764b97 100644 --- a/docs/advanced/macos-startup.md +++ b/docs/advanced/macos-startup.md @@ -24,7 +24,7 @@ Double-click `.local/start-epar.command` in Finder or run it from Terminal: The script: - finds the EPAR source folder, such as when the script lives at `.local/start-epar.command`; -- delegates to `./start`, which runs EPAR with `go run ./cmd/ephemeral-action-runner` if Go is installed and working, or a containerized `go run` if not (see [No Go Install](#no-go-install)); +- delegates to `./start`, which runs EPAR with `go run ./cmd/ephemeral-action-runner` if Go is installed and working, or uses a containerized toolchain to build and cache a native controller if not (see [No Go Install](#no-go-install)); - uses `.local/config.yml` by default; - waits for Docker to become ready before starting EPAR; - starts an existing `epar-dockerhub-cache` mirror container if one exists; @@ -56,7 +56,7 @@ If you use the optional Docker registry mirror, create the mirror container sepa ## No Go Install -`start-epar.command` delegates to `./start` at the repo root. If `go` isn't on `PATH`, or the `go` found there doesn't actually run (stale/wrong-architecture installs happen — see below), `./start` runs EPAR straight from source with a containerized Go toolchain (`scripts/run-with-docker.sh`) instead of failing. No binary is built or left on disk either way. Docker is required in both cases. +`start-epar.command` delegates to `./start` at the repo root. If `go` isn't on `PATH`, or the `go` found there doesn't actually run (stale/wrong-architecture installs happen — see below), `./start` uses a containerized Go toolchain through `scripts/run-with-docker.sh` to cross-compile a CGO-disabled native controller, cache it under `.local/bin`, and run it on the host. Docker is required for this fallback build path. To force this path even when Go is installed (for example, to avoid rebuilding via `go run` on every start), set in your local copy: @@ -124,6 +124,6 @@ launchctl bootout "gui/$(id -u)" ~/Library/LaunchAgents/com.example.epar.plist - `start` cleans up prefixed instances when it exits. Use `--keep-on-exit` only for debugging. - The first run can take a while because `start` may build or refresh the configured image before starting runners. -- If Docker Desktop, OrbStack, or Docker Engine cannot start, the script exits before EPAR starts. -- For Docker-DinD, the host Docker runtime must support privileged containers. +- If Docker cannot start, the script exits before EPAR starts. +- For Docker Container, the host Docker runtime must support privileged containers. - For Tart-only pools that do not use host Docker or a local registry mirror, disable the Docker wait in your local copy. diff --git a/docs/advanced/no-go-install.md b/docs/advanced/no-go-install.md index a9ec253..a8f79d6 100644 --- a/docs/advanced/no-go-install.md +++ b/docs/advanced/no-go-install.md @@ -1,10 +1,10 @@ # Running EPAR Without Installing Go -The standard path is to download GitHub's automatic **Source code (zip)** or **Source code (tar.gz)** from the [EPAR Releases page](https://github.com/solutionforest/ephemeral-action-runner/releases) and run `go run ./cmd/ephemeral-action-runner` from the extracted source folder. If you do not want Go installed on the host, use EPAR's Docker-based source runner instead. The default Docker-DinD provider still needs Docker. +The standard path is to download GitHub's automatic **Source code (zip)** or **Source code (tar.gz)** from the [EPAR Releases page](https://github.com/solutionforest/ephemeral-action-runner/releases) and run `./start` from the extracted source folder. The wrapper uses local Go when it works; if you do not want Go installed on the host, it uses EPAR's Docker-based native-controller builder instead. Docker remains required for that build toolchain. ## Run With Docker -Run `./start` at the source folder root on macOS, Linux, WSL, or Git Bash, or `./start.ps1` / `start.cmd` in native Windows PowerShell or cmd. The wrapper uses local Go when it is installed and runnable; otherwise it runs EPAR from the checked-out source with a containerized Go toolchain: +Run `./start` at the source folder root on macOS, Linux, WSL, or Git Bash, or `./start.ps1` / `start.cmd` in native Windows PowerShell or cmd. The wrapper uses local Go when it is installed and runnable; otherwise a containerized Go toolchain cross-compiles a CGO-disabled host-native binary at `.local/bin/ephemeral-action-runner` (or `.exe` on Windows). Its adjacent `ephemeral-action-runner.manifest` records the deterministic source, platform, and toolchain fingerprint: ```bash ./start --config .local/config.yml --instances 2 @@ -14,25 +14,19 @@ Run `./start` at the source folder root on macOS, Linux, WSL, or Git Bash, or `. .\start.ps1 --config .local\config.yml --instances 2 ``` -Under the hood, the wrapper calls `scripts/run-with-docker.sh` (or `scripts/run-with-docker.ps1` on Windows), builds a small local image with the Go toolchain and Docker CLI from `scripts/docker/dev.Dockerfile`, and runs `go run ./cmd/ephemeral-action-runner ...`. This stays source-based; it does not download or use a separately packaged EPAR executable. The Docker CLI in the toolchain image lets EPAR's own Docker operations reach the host daemon through the mounted socket. +Under the hood, the wrapper calls `scripts/run-with-docker.sh` or `scripts/run-with-docker.ps1`, builds the small toolchain image from `scripts/docker/dev.Dockerfile`, compiles EPAR with `CGO_ENABLED=0`, and runs the cached host-native binary. This stays source-based and does not download a separately packaged EPAR executable. Native execution is mandatory for Docker Sandboxes because its management endpoint and host security state are not exposed to the build container. -### Host trust bridge +### Host trust -When the selected config enables `image.hostTrustMode: overlay`, the controller must inherit trust from the real Windows, macOS, or Linux host, not from the temporary Linux Go-toolchain container. The official wrappers handle this boundary automatically: +Before compiling the native controller, the wrapper publishes a short-lived system-trust feed from the real Windows, macOS, or Linux host, validates its freshness, certificate hashes, CA constraints, and distrust entries in an offline container, and mounts the resulting bundle read-only into the Go toolchain container. This operational trust is automatic even when `image.hostTrustMode` is disabled and is not copied into runners. After a native controller is available, it reads the host trust stores directly; only the legacy containerized-controller path uses the separate short-lived `EPAR_BUILD_TRUST_FEED` and optional `EPAR_HOST_TRUST_FEED` bridge. If bootstrap TLS still fails, the wrapper preserves `work/logs/epar-native-controller-build.log` and prints a certificate diagnostic without disabling verification or retrying insecurely. -1. A native host helper reads the configured host trust scopes. -2. It rejects an empty collection and publishes a fresh, content-addressed snapshot in the host user's cache. -3. A watcher refreshes the snapshot every 10 seconds. -4. The wrapper bind-mounts only that feed read-only at `/run/epar-host-trust` and identifies the real controller host OS. -5. The containerized controller validates certificate hashes, scopes, host OS, and the 30-second feed expiry before using it. +The explicit legacy path `EPAR_LEGACY_CONTROLLER_IN_DOCKER=1` remains available only for compatible providers. That path uses the existing short-lived host-trust bridge and rejects `provider.type: docker-sandboxes`; it is not an automatic fallback. -On a first `start` with no config, the wrapper performs this as two phases: it runs interactive initialization with the real host OS identity, validates and publishes the selected host roots, then starts the controller again with the new feed mounted. A failed collection leaves the generated config disabled and does not start a controller with the toolchain container's trust store. +On a first `start` with no config, the native controller performs interactive initialization with the real host identity. A failed host-root collection stops initialization rather than using roots from the toolchain container. On Windows the helper reads local-machine and current-user root stores and excludes Windows-disallowed certificates. On macOS it evaluates the system, administrator, and selected user's native trust settings for TLS server use, with explicit deny taking precedence. On Linux it reads the distribution-generated system CA bundle; set `EPAR_HOST_TRUST_BUNDLE` to a readable generated PEM bundle when the host uses an unsupported layout. -The watcher and controller are tied to the wrapper lifecycle. If the host helper stops, its feed expires and EPAR stops authorizing stale idle runners. The feed is never mounted into the disposable runner containers. - -Do not replace the official wrapper with a bare `docker run` when overlay mode is enabled. A containerized controller without `EPAR_CONTROLLER_HOST_OS`, a fresh `EPAR_HOST_TRUST_FEED`, and the corresponding read-only mount fails closed; it does not fall back to the toolchain container's CA store. +Do not replace the official wrapper with a bare `docker run` for Docker Sandboxes. EPAR rejects the legacy controller-in-Docker path for that provider. You can run the Docker wrapper directly instead of through `./start`: @@ -41,22 +35,24 @@ scripts/run-with-docker.sh version scripts/run-with-docker.sh start --config .local/config.yml ``` -Set `EPAR_USE_DOCKER_RUN=1` to force `./start` down this path even when Go is installed, or `=0` to force local `go run` and error instead of falling back. A Docker volume caches Go modules and build output across runs, so repeat starts are fast. +Set `EPAR_USE_DOCKER_RUN=1` to force `./start` to use the containerized compiler even when Go is installed, or `=0` to force local `go run` and error instead of falling back. Docker volumes cache Go modules and build output, while unchanged source reuses the one native binary. A changed source or toolchain rebuilds it atomically. If an existing EPAR process is still using the prior binary, stop that process before retrying; the wrapper does not retain another historical binary as a fallback. + +Before containerized compilation, the wrapper requires 1 GiB free on the source-folder filesystem so bootstrap does not begin when the host is already critically constrained. `EPAR_BOOTSTRAP_MIN_FREE_BYTES` may raise this reserve for managed installations. -The Docker wrapper passes the real host name into the toolchain container as `EPAR_HOST_NAME` so first-run defaults and generated host labels describe the machine running EPAR, not the temporary Go container. Set `EPAR_HOST_NAME` before launching EPAR to override that identity. +The wrapper passes the real host name to the native controller as `EPAR_HOST_NAME` so first-run defaults and generated host labels describe the machine running EPAR. Set `EPAR_HOST_NAME` before launching EPAR to override that identity. The wrapper also defaults `DOCKER_CLI_HINTS=false` for its Docker calls. This suppresses Docker Desktop hint text that can otherwise appear after a normal Ctrl-C shutdown. Set `DOCKER_CLI_HINTS=true` before launching EPAR if you want Docker CLI hints during wrapper runs. ### Linux file ownership -The container runs as root because the Go toolchain image requires it to write module and build-cache directories. On Docker Desktop for macOS, bind-mounted files remain owned by your normal user. On native Linux hosts, `scripts/run-with-docker.sh` returns ownership of `.local/` and `work/` to the invoking user after each run; do not rely on the same behavior for other repository directories. +The compiler container mounts the source read-only and writes only the temporary native binary output plus named Go caches. The controller itself runs as the invoking host user, so its `.local/` and `work/` state is not created by a root controller process. ### Windows: WSL versus native PowerShell -- From WSL2 or Git Bash, use `./start`. It behaves like the Linux case and needs Docker Desktop's WSL2 integration when run from WSL2. +- From WSL2 or Git Bash, use `./start`. It behaves like the Linux case and needs Docker to be available and working in that environment. - From native PowerShell or cmd, use `./start.ps1` or `start.cmd`, which use `scripts/run-with-docker.ps1` instead of the Bash script. -`start.ps1` and `scripts/run-with-docker.ps1` are less exercised than the Bash/macOS path. If you hit an issue, check whether Docker Desktop file sharing is enabled for the drive that holds the source folder. +`start.ps1` and `scripts/run-with-docker.ps1` are less exercised than the Bash/macOS path. If you hit an issue, check whether the host runtime can bind-mount the drive that holds the source folder. ## macOS Login Item Startup diff --git a/docs/advanced/windows-startup.md b/docs/advanced/windows-startup.md index 325deb0..acb3705 100644 --- a/docs/advanced/windows-startup.md +++ b/docs/advanced/windows-startup.md @@ -47,13 +47,13 @@ Create a user logon task: 1. Open **Task Scheduler**. 2. Choose **Create Task**. -3. On **Triggers**, add **At log on**. Add a short delay if Docker Desktop or another Docker daemon needs time to start. +3. On **Triggers**, add **At log on**. Add a short delay if Docker needs time to start. 4. On **Actions**, choose **Start a program**. 5. Set **Program/script** to `D:\path\to\ephemeral-action-runner\bin\ephemeral-action-runner.exe`. 6. Set **Add arguments** to `start --config .local\config.yml`. 7. Set **Start in** to `D:\path\to\ephemeral-action-runner`. -For Docker Desktop, keep the task as a user logon task. Docker Desktop is usually tied to the user session, so a boot-time system task may start too early or without the expected Docker context. +If the host runtime is tied to the user session, keep the task as a user logon task. A boot-time system task may start too early or without the expected Docker context. PowerShell equivalent: @@ -84,6 +84,6 @@ Unregister-ScheduledTask -TaskName "EPAR" -Confirm:$false ## Notes - Stop the foreground EPAR process or scheduled task to trigger normal cleanup. -- For Docker-DinD, the host Docker runtime must support privileged Linux containers. +- For Docker Container, the host Docker runtime must support privileged Linux containers. - For WSL, make sure WSL2 is installed and the configured WSL image has been built or can be built by `start`. - If the selected provider needs Docker, use a logon trigger with a delay so Docker has time to become ready. diff --git a/docs/background.md b/docs/background.md deleted file mode 100644 index b70b89b..0000000 --- a/docs/background.md +++ /dev/null @@ -1,25 +0,0 @@ -# Background - -The original macOS-runner direction used macOS VM images because those tools are commonly found when searching for Mac self-hosted runners. That is a poor fit for Docker Compose jobs inside the guest: Docker Desktop and OrbStack on macOS rely on their own Linux VM, so using them inside a macOS Tart VM becomes nested virtualization. - -For Docker container actions, service containers, and Compose-heavy jobs, use an EPAR image built with `scripts/guest/ubuntu/install-docker-browser.sh` or `scripts/guest/ubuntu/install-web-e2e.sh` so Docker Engine is installed directly inside the Linux guest. On an M3 Mac that means Ubuntu ARM64. Workflows must target self-hosted ARM labels such as: - -```yaml -runs-on: [self-hosted, linux, ARM64, m3-ubuntu-24.04-docker] -``` - -Do not label these runners as `ubuntu-latest`. GitHub-hosted `ubuntu-latest` is a GitHub-managed image environment, and x64 assumptions may break on ARM64. - -For Docker-heavy Linux CI, Docker-DinD is the recommended first path when the host already has a Docker runtime that supports privileged containers. It keeps workflow Docker resources inside a private inner daemon per runner instance, so existing Compose stacks with fixed project names or ports usually need fewer repository changes. - -WSL on Windows x64 remains a good EPAR provider for workflows that need native x64 Linux Docker images. Tart on Apple Silicon remains useful when you specifically want VM-based runners on a Mac host, but the VM is ARM64 unless you opt into and validate Rosetta support. Workflows that depend on amd64-only images should target a distinct label such as a Docker-DinD label with verified amd64 emulation, `epar-tart-rosetta-amd64`, a WSL x64 label, or another x64 Linux runner label. - -On Apple Silicon hosts, amd64 containers inside Docker-DinD depend on the host runtime's emulation support; validate `docker run --platform linux/amd64 alpine:3.20 uname -m` inside a running EPAR instance before routing amd64-only workflows there. - -## OCI Clarification - -OCI is a registry and artifact ecosystem, not a guarantee that an artifact can run as both a container and a VM. Docker container images and Tart VM images can both live in OCI-compatible registries, but their contents are different. Tart can pull Tart-created VM images from OCI registries; it cannot run arbitrary Docker container images as VMs. - -## Browser Caveat - -GitHub's upstream `install-google-chrome.sh` currently assumes x64 Linux Chrome/Chromium artifacts in important places. Docker/browser-enabled EPAR images therefore use that upstream script on x64 only. On ARM64, Ubuntu's `chromium-browser` package redirects to snap and can hang when the snap store is unreachable, so EPAR installs Playwright-managed Chromium and exposes it through `epar-browser`, `chromium`, and `chromium-browser` wrappers. Runtime validation exercises a real headless Chromium browser against a locally generated marker page when the Docker/browser feature marker is present. Network and TLS behavior remains the responsibility of each workflow. diff --git a/docs/configuration.md b/docs/configuration.md index 120b57f..aafdcda 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1,191 +1,252 @@ # Configuration -EPAR stores local settings in `.local/config.yml` by default. The first run creates that file for the default Docker-DinD setup when it does not exist. - -Use `.local/config.yml` for real GitHub App values, local paths, labels, and runner counts. Tracked files under `configs/` are examples. - -## Config Lookup - -EPAR looks for config in this order: - -1. `--config ` -2. `EPAR_CONFIG` -3. `./.local/config.yml` -4. `~/.config/ephemeral-action-runner/config.yml` - -## Sections - -| Section | Purpose | -| --- | --- | -| `github` | GitHub App ID, organization, private key path, and optional GitHub API/web URLs. | -| `provider` | How EPAR creates disposable runners: `docker-dind`, `wsl`, or `tart`. | -| `image` | Source image/rootfs, output image, runner version, and optional install scripts. | -| `pool` | Runner count, instance name prefix, and replacement retry policy. | -| `logging` | Manager and transcript sinks, formats, rotation, retention, and log directory. | -| `runner` | GitHub Actions labels, runner group, default-label policy, and whether to add the host-machine label. | -| `docker` | Optional Docker registry mirrors and Docker-DinD daemon proxy settings. | -| `timeouts` | Boot, GitHub online, and command timeout values in seconds. | +EPAR reads a small, strict YAML subset: indentation uses spaces, unknown sections and keys fail, comments outside quotes are ignored, and list values may use either `[one, two]` or an indented list. Put real credentials and machine-specific paths in `.local/config.yml`; tracked configuration files are examples. + +## Contents + +- [Lookup and parser rules](#lookup-and-parser-rules) +- [Provider matrix](#provider-matrix) +- [Configuration reference](#configuration-reference) +- [Cross-field rules](#cross-field-rules) +- [Provider defaults](#provider-defaults) +- [Short recipes](#short-recipes) + +## Lookup and parser rules + +EPAR chooses the first available configuration path in this order: `--config `, `EPAR_CONFIG`, `./.local/config.yml`, then `~/.config/ephemeral-action-runner/config.yml`. A relative file path in configuration is resolved from the project root when EPAR consumes it. `~` and `~/...` are expanded for the configuration path, `github.privateKeyPath`, and each `image.trustedCaCertificatePaths` entry; do not assume they expand in other configuration properties. + +The canonical config path owns its lifecycle-state namespace and may have only one active controller, even if its contents change while that controller runs. A separate host-wide prefix reservation prevents another config or project from using the same normalized `pool.namePrefix`. Distinct configs with distinct prefixes may run concurrently; use unique routing labels and separate log directories so jobs and diagnostics remain unambiguous. + +`github`, `image`, `pool`, `storage`, `logging`, `runner`, `security`, `provider`, `docker`, `dockerSandboxes`, and `timeouts` are the only accepted top-level sections. `security` contains only the `runnerGroup` subsection. Values are strings unless this reference says integer, number, boolean, or list. Quote a value when it needs YAML-like punctuation; EPAR removes one matching pair of single or double quotes. + +`pool.logDir` is a deprecated compatibility input. If `logging.directory` is absent, EPAR uses it and emits a warning; using both is rejected. `pool.vmPrefix` is an accepted alias for `pool.namePrefix`. `image.profile` and the old `docker-socket` provider are rejected rather than silently migrated. + +## Provider matrix + +| Provider | Host and artifact model | Image defaults | Provider-only configuration | +| --- | --- | --- | --- | +| `docker-container` | A Docker-compatible host creates an outer disposable runner with its own inner Docker daemon. | `docker-image`, `ghcr.io/catthehacker/ubuntu:full-latest`, output `epar-docker-container-catthehacker-ubuntu`. | Optional `provider.platform`; `docker` proxy and mirror settings apply to its private daemon. | +| `docker-sandboxes` | A host with healthy `sbx diagnose --output json` results builds and imports a Linux runner template. It is preview-only until the exact host/platform combination has independent live evidence. | `image.sourceImage` selects a `ghcr.io/catthehacker/ubuntu` tag; EPAR records the exact artifact in local state. | `dockerSandboxes` is required; `provider.platform` is `linux/amd64` or `linux/arm64`; runner-group enforcement must be `enforce`. | +| `wsl` | Windows WSL2 imports a Docker image or rootfs tar into disposable Linux distros. | Docker source defaults to Catthehacker full Ubuntu, x64, with output under `work/images/`. | `provider.installRoot` controls WSL storage. | +| `tart` | Experimental Apple Silicon Linux VM path. | `ghcr.io/cirruslabs/ubuntu:latest`, output `epar-ubuntu-24-arm64`. | `provider.network` and optional `provider.rosettaTag`. Validate the exact workload before relying on Rosetta. | + +Docker Sandboxes never falls back to Docker Container. Its wizard is available when the required tooling, daemon diagnostics, and host-platform mapping pass; storage is estimated after image selection but never blocks provider selection or configuration creation. After the configuration is saved, ordinary startup performs storage admission, source resolution, policy fingerprinting, template construction, import, and exact readback before any runner starts. Warnings and skipped diagnostics remain visible, and provisioning failures preserve the desired configuration without silently running an old artifact. + +## Configuration reference + +### `github` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `appId` | integer; no default | Required for GitHub operations. | GitHub App ID used to request short-lived runner registration tokens. | +| `organization` | string; no default | Required for GitHub operations. | GitHub organization that owns the runner group and runner records. | +| `privateKeyPath` | string; no default | Required for GitHub operations. | Private-key file readable by the EPAR process. Keep it under ignored `.local/` storage. | +| `apiBaseUrl` | string; `https://api.github.com` | Optional, for GitHub Enterprise Server API endpoints. | Trailing `/` is removed. | +| `webBaseUrl` | string; `https://github.com` | Optional, for GitHub Enterprise Server web endpoints. | Trailing `/` is removed. | + +### `image` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `sourceImage` | string; provider default | Image-building providers. | Docker image reference or WSL rootfs-tar path, selected by `sourceType`. | +| `sourceType` | `docker-image` or `rootfs-tar`; provider default | `rootfs-tar` is WSL-only in normal use. | A rootfs tar cannot use `sourcePlatform`. | +| `sourcePlatform` | Docker platform string; empty except WSL Docker source default `linux/amd64` | Only with `sourceType: docker-image`. | Requests the source image platform; pulling an image does not prove it can execute. See [Cross-architecture containers](advanced/cross-architecture-containers.md). | +| `outputImage` | string; provider default | Image-building providers. | EPAR-owned reusable runner artifact name or path. | +| `upstreamDir` | string; `third_party/runner-images` | Image builds that adapt upstream scripts. | Local checkout/cache location for the pinned upstream runner-image scripts. | +| `upstreamLock` | string; `third_party/runner-images.lock` | Image builds that adapt upstream scripts. | Lock file identifying the approved upstream revision. | +| `runnerVersion` | string; `latest` | Runner image builds. | Runner release selector. EPAR resolves it to an exact platform package and verified SHA-256 when a remote check is due. | +| `updateFrequency` | `daily`, `weekly`, `biweekly`, `monthly`, or `manual`; `weekly` | Mutable source-image tags and `runnerVersion: latest`. | Controls remote freshness checks only. Local configuration, scripts, trust inputs, EPAR assets, and missing or corrupt artifacts apply immediately. | +| `updateTime` | local 24-hour `HH:MM`; `"07:00"` | Automatic update frequencies. | Preferred local check time. Manual mode ignores it. | +| `customInstallScripts` | list of non-empty paths; empty | Optional image customization. | Scripts run while creating the runner image; treat them as trusted build input. | +| `trustedCaCertificatePaths` | list of non-empty paths; empty | Optional additional TLS roots. | PEM or DER CA files are validated, supplied to EPAR's operational builder, and installed in the runner artifact. They supplement, not replace, system or runner-overlay trust. | +| `hostTrustMode` | `disabled` or `overlay`; `disabled` | Optional host-root inheritance for ephemeral runners. | Controls runner inheritance only. `overlay` requires `runner.ephemeral: true`; it collects a current host-trust generation before registration and fails closed on an invalid or stale result. EPAR's owned builder independently receives operational system trust. | +| `hostTrustScopes` | list of `system`, `user`; `[system]` | Required and non-empty with `hostTrustMode: overlay`. | Windows/macOS may use `[system, user]`; Linux supports `[system]` only. This is root-anchor inheritance, not exact host TLS-policy emulation. | + +Runner host trust is a common ephemeral-runner contract, not a Docker Container-only configuration rule. The interactive Docker Container and Docker Sandboxes paths offer it, while the configuration validator applies the same overlay and ephemeral requirements independently of provider type. Operational builder trust is automatic and separate: system roots are supplied to EPAR's dedicated BuildKit builder even when runner overlay is disabled, user roots remain opt-in through runner overlay scope, and explicit CA paths apply to both paths. A host Docker daemon must separately trust a private registry before EPAR can pull an image; neither builder nor guest overlay can repair a failed host-daemon pull. + +The default update policy checks remotely mutable inputs weekly at 07:00 local time. Repeated starts before the next check verify and reuse local artifacts without contacting the image registry or Actions runner release API. Run `./start image update` for an immediate check; `image build` remains the force-build command. Scheduled failures keep an exactly verified current generation available with visible persisted retry state, while user-requested local changes remain fail-closed. + +### `pool` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `instances` | integer; `1` | All providers. Must be at least `1`. | Strict cap for provisioning, ready, draining, quarantined, and cleanup-pending local instances. | +| `namePrefix` | 2-40 character name; provider default or wizard-derived machine name plus random suffix | All providers. | Literal prefix for local and GitHub identities. Keep it unique per machine/config and organization. Docker Sandboxes additionally permits only lowercase letters, digits, `-`, and `.`. | +| `vmPrefix` | deprecated alias for `namePrefix` | Existing configs only. | Do not set both aliases with conflicting intent; use `namePrefix` for new configs. | +| `logDir` | deprecated string path; no default | Existing configs only. | Used as `logging.directory` with a warning only when `logging.directory` is absent; using both is rejected. | +| `replacementRetryInitialSeconds` | integer; `15` | All providers. Greater than `0`. | Initial retry delay for transient replacement allocation failures. | +| `replacementRetryMaxSeconds` | integer; `1800` | All providers. At least the initial delay. | Upper retry-delay cap. | +| `replacementRetryMultiplier` | number; `2` | All providers. At least `1`. | Backoff multiplier. | +| `replacementRetryJitterPercent` | integer; `20` | All providers. `0` through `100`. | Randomizes retry delay to avoid synchronized retries. | + +GitHub `429` and `5xx` responses and transient network failures back off replacement allocation; a longer `Retry-After` wins. Invalid configuration, authentication failures, and initial startup remain fail-fast after compensating rollback. + +### `storage` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `minimumFree` | positive byte size; `1GiB` | All providers. | Fixed provider-neutral physical free-space reserve. Existing explicit values remain authoritative; the default does not scale with volume size. | +| `gracePeriod` | positive Go duration; `168h` | Conservative housekeeping. | Minimum age before abandoned or incomplete EPAR temporary work can be removed. It does not delay cleanup of a verified superseded generation. | +| `keepPrevious` | integer; `0` | Conservative housekeeping. Must be non-negative. | `0` allows immediate exact retirement after replacement. A positive value defers automatic artifact retirement to the explicit storage-prune retention preview. | +| `automaticHousekeeping` | `conservative` or `disabled`; `conservative` | All providers. | Conservative mode reconciles interrupted exact-owned work at startup and removes unreferenced superseded resources after readback. It never runs a broad Docker or WSL prune. | +| `buildCacheLimit` | positive byte size; `20GiB` | Image-building cache. | Bounded EPAR BuildKit cache target. Existing explicit values remain authoritative. | +| `goCacheLimit` | positive byte size; `10GiB` | Native/no-Go Go build cache. | Bounded EPAR Go cache target. | + +### `logging` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `directory` | string; `work/logs` | All providers. Non-empty. | Root for manager, instance, build, error, and benchmark logs. | +| `managerSinks` | non-empty list of `console`, `file`; `[console]` | Manager events. | Choose user-facing manager event destinations. | +| `managerConsoleFormat` | `text` or `json`; `text` | Manager console sink. | Console encoding. | +| `managerConsoleTextFormat` | one-line template; empty | Only when manager console format is `text`. | May use `{time}`, `{level}`, `{message}`, `{attributes}` and must contain `{message}`. | +| `managerFileFormat` | `text` or `json`; `json` | Manager file sink. | File encoding. | +| `transcriptSinks` | non-empty list of `console`, `file`; `[file]` | Raw instance/build transcripts. | Default keeps verbose transcript events out of the console. | +| `transcriptConsoleFormat` | `text` or `json`; `text` | Transcript console sink. | Console encoding. | +| `transcriptConsoleTextFormat` | one-line template; empty | Only when transcript console format is `text`. | May use `{time}`, `{instance}`, `{component}`, `{stream}`, `{message}`, `{session}`, `{category}`, `{provider}`, `{attributes}` and must contain `{message}`. | +| `maxFileSizeMiB` | integer at least `1`; `100` | Rotated logs. | Per-file rotation threshold. | +| `maxBackups` | integer at least `1`; `3` | Rotated logs. | Number of rotated files to retain per stream. | +| `compressBackups` | boolean; `true` | Rotated logs. | Compresses rotated backups. | +| `retentionEnabled` | boolean; `true` | Log retention. | Enables periodic age and total-size retention. | +| `retentionMaxTotalMiB` | integer at least `1`; `1024` | Log retention. | Maximum retained logging size. | +| `managerMaxAgeDays` | integer at least `1`; `14` | Log retention. | Manager-log age limit. | +| `instanceMaxAgeDays` | integer at least `1`; `14` | Log retention. | Instance-transcript age limit. | +| `buildMaxAgeDays` | integer at least `1`; `14` | Log retention. | Build-transcript age limit. | +| `errorMaxAgeDays` | integer at least `1`; `30` | Log retention. | Error-report age limit. | +| `benchmarkMaxAgeDays` | integer at least `1`; `90` | Log retention. | Startup-benchmark age limit. | +| `retentionIntervalMinutes` | integer at least `1`; `60` | Log retention. | Periodic retention interval. | + +### `runner` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `labels` | non-empty list; provider default | All providers. Each label is at most 256 characters. | GitHub Actions routing labels. Keep architecture-sensitive workflows on an explicitly compatible label. | +| `group` | string; empty | Optional organization runner group. | Group must exist and pass the configured runner-group policy. | +| `includeHostLabel` | boolean; `true` | All providers. | Adds sanitized `epar-host-` unless already listed. Set false when workflows must not route by host. | +| `ephemeral` | boolean; `true` | All providers. | Required by Docker Sandboxes and host-trust overlay; each runner accepts one job. | +| `noDefaultLabels` | boolean; `false` | Optional GitHub registration behavior. | Omits GitHub's default self-hosted, OS, and architecture labels, so workflows must use explicitly configured labels. | + +### `security.runnerGroup` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `enforcement` | `warn` or `enforce`; `warn` | All providers; Docker Sandboxes requires `enforce`. | `warn` records a policy failure and continues; `enforce` blocks registration. | +| `requireExplicitGroup` | boolean; `true` | Runner-group preflight. | Requires `runner.group` rather than an implicit default group. | +| `requireNonDefaultGroup` | boolean; `true` | Runner-group preflight. | Rejects the organization default group when enforcement applies. | +| `requiredRepositoryAccess` | `selected`, `private`, or `all`; `selected` | Runner-group preflight. | Maximum allowed repository breadth: selected only, all-private or narrower, or any visibility. | +| `requirePublicRepositoriesDisabled` | boolean; `true` | Runner-group preflight. | Requires the group not to be usable by public repositories. | + +If the complete subsection is absent, EPAR warns and uses the strict recommended checks in `warn` mode. New wizard configurations write an explicit policy. See [Runner Group Security](runner-groups.md). + +### `provider` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `type` | `tart`, `wsl`, `docker-container`, or `docker-sandboxes`; `tart` before provider defaults | Required. | Selects the provider. `docker-socket` is intentionally rejected because EPAR uses a private daemon for Docker Container. | +| `sourceImage` | string; image output for image-building providers, empty for Docker Sandboxes | Required except Docker Sandboxes. | Reusable artifact cloned by Tart, WSL, or Docker Container. Docker Sandboxes rejects it. | +| `network` | string; `default` | Tart image build and runtime. | Tart network mode. Do not assume this configures Docker or Docker Sandboxes networking. | +| `rosettaTag` | simple virtiofs tag; empty | Tart only. | Enables the experimental Tart Rosetta path. Validate each amd64 workload and label it distinctly. | +| `installRoot` | string; `work/wsl` | WSL. | Project-relative WSL distribution storage root. | +| `platform` | Docker platform string; empty except Docker Sandboxes default `linux/amd64` | Docker Container or Docker Sandboxes only. | Docker Sandboxes accepts only `linux/amd64` or `linux/arm64`; it also determines the default architecture label. | + +### `docker` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `registryMirrors` | list of root `http`/`https` URLs; empty | Private Docker daemon users. | Mirrors must not include credentials, query, fragment, or a non-root path. See [Docker Registry Mirrors](advanced/docker-registry-mirrors.md). | +| `httpProxy` | root `http`/`https` URL; empty | Private Docker daemon users. | Becomes `HTTP_PROXY` for the outer Docker Container runner and its inner daemon. Credentials are rejected. | +| `httpsProxy` | root `http`/`https` URL; empty | Private Docker daemon users. | Becomes `HTTPS_PROXY`; credentials are rejected. | +| `noProxy` | comma-separated host/domain/IP/CIDR/`*`; empty | Private Docker daemon users. | Becomes `NO_PROXY`; whitespace, URLs, credentials, empty entries, and invalid CIDRs are rejected. | + +### `dockerSandboxes` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `image.sourceImage` | `ghcr.io/catthehacker/ubuntu:full-latest` | Required with Docker Sandboxes. | Desired Catthehacker source selector; EPAR builds and imports the runnable template automatically. | +| `policyGeneration` | lowercase `sha256:<64-hex>`; no default | Required with Docker Sandboxes. | Recorded fingerprint of the host-global Balanced policy. | +| `networkBaseline` | `open` or `balanced`; `open` | Docker Sandboxes. | `open` adds a sandbox-scoped public-egress rule while denying host aliases; it does not change the host-global policy. | +| `additionalAllow` | unique hostname or `*.domain`, optional port; empty | Docker Sandboxes. | Adds sandbox-scoped allow resources. With `open`, it cannot re-allow EPAR's host-alias deny guardrails. | +| `additionalDeny` | unique hostname or `*.domain`, optional port; empty | Docker Sandboxes. | Adds sandbox-scoped deny resources. A resource cannot be in both allow and deny lists. | +| `stagingRoot` | canonical project-relative `.local/...` path; `.local/docker-sandboxes-staging` | Docker Sandboxes. | Per-create staging root; cannot be absolute, escape `.local`, or overlap `.local/bin` or `.local/state`. | +| `cpus` | positive integer; `4` | Docker Sandboxes. | CPU allocation for each sandbox. | +| `memory` | positive byte size; `8GiB` | Docker Sandboxes. | Per-sandbox memory allocation written by the wizard. | +| `rootDisk` | `auto` or byte size at least `20GiB`; `auto` | Docker Sandboxes. | Sparse guest-root logical maximum. `auto` is recalculated for each artifact as the expanded image estimate plus 5 GiB build allowance and 20 GiB writable headroom, rounded up to 10 GiB. An explicit undersized value is rejected before creation. | +| `dockerDisk` | byte size at least `1GiB`; `50GiB` | Docker Sandboxes. | Independent sparse logical maximum for the Docker daemon inside the sandbox; it is workload capacity and is not derived from the base image. | +| `maxConcurrentCreates` | positive integer; `2` | Docker Sandboxes. | Limits concurrent sandbox creation to control capacity pressure. | + +### `timeouts` + +| Property | Type and default | Required or applies when | Effect and caution | +| --- | --- | --- | --- | +| `bootSeconds` | integer; `180` | All providers. | Time allowed for instance boot/readiness. | +| `githubOnlineSeconds` | integer; `180` | All providers. | Time allowed for GitHub runner online readiness. | +| `commandSeconds` | integer; `900` | All providers. | Default bound for provider command execution. | + +## Cross-field rules + +- `provider.sourceImage` is required for Tart, WSL, and Docker Container, and forbidden for Docker Sandboxes. +- `provider.rosettaTag` is accepted only for Tart. `provider.platform` is accepted only for Docker Container or Docker Sandboxes. Docker Sandboxes accepts only `linux/amd64` and `linux/arm64`. +- Docker Sandboxes requires `runner.ephemeral: true`, `security.runnerGroup.enforcement: enforce`, a valid desired Catthehacker image, policy generation, resource values, and a lowercase-compatible pool prefix. +- `image.sourcePlatform` requires `image.sourceType: docker-image`; all byte-size fields require a positive `B`, `KiB`, `MiB`, `GiB`, or `TiB` value. +- Host-trust overlay requires a non-empty, duplicate-free scope list and `runner.ephemeral: true`; `user` is not supported on Linux. +- `pool.namePrefix` is a host-wide controller and ownership boundary. Tart, WSL, and Docker Container use the configured prefix to select legacy owned resources; Docker Sandboxes uses its durable ledger of exact owned identities. EPAR rejects concurrent reuse across configs, projects, and providers; do not assume broad prefix cleanup is safe. +- `runner.labels` must never be empty, even when `runner.noDefaultLabels` is false. + +## Provider defaults + +The configuration loader starts with Tart defaults, then applies provider-specific defaults for WSL, Docker Container, and Docker Sandboxes only when the corresponding key was not set explicitly. The first-run wizard writes a concrete configuration and derives a machine-based pool prefix; use its generated values as the normal starting point. + +| Provider | Source and output | Default labels and prefix | +| --- | --- | --- | +| Docker Container | Catthehacker full Ubuntu to `epar-docker-container-catthehacker-ubuntu`. | `self-hosted`, `linux`, `epar-docker-container-catthehacker-ubuntu`; prefix `epar-docker-container`. | +| Docker Sandboxes | Desired image settings plus policy generation; exact template identities live in the local artifact receipt. | `self-hosted`, `linux`, matching `X64`/`ARM64`, `epar-docker-sandboxes`; prefix `epar-docker-sandboxes`. | +| WSL Docker source | Catthehacker full Ubuntu, `linux/amd64`, output `work/images/epar-wsl-catthehacker-ubuntu.tar`. | `self-hosted`, `linux`, `X64`, `epar-wsl-catthehacker-ubuntu`; prefix `epar-wsl`. | +| WSL rootfs tar | `work/images/ubuntu-24.04-clean.rootfs.tar`, output `work/images/epar-ubuntu-24-wsl.tar`. | `self-hosted`, `linux`, `X64`, `epar-wsl-ubuntu-24.04-base`; prefix `epar-wsl`. | +| Tart | `ghcr.io/cirruslabs/ubuntu:latest` to `epar-ubuntu-24-arm64`. | `self-hosted`, `linux`, `ARM64`, `epar-tart-ubuntu-24.04-base`; prefix `epar`. | -## Common Edits +## Short recipes -Change how many runners stay online: +Keep two Docker Container runners warm: ```yaml pool: instances: 2 -``` - -Set a unique instance name prefix for each machine/config in the same GitHub organization: - -```yaml -pool: namePrefix: buildbox01-a4f9c2 ``` -`pool.namePrefix` is both the prefix for generated GitHub runner names and the cleanup boundary for GitHub runner records. It must be 2-40 characters and should leave room for EPAR's generated `-YYYYMMDD-HHMMSS-###` suffix. Do not reuse the same prefix on different machines or for separate EPAR supervisors in the same organization. If two machines share a prefix, one machine's cleanup can delete the other machine's GitHub runner record, causing the other supervisor to report that the runner record is gone and replace a healthy runner. - -Configure replacement retry behavior after a transient GitHub or network outage: - -```yaml -pool: - replacementRetryInitialSeconds: 15 - replacementRetryMaxSeconds: 1800 - replacementRetryMultiplier: 2 - replacementRetryJitterPercent: 20 -``` - -These values default to `15`, `1800`, `2`, and `20`, so existing configurations remain valid without changes. `replacementRetryInitialSeconds` must be positive, `replacementRetryMaxSeconds` must be at least the initial delay, `replacementRetryMultiplier` must be at least `1`, and `replacementRetryJitterPercent` must be from `0` through `100`. - -The supervisor backs off only replacement allocation after transient network errors and GitHub HTTP `429` or `5xx` responses. The nominal delay doubles from 15 seconds to a 30-minute cap with the configured jitter; a longer GitHub `Retry-After` response takes precedence. Authentication and deterministic configuration failures remain fail-fast after safe rollback. Initial `pool up` startup also remains fail-fast rather than entering an unattended retry loop. - -`pool.instances` is an absolute local physical-instance cap, not only an online-runner target. Provisioning, ready, draining, quarantined, and cleanup-pending instances all count toward it. Host-trust generation rotation does not receive surge capacity: an old busy runner keeps its slot until it exits or is safely removed. - -Add or change workflow labels: - -```yaml -runner: - labels: - - self-hosted - - linux - - epar-docker-dind-catthehacker-ubuntu -``` - -Disable the automatic host-machine label: +Use an explicit runner group with enforced least-breadth access: ```yaml runner: - includeHostLabel: false + group: trusted-ci +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true ``` -Register runners in an organization runner group and omit GitHub's automatic -`self-hosted`, operating-system, and architecture labels: - -```yaml -runner: - group: epar-ci-canary - labels: [epar-core-unique-label] - includeHostLabel: false - noDefaultLabels: true -``` - -`runner.group` is optional. The group must already exist and allow the target -repository to use it. `runner.noDefaultLabels` defaults to `false`; when it is -`true`, workflows must target labels explicitly configured under -`runner.labels` (and may also target the runner group). - -Use a different config file: - -```bash -go run ./cmd/ephemeral-action-runner start --config .local/wsl.yml -``` - -Configure logging and retention in the top-level `logging` section. The complete schema and local/Kubernetes examples are in [Logging](logging.md). Unknown configuration keys are rejected. For compatibility, a legacy `pool.logDir` value is used as `logging.directory` with a migration warning when the new key is absent; the file is not rewritten automatically. A configuration containing both keys is rejected as ambiguous. - -### Host trust inheritance - -Docker-DinD runners can inherit the host's trusted TLS root anchors: +Add a host and explicit enterprise root without weakening TLS verification: ```yaml image: hostTrustMode: overlay hostTrustScopes: [system, user] + trustedCaCertificatePaths: [.local/enterprise-root.pem] ``` -`image.hostTrustMode` accepts `disabled` or `overlay`. Existing configs default -to `disabled`. A new interactive Docker-DinD initialization asks whether to -enable host trust inheritance; pressing Enter accepts the displayed `yes` -default. Enabling the policy is the one-time consent for EPAR to follow later -host root additions, removals, and rotations automatically. - -The supported scopes are: - -| Controller host | `system` | `user` | -| --- | --- | --- | -| Windows | Local-machine trusted roots, excluding Windows-disallowed certificates | Current-user trusted roots, excluding Windows-disallowed certificates | -| macOS | System Roots plus CA certificates explicitly trusted for TLS server use in the administrator domain, excluding explicit deny | CA certificates in the user's keychain search list explicitly trusted for TLS server use, excluding explicit deny | -| Linux | The distribution's generated system CA bundle | Not supported | - -Use `[system, user]` on Windows or macOS when the runner should inherit the -same two trust scopes as the account running EPAR. Linux configs must use -`[system]`. Overlay mode is supported only for `provider.type: docker-dind` and -requires `runner.ephemeral: true`. -If macOS has disabled user-level Trust Settings, the `user` scope contributes -no certificates until that host policy is enabled again. - -The resulting Ubuntu runner trust is additive: - -```text -Ubuntu default roots -+ host roots from the current EPAR generation -+ image.trustedCaCertificatePaths -``` - -This is root-anchor inheritance, not exact emulation of Windows or macOS TLS -policy. macOS positive trust settings constrained by hostname, application, or -allowed error are not promoted into Ubuntu's unconstrained global root store. -Removing a host root does not remove an independently bundled Ubuntu root or a -certificate still listed under `trustedCaCertificatePaths`. +On Linux, use `hostTrustScopes: [system]`. Do not disable certificate verification to work around a private CA. -### Explicit CA paths - -Trust an additional enterprise TLS inspection or private package-registry CA: - -```yaml -image: - trustedCaCertificatePaths: - - .local/enterprise-root.pem -``` - -Paths may be repository-relative, absolute, or under `~/`. EPAR validates PEM -or DER X.509 CA certificates before building, normalizes them to deterministic -`.crt` files, and installs them before any `apt` or `curl` step. These paths are -independent of host trust inheritance and remain trusted until removed from the -config. - -Route the private Docker-DinD daemon through an enterprise network proxy: +Configure a private Docker proxy and mirror without embedding credentials: ```yaml docker: + registryMirrors: [https://mirror.example.test] httpProxy: http://proxy.example.test:3128 httpsProxy: http://proxy.example.test:3128 noProxy: localhost,127.0.0.1,.example.test ``` -These optional values become `HTTP_PROXY`, `HTTPS_PROXY`, and `NO_PROXY` on the -outer Docker-DinD container, so its inner `dockerd` inherits them at first -startup. Proxy URLs must not contain credentials. Keep machine-specific proxy -addresses in ignored `.local/config.yml`, not tracked example files. - -## Provider Defaults - -For `provider.type: docker-dind`, EPAR defaults to Catthehacker's full Ubuntu runner image and creates a Docker-DinD image named `epar-docker-dind-catthehacker-ubuntu`. - -For `provider.type: wsl`, EPAR defaults to Catthehacker's full Ubuntu runner image, converts it into a WSL rootfs, and stores the output under `work/images/`. - -For the experimental `provider.type: tart`, EPAR defaults to `ghcr.io/cirruslabs/ubuntu:latest`, a basic Ubuntu ARM64 VM image. EPAR installs its runner lifecycle but does not add the broad tool and dependency set found in GitHub's hosted runner images. If you require a GitHub-runner-like environment, build and maintain a bootable Tart image yourself by adapting the scripts in [actions/runner-images](https://github.com/actions/runner-images), then set `image.sourceImage` to that Tart image. Rosetta-based amd64 execution also has compatibility limits and must be validated against the exact workflow. - -See the provider docs for details: - -- [Docker-DinD Provider](providers/docker-dind.md) -- [WSL Provider](providers/wsl.md) -- [Tart Provider](providers/tart.md) +For Docker Sandboxes, run `./start` and let the wizard select the desired image, provision the native-platform runner template, and write the policy fingerprint and capacity settings. Do not hand-copy generated template identities between hosts; EPAR records them in the local artifact receipt. diff --git a/docs/core-runner-verification.md b/docs/core-runner-verification.md deleted file mode 100644 index bdf3c61..0000000 --- a/docs/core-runner-verification.md +++ /dev/null @@ -1,161 +0,0 @@ -# Level 1 Core Runner Verification - -The `Core runner verification` GitHub Actions workflow proves EPAR's central -contract against GitHub: create an isolated Docker-DinD runner, register it for -one job, replace it after that job, run a second job on the replacement, and -clean up the runner records and outer containers. - -This is an infrastructure canary, not a language or framework compatibility -matrix. The workload checks checkout and artifact transfer, then exercises the -nested Docker daemon with Buildx, Docker Compose, a health check, and an HTTP -request. - -## Architecture - -The workflow has three jobs: - -1. `Core runner controller` runs on a fresh, trusted GitHub-hosted runner. It - builds EPAR and the pinned lightweight core image, pre-cleans the - `epar-ci-core` boundary, and supervises one ephemeral runner. -2. `Core canary 1` runs on that ephemeral runner, validates its basic runtime, - and uploads its runner name and a nonce. -3. `Core canary 2` waits for the first job, runs on the replacement runner, - verifies that the runner name changed, downloads the artifact, and exercises - Buildx and Compose. - -The controller reads the workflow-job records to confirm that both canaries -used the expected group and unique per-run label, ran on distinct -`epar-ci-core-*` runners, and succeeded. Workflow concurrency is serialized -because every run intentionally shares the fixed cleanup prefix. - -## Required GitHub Setup - -### GitHub-hosted controller - -The controller uses the standard GitHub-hosted `ubuntu-latest` runner. No -pre-existing self-hosted controller is required. Each run receives a fresh -Linux VM with Docker, Bash, curl, jq, and GNU `timeout`; the workflow installs -the Go version declared by `go.mod` before building EPAR. - -### Restricted ephemeral-runner group - -In the organization settings, create a runner group named -`epar-ci-canary` and restrict its repository access to -`solutionforest/ephemeral-action-runner`. - -EPAR registers the temporary runners in this group with no GitHub default -labels and only a per-run label such as `epar-core-123456-1`. The canary jobs -target both the group and that unique label, so unrelated self-hosted runners -cannot accept them. - -### GitHub App and protected environment - -The GitHub App must be installed in the target organization and have organization self-hosted-runner read/write permission. Create a GitHub Actions environment named `epar-live-ci`, choose **Selected branches and tags**, allow `develop`, `main`, and `refs/pull/*/merge`, and add: - -| Kind | Name | Value | -| --- | --- | --- | -| Environment secret | `EPAR_GITHUB_APP_PRIVATE_KEY` | The complete PEM private key generated for the GitHub App. | -| Environment variable | `EPAR_GITHUB_APP_ID` | The numeric App ID. | -| Environment variable | `EPAR_GITHUB_ORGANIZATION` | The organization login, for example `solutionforest`. | - -Keep the PEM as an environment secret, including its original line breaks. Do -not store it in repository variables, workflow YAML, a tracked config file, or -an artifact. The workflow materializes it in a mode-restricted temporary file -on the controller and removes that file during cleanup. - -Requiring an environment reviewer is possible, but every matching push and same-repository pull request will wait for that approval. - -## Trust Boundaries and Triggers - -The controller job is privileged and secret-bearing. It receives the GitHub App key and a workflow token with Actions write permission, and it can start privileged containers in its disposable GitHub-hosted VM. It runs only for trusted repository changes: pushes to `develop` or `main`, and pull requests whose source branch is in this repository. Fork pull requests still trigger the workflow, but all three core-verification jobs skip before they can receive environment secrets or run on the canary group. - -The two canary jobs do not receive the GitHub App key. They receive only the -minimum workflow permissions needed for checkout and artifact operations, and -their Docker workloads run in the private daemon inside a disposable outer -container. - -The workflow runs on: - -- pull requests targeting `develop` or `main` (fork pull requests still trigger this workflow, but core-verification jobs run only when the source branch belongs to this repository) -- pushes to `develop` or `main` -- manual `workflow_dispatch` after the workflow is present on the repository's default branch - -The workflow does not use `pull_request_target`. Keep the same-repository job guard on every core-verification job so code from a forked pull request cannot reach the controller, environment secrets, or privileged Docker host. - -## Expected Result and Cleanup - -A successful run reports both canary runner names in the job summary. They must -be different, begin with `epar-ci-core-`, belong to `epar-ci-canary`, and carry -the run's `epar-core--` label. The second canary must also pass -the artifact, Buildx, Compose, container health, and HTTP checks. - -The controller performs cleanup before and after the canaries. It stops the -pool supervisor, deletes GitHub runner registrations within the exact -`epar-ci-core` prefix boundary, removes matching outer Docker-DinD containers, -and deletes its temporary key, generated config, and logs. Before failed-run -cleanup deletes those logs, the controller prints a sanitized final 200 lines -from the pool-supervisor log and each available runner log. Runner launch or -online readiness failures first append bounded process state, `run.log`, -latest `Runner_*.log`, and Docker-DinD daemon tails to the host guest log, so -those diagnostics pass through the same sanitizer before cleanup. A controller -failure then attempts cleanup before canceling the workflow so queued canary -jobs do not remain indefinitely. The next run also pre-cleans the same boundary. - -A sudden controller failure can bypass application cleanup. GitHub discards the -hosted VM and its Docker containers, while the next run pre-cleans stale GitHub -runner registrations within the same prefix boundary. - -## Troubleshooting - -### Controller job remains queued - -- Confirm the organization and repository allow standard GitHub-hosted - runners. -- Check GitHub Actions service status and the account's hosted-runner - concurrency. - -### Controller starts, but canaries remain queued - -- Open the controller log and check image-build errors plus the grouped, - sanitized pool-supervisor and runner-log tails printed before cleanup. -- Confirm `epar-ci-canary` exists with that exact spelling and permits this - repository. -- Confirm the GitHub App is installed in the organization and can administer - organization self-hosted runners. -- Confirm the environment's App ID is numeric, organization value is the login - rather than a display name, and the private-key secret contains the complete - PEM. -- In the organization runner list, look for an online - `epar-ci-core-*` runner carrying the unique label shown in the workflow log. - -### Image build or Docker workload fails - -- Check controller disk space and Docker health. -- Confirm outbound access to the pinned Catthehacker and BusyBox images. -- For nested-Docker startup failures, inspect the grouped Docker-DinD runner-log - tail in the controller output. - -### Workflow is canceled after a controller error - -This is expected failure behavior. The controller cancels the run after its -bounded wait or another fatal orchestration error so unmatched canary jobs do -not stay queued. Diagnose the first controller error rather than treating the -cancellation itself as the root cause. - -## Manual Cleanup - -Use a local, untracked config based on -`configs/docker-dind.core.example.yml`, with the same organization, GitHub App -key path, and `pool.namePrefix: epar-ci-core`, then run: - -```bash -go run ./cmd/ephemeral-action-runner cleanup \ - --config .local/core-cleanup.yml \ - --project-root . -``` - -This removes matching organization runner records. Docker containers from the -controller run existed only on its disposable GitHub-hosted VM. If cleanup -still reports an error, remove remaining `epar-ci-core-*` records from the -organization's Actions runner settings. Do not use a broader prefix: EPAR -cleanup is deliberately bounded to `epar-ci-core` and `epar-ci-core-*`. diff --git a/docs/design.md b/docs/design.md deleted file mode 100644 index 5126491..0000000 --- a/docs/design.md +++ /dev/null @@ -1,82 +0,0 @@ -# Design - -EPAR has three main layers: - -- `cmd/ephemeral-action-runner`: CLI for image builds, pool lifecycle, verification, cleanup, and status. -- `internal/provider`: a local instance provider interface. Tart, WSL, and Docker-DinD are implemented providers. -- `internal/github`: GitHub App authentication and self-hosted runner API calls. - -```mermaid -flowchart LR - CLI["CLI commands"] --> Manager["Pool manager"] - Manager --> Provider["Provider interface"] - Manager --> GitHub["GitHub App client"] - Provider --> Tart["Tart"] - Provider --> WSL["WSL2"] - Provider --> DinD["Docker-DinD"] - GitHub --> API["GitHub Actions runner APIs"] -``` - -## Lifecycle - -For each runner instance: - -1. Clone or create an instance from `provider.sourceImage`. -2. Start the instance headless when the provider supports that distinction. -3. Wait for provider-level reachability. -4. Apply optional Docker daemon registry mirror settings. -5. Run `/opt/epar/validate-runtime.sh`. -6. Fetch a short-lived GitHub registration token on the host. -7. Run `config.sh --ephemeral --unattended` inside the instance. -8. Start the runner process. VM and WSL images use systemd; Docker-DinD falls back to a PID-file managed background process. -9. Poll GitHub until the runner is ready. Verification-only flows require online/idle; supervised replacement pools also accept an online runner that is already busy with a queued job. -10. Monitor the runner service and GitHub runner record. -11. Delete the instance after the ephemeral runner exits, then create a replacement to maintain pool size. - -```mermaid -sequenceDiagram - participant M as Manager - participant P as Provider - participant I as Instance - participant G as GitHub - M->>P: Clone provider.sourceImage - M->>P: Start instance - M->>I: Apply optional Docker registry mirrors - M->>I: Validate runtime - M->>G: Request registration token - M->>I: Configure ephemeral runner - M->>I: Start runner process - I-->>G: Runner online - G-->>I: Assign one job - I-->>G: Runner exits after job - M->>P: Stop and delete instance - M->>G: Delete stale runner record if needed -``` - -## Multi-Instance Behavior - -`pool verify --instances 2 --register-only --cleanup` creates two instances concurrently, registers two ephemeral runners, verifies both are online/idle, and removes them. A supervised `pool up --replace-completed` accepts an online runner that is already busy, because requiring an idle observation would race immediate job assignment; the supervisor observes its completion and creates the replacement. - -`pool up --instances 2` keeps two runners available in the foreground. Replacement names use `pool.namePrefix` plus a timestamp and sequence suffix, for example `epar-20260703-010530-003`. - -The supervisor tracks every prefix-owned local resource as provisioning, ready, draining, quarantined, or cleanup-pending. All five states count against `pool.instances`, which is an absolute physical-instance cap. A cleanup failure, unknown remote state, or busy runner therefore blocks allocation rather than allowing a capacity surge; host-trust rotation follows the same rule. - -Reconciliation runs at startup and before each replacement allocation. It adopts healthy exact-name local/GitHub pairs, deletes proven stopped or unregistered local resources, and deletes exact stale GitHub records when GitHub is reachable. A pre-listener provisioning failure is locally rolled back immediately and queued for later exact-name GitHub reconciliation because registration may have partially succeeded. After the listener starts, uncertain GitHub readiness becomes quarantine rather than deletion until the remote state can be resolved. - -Supervised replacement treats network errors and GitHub HTTP `429` or `5xx` responses as transient: allocation pauses behind the configured exponential retry policy while monitoring and cleanup continue. A fully online replacement or successful adoption resets the retry delay. Initial pool startup remains fail-fast after safe rollback, and authentication or deterministic configuration failures are terminal rather than retryable. - -## Liveness Model - -The foreground supervisor checks each instance every 15 seconds by default. A runner is considered healthy when: - -- The matching GitHub runner record exists and reports `online`. -- A GitHub runner with `busy=true` is kept alive even if the local service check is temporarily inconclusive. -- When the runner is idle, `/opt/epar/check-runner.sh` reports the runner process is active. The script checks `actions-runner.service` on systemd instances and `/var/run/actions-runner.pid` on non-systemd instances such as Docker-DinD containers. - -The instance is retired when an idle runner process exits, the runner record disappears, or the runner reports a non-online status. Runner process exit is expected after an ephemeral runner finishes one job. - -## Provider Boundary - -The controller depends on provider operations for clone/create, start, exec, address discovery, stop, delete, and list. Provider implementations own host-specific details such as Tart VM names, WSL distro names, Docker-DinD container names, or future Hyper-V VM names. - -Docker-DinD is intentionally modeled as an instance provider, not a host Docker socket shortcut. Each instance is a privileged outer container that starts its own Docker daemon. Workflow `docker compose` resources are created inside that private daemon and disappear when EPAR removes the runner container with its volumes. diff --git a/docs/development/README.md b/docs/development/README.md new file mode 100644 index 0000000..0deebdd --- /dev/null +++ b/docs/development/README.md @@ -0,0 +1,11 @@ +# Development + +These documents are for contributors and maintainers. For installing, configuring, or operating EPAR, start with the [user documentation](../). + +- [Development and Extension Principles](principles.md) defines the product-wide contracts that changes must preserve. +- [Design](design.md) explains the provider-neutral controller and lifecycle responsibilities. +- [Adding a Provider](adding-provider.md) is the implementation and validation checklist for a new provider. +- [Core Runner Verification](core-runner-verification.md) documents the privileged live CI canary. +- [Releases](releases.md) documents the source-only release process. + +Read [Contributing](../../CONTRIBUTING.md) before opening a pull request. diff --git a/docs/development/adding-provider.md b/docs/development/adding-provider.md new file mode 100644 index 0000000..51c13c1 --- /dev/null +++ b/docs/development/adding-provider.md @@ -0,0 +1,16 @@ +# Adding A Provider + +Read [Development and Extension Principles](principles.md) and [Design](design.md). + +Put provider commands and host integration in `internal/provider/`. Shared onboarding, naming, image, pool lifecycle, GitHub, state, capacity, and retention behavior stays in its common package. + +A provider is complete only when it: + +- Registers its constructor, configuration rules, wizard contribution, reusable-artifact capabilities, and platform status in the provider registry. +- Implements every required lifecycle, exact inventory, artifact-requirement, storage-surface, ownership receipt, and crash-recovery contract without silent fallback. +- Uses the wizard’s complete machine-derived prefix and the shared `pool.RunnerName` format. +- Preserves strict `pool.instances`, durable exact identities, quarantine on uncertainty, resumable cleanup, diagnostics, and replacement. +- Reuses Catthehacker defaults when it consumes Docker runner images, unless an intentional exception is documented and tested. +- Adds provider contract tests, configuration and wizard tests, race tests, wrapper checks, and relevant live-platform evidence. + +Provider cleanup must target exact identities and record enough immutable evidence for common startup housekeeping to distinguish current, superseded, incomplete, and unknown resources. Never replace an unavailable exact operation with a prefix deletion, wildcard, broad prune, reset, or deletion of an unknown/shared resource. diff --git a/docs/development/core-runner-verification.md b/docs/development/core-runner-verification.md new file mode 100644 index 0000000..64d11ea --- /dev/null +++ b/docs/development/core-runner-verification.md @@ -0,0 +1,98 @@ +# Core Runner Verification + +The **Core runner verification** workflow proves EPAR's central contract against GitHub: create an isolated Docker Container runner, register it for one job, replace it after that job, run a second job on the replacement, and clean up the runner records and outer containers. + +This is an infrastructure canary, not a language or framework compatibility matrix. Its workload checks checkout and artifact transfer, then exercises the private Docker daemon with Buildx, Docker Compose, a health check, and an HTTP request. + +## Architecture + +The workflow has three jobs: + +1. **Core runner controller** runs on a fresh trusted GitHub-hosted runner, builds EPAR and the pinned lightweight image, pre-cleans the `epar-ci-core` boundary, and supervises one ephemeral runner. +2. **Core canary 1** runs on that runner, validates its runtime, and uploads its runner name and a nonce. +3. **Core canary 2** runs on the replacement, confirms the runner name changed, downloads the artifact, and exercises Buildx and Compose. + +The controller confirms that both canaries used the expected group and per-run label, ran on distinct `epar-ci-core-*` runners, and succeeded. Workflow concurrency is serialized because every run shares the fixed cleanup prefix. + +## Required GitHub Setup + +### Controller + +The controller uses GitHub's `ubuntu-latest` runner. No persistent self-hosted controller is required. + +### Runner Group + +Create an organization runner group named `epar-ci-canary` and restrict it to `solutionforest/ephemeral-action-runner`. + +The workflow registers temporary runners with no GitHub default labels and only a per-run label such as `epar-core-123456-1`. Canary jobs target both the group and that unique label. + +This repository is public, so the canary config deliberately sets `security.runnerGroup.requirePublicRepositoriesDisabled: false` while retaining enforcement, explicit naming, non-default-group use, and selected-repository access. This narrow exception is for the protected canary only and is not the recommended public-repository deployment model. + +### GitHub App And Protected Environment + +Install a GitHub App with organization self-hosted-runner read/write permission. Create an environment named `epar-live-ci`, restrict it to `develop`, `main`, and `refs/pull/*/merge`, and add: + +| Kind | Name | Value | +| --- | --- | --- | +| Environment secret | `EPAR_GITHUB_APP_PRIVATE_KEY` | Complete GitHub App PEM private key | +| Environment variable | `EPAR_GITHUB_APP_ID` | Numeric App ID | +| Environment variable | `EPAR_GITHUB_ORGANIZATION` | Organization login | + +Keep the PEM in the environment secret with its original line breaks. The workflow writes it to a restricted temporary file on the disposable controller and removes it during cleanup. + +## Trust Boundaries And Triggers + +The controller is privileged and secret-bearing. It receives the GitHub App key and an Actions write token, and it starts privileged containers in its disposable GitHub-hosted VM. + +The controller and canaries run only for trusted repository changes: pushes to `develop` or `main`, same-repository pull requests targeting those branches, and manual dispatch after the workflow exists on the default branch. Fork pull requests trigger the workflow but skip all three verification jobs before they can receive environment secrets or use the canary group. + +The workflow does not use `pull_request_target`. Keep the same-repository guard on every verification job. + +The canaries never receive the GitHub App key. They receive only the permissions needed for checkout and artifact operations. + +## Expected Result And Cleanup + +A successful run reports two different runner names beginning with `epar-ci-core-`. Both belong to `epar-ci-canary`, carry the run's unique label, and complete successfully. The second also passes the artifact, Buildx, Compose, health, and HTTP checks. + +The controller cleans before and after the canaries. It stops the supervisor, deletes GitHub runner registrations within the exact `epar-ci-core` prefix boundary, removes matching outer containers, and deletes its temporary key, config, and logs. + +Before failed-run cleanup removes logs, the controller prints sanitized bounded tails from the supervisor and runner logs. A controller failure attempts cleanup before canceling the workflow so unmatched canary jobs do not remain queued. + +A sudden controller loss can bypass application cleanup. GitHub discards the hosted VM and its containers; the next run pre-cleans stale GitHub runner registrations within the same prefix. + +## Troubleshooting + +### Controller Remains Queued + +- Confirm the repository can use standard GitHub-hosted runners. +- Check GitHub Actions service status and hosted-runner concurrency. + +### Canaries Remain Queued + +- Check the controller's sanitized supervisor and runner-log tails. +- Confirm `epar-ci-canary` exists and permits this repository. +- Confirm the GitHub App is installed and can manage organization runners. +- Confirm the environment values contain the numeric App ID, organization login, and complete PEM. +- Check for an online `epar-ci-core-*` runner carrying the unique label shown in the workflow. + +### Image Or Docker Workload Fails + +- Check controller disk space and Docker health. +- Confirm outbound access to the pinned Catthehacker and BusyBox images. +- Inspect the grouped Docker Container runner-log tail in the controller output. + +### Workflow Is Canceled + +Cancellation after a controller error is expected. Diagnose the first controller error; cancellation prevents unmatched canary jobs from remaining queued. + +## Manual Cleanup + +Create an untracked config based on `configs/docker-container.core.example.yml` with the same organization, GitHub App key path, and `pool.namePrefix: epar-ci-core`, then run: + +```bash +go run ./cmd/ephemeral-action-runner cleanup \ + --config .local/core-cleanup.yml \ + --project-root . +``` + +This removes matching organization runner records. The controller's Docker containers existed only on its discarded hosted VM. If cleanup still fails, remove remaining `epar-ci-core-*` records from the organization's Actions runner settings. Do not use a broader prefix. diff --git a/docs/development/design.md b/docs/development/design.md new file mode 100644 index 0000000..49d8896 --- /dev/null +++ b/docs/development/design.md @@ -0,0 +1,43 @@ +# Design + +EPAR has one provider-neutral control flow: + +```mermaid +flowchart LR + CLI["CLI and ./start wizard"] --> Pool["Common pool lifecycle"] + Pool --> Provider["Provider contracts"] + Pool --> GitHub["GitHub runner API"] + Pool --> Storage["Capacity and retention"] + Provider --> Implementations["Tart, WSL, Docker Container, Docker Sandboxes"] +``` + +## Responsibilities + +- `cmd/ephemeral-action-runner` owns command routing and the missing-config wizard. +- `internal/pool` owns naming, capacity admission, GitHub registration, readiness, strict instance limits, replacement, reconciliation, diagnostics, status, and exact instance cleanup. +- `internal/provider` defines required contracts; `internal/provider/` owns provider commands and host integration. +- `internal/image` owns reusable runner artifact acquisition, manifests, builds, and updates. +- `internal/storage` owns storage measurements, artifact ownership, retention plans, and exact cleanup execution. + +Provider code must not implement a second pool lifecycle. A capability that every provider needs belongs in a common contract; genuinely optional behavior uses an explicit capability interface. + +## Instance Lifecycle + +For every provider, the common controller: + +1. Verifies configuration, runner-group policy, reusable artifacts, and every required storage surface. +2. Allocates one exact instance using the shared pool prefix and `pool.RunnerName`. +3. Verifies runtime isolation, trust, diagnostics, and provider admission rules. +4. Requests a short-lived GitHub token and configures one ephemeral runner. +5. Tracks the exact provider and GitHub identities in durable state. +6. Removes the completed instance, verifies absence, and creates a replacement without exceeding `pool.instances`. + +Unknown ownership, unavailable dependencies, failed cleanup, and uncertain remote state consume capacity and block new allocation. EPAR does not silently fall back to another provider or broaden cleanup from an exact identity to a prefix, wildcard, prune, or reset. + +## Storage Lifecycle + +Each provider reports the storage surfaces and temporary expansion required by bootstrap, artifact builds, instance creation, and replacement. The common preflight requires enough space for the operation plus the configured free-space reserve. + +Artifacts are classified as active, current reusable, superseded EPAR-owned, incomplete temporary, or shared/unknown. At startup and after activation, conservative housekeeping reconciles interrupted work and removes only unreferenced, exactly owned superseded resources after live readback. It retains resources used by another configuration, lease, container, sandbox, distribution, or builder. Prefix-only and shared resources require an explicit previewed prune; cleanup never expands to a broad prune, reset, or wildcard. + +See [Adding a Provider](adding-provider.md) for the extension checklist. diff --git a/docs/development/principles.md b/docs/development/principles.md new file mode 100644 index 0000000..b1f7d7f --- /dev/null +++ b/docs/development/principles.md @@ -0,0 +1,15 @@ +# Development and Extension Principles + +EPAR extensions preserve the existing user flow and controller design. + +## First Run + +`./start` is the Quick Start and general source entry point. With no command, or with start flags only, it runs the `start` command and opens the missing-configuration wizard when needed; with an explicit command, it must forward that command and all arguments exactly as the binary and `go run ./cmd/ephemeral-action-runner` do. The local-Go and no-Go native-controller paths must behave the same, including native-host operational build trust, startup reconciliation of exactly owned stale work, and user-facing remediation commands that use the entry point the user actually invoked. A no-Go bootstrap TLS failure must preserve the native build transcript and report the requested host and presented certificate metadata without disabling verification or retrying insecurely. A selectable provider must appear in the wizard with its tooling and daemon prerequisite status; storage estimates never make provider selection unavailable or prevent configuration creation. Docker Container, Docker Sandboxes, and WSL use the same Catthehacker image, custom-script, and update-policy onboarding flow. The wizard writes the desired configuration first, then an embedded `./start` continues through the ordinary artifact provisioning and storage-admission path. Local artifact inputs always apply immediately; provider-neutral scheduling may defer only remote mutable source and Actions runner observations. Reject an unavailable platform or invalid image clearly and never silently switch a configured provider or artifact. Builder operational trust and optional runner trust are separate contracts: system roots always support the owned builder, while `image.hostTrustMode` controls only runner inheritance. + +## Runner Names + +The wizard defaults `pool.namePrefix` to `-`, capped at 40 characters. The machine name shows where a runner belongs, the random suffix reduces collisions, and the cap leaves room for the GitHub runner suffix. Every provider must use the shared `internal/pool.RunnerName` function and keep the configured prefix literal. + +## Components And Flow + +Keep the flow `CLI -> pool manager -> provider -> guest/GitHub`. Provider-specific host operations belong in `internal/provider/`; the pool manager owns shared registration, readiness, replacement, status, and cleanup flow. Extend the shared interfaces instead of creating a second control path. diff --git a/docs/development/releases.md b/docs/development/releases.md new file mode 100644 index 0000000..15ca590 --- /dev/null +++ b/docs/development/releases.md @@ -0,0 +1,23 @@ +# Releases + +EPAR releases are manually dispatched from GitHub Actions and contain no uploaded binaries. GitHub automatically provides **Source code (zip)** and **Source code (tar.gz)** for each release tag. + +## Create A Release + +1. Create and push an annotated tag: + + ```bash + git tag -a v0.1.0-beta.1 -m "v0.1.0-beta.1" + git push origin v0.1.0-beta.1 + ``` + +2. Confirm immutable releases are enabled in the repository settings. +3. Open **Actions**, choose **Release**, and run the workflow. +4. Enter an existing remote tag matching `[v]MAJOR.MINOR.PATCH` or `[v]MAJOR.MINOR.PATCH-(alpha|beta|rc).N`. +5. Type `publish source-only release` exactly. + +The workflow verifies the tag, confirms its commit is reachable from `origin/main`, checks out and tests that commit, and refuses to overwrite an existing release. Alpha, beta, and release-candidate tags are published as prereleases and are not marked latest. + +## Promote A Prerelease + +To promote a prerelease without changing its commit, create the stable tag at the same commit and provide the existing prerelease tag in `promotion_from`. The workflow verifies that both tags have the same normalized version core and point to the same commit, then creates stable promotion notes instead of a generated change summary. diff --git a/docs/github-app.md b/docs/github-app.md index 347f303..5415e01 100644 --- a/docs/github-app.md +++ b/docs/github-app.md @@ -30,10 +30,11 @@ github: `github.organization` must be the organization where the app is installed. `privateKeyPath` is resolved relative to the project root unless it is absolute. -Image-only commands do not use GitHub credentials. Runner registration, GitHub-backed status, and GitHub cleanup do. +Image-only commands do not use GitHub credentials. The initializer reads runner groups and selected repositories. Runner registration, GitHub-backed status, and GitHub cleanup also use the App. The existing **Self-hosted runners** read and write permission covers these operations. References: - [Registering a GitHub App](https://docs.github.com/en/apps/creating-github-apps/registering-a-github-app/registering-a-github-app) - [Managing private keys for GitHub Apps](https://docs.github.com/en/apps/creating-github-apps/authenticating-with-a-github-app/managing-private-keys-for-github-apps) - [Organization self-hosted runner registration token API](https://docs.github.com/en/rest/actions/self-hosted-runners?apiVersion=2022-11-28#create-a-registration-token-for-an-organization) +- [Organization runner-group API](https://docs.github.com/en/rest/actions/self-hosted-runner-groups) diff --git a/docs/image-build.md b/docs/image-build.md index 7c0bedb..e520bc2 100644 --- a/docs/image-build.md +++ b/docs/image-build.md @@ -1,161 +1,33 @@ -# Image Build +# Image Customization -EPAR builds a reusable Ubuntu runner image for the selected provider. The image contains the GitHub Actions runner plus whatever tools you choose through install scripts. - -For Tart, the image build has two image names: - -- `image.sourceImage`: clean upstream VM image, default `ghcr.io/cirruslabs/ubuntu:latest`. -- `image.outputImage`: reusable runner base image, default `epar-ubuntu-24-arm64`. - -These are Tart VM image names. They are stored in Tart's local VM registry and are visible with `tart list`; they are not emitted as repository-local files. - -> [!WARNING] -> Tart support is experimental, and its default source is a basic Ubuntu ARM64 OS image rather than a GitHub-hosted runner image. It does not include the usual dependency inventory from [`actions/runner-images`](https://github.com/actions/runner-images). For a GitHub-runner-like Tart environment, adapt those upstream build scripts to produce and maintain your own bootable Tart VM image, then configure EPAR to use it; EPAR does not automate that image conversion. - -For WSL, the image build produces a rootfs tar. It can start from either a Docker image or an existing rootfs tar: - -- `image.sourceType`: `docker-image` or `rootfs-tar`, default `docker-image` for WSL. -- `image.sourceImage`: source Docker image or rootfs tar, default `ghcr.io/catthehacker/ubuntu:full-latest`. -- `image.sourcePlatform`: Docker platform used when `sourceType` is `docker-image`, default `linux/amd64`. -- `image.outputImage`: reusable EPAR runner rootfs tar, default `work/images/epar-wsl-catthehacker-ubuntu.tar`. - -For `docker-image` sources, EPAR pulls the source image, creates a temporary container, exports that container filesystem to an intermediate rootfs tar, and captures the image environment metadata. Later builds reuse the intermediate rootfs only when the cached source manifest still matches the source image, platform, and digest. The WSL build then imports the rootfs into a temporary distro, enables systemd, installs the runner runtime, writes the captured image env under `/opt/epar`, validates it, exports the reusable tar, and unregisters the temporary distro. Pool instances import from `provider.sourceImage`, which should point at the built reusable tar. - -For Docker-DinD, the image build uses Docker images: - -- `image.sourceType`: `docker-image`. -- `image.sourceImage`: maintained Catthehacker Ubuntu runner image, default `ghcr.io/catthehacker/ubuntu:full-latest`. -- `image.outputImage`: reusable EPAR runner container image tag, default `epar-docker-dind-catthehacker-ubuntu`. - -Docker-DinD builds a thin wrapper over the source image, installs the GitHub Actions runner, reuses Docker Engine/CLI/Compose/Buildx from the base image when they are already present, adds a container entrypoint that starts `dockerd`, then runs configured install scripts and validation. Pool instances are privileged containers created from `provider.sourceImage`, which should match the built reusable Docker image tag. - -Every build writes an EPAR image manifest. The `start` command compares that manifest with the current config and source image identity before creating runners. If the image is missing, has no manifest, or no longer matches, `start` rebuilds it with replace enabled. The lower-level `image build` command keeps its explicit safety behavior and still requires `--replace` when the output already exists. - -```mermaid -flowchart TD - Source["Clean Ubuntu source
Tart image, WSL source, or Docker image"] --> Build["Temporary build instance or Docker build"] - Build --> Scripts["EPAR guest scripts"] - Scripts --> Runner["GitHub Actions runner"] - Runner --> WSLFull["WSL Docker-source env and Docker validation"] - WSLFull --> DockerDind["Docker-DinD-only private daemon layer"] - DockerDind --> Rosetta["Optional Tart Rosetta layer"] - Rosetta --> Custom["Optional custom install scripts"] - Custom --> Validate["Runtime validation"] - Validate --> Output["Reusable runner image"] - Output --> Pool["Disposable pool instances"] -``` - -## Image Install Scripts - -Several layers control what is pre-installed in the Ubuntu image: - -1. `image.hostTrustMode: overlay`, when enabled for Docker-DinD, snapshots and installs the controller host's trusted root anchors. -2. `image.trustedCaCertificatePaths`, when configured, installs additional explicitly trusted enterprise CA certificates. -3. EPAR installs both CA sources before guest install steps access the network. -4. `/opt/epar/install-base.sh` is intentionally lean. It does not install Docker, browsers, language runtimes, or project tools. -5. `/opt/epar/install-runner.sh` always installs the GitHub Actions runner. -6. WSL builds from Docker-image sources validate Docker Engine from the base image and preserve source image environment metadata for runner jobs. -7. Docker-DinD builds validate or install Docker Engine and add the private `dockerd` entrypoint. -8. Tart builds with `provider.rosettaTag` install the optional Rosetta amd64 container layer. -9. `image.customInstallScripts` adds optional tool layers. +EPAR prepares a reusable runner artifact before it creates disposable instances. The artifact differs by provider: Docker Container uses a Docker image, WSL2 uses a rootfs tar, Tart uses a Tart VM image, and Docker Sandboxes uses a built, imported, and read-back runner template. ```mermaid flowchart LR - Base["Runner-only base"] --> Runner["GitHub Actions runner"] - Runner --> WSLFull["WSL Docker-source env and Docker validation"] - WSLFull --> DinD["Docker-DinD-only daemon layer"] - DinD --> Rosetta["Optional Tart Rosetta layer"] - Rosetta --> Optional["customInstallScripts"] - Optional --> Docker["Docker/browser layer"] - Optional --> Web["web/E2E layer"] - Optional --> Yours["your scripts"] -``` - -The public WSL and Docker-DinD default examples start from `ghcr.io/catthehacker/ubuntu:full-latest`, so they inherit Docker plus the broader Catthehacker runner tool stack. Tart and the WSL lean examples leave `image.customInstallScripts` empty, producing a runner-only Ubuntu image. Docker-DinD always needs Docker Engine because the provider depends on a private inner Docker daemon. - -EPAR ships reusable install scripts for common cases: - -- `scripts/guest/ubuntu/install-docker-browser.sh` installs Docker Engine, Docker CLI, Compose v2, Buildx, and a Chromium-compatible browser. -- `scripts/guest/ubuntu/install-web-e2e.sh` includes Docker/browser support and ensures Node.js/npm, `zip`, `rsync`, and `mysql-client` for common web app and browser E2E workflows. - -Built-in install scripts call `scripts/guest/ubuntu/wait-apt-ready.sh` before package installs. It stops active `apt-daily` jobs for the current boot only, waits for dpkg locks to clear, and leaves Ubuntu's normal apt timer enablement unchanged in the finalized image. - -### Host trust overlay - -Docker-DinD can snapshot all trusted root anchors visible in configured host -scopes and add them to Ubuntu's default CA store: - -```yaml -image: - hostTrustMode: overlay - hostTrustScopes: [system, user] + Source["Provider source"] --> Runner["EPAR runner layer"] + Runner --> Trust["Optional CA and host-trust layer"] + Trust --> Custom["Optional install scripts"] + Custom --> Verify["Build validation"] + Verify --> Artifact["Reusable artifact"] + Artifact --> Pool["Disposable runners"] ``` -Windows and macOS support `system` and `user`; Linux supports `system` only. -Each canonical root set has a content generation recorded in the EPAR image -manifest. An addition, removal, or rotation creates a new generation, invalidates -image reuse, and causes idle runners from the earlier generation to be replaced. -A job already running keeps its immutable starting generation. +## Choose A Starting Point -The pool keeps its normal 15-second liveness interval. With overlay mode -enabled, a native controller recollects host trust every 15 seconds; a -containerized controller checks its read-only feed and refreshes idle-runner -leases every 5 seconds. The official no-Go wrappers run a host-side collector -every 10 seconds. Feed data must be no more than 30 seconds old. A runner receives a -20-second controller lease, and its synchronous -pre-job hook fails closed if the lease is missing, expired, or belongs to a -different generation. A rare assignment racing runner retirement can therefore -fail before workflow steps rather than run with stale trust. +| Provider | Default source | Reusable artifact | Build command | +| --- | --- | --- | --- | +| Docker Container | `ghcr.io/catthehacker/ubuntu:full-latest` | Docker image tag | `image build --replace` | +| WSL2 | Catthehacker Docker image converted to rootfs | Rootfs tar | `image build --replace` | +| Tart | `ghcr.io/cirruslabs/ubuntu:latest` | Tart VM image | `image build --replace` | +| Docker Sandboxes | Selected Catthehacker source | Verified imported runner template | `image build` | -Host trust is copied into the image build context; the host trust feed is not -mounted into job containers. EPAR does not alter a running job's CA store. +The first-run wizard gives Docker Container, Docker Sandboxes, and WSL the same ordered Catthehacker choices (`full-latest`, `act-latest`, `dotnet-latest`, `js-latest`, or another validated tag), platform resolution, optional custom-script collection, update policy, and storage estimate. `./start` always verifies local inputs and the active artifact, but checks mutable source tags and `runnerVersion: latest` only when the configured schedule is due. The default is weekly at 07:00 local time. Docker Sandboxes imports a replacement into its template cache and activates it only after exact readback succeeds. -This is an additive Ubuntu overlay. It does not reproduce every Windows or -macOS trust-policy constraint and does not remove an independently bundled -Ubuntu root. See [Docker-DinD Host Trust Inheritance](providers/docker-dind.md#host-trust-inheritance). +Use `./start image update` for an immediate remote check that rebuilds only when an immutable source or Actions runner identity changed. Use `./start image build` to force a build. Actions runner packages are selected by exact platform, downloaded into a content-addressed cache by the native controller, and SHA-256 verified before entering any provider build; guests do not resolve `latest`. -### Enterprise CA certificates +## Add Tools -If HTTPS traffic is inspected by an enterprise proxy, or install scripts use a -private registry with an internal CA, provide the trusted CA explicitly: - -```yaml -image: - trustedCaCertificatePaths: - - .local/enterprise-root.pem - - ~/company/intermediate.crt -``` - -Each path must be a readable PEM or DER X.509 CA certificate file, not a -directory. Bundled PEM CA files are supported. EPAR rejects invalid or non-CA -content before the build, normalizes certificate filenames from certificate -fingerprints, and runs Ubuntu's `update-ca-certificates` before `apt`, `curl`, -or runner downloads. Certificate paths and content digests are part of the EPAR -image manifest, so certificate rotation invalidates image reuse. - -This keeps TLS verification enabled. Do not work around certificate errors with -`curl -k`, `NODE_TLS_REJECT_UNAUTHORIZED=0`, or package-manager verification -disabling. - -```yaml -image: - customInstallScripts: - - scripts/guest/ubuntu/install-web-e2e.sh -``` - -The built-in `install-web-e2e.sh` script reuses base-image Node.js/npm only when its numeric Node major version meets the pinned toolset's `.node.default`; otherwise, it installs Node.js/npm through the pinned GitHub runner-image `install-nodejs.sh` script. It also adds `zip`, `rsync`, and `mysql-client`. It does not install MySQL server, project dependencies, `node_modules`, Playwright test packages, Playwright browser cache, Docker credentials, or application runtime secrets. - -Use the web/E2E examples when workflows need this larger toolset: - -```bash -cp configs/tart.web-e2e.example.yml .local/config.yml -``` - -```powershell -Copy-Item configs/wsl.web-e2e.example.yml .local/config.yml -``` - -`image.customInstallScripts` is a list of extra shell scripts: +Use `image.customInstallScripts` for non-secret additions that every runner from the artifact should contain: ```yaml image: @@ -164,189 +36,56 @@ image: - examples/custom-install/install-extra-apt-tools.sh ``` -Relative paths are resolved from the repository root and must stay inside the repository; absolute paths are also accepted. EPAR copies and runs these scripts as root, in the order listed, after the GitHub Actions runner is installed and before image validation/finalization. On Tart, if `provider.rosettaTag` is set, EPAR installs the Rosetta guest support before these custom scripts run. +Scripts run as root in listed order after the Actions runner is installed and before final validation. Keep custom scripts in the repository when practical, assign the customized artifact a distinct name and workflow label, and test it with `pool verify` before normal use. Do not bake GitHub tokens, private keys, registry credentials, application source, dependency caches, or other workflow secrets into an image. -Example script: - -```bash -#!/usr/bin/env bash -set -euo pipefail - -export DEBIAN_FRONTEND=noninteractive - -apt-get update -apt-get install -y --no-install-recommends \ - make \ - pkg-config \ - shellcheck -``` - -The same script can be used by Tart, WSL, and Docker-DinD because it runs inside Ubuntu. If the customized image changes workflow capabilities, give it distinct image names and labels so workflows can opt into it explicitly: - -```yaml -image: - outputImage: work/images/epar-ubuntu-24-wsl-web-e2e-extra.tar - customInstallScripts: - - scripts/guest/ubuntu/install-web-e2e.sh - - examples/custom-install/install-extra-apt-tools.sh - -runner: - labels: [self-hosted, linux, X64, epar-wsl-ubuntu-24.04-web-e2e-extra] - -provider: - sourceImage: work/images/epar-ubuntu-24-wsl-web-e2e-extra.tar -``` - -Do not bake secrets, private keys, Docker credentials, project `node_modules`, language package caches, or application runtime artifacts into the image. Those belong in the workflow, repository dependency lock files, or GitHub secrets. - -Docker registry mirrors are runtime configuration, not image content. Keep them in local config under `docker.registryMirrors`; EPAR applies them to each disposable instance before validation. See [Docker Registry Mirrors](advanced/docker-registry-mirrors.md). - -## Upstream Runner Images - -Runner-only Tart images and the default WSL and Docker-DinD images do not require EPAR's pinned `actions/runner-images` checkout. The default WSL and Docker-DinD images start from `ghcr.io/catthehacker/ubuntu:full-latest`, which already includes Docker Engine, Compose, Buildx, Node/npm, and the broader Catthehacker runner tool stack. - -The built-in Docker/browser and web/E2E scripts require a pinned checkout of `actions/runner-images`: +The built-in `install-web-e2e.sh` adds browser/E2E tooling. It needs EPAR's pinned `actions/runner-images` checkout: ```bash ephemeral-action-runner image update-upstream +ephemeral-action-runner image build --replace ``` -That writes the checked-out commit to `third_party/runner-images.lock`. The checkout directory itself is ignored by Git. - -When one of those built-in scripts is selected, the build copies only the required upstream Ubuntu script subset into the guest or Docker build context: - -- `images/ubuntu/scripts/helpers` -- `images/ubuntu/scripts/build/install-docker.sh` -- `images/ubuntu/scripts/build/install-google-chrome.sh` -- `images/ubuntu/scripts/build/install-nodejs.sh` -- `images/ubuntu/toolsets` - -## Docker-DinD Images - -Use `configs/docker-dind.example.yml` for the default full Catthehacker runner container with a private Docker daemon. For smaller Docker-focused workloads, use `configs/docker-dind.act.example.yml`, which starts from `ghcr.io/catthehacker/ubuntu:act-latest` without custom install scripts. It includes Node plus Docker Engine/CLI/Compose/Buildx, but does not guarantee browser dependencies. Use `configs/docker-dind.web-e2e.example.yml` when workflows need Playwright or another browser workload; it starts from the same Act base and layers the web/E2E add-on. - -```bash -cp configs/docker-dind.example.yml .local/config.yml -./bin/ephemeral-action-runner image build --replace -``` - -Run `image update-upstream` first when using `configs/docker-dind.web-e2e.example.yml`, because that optional layer installs browser and Node.js pieces from the pinned upstream runner-images scripts. - -The output image is a Docker image tag: - -```bash -docker image ls epar-docker-dind-catthehacker-ubuntu -``` - -The provider creates each runner instance with `docker create --privileged` and no host socket mount. The image entrypoint starts a private `dockerd`, waits for `docker info`, and keeps the container alive while EPAR configures and monitors the GitHub runner process. Workflow Docker resources live inside that inner daemon. The inner daemon defaults to the `vfs` storage driver because it is reliable for nested Docker across Docker Desktop, OrbStack, and Linux Docker hosts; users can bake a different `EPAR_DOCKERD_STORAGE_DRIVER` into a derived image after validating the host. - -## WSL Images - -Use `configs/wsl.example.yml` for the default full Catthehacker runner image converted into WSL: - -```powershell -Copy-Item configs/wsl.example.yml .local/config.yml -./bin/ephemeral-action-runner image build --replace -``` - -The default WSL build uses Docker on the Windows host only to prepare the source rootfs. It runs `docker pull`, `docker create`, `docker export`, and cleanup for `ghcr.io/catthehacker/ubuntu:full-latest`, then imports the exported rootfs into WSL and applies EPAR's normal lifecycle layer. If Docker is unavailable, use Docker Desktop, Docker Engine, or switch to `image.sourceType: rootfs-tar` with a prepared rootfs tar. +The default Catthehacker sources and runner-only Tart builds do not require that checkout. Use the exact configuration and provider guide to decide whether a selected script needs it. -The output image is a WSL-importable rootfs tar: - -```text -work/images/epar-wsl-catthehacker-ubuntu.tar -``` +## Trust And Enterprise CAs -EPAR also writes an intermediate source tar and env cache beside the output, for example: +Use an explicit CA path when a required CA is independent of the host trust store: -```text -work/images/epar-wsl-catthehacker-ubuntu.source.rootfs.tar -work/images/epar-wsl-catthehacker-ubuntu.source.rootfs.tar.env -work/images/epar-wsl-catthehacker-ubuntu.source.rootfs.tar.source.json -work/images/epar-wsl-catthehacker-ubuntu.tar.epar-manifest.json +```yaml +image: + trustedCaCertificatePaths: + - .local/enterprise-root.pem ``` -Later builds reuse that source cache when its source manifest still matches. Delete those files when you intentionally want to reconvert the Docker image. - -WSL runner startup sources `/opt/epar/source-image.env` before launching the GitHub Actions runner. That preserves image metadata such as `ImageOS`, `ImageVersion`, runner tool cache paths, browser paths, and Java paths from the Docker image source. WSL does not use Docker-DinD's container entrypoint; it keeps the systemd and keepalive model used by other WSL images. - -Use `configs/wsl.lean.example.yml` when you want the old smaller rootfs-tar path. That config expects you to export a clean Ubuntu 24.04 WSL distro to `work/images/ubuntu-24.04-clean.rootfs.tar`. - -## Installed Runtime - -The default WSL and Docker-DinD builds use `ghcr.io/catthehacker/ubuntu:full-latest` as the source image. It is larger than the medium Catthehacker act image, but it is the recommended default for public users because common tools such as Node/npm are already present. The WSL lean and web/E2E examples keep demonstrating smaller custom paths that layer only selected dependencies. - -Catthehacker's `full-latest` and `act-latest` tags are rolling references. EPAR records the resolved source digest in its image manifest so a built image remains auditable, but a later `image build --replace` can consume a newer upstream digest. Pin `image.sourceImage` to a tested digest when rebuild reproducibility is required. - -The default build installs: +EPAR validates PEM or DER CA certificates and incorporates their hashes into artifact freshness. Explicit certificates are available to both the operational image build and the resulting runner artifact. Keep TLS verification enabled. -- GitHub Actions runner Linux package from `actions/runner` -- minimal OS packages required by that runner package -- additional tools selected by `image.customInstallScripts` +EPAR's project-owned BuildKit builder always receives current host system roots so image acquisition can operate behind authorized HTTPS inspection. This operational trust is independent of `image.hostTrustMode` and is not copied into runners. If runner overlay explicitly includes the `user` scope, those user roots are also available to the builder for the same invocation. -The optional `install-docker-browser.sh` layer installs: +The wizard can enable `image.hostTrustMode: overlay` for Docker Container and Docker Sandboxes when runners themselves must inherit selected host root anchors. Omitted or `disabled` mode creates a Docker Sandboxes template with an explicit disabled-policy marker and no job-start trust hook. Overlay mode is additive to Ubuntu roots and explicit CA paths, not an emulation of every Windows or macOS trust policy. Use `[system, user]` on Windows/macOS or `[system]` on Linux. See [Configuration](configuration.md) and [Security](security.md) before enabling it. -- Docker through upstream `install-docker.sh` -- upstream Google Chrome on x64 -- Playwright-managed Chromium on ARM64, exposed as `epar-browser`, `chromium`, and `chromium-browser` +## Provider Differences -The WSL default and Docker-DinD provider validate Docker Engine/CLI/Compose/Buildx through `scripts/guest/ubuntu/install-docker-engine.sh`. Docker-DinD then starts the daemon at container runtime from `/opt/epar/container-entrypoint.sh`; WSL starts Docker through systemd inside the distro. Set `EPAR_FORCE_UPSTREAM_DOCKER_INSTALL=true` inside non-WSL-default image builds only if you intentionally want to replace the base image's Docker packages with the pinned upstream `actions/runner-images` Docker install harness. +### Docker Container -The ARM64 Docker harness prefers upstream `toolset-2404-arm64.json`. If an older upstream checkout does not contain that file, EPAR falls back to a minimal ARM-aware Docker toolset. +The output is a Docker image named by `image.outputImage`; `provider.sourceImage` must point to it. The provider starts a private `dockerd` inside each privileged runner container. Use `configs/docker-container.act.example.yml` for a smaller Docker-focused base or `configs/docker-container.web-e2e.example.yml` for the browser/E2E layer. -The harness skips upstream Docker image cache pulls by default. Set `EPAR_SKIP_UPSTREAM_DOCKER_IMAGE_CACHE=false` inside the guest environment before `install-docker-browser.sh` if exact upstream cache behavior is required. +### WSL2 -## Tart Rosetta Layer +The default source is converted from a Docker image to an intermediate rootfs tar, then EPAR produces `image.outputImage` as the reusable WSL tar. Docker is required during this conversion. For `image.sourceType: rootfs-tar`, export a clean Ubuntu WSL distribution once and use that tar as `image.sourceImage`; see [WSL2 Provider](providers/wsl.md). -Set `provider.rosettaTag: rosetta` on a Tart config to start image-build and pool instances with `tart run --rosetta rosetta`. During Tart image build EPAR installs: +### Tart -- `/opt/epar/setup-rosetta.sh` -- `epar-rosetta.service` -- `/opt/epar/features/rosetta-amd64` +The output is a local Tart VM image. The default is intentionally lean; use a custom bootable Ubuntu source image or focused install scripts when a workflow needs more tooling. EPAR builds and verifies a content-named candidate, keeps a rollback clone until the configured output passes immutable identity readback, and disables Tart's unrelated automatic cache pruning for its clone operations. `provider.rosettaTag` is an opt-in, experimental Tart-only layer for selected Linux amd64 user-space workloads. -The setup script mounts the Tart Rosetta virtiofs share at `/run/rosetta`, mounts `binfmt_misc` if needed, and registers x86_64 ELF execution through `/run/rosetta/rosetta` with `OCF` flags. +### Docker Sandboxes -Rosetta is not installed for WSL builds and is not enabled for Tart unless `provider.rosettaTag` is non-empty. +Docker Sandboxes uses `image.sourceImage`, `image.sourcePlatform`, and `image.customInstallScripts` as the desired template inputs. `./start` and `./start image build` share the same build/import implementation. BuildKit writes an attestation-free Docker-compatible archive directly because that archive format cannot carry the provenance/SBOM manifest list; EPAR verifies the archive without creating a Docker staging image, imports it, and records the exact Sandbox cache identity under `.local/state/image//docker-sandboxes/active.json`. A separate cache-backed BuildKit evidence operation produces max-mode provenance and the SBOM without loading the runner image into Docker Engine. The large archive and full SBOM workspace are transient, while compact metadata, provenance, compatibility, inventory, and SBOM descriptor evidence remain. A failed desired update leaves the previous receipt and artifact intact but does not run it as a fallback. -At the end of a build, `/opt/epar/finalize-image.sh` stops Docker/containerd if they exist, clears Docker's persisted default bridge database, removes temporary validation files, and syncs the filesystem. This avoids cloned instances inheriting stale `docker0` bridge metadata from build-time validation. - -Runtime validation always verifies the base runner user and runner files. If the Docker/browser feature marker is present, validation also starts Docker and verifies Docker access as the same Linux user that runs the GitHub Actions runner: +## Verify A Customized Artifact ```bash -sudo -u runner -H docker version -sudo -u runner -H docker compose version -sudo -u runner -H docker buildx version -sudo -u runner -H docker run --rm hello-world -printf '%s\n' '

EPAR browser validation marker

' >/tmp/epar-browser-validation.html -sudo -u runner -H chromium --headless --no-sandbox --dump-dom file:///tmp/epar-browser-validation.html -``` - -If the Rosetta feature marker is present, validation also verifies a real amd64 Linux container: - -```bash -sudo -u runner -H docker run --rm --platform linux/amd64 alpine:3.20 sh -c 'uname -m' -``` - -The expected output is `x86_64`. - -The bundled `scripts/guest/ubuntu/install-web-e2e.sh` script creates a feature marker so `pool verify` also validates `node`, `npm`, `zip`, `unzip`, `tar`, `rsync`, and `mysql` on cloned instances. - -## WSL Bootstrap - -The default WSL config does not require a manually exported Ubuntu tar. It uses Docker to convert `ghcr.io/catthehacker/ubuntu:full-latest` into a rootfs tar during `image build`. - -For lean `image.sourceType: rootfs-tar` configs, create the clean Ubuntu tar once before `image build`. The supported path is to install an Ubuntu 24.04 WSL distro, export it, then use that tar as `image.sourceImage`: - -```powershell -New-Item -ItemType Directory -Force work/images -wsl --install -d Ubuntu-24.04 --no-launch -wsl --export Ubuntu-24.04 work/images/ubuntu-24.04-clean.rootfs.tar +ephemeral-action-runner pool verify --instances 1 --cleanup +ephemeral-action-runner pool verify --instances 1 --register-only --cleanup ``` -After the export exists, EPAR uses disposable imported distros for image builds and runner instances. The WSL provider uses `provider.installRoot`, default `work/wsl`, for those imported distro files. - -References: - -- [WSL basic commands](https://learn.microsoft.com/en-us/windows/wsl/basic-commands) -- [Systemd support in WSL](https://learn.microsoft.com/en-us/windows/wsl/systemd) -- [GitHub Ubuntu runner image build scripts](https://github.com/actions/runner-images/tree/main/images/ubuntu/scripts/build) +The first command checks an unregistered disposable instance. The second also checks GitHub registration. Provider-specific runtime checks run when their feature markers are present; for example, Docker-enabled images validate Docker, Compose, Buildx, and a real container. diff --git a/docs/logging.md b/docs/logging.md index 4fea523..81dd13d 100644 --- a/docs/logging.md +++ b/docs/logging.md @@ -1,10 +1,10 @@ # Logging -EPAR uses `work/logs` under the project root by default. Set `logging.directory` to an absolute path to store logs elsewhere, or to another relative path to resolve it from the project root. +EPAR writes logs under `work/logs` by default. Use `ephemeral-action-runner logs path` to print the resolved directory. ```text work/logs/ -├── epar.log # when managerSinks includes file +├── epar.log ├── epar-last-error.log ├── errors/ ├── instances/ @@ -12,71 +12,60 @@ work/logs/ └── benchmarks/ ``` -Manager events and command transcripts have independent sinks. `console` and `file` may be selected separately or together. Console manager events route debug and info to stdout and warnings and errors to stderr. File manager events go to `epar.log`. File transcripts remain raw command output; console transcripts frame every completed stdout or stderr line with timestamp, instance, component, and stream context. JSON console mode emits one JSON object per line. +## What Goes Where -The local default keeps manager events on the console and command transcripts in files: +Manager events describe EPAR decisions and progress. Command transcripts contain raw instance, provider, and image-build output. + +The local default sends manager events to the console and transcripts to files: ```yaml logging: - directory: work/logs managerSinks: [console] managerConsoleFormat: text - managerConsoleTextFormat: "{time} [{level}] {message}" - managerFileFormat: json transcriptSinks: [file] - transcriptConsoleFormat: text - maxFileSizeMiB: 100 - maxBackups: 3 - compressBackups: true - retentionEnabled: true - retentionMaxTotalMiB: 1024 - managerMaxAgeDays: 14 - instanceMaxAgeDays: 14 - buildMaxAgeDays: 14 - errorMaxAgeDays: 30 - benchmarkMaxAgeDays: 90 - retentionIntervalMinutes: 60 ``` -The default manager console line is compact and human-readable: +Manager file events use `epar.log`. Instance transcripts use `instances/`, build and source transcripts use `builds/`, startup timing records use `benchmarks/`, and timestamped error reports use `errors/`. `epar-last-error.log` always points to the latest error report and is never removed by retention. -```text -2026-07-16T00:14:58.108+08:00 [INFO] cloning instance -``` +Runner logs inside an Ubuntu guest are normally: -When `managerConsoleFormat` is `text`, `managerConsoleTextFormat` accepts `{time}`, `{level}`, `{message}`, and `{attributes}`. `{message}` is required. The default deliberately omits structured attributes for concise local output. To include them, use `"{time} [{level}] {message}{attributes}"`; `{attributes}` expands to structured fields with its own leading space when fields exist. +- `/var/log/actions-runner/run.log` +- `/opt/actions-runner/_diag` +- `/var/log/epar-dockerd.log` when the runner has a private Docker daemon -When `transcriptConsoleFormat` is `text`, the optional `transcriptConsoleTextFormat` accepts `{time}`, `{instance}`, `{component}`, `{stream}`, `{message}`, `{session}`, `{category}`, `{provider}`, and `{attributes}`. Its default is `"{time} {stream} {instance} {component} {message}{attributes}"`. +When runner launch or GitHub readiness fails, EPAR appends bounded process and runner diagnostics to the matching host-side instance transcript. -Custom text formats are rejected when the corresponding console format is `json`. JSON records keep their fixed structured schema so downstream parsers and log shippers can rely on it. +## Console And File Formats -For Kubernetes, use console sinks so the container runtime can collect stdout and stderr: +Console and file sinks can use `text` or `json`. For Kubernetes or another runtime that already collects standard output, send both event types to the console: ```yaml logging: - directory: work/logs managerSinks: [console] managerConsoleFormat: json - managerFileFormat: json transcriptSinks: [console] transcriptConsoleFormat: json - maxFileSizeMiB: 100 - maxBackups: 3 - compressBackups: true - retentionEnabled: true - retentionMaxTotalMiB: 1024 - managerMaxAgeDays: 14 - instanceMaxAgeDays: 14 - buildMaxAgeDays: 14 - errorMaxAgeDays: 30 - benchmarkMaxAgeDays: 90 - retentionIntervalMinutes: 60 ``` -Benchmark JSONL and error reports remain file artifacts in every sink mode. EPAR rotates active manager and transcript files at the configured size, retains the configured number of gzip-compressed backups, and applies category age limits before the aggregate size budget. It protects active files across EPAR processes, does not follow links or reparse points, ignores unknown files, and never removes `epar-last-error.log`. +Text templates can change the human-readable console layout. Manager templates support `{time}`, `{level}`, `{message}`, and `{attributes}`. Transcript templates also support instance, component, stream, session, category, and provider fields. A template must contain `{message}` and is invalid when the corresponding console format is JSON. + +See [Configuration](configuration.md#logging) for every logging property, default, and validation rule. + +## Rotation And Retention + +EPAR rotates active manager and transcript files at `logging.maxFileSizeMiB`, retains the configured number of compressed backups, applies category age limits, then enforces the total retained-size budget. It protects active files across EPAR processes, does not follow links or reparse points, and ignores unknown files. + +Inspect or preview recognized log maintenance with: + +```bash +ephemeral-action-runner logs list +ephemeral-action-runner logs prune --dry-run +``` + +Remove `--dry-run` only after reviewing the exact retention plan. -Wrapper control files and command result files are state rather than logs. New wrappers place them under `work/state`, outside retention scope. +Wrapper control files and command results are state, not logs. Current wrappers place them under `work/state`, outside log retention. -Use `ephemeral-action-runner logs path` to find the resolved root, `ephemeral-action-runner logs list` to inspect recognized artifacts, and `ephemeral-action-runner logs prune --dry-run` to preview retention. Remove `--dry-run` to prune immediately. +## Shipping Logs -EPAR intentionally does not embed OTLP or vendor-specific clients. To ship file artifacts, use an external agent such as the OpenTelemetry Collector `filelog` receiver. See [`examples/observability`](../examples/observability/README.md). +EPAR does not embed vendor-specific or OTLP clients. Use an external collector when logs must leave the host. The [`examples/observability`](../examples/observability/README.md) directory includes local file, Kubernetes console, and OpenTelemetry Collector examples. diff --git a/docs/operations.md b/docs/operations.md index 132d9fd..9c9e5af 100644 --- a/docs/operations.md +++ b/docs/operations.md @@ -1,76 +1,75 @@ # Operations -## Logs +EPAR is a foreground supervisor. Keep it running while the pool should accept jobs; it creates, monitors, retires, and replaces disposable runners within the configured capacity. + +```mermaid +flowchart LR + Start["Start or resume pool"] --> Ready["Maintain ready runners"] + Ready --> Job["One runner accepts one job"] + Job --> Retire["Retire completed runner"] + Retire --> Ready + Retire -->|"Cleanup or ownership is uncertain"| Quarantine["Quarantine; capacity remains occupied"] + Quarantine --> Reconcile["Reconcile exact local and GitHub state"] + Reconcile --> Ready + Ready --> Stop["Ctrl-C or pool down"] + Stop --> Cleanup["Reconcile and clean owned resources"] +``` -Host-side logs go under `work/logs` by default. Run `ephemeral-action-runner logs path` to print the resolved directory. When `managerSinks` includes `file`, manager events use `work/logs/epar.log`; the default manager sink is console-only. Instance transcripts use `work/logs/instances/`, build/source transcripts use `work/logs/builds/`, startup timing JSONL uses `work/logs/benchmarks/`, and timestamped error reports use `work/logs/errors/`. See [Logging](logging.md) for sink, rotation, retention, and shipping configuration. Runner logs inside the Ubuntu guest are under: +## Start and stop deliberately -- `/var/log/actions-runner/run.log` -- `/opt/actions-runner/_diag` +Use `./start` for normal operation because it verifies the configured image or sandbox template before starting the pool. Press `Ctrl-C` once to request a clean stop, then wait for cleanup to finish before closing the terminal. `--keep-on-exit` is a debugging option that deliberately leaves owned runner resources running after the supervisor exits. -Guest provisioning command output is streamed to `work/logs/instances/.guest.log`. If runner launch or GitHub online readiness fails, EPAR first appends bounded diagnostics to that host guest log: -runner PID/process state, tails from `run.log` and the latest `Runner_*.log`, -and the Docker-DinD daemon log when present. Diagnostic collection is -best-effort and does not replace the original readiness error. +The supervisor reports when GitHub assigns a job and when the ephemeral runner finishes or is released. GitHub Actions remains the source of truth for whether the job succeeded or failed. -On systemd instances, the runner process is launched with `systemd-run` as `actions-runner.service` so provider `exec` calls return immediately after the service starts. On non-systemd instances such as Docker-DinD containers, EPAR starts `run.sh` in the background, writes `/var/run/actions-runner.pid`, and appends output to `/var/log/actions-runner/run.log`. +`pool up` is the lower-level command for a prepared image or template. While no supervisor is running, EPAR cannot retire completed ephemeral runners or create replacements. -Docker-DinD containers also write inner Docker daemon logs to `/var/log/epar-dockerd.log` inside the runner container. Host-side Docker commands only show the outer runner container; job-created Compose resources live in the inner daemon. +## Capacity and replacement -When `docker.registryMirrors` is configured, EPAR writes `/etc/docker/daemon.json` inside each instance before runtime validation. For Docker-DinD, inspect both `/etc/docker/daemon.json` and `/var/log/epar-dockerd.log` inside the outer runner container. +`pool.instances` is a strict cap on local physical instances, not just ready GitHub runners. Provisioning, ready, draining, quarantined, and cleanup-pending instances all consume a slot. A busy runner retained during a trust-generation change also keeps its slot until it finishes or can be safely removed. -## Supervisor Exit +Only one controller may manage a canonical configuration path on a host, even if that file is edited to select a different provider or prefix while the first controller is running. A second host-wide lock also reserves the normalized `pool.namePrefix`, so separate configs and projects can run concurrently only when every independent pool has a distinct prefix. Lock failures report the current owner metadata without exposing configuration contents. -`pool up` cleans up prefixed instances and GitHub runner records when it exits. Use `--keep-on-exit` only when intentionally debugging a live instance after the supervisor stops. While the supervisor is not running, EPAR cannot retire or replace completed runners. +Before upgrading an existing checkout to a release that introduces or changes controller locking or lifecycle-state identity, stop the older controller for that same project/config and wait for its normal shutdown to finish. A pre-change process cannot participate in a lock protocol it does not implement, so starting the new binary concurrently could migrate state beneath it. This restriction is per managed config and prefix; an unrelated controller in another checkout with a distinct prefix does not need to be stopped. -## Capacity, Reconciliation, and Outage Recovery +At startup and before a replacement, EPAR compares provider inventory with exact GitHub runner records. Healthy pairs are adopted. Proven stopped or unregistered resources are removed. An ambiguous resource is quarantined and consumes capacity instead of being deleted or replaced. -`pool.instances` is a strict cap on prefix-owned local resources. EPAR counts provisioning, ready, draining, quarantined, and cleanup-pending instances, so a GitHub outage or a failed cleanup cannot create a replacement storm. Host-trust rotation has no temporary surge allowance; old busy-generation instances retain their slots until they finish or can be safely retired. +For transient GitHub or network failures during replacement, including `429` and `5xx` responses, EPAR pauses allocation and retries with configured exponential backoff while monitoring and cleanup continue. Authentication failures and invalid configuration remain fail-fast. See [Configuration](configuration.md) to adjust the retry settings. -Only one controller may mutate a given canonical config, provider, and `pool.namePrefix` at a time. If another EPAR controller holds that lock, stop or reconfigure the other controller instead of forcing concurrent starts; use a distinct unique prefix for intentionally independent pools. +## Inspect status and logs -At startup and before allocating a replacement, EPAR reconciles the provider inventory with exact-name GitHub runner records. Healthy pairs are adopted, stopped or proven unregistered local resources are removed, and exact stale GitHub records are deleted when GitHub is reachable. When GitHub is unavailable, an ambiguous local instance is quarantined and continues to consume capacity instead of being deleted or replaced. If old resources already exceed the configured cap, the supervisor creates nothing until safe cleanup or normal draining restores capacity. +```bash +ephemeral-action-runner status +ephemeral-action-runner logs path +ephemeral-action-runner logs list +``` -For transient network failures and GitHub `429` or `5xx` responses during supervised replacement, EPAR pauses allocation and retries with exponential backoff. The default nominal delays are 15, 30, 60, 120, 240, 480, 960, and 1800 seconds, each with ±20% jitter; a longer `Retry-After` response wins. Monitoring and housekeeping continue during that pause, and a successful adoption or fully online replacement resets the delay. Startup remains fail-fast after compensating rollback, and invalid configuration or GitHub authentication failures do not retry indefinitely. +Add `--no-github` to `status` when you need a local-only view. By default, host logs live under `work/logs`; manager events are console-first and instance/build transcripts are file artifacts. A failed launch or readiness check appends bounded guest diagnostics to the relevant instance log. See [Logging](logging.md) for locations, formats, retention, and shipping. -## Cleanup Safety +When multiple configs run concurrently, give each one a distinct `logging.directory` as well as a distinct prefix and workflow-routing label. Config-specific lifecycle state and build workspaces remain isolated, while the host resource catalog retains exact shared-artifact references. -Cleanup only touches local instances and GitHub runner records matching `pool.namePrefix`: +## Clean up safely -```yaml -pool: - namePrefix: epar-tart +```bash +ephemeral-action-runner cleanup +ephemeral-action-runner pool down ``` -Generated names look like: +`pool down` is an alias for `cleanup`. Cleanup is intentionally bounded: Docker Sandboxes uses the durable ledger of exact owned identities, while legacy providers use the configured `pool.namePrefix` boundary. Unknown, shared, or identity-drifted resources are report-only rather than broad deletion targets. Do not reuse a prefix across machines or independent supervisors in the same GitHub organization. + +Use `cleanup --no-github` only when you intentionally want to leave GitHub runner records untouched. After a failed Docker Sandboxes diagnostic check, review the retained evidence before using `--acknowledge-failed-diagnostics` to allow its exact cleanup. -```text -epar-tart-20260703-010500-001 +## Maintain storage and retention + +```bash +ephemeral-action-runner storage status +ephemeral-action-runner storage prune +ephemeral-action-runner storage prune --execute +ephemeral-action-runner storage prune --legacy +ephemeral-action-runner logs prune --dry-run ``` -Do not set `namePrefix` to a broad value such as `ubuntu` or `runner`. Keep it within 40 characters so EPAR can append its generated runner-name suffix. -Also do not reuse the same `namePrefix` on different machines or for separate EPAR supervisors in the same GitHub organization. GitHub cleanup is prefix-based, so a shared prefix lets one machine delete another machine's runner records. - -For Docker-DinD, cleanup removes the outer runner container with `docker rm -f -v`. That also removes the private inner Docker daemon's containers, networks, volumes, and image cache for that EPAR instance. - -## Troubleshooting - -This section is a compact checklist. For symptom-first diagnostics with host/provider-specific commands, see [Troubleshooting](troubleshooting.md). - -- If a Docker/browser or web/E2E image build fails before package installation, run `image update-upstream`. -- If an image build fails with `E: You don't have enough free space in /var/cache/apt/archives/.`, check the Docker daemon or VM storage with `docker system df` and `docker run --rm ghcr.io/catthehacker/ubuntu:full-latest df -h /`. On Windows Docker Desktop with WSL2, the container-visible disk can be much smaller than Windows Explorer free space; see [Windows Docker Desktop WSL2 Disk Is Smaller Than Expected](troubleshooting.md#windows-docker-desktop-wsl2-disk-is-smaller-than-expected). -- If Docker validation fails for a Docker-enabled image, inspect `work/logs/builds/.guest.log`. -- If browser validation fails on ARM64, confirm `epar-browser` exists inside the guest and inspect `/opt/epar/browser`. -- If a Docker Compose job uses an amd64-only runtime image on an ARM64 Tart runner and fails with `exec format error` or repeated container exits such as status `139`, use a runner label that supports that image instead of changing application runtime settings only for runner compatibility. Suitable targets include Docker-DinD with verified `linux/amd64` emulation, WSL x64, an x64 Linux host, or a Tart image with Rosetta enabled and validated. -- If a workflow uses fixed Compose project names, fixed container names, or fixed ports, Docker-DinD is often a better fit than a shared host Docker socket because each runner gets a private inner Docker daemon. Verify by starting two unregistered instances, running the same compose stack in both, and confirming host Docker only shows the outer EPAR runner containers. -- If repeated jobs still pull slowly after configuring a registry mirror, verify the mirror is reachable from inside the runner instance and that it supports the requested registry, image platform, and authentication model. Docker daemon mirrors primarily target Docker Hub; other registry caches may require workflow image references to use the cache registry URL. -- If a mirrored workflow only improves modestly, check where the time is going. Registry mirrors mainly reduce image pull time; container startup, Compose health checks, database initialization, volume sync, browser tests, private image authentication, and CPU-bound or emulated workloads can still dominate the total job time. -- If GitHub registration fails, confirm the app has permission to manage organization self-hosted runners and that the private key path is readable by the host user. -- If GitHub returns transient `500`, `502`, `503`, or `429` errors while the supervisor is replacing a runner, leave the supervisor running so it can retain the strict cap, reconcile exact runner names after recovery, and retry on its configured backoff. Do not manually start a second supervisor with the same `pool.namePrefix`. -- If registration fails before the runner listener starts, EPAR rolls back the local candidate immediately and later reconciles a possible exact-name GitHub record. If the listener already started but GitHub readiness is uncertain, EPAR quarantines that candidate; inspect the instance transcript and wait for GitHub recovery before manually deleting it. -- If stale runners remain, run `ephemeral-action-runner cleanup`. -- If using Tart `softnet`, verify the host has the privileges Tart requires. -- If default WSL image build fails before import, confirm Docker Desktop, Docker Engine, or another Docker daemon is reachable so EPAR can export `ghcr.io/catthehacker/ubuntu:full-latest` into a rootfs tar. For lean WSL configs, confirm the clean Ubuntu rootfs was exported from an Ubuntu 24.04 WSL distro. -- If WSL image build fails before systemd is ready, confirm WSL2 is enabled and inspect `work/logs/builds/.guest.log`. -- If Docker-DinD startup fails, confirm the host Docker runtime supports privileged containers and inspect `/var/log/epar-dockerd.log` inside the runner container. -- If Docker-DinD `docker run` fails with nested overlay mount errors, keep the default `EPAR_DOCKERD_STORAGE_DRIVER=vfs`. Only switch to `overlay2` or `auto` in a derived image after proving that storage driver works on the exact host runtime. -- If the default WSL or Docker-DinD build cannot validate Docker, confirm the source image still provides `docker`, `dockerd`, Compose, Buildx, and `iptables`. If you intentionally use a clean Ubuntu source image instead of Catthehacker's runner image, run `image update-upstream` first and use a config that installs Docker from EPAR's pinned `actions/runner-images` Docker install harness. +Normal `./start` reconciles interrupted exact-owned work and retires unreferenced superseded artifacts. `storage prune` is a preview until `--execute` is supplied. `storage prune --legacy` reports prefix-era resources and requires its displayed plan hash for execution. Log pruning is separate. EPAR does not run broad Docker prune, Docker Sandboxes reset, WSL reset, Docker Desktop reset, or VHDX compaction. Read [Storage](storage.md) before reclaiming capacity. + +## Get help from the right page + +Use [Troubleshooting](troubleshooting.md) for symptom-led diagnosis and host/provider commands. Use [Support](../SUPPORT.md) when you need help and [Security](security.md) for trust-boundary guidance or private vulnerability reporting. diff --git a/docs/providers/adding-provider.md b/docs/providers/adding-provider.md deleted file mode 100644 index f3e5e49..0000000 --- a/docs/providers/adding-provider.md +++ /dev/null @@ -1,7 +0,0 @@ -# Adding A Provider - -Providers implement the shared interface in `internal/provider`. The controller expects a provider to clone or create an instance, start it, execute guest commands, return an address when available, stop/delete it, and list existing instances for prefix-safe cleanup. Tart, WSL, and Docker-DinD are the current examples of that boundary. - -Keep provider behavior idempotent where possible. Cleanup must only remove instances whose names match `pool.namePrefix`. - -Use provider-specific docs for host setup, image format, and isolation caveats. New providers should also document whether runner liveness uses systemd or the PID-file fallback in `/opt/epar/check-runner.sh`. diff --git a/docs/providers/docker-container.md b/docs/providers/docker-container.md new file mode 100644 index 0000000..9d30ec4 --- /dev/null +++ b/docs/providers/docker-container.md @@ -0,0 +1,73 @@ +# Docker Container Provider + +Docker Container creates one privileged Ubuntu runner container per EPAR instance. The outer container starts a private inner Docker daemon, so workflow containers, networks, volumes, and image cache stay inside that disposable runner. + +## When To Use It + +Choose Docker Container on a Docker-capable host when privileged containers are acceptable and you want strong per-runner Docker resource separation. It is a practical fit for Compose-heavy jobs, including jobs that reuse fixed Compose project names or ports. EPAR does not support a host-Docker-socket provider. + +## Support Status + +This is a supported provider on hosts whose Docker runtime can run privileged Linux containers. It is trusted-job infrastructure: `--privileged` weakens the normal container boundary, so do not use it for arbitrary untrusted workflow code. + +## Prerequisites + +- Docker installed and running with support for `docker run --privileged`. +- Enough host Docker storage for the reusable image and the requested disposable runners. +- A GitHub App and runner group configured as described in [Runner Group Security](../runner-groups.md). + +## Minimal Configuration + +Start with [`configs/docker-container.example.yml`](../../configs/docker-container.example.yml): + +```yaml +image: + sourceType: docker-image + sourceImage: ghcr.io/catthehacker/ubuntu:full-latest + outputImage: epar-docker-container-catthehacker-ubuntu + updateFrequency: weekly + updateTime: "07:00" + +provider: + type: docker-container + sourceImage: epar-docker-container-catthehacker-ubuntu + network: default +``` + +`provider.platform` is optional and maps to Docker's `--platform` for the reusable image and runner containers. Give cross-architecture configurations a distinct workflow label and verify actual execution on the intended host. + +Use `configs/docker-container.act.example.yml` for a smaller Docker-focused Catthehacker base, or `configs/docker-container.web-e2e.example.yml` when browser/E2E tooling is required. The full configuration, host-trust settings, proxies, and registry mirrors belong in [Configuration](../configuration.md). + +## Normal Workflow + +1. Create a local configuration with the wizard or copy an example. +2. Run `./start`. EPAR immediately applies local input changes and checks mutable upstream identities when the configured schedule is due, then starts the pool. +3. Target the configured label in a workflow, for example `runs-on: [self-hosted, linux, epar-docker-container-catthehacker-ubuntu]`. + +The outer container has no host Docker socket mount and does not publish host ports by default. The inner daemon defaults to the reliable nested-Docker `vfs` storage driver. Use a different `EPAR_DOCKERD_STORAGE_DRIVER` only in a derived image after validating the exact host runtime. + +## Limitations + +- The inner Docker daemon is private, but its CPU, memory, and disk use still comes from the host. +- The runner's inner image cache disappears with the instance. +- Docker-compatible host runtimes can differ in privileged-container and foreign-architecture behavior; verify both on the host you intend to use. +- Registry mirrors and proxy services are external infrastructure; EPAR configures the runner daemon but does not operate or secure those services. + +## Verification + +```bash +ephemeral-action-runner pool verify --instances 1 --cleanup +ephemeral-action-runner pool verify --instances 1 --register-only --cleanup +``` + +For an ARM64 host that must run amd64 Docker images, verify execution inside a live EPAR runner rather than relying on image pull success: + +```bash +docker exec docker run --rm --platform linux/amd64 alpine:3.20 uname -m +``` + +Expected output is `x86_64`. + +## Troubleshooting + +See [Troubleshooting](../troubleshooting.md) for privileged-container checks, nested-Docker storage-driver failures, architecture emulation, disk pressure, and TLS errors. diff --git a/docs/providers/docker-dind.md b/docs/providers/docker-dind.md deleted file mode 100644 index cde6373..0000000 --- a/docs/providers/docker-dind.md +++ /dev/null @@ -1,220 +0,0 @@ -# Docker-DinD Provider - -The Docker-DinD provider creates one privileged Ubuntu-based runner container per EPAR instance. That outer container starts its own private Docker daemon, and the GitHub Actions runner executes inside the same container. - -This is useful when a host already has a reliable Docker runtime and you want disposable runner environments without creating full VMs. It is also useful for Docker Compose-heavy jobs because each runner instance has a separate inner Docker daemon. Deleting the EPAR container deletes that instance's job containers, networks, volumes, and inner image cache. - -EPAR does not support a host Docker socket provider. - -## When To Choose It - -Choose Docker-DinD first for Docker-heavy Linux workflows when privileged containers are acceptable on the host. It is especially useful when the target repository already has Compose scripts, fixed project names, fixed internal ports, or amd64-only runtime images. In those cases, selecting a compatible runner label and Docker platform is usually cleaner than changing application runtime settings for CI compatibility. - -Choose Tart or WSL instead when you specifically need their host model: Tart for VM-based Apple Silicon runners, WSL for Windows-hosted Linux runners, and x64 WSL/Linux hosts for native amd64 performance. - -## Configuration - -Use `configs/docker-dind.example.yml` for a base runner image: - -```yaml -image: - sourceType: docker-image - sourceImage: ghcr.io/catthehacker/ubuntu:full-latest - outputImage: epar-docker-dind-catthehacker-ubuntu - -provider: - type: docker-dind - sourceImage: epar-docker-dind-catthehacker-ubuntu - network: default -``` - -Use `configs/docker-dind.act.example.yml` for a smaller Docker-focused runner. Its Catthehacker Act base includes Node plus Docker Engine/CLI/Compose/Buildx, but does not guarantee a browser runtime: - -```yaml -image: - sourceType: docker-image - sourceImage: ghcr.io/catthehacker/ubuntu:act-latest - outputImage: epar-docker-dind-catthehacker-act - -runner: - labels: [self-hosted, linux, epar-docker-dind-catthehacker-act] - -provider: - sourceImage: epar-docker-dind-catthehacker-act -``` - -Use `configs/docker-dind.web-e2e.example.yml` as a smaller customized-image example. It starts from `ghcr.io/catthehacker/ubuntu:act-latest` and layers only the web/E2E add-on: - -```yaml -image: - sourceType: docker-image - sourceImage: ghcr.io/catthehacker/ubuntu:act-latest - outputImage: epar-docker-dind-catthehacker-ubuntu-web-e2e - customInstallScripts: - - scripts/guest/ubuntu/install-web-e2e.sh - -runner: - labels: [self-hosted, linux, epar-docker-dind-catthehacker-ubuntu-web-e2e] - includeHostLabel: true - -provider: - sourceImage: epar-docker-dind-catthehacker-ubuntu-web-e2e -``` - -`provider.platform` is optional and maps to Docker's `--platform` flag for image builds and runner containers. Use a label that reflects the actual platform your workflows should target. - -Optional Docker registry mirrors are configured under the provider-neutral `docker` section: - -```yaml -docker: - registryMirrors: - - http://host.docker.internal:5050 -``` - -When a Docker-DinD mirror URL uses `host.docker.internal`, EPAR adds Docker's `host-gateway` alias to the outer runner container so the inner daemon can reach a host-published mirror on Linux Docker Engine. See [Docker Registry Mirrors](../advanced/docker-registry-mirrors.md). - -If the inner daemon must use an enterprise HTTP proxy, configure its startup -environment explicitly: - -```yaml -docker: - httpProxy: http://proxy.example.test:3128 - httpsProxy: http://proxy.example.test:3128 - noProxy: localhost,127.0.0.1,.example.test -``` - -EPAR sets these values on the outer container before it starts, allowing -`dockerd` to inherit them on its first launch. Empty values preserve direct -networking. Proxy URLs are limited to credential-free HTTP(S) roots; use network -controls rather than embedding proxy passwords. Put host-specific endpoints in -ignored `.local/config.yml`. If the proxy performs TLS inspection, enable host -trust inheritance or configure the authorized root under -`image.trustedCaCertificatePaths` so verified HTTPS continues to work. - -## Host Trust Inheritance - -On Windows, macOS, and Linux controller hosts, Docker-DinD can add the host's -trusted TLS root anchors to each disposable Ubuntu runner: - -```yaml -image: - hostTrustMode: overlay - hostTrustScopes: [system, user] -``` - -Use `[system, user]` on Windows or macOS. Use `[system]` on Linux; Linux has no -portable per-user TLS root store. Overlay mode requires ephemeral Docker-DinD -runners. Existing configs remain disabled. New interactive Docker-DinD setup -shows a yes/no prompt and defaults to enabling inheritance. - -EPAR treats the host's root set as a versioned generation. It installs that -generation alongside Ubuntu's default roots and any certificates from -`image.trustedCaCertificatePaths`. The pool keeps its normal 15-second liveness -interval. A native controller recollects host trust every 15 seconds; a -containerized controller checks its read-only feed and refreshes idle-runner -leases every 5 seconds. Official no-Go launchers keep collection outside the -Linux toolchain container: their host-side watcher refreshes a read-only feed -every 10 seconds. The wrapper fails rather than use -the toolchain container's unrelated CA bundle when the configured feed is -missing, empty, invalid, or older than 30 seconds. - -When a generation changes, EPAR stops leasing old idle runners, removes them, -builds the replacement image, and registers replacement capacity. A busy runner -finishes its current job without having its trust changed. A synchronous -`ACTIONS_RUNNER_HOOK_JOB_STARTED` gate requires a current 20-second controller -lease and matching image generation. If GitHub assigns an old runner while it -is being retired, the gate fails before repository workflow steps run; the job -can still appear failed because GitHub assignment has already happened. - -The host trust snapshot and no-Go feed are controller inputs only. They are not -mounted into runner containers, and workflow code cannot use the feed as a host -filesystem channel. - -This feature inherits root anchors, not every host TLS-policy rule. The overlay -is additive, so removing a host root does not remove a matching root already in -Ubuntu or explicitly configured by path. Host Docker daemon trust is also -separate: a source-image pull can fail before EPAR can build the guest overlay. -Applications with private trust stores, including Java keystores, can require -additional configuration. - -## Image Build - -Docker-DinD images are Docker image tags, not Tart images or rootfs tar files: - -```bash -./bin/ephemeral-action-runner image build --replace -docker image ls epar-docker-dind-catthehacker-ubuntu -``` - -The default build starts from `ghcr.io/catthehacker/ubuntu:full-latest`, installs the GitHub Actions runner and EPAR helper scripts, and reuses the base image's Docker Engine/CLI/Compose/Buildx. The generated image also includes `/opt/epar/container-entrypoint.sh`, which starts the private inner `dockerd` when the runner container starts. - -Run `image update-upstream` only when selected install scripts need EPAR's pinned `actions/runner-images` checkout, such as the web/E2E script. - -## Runtime Behavior - -EPAR maps provider operations to Docker commands: - -- clone/create: `docker create --privileged --label epar.provider=docker-dind ...` -- start: `docker start`, then wait for inner `docker info` -- exec: `docker exec` -- address: `docker inspect` -- stop: `docker stop` -- delete: `docker rm -f -v` -- list: `docker ps -a --filter label=epar.provider=docker-dind` - -The provider does not mount `/var/run/docker.sock`, an OrbStack socket, or any host Docker socket into the runner container. It also does not publish host ports by default. If two jobs use the same Docker Compose project name or container ports, they are separated by their private inner Docker daemons. - -The inner daemon starts with `EPAR_DOCKERD_STORAGE_DRIVER=vfs` by default. `vfs` is slower than `overlay2`, but it is the most reliable default for nested Docker on Docker Desktop, OrbStack, and other privileged-container hosts where overlay mounts can fail inside the runner container. Advanced users can bake `EPAR_DOCKERD_STORAGE_DRIVER=overlay2` or `EPAR_DOCKERD_STORAGE_DRIVER=auto` into a derived image after validating that the exact host runtime supports it. - -On Apple Silicon hosts using Docker Desktop or OrbStack, the inner daemon may be able to run `linux/amd64` containers through the host runtime's emulation support. Validate this on the exact host before routing amd64-only workflows to Docker-DinD: - -```bash -docker exec docker run --rm --platform linux/amd64 alpine:3.20 uname -m -``` - -Expected output: - -```text -x86_64 -``` - -The runner process uses EPAR's non-systemd fallback: - -- `/opt/epar/run-runner.sh` starts `/opt/actions-runner/run.sh` in the background. -- `/var/run/actions-runner.pid` records the runner PID. -- `/opt/epar/check-runner.sh` reports liveness. -- `/var/log/actions-runner/run.log` records runner output. - -## Verification - -Local runtime check without GitHub registration: - -```bash -./bin/ephemeral-action-runner pool verify --instances 1 --cleanup -``` - -Full registration check: - -```bash -./bin/ephemeral-action-runner pool verify --instances 2 --register-only --cleanup -``` - -Dry-run command construction: - -```bash -./bin/ephemeral-action-runner pool verify --dry-run --instances 1 -``` - -The dry run should show `docker create` with `--privileged` and no host socket mount. - -For Docker Compose-heavy jobs that use fixed project names or ports, a useful isolation smoke test is to start two unregistered Docker-DinD instances, run the same compose stack in both with the same project name, and confirm the host Docker daemon only shows the two outer EPAR containers. The job-created containers should appear only when you run `docker exec docker ps` against each instance. - -## Caveats - -- Docker-DinD requires privileged containers. Treat it as trusted-job infrastructure. -- It is not a security boundary for hostile code. -- Inner Docker image cache is per runner instance and disappears on cleanup. -- Optional registry mirrors can reduce repeated pull time, but they are external services that must be secured and monitored separately. -- Cross-architecture containers, for example `linux/amd64` images on an ARM64 host, depend on the host Docker runtime's emulation support. -- Host Docker resource usage still matters because each runner container and inner daemon consumes CPU, memory, and disk on the same host. -- Docker Desktop, OrbStack, and Linux Docker Engine can have different privileged-container behavior. Validate on the exact host runtime you plan to use. diff --git a/docs/providers/docker-sandboxes.md b/docs/providers/docker-sandboxes.md new file mode 100644 index 0000000..ddc8282 --- /dev/null +++ b/docs/providers/docker-sandboxes.md @@ -0,0 +1,149 @@ +# Docker Sandboxes Provider + +Docker Sandboxes places each GitHub Actions listener inside a dedicated microVM sandbox with a private guest filesystem and Docker daemon. This is EPAR's strongest current host-isolation boundary. Its protection still depends on the installed Docker Sandboxes runtime, the host platform, EPAR's configuration, and the resources deliberately exposed to the workflow. + +```mermaid +flowchart TB + subgraph Host["Native controller host"] + EPAR["EPAR controller"] + SBX["Docker Sandboxes"] + Ledger["Exact ownership ledger"] + Cache["Shared template cache"] + end + subgraph Sandbox["One disposable microVM"] + Runner["Ephemeral Actions runner"] + Docker["Private Docker daemon"] + Jobs["Workflow and service containers"] + end + EPAR --> SBX + EPAR --> Ledger + Cache --> SBX + SBX --> Runner + Runner --> Docker + Docker --> Jobs +``` + +## When To Use It + +Choose Docker Sandboxes when its local checks pass and you want a microVM boundary around the runner and its Docker workload. The first-run wizard recommends it when the supported-platform, Docker, and machine-readable `sbx` readiness checks pass. Startup then performs the remaining storage, template, policy-rule, runtime, and registration admission checks and fails closed. A configured Docker Sandboxes pool never silently falls back to Docker Container or another provider. + +## Support Status + +EPAR recommends this provider in the wizard by capability, not by an operating-system allowlist: Docker must work, `sbx diagnose --output json` must report at least one passing check and no failed checks, and the controller architecture must have a matching native guest template. After configuration is saved, ordinary startup additionally requires storage and template admission before any runner starts. Windows x86_64 has the recorded real-host lifecycle evidence. The ARM64 implementation is architecture-complete, but equivalent real-host build, load, lifecycle, and independent-certification evidence has not yet been recorded. macOS and Linux also lack equivalent EPAR real-host evidence in this repository. + +## Prerequisites + +- Docker installed and running. +- Docker Sandboxes CLI whose `sbx diagnose --output json` result reports at least one passing check and no failed checks. Before the first-run provider assessment, the wizard runs `sbx daemon start --detach` when the `sbx` executable is installed, then runs diagnostics. Warnings and skipped checks do not make the provider unavailable. +- A native `amd64` or `arm64` controller with matching `linux/amd64` or `linux/arm64` image support. EPAR does not use emulation to admit a mismatched template. +- Enough capacity to resolve, build, export, import, and retain the selected runner template. +- Enough physical backing storage for the estimated incremental template and sandbox bootstrap work while retaining `storage.minimumFree`. Sparse root and inner-Docker logical maxima are reported separately and are not counted as immediate host allocation. +- A GitHub runner group that meets enforced policy. Docker Sandboxes requires `security.runnerGroup.enforcement: enforce` and `runner.ephemeral: true`. + +The wizard builds and imports the template. The recipes in `templates/docker-sandboxes` are build inputs, not prebuilt images. + +## Private Filesystem and VM Helper Approval + +Every sandbox receives a private Docker daemon backed by its own Linux filesystem image and a read-only template filesystem. During first creation on macOS or Linux, Docker Sandboxes may launch helpers including `mkfs.ext4` to format the private Docker disk image, `mkfs.erofs` to construct an EROFS template snapshot, and `containerd-shim-nerdbox-v1` to launch and manage the sandbox VM. The operating system, endpoint-security software, or application-control policy may ask the signed-in user to approve each helper; macOS Gatekeeper may say that the executable “is an app downloaded from the Internet.” These commands are launched by the Docker Sandboxes runtime, not by a workflow and not directly by EPAR. With the current Homebrew `sbx` package on macOS, the runtime and shim are installed beneath `/opt/homebrew/Caskroom/sbx//`, sandbox state is beneath Docker Sandboxes' user data directories, the ext4 target resembles `~/.sbx/run/d/containerd/.../images/-docker.img`, and the EROFS target resembles `~/.sbx/run/d/containerd/.../snapshots//layer.erofs`. + +Review the complete executable and any target shown by every prompt. Approve it only when the executable belongs to the Docker Sandboxes installation you intentionally installed and any file target is beneath that runtime's sandbox data directory. A formatter must not target a physical device such as `/dev/disk*`, another user-data path, or an unrelated file. The configured `dockerSandboxes.dockerDisk` value is the sparse logical maximum for the private Docker filesystem, not an immediate allocation of that entire size. Denying or blocking any required helper prevents that sandbox from starting and commonly surfaces through `sbx` as `500 Internal Server Error: failed to run sandbox container`; EPAR then fails closed without registering the runner. Correct the host approval policy and retry the exact EPAR command rather than running a formatter or shim manually. + +## Minimal Configuration + +Start with `./start` or [`configs/docker-sandboxes.example.yml`](../../configs/docker-sandboxes.example.yml). Configuration expresses the desired source; EPAR stores immutable build and cache identities in its local receipt. + +```yaml +provider: + type: docker-sandboxes + platform: linux/amd64 + +image: + sourceType: docker-image + sourceImage: ghcr.io/catthehacker/ubuntu:full-latest + sourcePlatform: linux/amd64 + runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" + customInstallScripts: + # - examples/custom-install/install-extra-apt-tools.sh + +dockerSandboxes: + policyGeneration: sha256: + networkBaseline: open + stagingRoot: .local/docker-sandboxes-staging + cpus: 4 + memory: 8GiB + rootDisk: auto + dockerDisk: 50GiB + maxConcurrentCreates: 2 +``` + +`provider.sourceImage` is invalid for this provider; use the common `image` section. `rootDisk: auto` derives a sparse logical root maximum from the selected artifact. `dockerDisk` is an independent sparse workload limit whose default is 50 GiB and minimum is 1 GiB. Neither virtual maximum is treated as immediately consumed host space; the only physical reserve is `storage.minimumFree`, whose generated default is 1 GiB. See [Configuration](../configuration.md) for the complete schema. + +`networkBaseline: open` adds EPAR-owned sandbox-scoped public egress plus deny-wins guardrails for host aliases; it does not change the host-global Docker Sandboxes policy. Use `balanced` with `additionalAllow` for default-deny public egress. Additional allow/deny entries are exact hostnames or `*.domain[:port]`; they cannot override the Open host-alias denies. + +## Normal Workflow + +1. Run `./start` with no config and select Docker Sandboxes when its tooling and diagnostics pass. Choose a Catthehacker profile or tag and optional custom install scripts; review the non-blocking physical-growth estimate, sparse logical limits, reserve, confidence, and expected duration. +2. The wizard writes the desired configuration. Embedded `./start` then enters the ordinary provisioning path, performs authoritative storage admission, builds and imports the template, and activates it only after exact readback. On macOS or Linux, review the narrowly scoped helper prompts described in [Private Filesystem and VM Helper Approval](#private-filesystem-and-vm-helper-approval) if the host presents them. +3. Prewarm the selected template without GitHub registration: + + ```powershell + powershell.exe -NoProfile -ExecutionPolicy Bypass -File scripts/build-native-controller.ps1 pool verify --config .local/docker-sandboxes.yml --project-root . --instances 1 --cleanup + ``` + +4. Start the pool with `./start`. EPAR reuses the verified imported template without a registry check until the configured update schedule is due; local input changes and missing templates still rebuild immediately. + +Each allocation receives an empty owner-restricted staging directory, but Actions `_work` stays on the guest filesystem. EPAR verifies the guest, confirms that the configured sandbox-scoped policy rules are present, and verifies the private daemon and runner trust policy before requesting a short-lived registration token. With `image.hostTrustMode: overlay`, the common pool lifecycle installs the selected roots, verifies the immutable generation, and maintains the job-start lease. With the setting omitted or disabled, the template carries an explicit disabled-policy marker and does not install the trust hook. The token remains on the native host except for registration through `sbx exec` standard input. + +The listener identity is explicit and self-consistent: `agent` owns its home, XDG, runtime, and Docker configuration directories, and every workflow action and shell command inherits those exact paths. Template construction removes Docker credentials inherited from source-image user homes and verification rejects reusable artifacts that retain registry authentication or point identity-derived paths at another user. A workflow login can therefore write only to the disposable sandbox's Docker client configuration, and that file disappears with the sandbox. Registry authorization can still be changed by Docker Sandboxes' host-side credential proxy as described below. + +Docker Sandboxes can automatically forward the host SSH agent when its shared daemon inherits `SSH_AUTH_SOCK`. That would expose a host credential capability to every sandbox created by that daemon, so EPAR rejects any guest containing `SSH_AUTH_SOCK`, `SSH_AUTH_SOCK_GATEWAY`, `SSH_AGENT_PID`, or `/run/ssh-agent.sock`. EPAR removes these variables when it launches Docker Sandboxes commands, but it cannot repair an already-running daemon that another shell or tool started with forwarding enabled. Coordinate with other Docker Sandboxes users on the host, stop the shared daemon, and restart it from a sanitized environment before retrying EPAR: + +```sh +sbx daemon stop +env -u SSH_AUTH_SOCK -u SSH_AUTH_SOCK_GATEWAY -u SSH_AGENT_PID sbx daemon start --detach +``` + +Do not relax the verification or merely delete the relay socket: the gateway setting is itself a forwarding capability and the daemon's inherited environment is authoritative for subsequently created sandboxes. + +## Docker Hub Credentials and Transparent Egress + +Docker Sandboxes v0.37.1 provides three HTTP(S) egress paths. Its `forward` path can terminate TLS and replace a guest registry `Authorization` header with the host `sbx login` credential; `forward-bypass` and `transparent` do not inject credentials. All three remain subject to Docker Sandboxes network policy. A runner using the forward path can therefore report `Login Succeeded`, have a correct `/home/agent/.docker/config.json`, and still receive `insufficient_scope: authorization failed` because the host identity, not the workflow identity, performs the private pull. + +EPAR configures the sandbox-private Docker daemon with Docker Engine's normal daemon proxy object and `no-proxy: "*"`. The forward proxy address remains explicit, but the wildcard makes dockerd use Docker Sandboxes' transparent interception for every registry and changing CDN hostname. EPAR also starts runner registration and the Actions listener from an allowlisted clean environment that does not inherit `HTTP_PROXY`, `HTTPS_PROXY`, or their lowercase forms. Ordinary workflow and nested-Docker traffic therefore defaults to policy-enforced transparent egress, where the workflow's own `docker login` remains authoritative. Template verification requires `/etc/docker/daemon.json` to be root-owned and non-symlinked, preserves the exact proxy object while permitting only EPAR's validated optional `registry-mirrors` merge, and requires the running daemon to report `NoProxy=*`. + +The host variables `DOCKER_SANDBOXES_NO_PROXY` and `NO_PROXY` described in Docker Sandboxes' architecture documentation are not substitutes for this guest configuration. They only choose whether the host-side Sandbox proxy reaches its next hop directly or through an optional upstream proxy; they do not bypass the Sandbox proxy or its credential interceptor. Likewise, changing `DOCKER_CONFIG`, combining `docker login` and `docker pull` in one step, or replacing `docker/login-action` with a shell command cannot repair an old template that still sends dockerd through the forward path. + +EPAR continues to reject global Docker Sandboxes service and registry secrets and host SSH-agent forwarding. It never copies workflow secrets back to the host. The transparent default is an operational isolation control, not a hard boundary against a hostile root-capable workflow: a job with root inside the microVM can deliberately configure a client to reconnect to the Sandbox forward proxy, and Docker Sandboxes v0.37.1 documents no per-sandbox switch that disables the credential interceptor itself. Keep the mandatory host `sbx login` identity least-privileged. Use Docker Container or another provider if even deliberate use of that residual host credential capability is outside the workflow trust boundary. + +For diagnosis, `sbx policy log ` should report `transparent` for `registry-1.docker.io`, `auth.docker.io`, and current Docker Hub blob hosts. A daemon log saying it is overriding a client-supplied registry credential, or policy entries showing `forward` for Docker Hub, identifies an old or altered template. Aligning `sbx login` with a workflow account can prove the interception diagnosis, but it establishes only shared-host authorization and is not EPAR's remediation for per-job credentials. + +Template construction uses two independent trust paths. EPAR's project-owned BuildKit builder automatically receives host system roots for Docker Hub, GHCR, and the other pinned registries used by the build. The native controller downloads the locked Actions runner and `tini`, verifies their SHA-256 values, and then supplies them as local build inputs; the Dockerfile does not perform remote HTTPS downloads. + +BuildKit streams the runner template directly to an attestation-free, verified archive; Docker Sandboxes does not require or retain a Docker staging image. Separate cache-backed BuildKit targets produce the max-mode provenance, SBOM, and software inventory without loading the runner image into Docker Engine. After `sbx template load` succeeds and EPAR reads back the exact imported template, startup housekeeping removes the transient archive workspace while retaining the active template and compact receipt evidence. Initial creation and every replacement trust the authoritative Sandbox cache readback, so the expected absence of a Docker image does not block a runner. Superseded templates are removed only after no configuration, lease, or live sandbox references them. + +The opt-in `TestLiveRunnerTemplateIsolation` proof also exercises authenticated Docker-client state across separate commands. Set `EPAR_LIVE_DOCKER_SANDBOXES_REGISTRY_IMAGE` to an immutable Distribution Registry image reference and `EPAR_LIVE_DOCKER_SANDBOXES_HTPASSWD_IMAGE` to an immutable image containing `htpasswd`, in addition to the existing live-test template, digest, and staging variables. The test generates credentials in memory, creates a registry inside the sandbox-private daemon, logs in through standard input, pushes and separately pulls a private image, logs out, verifies the registry auth entry is absent, and exactly removes its registry container and images. + +## Limitations + +- `networkBaseline: open` permits public egress, which can exfiltrate secrets or data exposed to the workflow. Use least-privilege runner groups and secrets, and choose `balanced` with narrow allow rules for higher-risk workloads. +- Docker Sandboxes template cache storage is shared host state; it is not a per-sandbox root-disk measurement. +- macOS ARM64 remains preview-only while Docker Sandboxes and its host-authentication contract continue to evolve. Current v0.37.1 evidence includes three consecutive private Docker Hub pulls with intentionally different host and workflow identities, transparent registry/auth/blob routing, AMD64 image pulls from an ARM64 guest daemon, ephemeral replacement, and exact sandbox cleanup. +- Transparent egress preserves ordinary per-job Docker Hub credentials, but a root-capable workflow can deliberately opt back into the Docker Sandboxes forward proxy. Use Docker Container when that residual host credential capability is outside the trust boundary. +- A stopped sandbox is diagnostic state, not proof of deletion. Unknown state consumes capacity and blocks replacement. +- `EPAR_DISABLE_DOCKER_SANDBOXES=1` fails admission closed during an incident or compatibility problem. + +## Verification + +Use the prewarm command above for an unregistered lifecycle check. To include GitHub registration, run: + +```bash +ephemeral-action-runner pool verify --config .local/docker-sandboxes.yml --instances 1 --register-only --cleanup +``` + +The shared pool treats provisioning, ready, draining, quarantined, and cleanup-pending instances as capacity-consuming states. Cleanup uses durable exact sandbox, GitHub runner, and staging-directory identities; it never uses an `sbx` reset or broad prefix deletion. + +## Troubleshooting + +For symptoms and recovery, see [Troubleshooting](../troubleshooting.md). If failed diagnostics retain a sandbox, inspect `status` and preserve the reported evidence before using `cleanup --acknowledge-failed-diagnostics`; ordinary cleanup never applies that override. diff --git a/docs/providers/tart.md b/docs/providers/tart.md index 30901f1..b200f19 100644 --- a/docs/providers/tart.md +++ b/docs/providers/tart.md @@ -1,51 +1,69 @@ # Tart Provider (Experimental) -The Tart provider is experimental. It targets Apple Silicon macOS hosts and currently supports Ubuntu ARM64 guests. Tart itself can run macOS ARM64 VMs, but EPAR's image build, runner service, validation, and cleanup scripts currently depend on Ubuntu, systemd, and Linux process interfaces, so macOS guests are not yet an EPAR provider mode. +Tart runs each disposable EPAR runner in an Ubuntu ARM64 VM on Apple Silicon macOS. EPAR's Tart integration is experimental. -> [!WARNING] -> The default Tart source, `ghcr.io/cirruslabs/ubuntu:latest`, is a basic Ubuntu ARM64 OS image. It does not contain the broad dependency set normally present in GitHub's hosted images from [`actions/runner-images`](https://github.com/actions/runner-images), including many language SDKs, CLIs, browsers, and build tools. Tart uses Apple's Virtualization framework, so an Apple Silicon host runs an ARM64 VM. Rosetta translates supported x86_64 Linux user-space programs inside that ARM64 guest; it does not create an x64 VM, and not every amd64 image or workload is compatible. +## When To Use It -EPAR currently validates the Ubuntu path: +Choose Tart only when you have Apple Silicon macOS and specifically want Linux ARM64 VMs. Use Docker Container or WSL/x64 Linux when native amd64 workflow compatibility is more important. -- clone a reusable Tart image -- start the VM headless -- use the Tart guest agent for `exec` and IP discovery -- validate the base GitHub Actions runner runtime -- register an ephemeral GitHub runner from the host -- delete the VM after the runner exits +## Support Status -Use `configs/tart.example.yml` for the basic runner-only Ubuntu image or `configs/tart.web-e2e.example.yml` for the existing opt-in web/E2E and Rosetta experiment. EPAR installs the GitHub Actions runner and its lifecycle scripts, but the default does not install Docker, .NET, PowerShell, Go, browsers, or the rest of GitHub's hosted-runner tool inventory. +EPAR supports Ubuntu ARM64 guests, not macOS guests. The default source is a basic Ubuntu VM image, not a GitHub-hosted runner image; it does not include the broad language, browser, Docker, and CLI inventory associated with `actions/runner-images`. -If a workflow needs a GitHub-runner-like environment, build and maintain your own bootable Tart source image. Adapt the Ubuntu build scripts and tool definitions from [`actions/runner-images`](https://github.com/actions/runner-images) to that image, validate the resulting ARM64 tools, push it as a Tart VM image, and set `image.sourceImage` to it. Alternatively, add narrowly scoped `image.customInstallScripts` for only the dependencies your workflows require. EPAR does not convert the Catthehacker Docker image or automatically reproduce the complete GitHub-hosted image for Tart. +## Prerequisites -When Docker/browser support is selected on ARM64, EPAR exposes a Chromium-compatible browser through `epar-browser`, `chromium`, and `chromium-browser`; it is not guaranteed to be Google Chrome. +- Apple Silicon macOS and a working `tart` CLI. +- A bootable Ubuntu ARM64 Tart source image. +- Enough local VM storage for a reusable image and active disposable VMs. -The default network mode is Tart NAT. `softnet` is accepted by the provider, but it can require host-side privileges. +## Minimal Configuration -If Docker is installed in a custom Tart guest, optional `docker.registryMirrors` settings are applied to the guest Docker daemon when each disposable VM starts. Use a mirror URL that is reachable from inside the Tart VM; `host.docker.internal` is Docker-container-specific and may not resolve in Tart guests. See [Docker Registry Mirrors](../advanced/docker-registry-mirrors.md). - -## Experimental Rosetta Support For Linux Amd64 Containers - -Tart on Apple Silicon runs ARM64 VMs, but Tart can expose Apple's Linux Rosetta runtime to the guest with `tart run --rosetta `. EPAR supports this as an opt-in Tart-only setting: +Start with [`configs/tart.example.yml`](../../configs/tart.example.yml): ```yaml +image: + sourceImage: ghcr.io/cirruslabs/ubuntu:latest + outputImage: epar-ubuntu-24-arm64 + updateFrequency: weekly + updateTime: "07:00" + provider: type: tart - rosettaTag: rosetta + sourceImage: epar-ubuntu-24-arm64 + network: default ``` -When `provider.rosettaTag` is set, EPAR starts Tart instances with `--rosetta rosetta`, installs `/opt/epar/setup-rosetta.sh` during image build, enables `epar-rosetta.service`, and registers an x86_64 Linux `binfmt_misc` handler inside the guest. Images with the Rosetta feature marker validate: +Use a distinct image and label when adding tools through `image.customInstallScripts`. If workflows need a GitHub-runner-like environment, build and maintain your own bootable Ubuntu Tart image; EPAR does not convert Catthehacker Docker images into Tart VMs. + +## Normal Workflow + +1. Create the configuration and run `./start` to build the reusable Tart image and start the pool. +2. Target the ARM64 Tart label in workflows. +3. Use `pool verify` before routing a new workload to the provider. + +Tart clones the reusable image, starts the VM headless, uses the guest agent for command execution and IP discovery, and removes the VM after the ephemeral runner exits. + +## Limitations + +- This provider is experimental. It uses a per-runner VM boundary, but workflows still control the guest and any secrets or services exposed to the job. +- The default image is runner-only. Add only the dependencies your workflows need, or maintain a fuller source image yourself. +- `provider.network: softnet` may require additional host privileges; NAT is the default. +- Rosetta can translate some Linux amd64 user-space workloads in an ARM64 guest, but it does not turn the VM into an x64 VM or guarantee every amd64 workload. + +## Verification ```bash -sudo -u runner -H docker run --rm --platform linux/amd64 alpine:3.20 sh -c 'uname -m' +ephemeral-action-runner pool verify --instances 1 --cleanup ``` -The expected output is `x86_64`. +For the optional Rosetta experiment, use `configs/tart.web-e2e.example.yml` or set a distinct `provider.rosettaTag`. Verify a real container execution before routing amd64 workflows: + +```bash +docker run --rm --platform linux/amd64 alpine:3.20 uname -m +``` -Host prerequisites: +Run that command inside the Tart guest; expected output is `x86_64`. -- Apple Silicon macOS -- Tart version with `--rosetta` support -- Apple's Rosetta package installed on the macOS host +## Troubleshooting -This is experimental support for Linux amd64 user-space containers. It is not nested virtualization and it does not make the Ubuntu VM an x64 VM. For native amd64 performance and compatibility, use a Windows WSL x64 provider or another x64 Linux host. For Tart Rosetta-capable runners, expose a distinct label such as `epar-tart-rosetta-amd64` so workflows can opt into the behavior explicitly. +See [Troubleshooting](../troubleshooting.md) for platform and runtime failures. Keep Rosetta-capable runners behind a dedicated label so workflows opt in deliberately. diff --git a/docs/providers/wsl.md b/docs/providers/wsl.md index 1a62e9c..d067c85 100644 --- a/docs/providers/wsl.md +++ b/docs/providers/wsl.md @@ -1,21 +1,24 @@ -# WSL Provider +# WSL2 Provider -The WSL provider targets Windows hosts running WSL2. It manages disposable Ubuntu distros for trusted GitHub Actions jobs. +WSL2 runs each disposable GitHub Actions runner in an imported Ubuntu WSL distribution on a native Windows host. EPAR uses the shared pool lifecycle, including strict capacity, registration, replacement, and cleanup. -The provider maps EPAR lifecycle operations to `wsl.exe`: +## When To Use It -- clone/create: `wsl --import --version 2` -- start/exec: `wsl -d --user root --exec ` -- stop: `wsl --terminate ` -- delete: `wsl --unregister ` -- export image: `wsl --export ` -- list: `wsl --list --verbose` +Choose WSL2 for Windows-hosted Linux runners when you want native x64 Linux execution and a WSL distribution per runner. It is often the clearest choice for workflows that need native amd64 Docker execution on Windows. -When a disposable runner is started, EPAR also keeps a quiet host-side `wsl.exe -d ` process open. This prevents WSL from auto-stopping an imported distro that is otherwise only running systemd services. `pool up`, `pool verify --cleanup`, and `cleanup` terminate that keepalive by terminating or unregistering the distro. +## Support Status -## Configuration +WSL2 is supported only on native Windows with WSL default version 2. It is not equivalent to one full VM per job: distros share the WSL kernel and host integration surface, so use it for trusted jobs. -Use `configs/wsl.example.yml` as the starting point: +## Prerequisites + +- Native Windows, `wsl.exe --status` reporting default version 2, and an Ubuntu-compatible WSL environment. +- Docker installed and running for the default Catthehacker Docker-image source during `image build`; later runner startup does not require Docker on the host. +- Enough storage for the pulled source image, intermediate rootfs tar, temporary WSL build distro, reusable tar, and active pool. + +## Minimal Configuration + +Start with [`configs/wsl.example.yml`](../../configs/wsl.example.yml): ```yaml image: @@ -23,12 +26,8 @@ image: sourceImage: ghcr.io/catthehacker/ubuntu:full-latest sourcePlatform: linux/amd64 outputImage: work/images/epar-wsl-catthehacker-ubuntu.tar - customInstallScripts: - # - examples/custom-install/install-extra-apt-tools.sh - -runner: - labels: [self-hosted, linux, X64, epar-wsl-catthehacker-ubuntu] - includeHostLabel: true + updateFrequency: weekly + updateTime: "07:00" provider: type: wsl @@ -36,71 +35,34 @@ provider: installRoot: work/wsl ``` -`image.sourceType: docker-image` tells EPAR to convert the source Docker image into a WSL-importable rootfs tar during `image build`. `image.outputImage` is the reusable runner tar produced by `image build`. `provider.sourceImage` is the tar imported for disposable runner instances. - -Use `configs/wsl.lean.example.yml` when you want the smaller tar-first path. Existing WSL configs that point `image.sourceImage` at a `.tar`, `.tar.gz`, or `.tgz` file are treated as `image.sourceType: rootfs-tar` for backward compatibility. Use `configs/wsl.web-e2e.example.yml` when workflows need the larger lean web/E2E install script and its `epar-wsl-ubuntu-24.04-web-e2e` label. - -## Docker Image Sources - -For the default full WSL image, EPAR uses Docker on the Windows host during `image build`: - -1. `docker pull --platform linux/amd64 ghcr.io/catthehacker/ubuntu:full-latest` -2. `docker create` a temporary stopped container -3. `docker container inspect` to capture image environment metadata -4. `docker export` the container filesystem into an intermediate rootfs tar -5. `docker rm -f -v` cleanup +Use `configs/wsl.lean.example.yml` with a pre-exported Ubuntu rootfs tar for a smaller path, or `configs/wsl.web-e2e.example.yml` for browser/E2E tooling. See [Configuration](../configuration.md) for labels, runner-group policy, capacity, and mirrors. -That exported rootfs is then imported into a temporary WSL distro. EPAR copies `/opt/epar` scripts, enables systemd, installs the GitHub Actions runner, writes the captured env to `/opt/epar/source-image.env`, validates Docker Engine from the base image, finalizes the image, and exports `image.outputImage`. +## Normal Workflow -The intermediate source tar and env metadata are cached beside `image.outputImage`. Delete `*.source.rootfs.tar` and `*.source.rootfs.tar.env` when you intentionally want to reconvert the Docker image. +1. Run `./start`; EPAR converts the default Docker source into a rootfs tar, builds the reusable WSL runner tar, and starts the pool. +2. Target the configured WSL label, normally `epar-wsl-catthehacker-ubuntu`. +3. Stop the supervisor with Ctrl-C, then wait for cleanup to finish before closing the terminal. -Docker is required only for the Docker-image conversion step. Running WSL pool instances afterward does not require Docker Desktop unless your workflows need Docker Desktop or another host-side Docker service. +EPAR imports each runner from `provider.sourceImage`, enables systemd in the reusable image, and keeps a quiet host-side WSL process alive while a runner waits for work. That process is intentional: it prevents an otherwise idle systemd distro from stopping automatically. -## Systemd And Docker +## Limitations -The WSL image build writes `/etc/wsl.conf` with systemd enabled and `appendWindowsPath=false`, restarts the temporary distro, then installs the GitHub Actions runner inside the distro. Disabling Windows PATH injection keeps validation and jobs from accidentally resolving host-installed tools such as Windows Docker or Node. +- The default full image needs Docker only for source conversion; an image build can fail when the host daemon's storage is full even if Windows has free disk space. +- EPAR does not install cross-architecture emulation. An x64 WSL runner can pull an ARM64 image but cannot execute it natively. +- The default Docker-enabled runner uses Docker Engine inside WSL, not a mounted Windows Docker socket. -The default WSL full image expects Docker Engine, dockerd, Compose v2, Buildx, and iptables to already exist in `ghcr.io/catthehacker/ubuntu:full-latest`. EPAR validates those tools and marks the image with `/opt/epar/features/docker-engine` so `pool verify` proves: +## Verification -```bash -sudo -u runner -H docker version -sudo -u runner -H docker compose version -sudo -u runner -H docker buildx version -sudo -u runner -H docker run --rm hello-world +```powershell +ephemeral-action-runner pool verify --instances 1 --cleanup ``` -Browser support is validated only when `image.customInstallScripts` includes `scripts/guest/ubuntu/install-docker-browser.sh` or `scripts/guest/ubuntu/install-web-e2e.sh`: +For a lean rootfs source, export Ubuntu 24.04 once before building: -```bash -sudo -u runner -H docker version -sudo -u runner -H docker compose version -sudo -u runner -H docker buildx version -sudo -u runner -H docker run --rm hello-world -printf '%s\n' '

EPAR browser validation marker

' >/tmp/epar-browser-validation.html -sudo -u runner -H chromium --headless --no-sandbox --dump-dom file:///tmp/epar-browser-validation.html +```powershell +wsl --export Ubuntu-24.04 work/images/ubuntu-24.04-clean.rootfs.tar ``` -The provider does not mount the Windows Docker Desktop socket. Docker-enabled jobs run against Docker Engine inside the WSL distro. - -Runner startup sources `/opt/epar/source-image.env` before launching `/opt/actions-runner/run.sh`. This lets GitHub Actions jobs inherit source image variables such as `ImageOS`, `ImageVersion`, `RUNNER_TOOL_CACHE`, browser paths, and Java paths. WSL keeps its own systemd and host keepalive model; it does not reuse Docker-DinD's container entrypoint. - -WSL x64 is the preferred EPAR target for workflows that pull amd64-only Docker runtime images. - -An x64 WSL runner can store an ARM64 image with `docker pull` or `docker load`, but it cannot run that image natively. Running ARM64 containers requires an explicitly configured emulation layer such as QEMU registered through `binfmt_misc`, or a native ARM64 runner. EPAR does not install cross-architecture emulation in WSL by default, so validate both the architecture and execution path before routing ARM64-dependent jobs to an x64 WSL label. - -If `docker.registryMirrors` is configured, EPAR applies it to Docker Engine inside each disposable WSL distro before validation. Use a mirror URL reachable from inside WSL, such as an organization DNS name or a host/LAN address. See [Docker Registry Mirrors](../advanced/docker-registry-mirrors.md). - -## Caveats - -- WSL2 is not the same isolation boundary as a full VM per job. -- WSL distros share the WSL kernel and host integration surface. -- Use this provider for trusted internal jobs unless your environment has reviewed and accepted the isolation model. -- The default Docker-image source needs Docker Desktop, Docker Engine, or another reachable Docker daemon during `image build`. -- The full Catthehacker runner image is large and needs enough disk for the pulled Docker image, the intermediate source rootfs tar, the temporary WSL import, and the final WSL tar. -- Expect one long-lived host `wsl.exe` process per running disposable runner. This is intentional and keeps the WSL distro alive while it waits for jobs. -- Cleanup only unregisters distros whose names match `pool.namePrefix`. - -References: +## Troubleshooting -- [WSL basic commands](https://learn.microsoft.com/en-us/windows/wsl/basic-commands) -- [Systemd support in WSL](https://learn.microsoft.com/en-us/windows/wsl/systemd) +See [Troubleshooting](../troubleshooting.md) for Docker source conversion, WSL import failures, WSL/Docker storage, and architecture errors. Do not use `wsl --unregister` on an unrelated distro as general EPAR cleanup. diff --git a/docs/runner-groups.md b/docs/runner-groups.md new file mode 100644 index 0000000..83cc36a --- /dev/null +++ b/docs/runner-groups.md @@ -0,0 +1,54 @@ +# Runner Group Security + +GitHub decides which repositories can route jobs to a self-hosted runner through its runner group before EPAR receives a job. EPAR therefore verifies the configured group policy before creating a registration token; it does not accept a job first and reject it afterward. + +## Recommended GitHub Setup + +1. Open the organization’s **Settings**, then **Actions**, **Runners**, and **Runner groups**. +2. Create a dedicated group for EPAR runners. +3. Choose **Selected repositories** and add only repositories whose workflows are trusted to run on the EPAR host. +4. Keep public repository access disabled. +5. Run `./start` or `ephemeral-action-runner init` and select that group when the wizard lists the organization’s live runner groups. + +See GitHub’s [runner-group access documentation](https://docs.github.com/en/actions/how-tos/manage-runners/self-hosted-runners/manage-access) for the organization and enterprise controls. + +Any repository with access to the group can route a matching job to its runners. Broad access also applies to repositories created later, and public repositories can expose self-hosted runners to untrusted pull request or fork workflows. Provider isolation does not replace this authorization boundary: Docker Sandboxes isolates each runner inside a dedicated microVM, but a job can still access its assigned secrets and reachable services; Docker Container and WSL should remain limited to trusted workflows. + +## Wizard Decisions + +The wizard requires an explicit numbered selection; pressing Enter does not select the default group. It blocks groups that permit public repositories under the generated safety policy. + +GitHub's runner-group API does not provide a creation timestamp, so the wizard cannot order groups by age. It orders them by security posture instead: groups with public repository access disabled appear first, then selected-repository access before all-private access before all-repository access. Within equivalent policies, non-default organization-managed groups appear before default or inherited alternatives. The wizard explains each access mode in plain language and labels groups as recommended, requiring review, not recommended, or blocked. + +The default group and groups available to all private or all repositories are not automatically forbidden. The wizard explains their broader and potentially future access, then offers **Continue with this group** or **Back to group selection**. Continuing records that deliberate choice in the policy. Inherited enterprise groups receive an additional warning because their policy must be managed at enterprise level. + +The wizard must read the live GitHub policy. It has no offline or unchecked group-name fallback, and it does not write or overwrite a config if the API call or selection fails. + +## Configuration + +```yaml +runner: + group: your-runner-group + +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + requirePublicRepositoriesDisabled: true +``` + +`enforcement` accepts `enforce` or `warn`. Enforcement blocks new runner registrations when GitHub cannot be checked or the policy violates a requirement. Warning mode reports the same conditions and continues; configs created before this feature use strict recommended requirements in warning mode until migrated. + +`requiredRepositoryAccess` is the maximum permitted breadth. `selected` permits only selected-repository access, `private` permits selected or all-private access, and `all` permits any GitHub repository-access setting. Tightening GitHub policy remains valid; broadening beyond the configured ceiling blocks registration. + +`requirePublicRepositoriesDisabled` is independent of repository breadth. Setting it to `false` deliberately allows public repositories but produces a runtime advisory. This override should be exceptional and should be paired with a documented trusted-workflow threat model. + +Setting `requireNonDefaultGroup` to `false` permits the default group but still produces an advisory when it is used. Setting `requireExplicitGroup` to `false` permits an empty `runner.group`; EPAR then resolves and checks the group GitHub marks as default. + +## Runtime Behavior + +EPAR checks policy before startup image work, before registration-enabled pool provisioning, and again immediately before each runner registration token request. Enforcement failures do not delete existing runners or interrupt jobs. EPAR does not modify GitHub policy and does not run a periodic policy audit. + +An administrator can change GitHub policy between the final check and registration because GitHub does not provide an atomic policy-check-and-register operation. Keeping group administration restricted remains necessary. diff --git a/docs/security.md b/docs/security.md index c8c277a..4e3f8bb 100644 --- a/docs/security.md +++ b/docs/security.md @@ -1,6 +1,6 @@ # Security -EPAR is intended for trusted jobs by default. It adds cleanup and isolation around GitHub self-hosted runners, but it does not make an existing host safe for arbitrary untrusted workflows. +EPAR provides disposable GitHub self-hosted runners with provider-dependent isolation. Docker Sandboxes places each runner inside a dedicated microVM sandbox and provides EPAR's strongest current host-isolation boundary. Docker Container and WSL remain trusted-workflow infrastructure; Tart uses a VM but is experimental. No provider is guaranteed to be universally safe for arbitrary hostile workflows. GitHub's self-hosted runner warning still applies: GitHub recommends using self-hosted runners only with private repositories because public repository forks can run code on the runner machine through pull request workflows. Read the official GitHub guidance before exposing any self-hosted runner to public or untrusted workflows: [Adding self-hosted runners](https://docs.github.com/actions/hosting-your-own-runners/adding-self-hosted-runners). @@ -14,23 +14,27 @@ If private reporting is unavailable, contact a repository maintainer privately t ## What EPAR Improves -Disposable instances reduce host pollution, stale runner state, and accidental cross-job interference. After a job completes, EPAR retires the instance and creates a replacement. For Docker-DinD, job-created containers, networks, volumes, and inner image cache live inside the runner container's private Docker daemon and are removed with that runner instance. +Disposable instances reduce host pollution, stale runner state, and accidental cross-job interference. After a job completes, EPAR retires the instance and creates a replacement. For Docker Container, job-created containers, networks, volumes, and inner image cache live inside the runner container's private Docker daemon and are removed with that runner instance. ## What EPAR Does Not Guarantee -A workflow controls the runner environment while it runs and can access any secrets exposed to that workflow. Ephemeral cleanup reduces persistence risk after the job, but it is not a hostile-code sandbox. +A workflow controls its runner environment while it runs and can access any secrets and reachable services exposed to that workflow. Ephemeral cleanup alone is not a sandbox boundary: Docker Sandboxes supplies a dedicated microVM boundary, while the other providers use their documented isolation models. Do not mount host source directories, Docker sockets, private keys, or long-lived cloud credentials into runner instances unless that is inside your trust boundary. -Use GitHub runner groups, repository restrictions, environment protections, and minimal secrets. Avoid routing public pull request workflows, forked contributions, or unknown third-party workflow code to EPAR runners. +Use GitHub runner groups, repository restrictions, environment protections, and minimal secrets. EPAR's [runner-group security preflight](runner-groups.md) checks the configured routing policy before registration, but it does not make public pull request workflows, forked contributions, or unknown third-party workflow code trustworthy. ## Provider Notes EPAR intentionally does not implement a Docker-socket provider. A runner that controls the host Docker socket can usually control the host. -Docker-DinD uses a privileged outer container with a private inner Docker daemon. That gives good cleanup and Docker resource separation for each job, but it is still trusted-job infrastructure because `--privileged` weakens container isolation. +Docker Container uses a privileged outer container with a private inner Docker daemon. That gives good cleanup and Docker resource separation for each job, but it is still trusted-job infrastructure because `--privileged` weakens container isolation. -Tart runs jobs inside VMs on Apple Silicon macOS. That is a stronger host boundary than Docker-DinD, but workflows still control the guest and any secrets exposed to the job. +Docker Sandboxes places the listener, guest filesystem, and private Docker daemon inside a dedicated microVM sandbox. It provides EPAR's strongest current host boundary and materially strengthens host isolation relative to Docker Container. The first-run wizard recommends it when the supported-platform, Docker, and machine-readable `sbx` readiness checks pass; startup then performs the remaining storage, template, policy-rule, runtime, and registration admission checks and fails closed. This selection rule is separate from independent platform certification and does not claim that every host combination has received the same real-host validation. + +Docker Sandboxes may forward a host SSH agent when its shared daemon inherits `SSH_AUTH_SOCK`. EPAR strips SSH-agent variables from child commands and rejects any sandbox exposing the socket, gateway, or agent PID; operators must restart an already-running daemon with those variables unset rather than weakening the check. + +Tart runs jobs inside VMs on Apple Silicon macOS. That is a stronger host boundary than Docker Container, but workflows still control the guest and any secrets exposed to the job. WSL2 has a weaker isolation story than one full VM per job. Treat the WSL provider as trusted-job infrastructure unless your environment has reviewed and accepted that model. @@ -38,23 +42,11 @@ WSL2 has a weaker isolation story than one full VM per job. Treat the WSL provid `image.customInstallScripts` run as root during image build and their effects are captured in the reusable image. Use them only for non-secret tooling and configuration. Do not bake Docker credentials, GitHub tokens, private keys, or project secrets into runner images. -Certificates configured through `image.trustedCaCertificatePaths` are embedded -in the reusable image and become public trust anchors for every process in its -runner instances. CA certificates are not treated as secrets. Add only CA roots -or intermediates that your organization has explicitly authorized, and rebuild -the image when they are rotated or revoked. +Certificates configured through `image.trustedCaCertificatePaths` are embedded in the reusable image and become public trust anchors for every process in its runner instances. CA certificates are not treated as secrets. Add only CA roots or intermediates that your organization has explicitly authorized, and rebuild the image when they are rotated or revoked. -`image.hostTrustMode: overlay` is a broader policy choice: after the operator -enables it, EPAR follows every root anchor in the configured host scopes, -including later additions, removals, and rotations. Windows and macOS user scope -can include roots installed by software running as that account. Enable it only -when the host trust administrators are also authorized to control runner trust. +`image.hostTrustMode: overlay` is a broader policy choice: after the operator enables it, EPAR follows every root anchor in the configured host scopes, including later additions, removals, and rotations. Windows and macOS user scope can include roots installed by software running as that account. Enable it only when the host trust administrators are also authorized to control runner trust. -Host trust inheritance is additive to Ubuntu's default roots and explicit CA -paths. It does not emulate every Windows or macOS certificate-policy constraint, -and removing a host root cannot revoke an identical Ubuntu-bundled or explicitly -configured anchor. EPAR applies host changes through immutable runner generations: -running jobs keep their starting trust, while stale idle runners are replaced. +Host trust inheritance is additive to Ubuntu's default roots and explicit CA paths. It does not emulate every Windows or macOS certificate-policy constraint, and removing a host root cannot revoke an identical Ubuntu-bundled or explicitly configured anchor. EPAR applies host changes through immutable runner generations: running jobs keep their starting trust, while stale idle runners are replaced. The GitHub App private key remains on the host. Guest instances receive only short-lived registration tokens at runtime. Do not bake tokens or private keys into runner images. @@ -64,4 +56,4 @@ Docker registry mirrors are optional infrastructure outside EPAR. Treat them as Do not assume a mirror makes private image pulls safe or anonymous. A private image still needs authorization from the workflow's `docker login` or from credentials configured on the mirror itself. If the mirror is configured with upstream registry credentials, secure the mirror because it may be able to serve private images that credential can access. -Host-side Docker login state is not copied into EPAR instances. Keep Docker Hub, cloud registry, and package registry credentials in GitHub secrets or in a deliberately secured mirror service. +Host-side Docker login state is not copied into EPAR instances. Keep Docker Hub, cloud registry, and package registry credentials in GitHub secrets or in a deliberately secured mirror service. Docker Sandboxes has a host forward proxy that can replace a guest's Docker Hub authorization with the host `sbx login` identity without copying that credential into the guest. EPAR configures its private Docker daemon and Actions listener to use Docker Sandboxes' policy-enforced transparent egress path by default, where credential injection is unavailable. This preserves ordinary per-job registry authentication but is not a hard boundary against a root-capable workflow deliberately reconnecting to the forward proxy; use a least-privilege `sbx` account and choose Docker Container when that residual host credential capability is outside the trust boundary. See the [Docker Sandboxes provider guide](providers/docker-sandboxes.md#docker-hub-credentials-and-transparent-egress). diff --git a/docs/storage.md b/docs/storage.md new file mode 100644 index 0000000..987963f --- /dev/null +++ b/docs/storage.md @@ -0,0 +1,44 @@ +# Storage + +EPAR checks the storage surfaces required by the selected provider before bootstrap, reusable-artifact work, instance creation, and replacement. An operation starts only when its estimated temporary expansion leaves at least `storage.minimumFree` available. + +```yaml +storage: + minimumFree: 1GiB + gracePeriod: 168h + keepPrevious: 0 + automaticHousekeeping: conservative + buildCacheLimit: 20GiB + goCacheLimit: 10GiB +``` + +`storage status` reports capacity, exact ownership, references, live blockers, cleanup-pending work, and reclaimable estimates. `storage prune` is always a preview unless `--execute` is supplied. + +```text +ephemeral-action-runner storage status +ephemeral-action-runner storage status --json +ephemeral-action-runner storage prune +ephemeral-action-runner storage prune --execute +ephemeral-action-runner storage prune --legacy +ephemeral-action-runner storage prune --legacy --execute --plan +``` + +With `automaticHousekeeping: conservative`, EPAR reconciles interrupted work at startup and after successful artifact activation. It immediately retires an unreferenced, superseded resource only when its catalog receipt and live readback prove that EPAR created or introduced it. The grace period applies to abandoned or incomplete temporary work, not to a successfully replaced generation. A resource referenced by another configuration, lease, container, sandbox, distribution, or builder remains protected. + +Setting `automaticHousekeeping: disabled` suppresses controller-managed artifact cleanup. The no-Go wrapper still maintains one correct stable native binary and removes only inactive legacy revision directories with valid ownership metadata. + +The exact host resource catalog lives in the platform's per-user state directory and coordinates all EPAR project directories on that account. References use canonical configuration paths, so separate configs may keep different generations and exact shared artifacts remain protected until the last reference disappears. + +Older prefix-era resources are never adopted automatically. Use `storage prune --legacy` to produce their exact preview; execution requires the preview plan hash. The Docker Sandboxes base template `docker/sandbox-templates:shell-docker` is always protected. + +EPAR image builds use a config-scoped Buildx builder with persisted ownership metadata, an exact registry/trust configuration digest, and BuildKit garbage collection capped by `storage.buildCacheLimit`. Its running BuildKit control container is intentional reusable build infrastructure, not a runner. EPAR prunes only that exact builder; it never selects, modifies, or prunes Docker's shared/default builder. Metadata, BuildKit configuration, and active CA material live under `.local/storage/buildx/`, `.local/storage/buildkit/`, and `.local/storage/buildkit-certs//`. Project-scoped builders recorded by metadata schema 1–3 are retained, reported by storage inventory, and never silently adopted by a config-scoped controller; remove them only through explicit storage cleanup after confirming no older EPAR controller uses them. The no-Go controller similarly uses project-scoped Go module and build-cache volumes, bounded by `storage.goCacheLimit`; shared or explicitly overridden caches are report-only. + +Docker Sandboxes builds directly to one transient archive, so no Docker staging image is expected. After the archive is imported and the exact Sandbox cache identity is read back, EPAR removes the archive workspace while retaining the active imported template and compact receipt evidence. A completely verified interrupted archive can resume at import; partial archives remain inactive and are reclaimed as owned temporary work. + +Physical host growth and logical virtual-disk limits are reported separately. A 300 GiB VHDX or `Docker.raw` maximum/apparent length does not mean 300 GiB of live Docker content or 300 GiB of new host space is required: Docker Desktop, WSL, and Docker Sandboxes use dynamically allocated or sparse backing storage. `docker system df` reports Docker-managed usage, not host free capacity. EPAR probes the physical filesystem that contains the backing file when it is measurable and treats an unexposed Docker Desktop internal-free value as advisory. + +Storage admission remains fail-closed during normal artifact provisioning, creation, and replacement. To accept the storage risk for one invocation while retaining every non-storage safety check, pass `--allow-insufficient-storage` to `start`, `pool up`, `pool verify`, `image update`, `image build`, or `image update-upstream`. + +Unknown, shared, prefix-only, custom-path, or identity-drifted resources are report-only. EPAR never turns storage cleanup into a broad Docker prune, Docker Sandboxes reset, Docker Desktop reset, WSL reset, or VHDX compaction. + +Deleting Docker data does not necessarily reduce Docker Desktop's VHDX file. VHDX compaction is separate offline host maintenance and is never performed by EPAR. diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 993f380..17b3dc0 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -1,57 +1,43 @@ # Troubleshooting -This page is organized by symptom, host OS, and EPAR provider. Start with the sections that match the machine and provider you are using. - -## Quick Diagnostics - -Start with the commands that match your host and provider. - -### All Hosts - -EPAR logs are written under `work/logs` by default. The latest top-level error report is usually: - -```text -work/logs/epar-last-error.log -``` - -Image build logs use provider-specific names, for example: - -```text -work/logs/builds/epar-docker-dind-catthehacker-ubuntu.docker-build.log -work/logs/builds/epar-wsl-catthehacker-ubuntu.wsl-build.log -``` - -Check the EPAR version and selected config: +Start with the symptom that most closely matches the failure. Regardless of provider, keep TLS verification enabled, preserve the first relevant log, and do not use broad Docker/WSL resets or prune commands as a first response. + +## Contents + +- [Quick diagnostics](#quick-diagnostics) +- [Windows no-Go startup prints an HTTP/2 named-pipe diagnostic](#windows-no-go-startup-prints-an-http2-named-pipe-diagnostic) +- [A Docker workload fails with an architecture error](#a-docker-workload-fails-with-an-architecture-error) +- [Docker Sandboxes is unavailable or its preflight fails](#docker-sandboxes-is-unavailable-or-its-preflight-fails) +- [Docker Sandboxes rejects template, policy, or capacity](#docker-sandboxes-rejects-template-policy-or-capacity) +- [Docker Sandboxes creation fails after a runtime-helper prompt](#docker-sandboxes-creation-fails-after-a-runtime-helper-prompt) +- [Docker Sandboxes rejects a staging workspace because SSH-agent forwarding is present](#docker-sandboxes-rejects-a-staging-workspace-because-ssh-agent-forwarding-is-present) +- [Docker Hub login succeeds but a private pull is denied in Docker Sandboxes](#docker-hub-login-succeeds-but-a-private-pull-is-denied-in-docker-sandboxes) +- [An idle runner reports GitHub or Sandbox health warnings](#an-idle-runner-reports-github-or-sandbox-health-warnings) +- [A scheduled image check or update fails](#a-scheduled-image-check-or-update-fails) +- [A runner is held for diagnostics or an acknowledgement](#a-runner-is-held-for-diagnostics-or-an-acknowledgement) +- [Docker image build runs out of space](#docker-image-build-runs-out-of-space) +- [Storage keeps growing after updates](#storage-keeps-growing-after-updates) +- [Docker image build fails with TLS certificate errors](#docker-image-build-fails-with-tls-certificate-errors) +- [Windows Docker Desktop WSL2 disk is smaller than expected](#windows-docker-desktop-wsl2-disk-is-smaller-than-expected) +- [Docker Container startup fails](#docker-container-startup-fails) +- [WSL provider image build fails early](#wsl-provider-image-build-fails-early) +- [GitHub runner registration fails](#github-runner-registration-fails) + +## Quick diagnostics + +EPAR writes logs under `work/logs` by default. Start with `work/logs/epar-last-error.log`, then inspect the matching build log in `work/logs/builds/` or instance transcript in `work/logs/instances/`. Manager events are console-only by default; raw transcripts are file-only unless `logging.transcriptSinks` includes `console`. + +Long Buildx operations show a bounded console summary with downloaded bytes, completed layers, the active BuildKit step, elapsed time, and growing direct-archive bytes when an export is in progress. The complete raw progress remains in the printed build-log path. ```bash ./start --help go run ./cmd/ephemeral-action-runner version -``` - -If you are running without local Go, use `./start --help`; the wrapper will run EPAR through the containerized Go toolchain. - -### Docker-Backed Workflows - -Use these on any host when the provider is Docker-DinD, when WSL image preparation starts from a Docker image, or when the no-Go wrapper is in use: - -```bash docker version docker info docker system df -docker image ls ``` -To see the free space available to containers on the active Docker daemon: - -```bash -docker run --rm ghcr.io/catthehacker/ubuntu:full-latest df -h / -``` - -For a custom source image, replace `ghcr.io/catthehacker/ubuntu:full-latest` with the value from `image.sourceImage`. - -### Windows Hosts - -For Windows hosts that use WSL2, Docker Desktop's WSL2 backend, or the WSL provider: +Without local Go, use `./start --help`; the wrapper selects the containerized toolchain. For Windows WSL2, a WSL-backed Docker daemon, or the WSL provider, also run: ```powershell wsl --version @@ -60,320 +46,283 @@ docker context ls docker run --rm ghcr.io/catthehacker/ubuntu:full-latest df -h / ``` -`ghcr.io/catthehacker/ubuntu:full-latest` is the default source image for EPAR's default config. If your config uses a custom source image, replace it with your configured `image.sourceImage`. - -Docker Desktop's WSL2 backend stores Docker data in a WSL-backed virtual disk. Windows Explorer free space and container-visible free space are related, but they are not the same number. - -### Linux Docker Engine Hosts - -On native Linux, Docker data usually lives under Docker's root directory. Check it directly: +Container-visible free space is the relevant value for Docker builds. Windows Explorer or Finder free space does not necessarily equal the free space in a Linux VM backing the daemon. -```bash -docker info --format '{{.DockerRootDir}}' -df -h "$(docker info --format '{{.DockerRootDir}}')" -``` +## Windows no-Go startup prints an HTTP/2 named-pipe diagnostic -### macOS Docker Hosts +### Symptom -Docker Desktop and OrbStack keep Linux container data inside their own VM/storage area. Use Docker's own view first: +The Windows no-Go bootstrap prints a line like: -```bash -docker system df -docker run --rm ghcr.io/catthehacker/ubuntu:full-latest df -h / +```text +http2: server: error reading preface from client //./pipe/dockerDesktopLinuxEngine: file has already been closed ``` -If the container-visible disk is full, adjust or clean the Docker/OrbStack storage from that product's settings. Finder free space by itself may not reflect the Linux VM storage available to containers. +### Diagnosis and remediation -## Docker Container Fails Because Its Architecture Does Not Match The Runner +Docker Desktop can emit this named-pipe transport diagnostic when a client connection closes. If the bootstrap `docker build` succeeds and the wizard or command continues, it is not an EPAR runner-group or GitHub API failure. The wrapper suppresses only this exact successful-build diagnostic and keeps full Docker stderr when the build fails. -### Symptoms +If startup stops, verify the selected engine and context: -A Docker or Docker Compose service may fail immediately with one of these messages: - -```text -exec /bin/sh: exec format error -exec user process caused: exec format error -cannot execute binary file: Exec format error +```powershell +docker version +docker info +docker context show ``` -Docker may also warn that the requested image platform does not match the detected host platform. That warning alone is not a failure: the container can still run when a compatible emulation handler is already registered. +Resolve a failed command or unhealthy engine; do not treat every named-pipe line as harmless when Docker returned a nonzero exit code. -These related messages point to different problems: +## A Docker workload fails with an architecture error + +### Symptom -- `no matching manifest for linux/arm64/v8` or `no matching manifest for linux/amd64` means the image does not publish the requested platform. QEMU cannot supply a missing image manifest; choose an available platform or publish a multi-platform image. -- `qemu-x86_64: Could not open '/lib64/ld-linux-x86-64.so.2'` means translation started but the expected foreign-architecture loader or userspace is unavailable or incompatible. Registration alone may not make that image work. -- Exit code `139` indicates a segmentation fault. Emulation can expose workload-specific incompatibilities, but this code by itself does not prove an architecture mismatch. +Docker or Compose exits with `exec format error`, `cannot execute binary file`, a platform-mismatch warning, `no matching manifest`, a QEMU loader error, or exit code `139`. -### Confirm The Host And Image Platforms +### Diagnosis and remediation -Check the runner architecture and the Docker daemon that will execute the container: +Inspect the runner, daemon, image manifest, and Compose platform setting: ```bash uname -m docker info --format '{{.OSType}}/{{.Architecture}}' +docker image inspect --format '{{.Os}}/{{.Architecture}}' IMAGE +docker buildx imagetools inspect IMAGE +docker compose config ``` -Inspect the locally selected image platform: +`no matching manifest` means the image does not publish the requested platform; emulation cannot create a missing manifest. A platform warning alone does not prove failure, and exit code `139` alone does not prove an architecture mismatch. Use the exact image and workload evidence. For the architecture model, QEMU setup, provider scope, and verification commands, see [Cross-architecture containers](advanced/cross-architecture-containers.md). -```bash -docker image inspect --format '{{.Os}}/{{.Architecture}}' IMAGE -``` +## Docker Sandboxes is unavailable or its preflight fails -Inspect all platforms published by a registry image: +### Symptom -```bash -docker buildx imagetools inspect IMAGE -``` +The wizard marks Docker Sandboxes unavailable, or `sbx diagnose --output json` reports failures. + +### Diagnosis and remediation -For Docker Compose, also inspect the resolved configuration and look for a service-level `platform:` value: +Check the diagnostic result before editing configuration: ```bash -docker compose config +sbx diagnose --output json ``` -An x64 Linux Docker daemon normally runs `linux/amd64` images natively, and an ARM64 daemon normally runs `linux/arm64` images natively. Pulling or loading a foreign image does not prove that the daemon can execute it. - -### Match GitHub-Hosted Linux Behavior With Explicit QEMU Setup +EPAR requires a controller architecture with an available Linux guest template and at least one diagnostic pass with zero failures. Diagnostic warnings and skipped checks remain visible but do not disable the provider. Review the failed item and its hint in the JSON output, fix the prerequisite, then choose Refresh in the provider menu to recheck availability; do not manually force a provider selection or substitute Docker Container for a configured Docker Sandboxes pool. -GitHub's Ubuntu runner image installs Docker, but its published installation script and software inventory do not promise pre-registered foreign-architecture emulators. When a trusted Linux job must run foreign-architecture containers, configure the requirement explicitly before the first such container starts: +## Docker Sandboxes rejects template, policy, or capacity -```yaml -jobs: - test: - runs-on: ubuntu-latest - steps: - - name: Set up ARM64 container emulation - uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4 - with: - image: docker.io/tonistiigi/binfmt@sha256:400a4873b838d1b89194d982c45e5fb3cda4593fbfd7e08a02e76b03b21166f0 - platforms: arm64 - - - name: Verify the foreign container - run: docker run --rm --platform linux/arm64 alpine:3.22 uname -m -``` - -The expected output is `aarch64`. Add only the platforms the workflow needs; emulated compilation and compute-heavy workloads can be substantially slower than native execution. Keep architecture-sensitive jobs on native runners when performance or full compatibility matters. +### Symptom -`docker/setup-qemu-action` registers user-mode QEMU interpreters through Linux `binfmt_misc`. It helps Linux containers launch foreign-architecture user-space executables; it does not change the runner's CPU architecture, create a foreign-architecture VM, or make arbitrary host executables and libraries compatible. The action uses a privileged helper container, so use it only in trusted workflows and pin reviewed action and helper-image revisions according to your dependency policy. +Startup reports a template identity/digest mismatch, policy-generation drift, an admission failure, or insufficient capacity. -Provider notes: +### Diagnosis and remediation -- Docker-DinD: run the setup action inside the EPAR job before Docker Compose or other foreign-image commands. It configures the disposable runner's Docker execution environment; no EPAR configuration switch is required. -- WSL: run the setup action inside the WSL runner when its Linux Docker daemon must execute a foreign image. An x64 WSL runner does not gain ARM64 container support merely by pulling or loading an ARM64 image. -- Tart: Tart runs an ARM64 VM on Apple Silicon. Its optional Rosetta path is experimental and is not equivalent to QEMU/binfmt compatibility. Prefer Docker-DinD or a native matching architecture when a workload is not compatible. -- GitHub-hosted Windows and macOS: GitHub documents Docker container actions and service containers as Linux-runner features. A Windows or macOS hardware label alone is therefore not a substitute for a Linux Docker daemon with emulation configured. +Docker Sandboxes resolves the configured source selector, records the exact OCI identities in a local artifact receipt, and verifies the host-global policy fingerprint. If the desired source, platform, scripts, template inputs, runner inputs, or trust inputs change, rerun `./start`; EPAR builds and imports a replacement and activates it only after exact readback succeeds. -Official references: +An imported Docker Sandboxes template does not require a matching Docker image. EPAR builds directly to a verified archive, imports that archive, and then removes the transient workspace. If startup reports a missing Docker staging image, the controller is stale; rebuild the native controller and rerun `./start`. If direct archive verification or `sbx template load` fails, use the printed Buildx transcript and archive error; EPAR does not fall back to the memory-heavy Docker load/save path. -- [Docker Setup QEMU action](https://github.com/docker/setup-qemu-action) -- [Docker multi-platform build strategies](https://docs.docker.com/build/building/multi-platform/) -- [GitHub-hosted runner labels and limitations](https://docs.github.com/en/actions/reference/runners/github-hosted-runners) -- [GitHub self-hosted runner container requirements](https://docs.github.com/en/actions/reference/runners/self-hosted-runners#requirements-for-self-hosted-runner-machines) -- [GitHub Ubuntu runner Docker installation](https://github.com/actions/runner-images/blob/main/images/ubuntu/scripts/build/install-docker.sh) +Capacity admission accounts for estimated incremental physical growth on each measurable backing filesystem plus the fixed `storage.minimumFree` reserve. Docker Sandboxes root and inner-Docker sizes are independent sparse logical maxima and are not added as immediate host usage. Inspect the reported physical surface, run the matching `storage status` and prune-preview commands, or deliberately retry only that invocation with `--allow-insufficient-storage`. Avoid broad cleanup commands: they can delete stopped containers and intentionally retained resources. -## Docker Image Build Runs Out Of Space +## Docker Sandboxes creation fails after a runtime-helper prompt ### Symptom -During `start` or `image build`, the log contains: +On macOS or Linux, host security asks whether to allow a Docker Sandboxes helper such as `mkfs.ext4`, `mkfs.erofs`, or `containerd-shim-nerdbox-v1`; macOS may say that the helper “is an app downloaded from the Internet.” After a required prompt is denied or blocked, EPAR reports `create docker sandbox failed`, and `sbx` may report `500 Internal Server Error: failed to run sandbox container`. The runner is neither registered nor marked ready. -```text -E: You don't have enough free space in /var/cache/apt/archives/. -``` +### Diagnosis and remediation -or another package install fails with `No space left on device`. +Docker Sandboxes uses `mkfs.ext4` to create an ext4 filesystem inside each sandbox's private Docker disk-image file, `mkfs.erofs` to construct the read-only template snapshot, and `containerd-shim-nerdbox-v1` to launch and manage the sandbox VM. Expected file targets are regular sandbox-owned files beneath the Docker Sandboxes runtime data directory—for example, current macOS releases may use `~/.sbx/run/d/containerd/.../images/-docker.img` and `~/.sbx/run/d/containerd/.../snapshots//layer.erofs`. With the current Homebrew `sbx` package, the runtime and shim are beneath `/opt/homebrew/Caskroom/sbx//`. A formatter must not target a physical device such as `/dev/disk*`, an EPAR checkout, a home-directory document, or another unrelated path. -### What It Means +If each executable belongs to the Docker Sandboxes installation you intentionally installed and any displayed target is the expected sandbox-owned file, allow the operation through the host's security or application-control prompt, then rerun the same EPAR start or verification command if creation already failed. Do not invoke a formatter or shim yourself, disable host security broadly, or approve a command with an unfamiliar target. The configured private Docker disk is sparse, so its logical maximum does not mean the formatter immediately consumes that amount of physical storage. -This error is raised inside the temporary container or guest that is building the runner image. It usually means the Docker daemon or VM backing that build is out of writable layer space. It does not necessarily mean the host OS drive has no free space. +If no prompt appeared, or approval still produces the 500 error, preserve the failed runner evidence and inspect the Docker Sandboxes daemon/client logs and `sbx diagnose --output json`; the same top-level error can also represent a runtime, capacity, or host-policy failure. See [Private Filesystem and VM Helper Approval](providers/docker-sandboxes.md#private-filesystem-and-vm-helper-approval) for the provider contract. -Check the active Docker daemon: +## Docker Sandboxes rejects a staging workspace because SSH-agent forwarding is present -```bash -docker run --rm ghcr.io/catthehacker/ubuntu:full-latest df -h / -docker system df -``` +### Symptom -If `/` inside the container is nearly full, clean up Docker data or increase the Docker VM/data-disk limit before retrying. +Sandbox creation reaches `verify dedicated docker sandbox staging workspace` and fails with a message that host SSH-agent forwarding is not permitted. Diagnostics may show `SSH_AUTH_SOCK=/run/ssh-agent.sock` or `SSH_AUTH_SOCK_GATEWAY=...` inside the guest even though the imported template does not define them. -### Cleanup Direction +### Diagnosis and remediation -Review Docker's usage first: +Docker Sandboxes may forward the host SSH agent when its shared daemon inherits the host's agent environment. EPAR rejects the resulting sandbox because the forwarded socket or gateway could let a workflow use host SSH credentials. This is not evidence that the staging mount is missing or read-only, and deleting only `/run/ssh-agent.sock` is insufficient when the forwarding gateway remains configured. -```bash -docker system df -docker system df -v +Coordinate the interruption with every process using the shared Docker Sandboxes daemon, then restart it with all forwarding variables removed and retry EPAR: + +```sh +sbx daemon stop +env -u SSH_AUTH_SOCK -u SSH_AUTH_SOCK_GATEWAY -u SSH_AGENT_PID sbx daemon start --detach ``` -Docker prune commands remove resources that Docker considers unused. Depending on the command, that can include stopped containers, unused images, build cache, unused networks, or unused volumes. Review the Docker command help and the data on the machine before pruning, especially if you expect to restart stopped containers or keep data in Docker volumes. +EPAR strips these variables from Docker Sandboxes commands it launches, but an already-running daemon retains the environment with which another shell or tool started it. Do not disable this admission check or forward an agent into a reusable runner template. If the failed creation predates the immutable-receipt fix, preserve its reported sandbox UUID and use exact provider cleanup; never delete a same-name resource by prefix alone. -## Docker Image Build Fails With SSL Certificate Errors +## Docker Hub login succeeds but a private pull is denied in Docker Sandboxes ### Symptom -During `start`, `image build`, or a CI job, HTTPS access fails with certificate errors such as: +A workflow's Docker login step reports `Login Succeeded`, but a later pull of a private Docker Hub image fails with `insufficient_scope: authorization failed`, `pull access denied`, or an equivalent authorization response. The same workflow and credentials may succeed with Docker Container or a GitHub-hosted runner. Using `docker --config /home/agent/.docker pull ...` produces the same denial. -```text -curl: (60) SSL certificate problem: unable to get local issuer certificate -``` +### Diagnosis and remediation -```text -Certificate verification failed: The certificate is NOT trusted. -The certificate issuer is unknown. Could not handshake: Error in the certificate verification. -``` +First verify only metadata, never credential contents: the listener should run as `agent` with `HOME=/home/agent` and `DOCKER_CONFIG=/home/agent/.docker`, and a post-login config should be owned by `agent` with restrictive permissions. If Docker reaches the registry and returns an authorization response, do not investigate CA copying unless an `x509` or TLS error is also present. -### What It Means +Inspect the host Docker Sandboxes daemon log for a message that the proxy is overriding a client-supplied registry credential with a host credential. When that message is present, the workflow credential was written correctly but dockerd used the credential-injecting forward path. An explicit guest config path cannot bypass that path. -Antivirus, endpoint-security, firewall, or corporate-proxy software may be inspecting HTTPS and re-signing certificates with a root CA trusted by the host but not present in the Ubuntu runner image. +On a current EPAR template, `docker info --format '{{.NoProxy}}'` must print `*`, `/etc/docker/daemon.json` must be a root-owned regular file, and `sbx policy log ` must report `transparent` for `registry-1.docker.io`, `auth.docker.io`, and the blob host used by the pull. Docker Sandboxes documents transparent traffic as policy-enforced without credential injection. If dockerd reports another no-proxy value or the policy log reports `forward`, stop the workspace controller, let its exact cleanup finish, rerun `./start` to build and import the changed template, and test on a newly created runner. Do not reuse the old sandbox. -### How to Fix +Do not set `DOCKER_SANDBOXES_NO_PROXY` expecting it to disable credential injection. That host variable only excludes destinations from an optional upstream proxy used after traffic reaches the mandatory Sandbox proxy. Replacing `docker/login-action` with `docker login`, combining login and pull in one shell step, or changing `DOCKER_CONFIG` also leaves an old daemon's forward route unchanged. -Do not disable certificate verification. Use EPAR's host trust overlay so the disposable Docker-DinD runners automatically inherit the host's trusted root CAs while retaining Ubuntu's standard roots. +Keep the host `sbx login` identity intentionally different from the workflow identity when proving this fix. A successful private pull together with transparent policy-log entries proves that the guest credential is authoritative. Changing the host login to match the workflow can diagnose the old interception behavior, but it is a shared-identity workaround rather than the fix. -New interactive Docker-DinD configurations enable the overlay by default. For an older Windows or macOS configuration, add: +EPAR rejects global `sbx` secrets and removes inherited proxy variables from runner registration and the Actions listener. A root-capable workflow can still deliberately reconnect a client to Docker Sandboxes' forward proxy, and v0.37.1 has no documented per-sandbox switch that disables the interceptor. Use a least-privilege host `sbx` account and choose Docker Container if that residual capability is outside the trust boundary. See [Docker Hub Credentials and Transparent Egress](providers/docker-sandboxes.md#docker-hub-credentials-and-transparent-egress). -```yaml -image: - hostTrustMode: overlay - hostTrustScopes: [system, user] -``` +## An idle runner reports GitHub or Sandbox health warnings -Linux supports only the system scope, so use `hostTrustScopes: [system]` instead. Rebuild the image after changing the configuration: +A GitHub 429/5xx response or an `sbx` command timeout makes runner health temporarily unknown; it does not prove that the Actions listener stopped. EPAR keeps the exact runner, lets a trust lease expire closed when it cannot refresh it, and retries. Cleanup for an inactive listener requires two consecutive guest probes that successfully execute and explicitly report the process stopped. Review the instance guest transcript when warnings repeat; do not delete the runner merely because one API or Sandbox inspection failed. -```powershell -go run ./cmd/ephemeral-action-runner image build --replace -``` +`networkBaseline: open` is a sandbox-scoped public-egress compatibility rule with EPAR host-alias deny guardrails. It does not alter the host-global policy. If a required service is blocked, use a narrow `additionalAllow` hostname rule; do not allow `host.docker.internal`, `gateway.docker.internal`, `kubernetes.docker.internal`, or `host.containers.internal` through the Open-policy guardrails. -Without Go installed, run `scripts\run-with-docker.ps1 image build --replace` on Windows or `scripts/run-with-docker.sh image build --replace` on macOS or Linux. Use the official wrapper because it collects trust from the real host rather than the temporary Linux toolchain container. +## A scheduled image check or update fails -EPAR uses the resulting consolidated Ubuntu trust bundle during both image construction and CI jobs. Node.js, Python Requests, and pip receive compatible runtime defaults without overwriting values already supplied by the source image or runner environment. +Run `./start status` to see the last successful remote check, next check or retry, pending immutable identity, deferred reason, and last error. A failed scheduled check or build keeps the previous exactly verified generation available and retries with bounded backoff; a missing artifact or changed local configuration still fails closed. Use `./start image update` to retry an immediate remote check, or correct local input errors and rerun `./start`. -Programs with private certificate stores can still require application-specific configuration. Java keystores remain a separate concern. +## A runner is held for diagnostics or an acknowledgement -### Advanced: Add a Certificate Explicitly +### Symptom -Use `image.trustedCaCertificatePaths` when a required CA is not trusted by the selected host stores or when the configuration must pin a specific CA independently of host trust. +An instance is retained, quarantined, or shown as requiring an acknowledgement after a provisioning, policy, or runtime failure. -Export the CA as PEM, Base-64 encoded X.509 `.CER`, or DER-encoded `.CER`, place it in the repository, and add it to the configuration: +### Diagnosis and remediation -```yaml -image: - hostTrustMode: overlay - hostTrustScopes: [system, user] - trustedCaCertificatePaths: - - .local/private-root.cer -``` +Preserve the instance and inspect `work/logs/instances/.guest.log`, the matching runner diagnostics, and controller output before acknowledging or removing it. EPAR deliberately keeps uncertain ownership, failed cleanup, and unverified remote state inside the strict `pool.instances` cap instead of creating a replacement storm. -Explicit certificates are validated and added to the same Ubuntu trust bundle; they are combined with, not substituted for, the host trust overlay. +If an incident requires stopping new work immediately, stop the controller with `Ctrl-C` or the service manager that launched it. This prevents replacement; it does not erase retained evidence. Use the configured EPAR cleanup command only after identifying the exact affected pool. Do not use a broad `docker system prune`, WSL unregister, or reset as an incident-disable switch. -### Host Trust Overlay Is Missing, Stale, Or Mismatched +Set `EPAR_DISABLE_DOCKER_SANDBOXES=1` before starting EPAR when Docker Sandboxes admission must remain disabled during an incident or compatibility investigation. This fails the provider closed without changing configuration or deleting evidence. -This section applies when a Docker-DinD config contains: +After reviewing retained Docker Sandboxes diagnostics, acknowledge that review only for the exact configured pool: -```yaml -image: - hostTrustMode: overlay +```bash +ephemeral-action-runner cleanup --acknowledge-failed-diagnostics ``` -EPAR fails closed when host collection returns no roots, the official no-Go bridge is missing, its feed is invalid or more than 30 seconds old, or a runner's 20-second lease does not match the image's trust generation. A pre-job mismatch can fail an already assigned GitHub job before repository steps run. Inspect the controller output, runner guest log, and the no-Go watcher's log in the host trust cache for the first collection, feed, or lease error. +## Docker image build runs out of space -Check these boundaries: +### Symptom -- Windows and macOS support `hostTrustScopes: [system, user]`; Linux supports `[system]` only. -- Overlay mode requires `provider.type: docker-dind` and `runner.ephemeral: true`. -- Use the official `./start`, `start.ps1`, or release launcher for the no-Go path. A bare Linux toolchain container cannot inspect Windows Certificate Stores or macOS Keychain and must not substitute its own CA bundle. -- On an uncommon Linux distribution, set `EPAR_HOST_TRUST_BUNDLE` to the distribution-generated PEM CA bundle before launching EPAR. -- Confirm host and guest clocks are correct; feed and lease expiry checks use timestamps and reject stale data. +`start` or `image build` reports `No space left on device` or `E: You don't have enough free space in /var/cache/apt/archives/.`. -Do not disable the pre-job gate or TLS verification. Restore host collection, then let EPAR build and register the current immutable trust generation. +### Diagnosis and remediation -Host Docker daemon trust is separate from runner trust. If `docker pull` of the source image fails, configure the authorized CA for Docker Desktop, OrbStack, or the host Docker Engine first; the Ubuntu overlay does not exist until after that pull succeeds. +The temporary guest or Docker writable layer is full; this does not necessarily mean the host OS drive is full. Inspect the active Docker daemon: -## Windows Docker Desktop WSL2 Disk Is Smaller Than Expected +```bash +docker run --rm ghcr.io/catthehacker/ubuntu:full-latest df -h / +docker system df +docker system df -v +``` + +Increase the relevant Docker/VM data-disk allocation or deliberately remove unneeded data after reviewing it. Docker prune commands can remove stopped containers, unused images, build cache, networks, and volumes; they are not a safe generic fix. + +## Storage keeps growing after updates ### Symptom -On a Windows machine where WSL2 storage was set up before 2021, the Docker container filesystem may report about 251 GB total: +Old EPAR images, Docker Sandboxes templates, staging archives, or no-Go controller files remain after an update or an interrupted start. -```powershell -docker run --rm ghcr.io/catthehacker/ubuntu:full-latest df -h / -``` +### Diagnosis and remediation -Example: +Run `./start` once more. Startup reconciles incomplete exact-owned work and retires an unreferenced superseded generation after its replacement passes readback. It does not delete shared images, prefix-only historical resources, active containers, active sandboxes, or resources referenced by another configuration. -```text -Filesystem Size Used Avail Use% Mounted on -overlay 251G 211G 28G 89% / +Inspect the exact classification before removing anything manually: + +```bash +./start storage status +./start storage prune +./start storage prune --legacy ``` -On a Windows machine where WSL2 storage was set up after the 2022 WSL default-size change, the same command may report about 1007 GB total: +Use normal `storage prune --execute` only for exact catalog-owned resources. Legacy prefix-era entries require the plan hash printed by `storage prune --legacy`; they are not removed automatically. Do not use broad Docker prune/reset commands or VHDX compaction as a substitute for this review. -```text -Filesystem Size Used Avail Use% Mounted on -overlay 1007G 127G 830G 14% / -``` +## Docker image build fails with TLS certificate errors -### Why It Happens +### Symptom -This is a Windows Docker Desktop / WSL2 storage detail, not an EPAR image-size issue. The command reports the size of Docker Desktop's Linux container storage, not the size of `ghcr.io/catthehacker/ubuntu:full-latest`. +HTTPS access fails with `curl: (60)`, `certificate verification failed`, or an unknown issuer during an EPAR build or job. -For Windows machines where WSL2 storage was set up before 2021, the default WSL2 virtual disk maximum may be about 256 GB. For WSL2 setups created after the WSL 0.58.0 change, released in 2022, Microsoft's documentation says the default maximum for each WSL2 VHD is 1 TB. This can explain why one Windows machine reports about 251 GB while another reports about 1007 GB for the container-visible filesystem. +### Diagnosis and remediation -Windows Explorer free space by itself is not enough to confirm Docker has build space. The container-visible filesystem must have enough free space for the image pull, build layers, package manager cache, and final runner image. +Do not disable certificate verification. First identify which trust boundary failed. -For more background, see: +For a no-Go native-controller build, `./start` automatically reads host system roots, excludes explicitly distrusted certificates, validates the short-lived feed in an offline container, and mounts only the resulting CA bundle into the Go compiler container. Runner CA inheritance remains independent. If the build still reports an unknown issuer, inspect `work/logs/epar-native-controller-build.log`: the wrapper prints the requested host, presented certificate subject and issuer, SHA-256 fingerprint, validity, and verification result, and on Windows it lists matching roots from `LocalMachine\Root` and `CurrentUser\Root`. A remaining failure means the expected issuer was absent, distrusted, malformed, expired, or not the certificate actually presented; EPAR never disables TLS verification or retries insecurely. -- -- -- +For an EPAR Buildx failure, leave `image.hostTrustMode` unchanged. EPAR automatically supplies host system roots to its config-owned builder and prints the full build transcript path before `docker buildx build`. The console and error report include a bounded redacted tail. Inspect the underlying `x509` line together with the registry host, builder identity, and active trust generation: -## Docker-DinD Build Fails With `unknown flag: --progress` +```powershell +docker buildx ls +Get-ChildItem .local/storage/buildx -Recurse -Filter metadata.json | Get-Content +Get-ChildItem .local/storage/buildkit -Recurse -Filter buildkitd.toml | Get-Content +``` -### Symptom +The owned metadata records the exact registry set, configuration digest, certificate bundle, and trust generation. Rerunning the same command reconciles that exact builder and preserves its BuildKit state; EPAR never changes Docker's shared/default builder. If the source-image `docker pull` itself fails before Buildx starts, configure the authorized CA in the host daemon because builder trust cannot repair host-daemon trust. -The Docker-DinD image build fails with: +Configure runner overlay only when jobs inside an ephemeral runner must inherit host roots: -```text -unknown flag: --progress +```yaml +image: + hostTrustMode: overlay + hostTrustScopes: [system, user] ``` -### What It Means +Use `[system]` on Linux. Overlay mode collects the current host roots, validates them before registration, and combines them with Ubuntu roots and any `image.trustedCaCertificatePaths`; it is root-anchor inheritance rather than exact Windows/macOS TLS-policy emulation. It requires `runner.ephemeral: true`. Omitted or disabled mode remains valid for Docker Sandboxes and does not install the job-start trust hook. -This happens when the Docker client used for the build routes `docker build` through the legacy builder, or when the client does not have Buildx-style build support. It is most visible when EPAR is run through a containerized Go toolchain whose bundled Docker client differs from the host `docker.exe`. +Use the normal host entry point so EPAR can inspect the real Windows certificate stores or macOS Keychain: -Current EPAR builds use legacy-builder-compatible Docker build arguments. If you still see this error, confirm you are running a revision that includes that fix and check which Docker client is actually executing the command: - -```bash -docker version -docker build --help -docker buildx version +```powershell +./start +go run ./cmd/ephemeral-action-runner image build --replace ``` -## Docker-DinD Startup Fails +On no-Go Windows, use `scripts\run-with-docker.ps1 image build --replace`; on macOS/Linux, use `scripts/run-with-docker.sh image build --replace`. The wrapper uses a native-host trust feed while compiling the native controller; the resulting native controller reads host trust directly for `start`, `image build`, `pool up`, and `pool verify`, even when runner overlay is disabled. The legacy containerized controller still requires the separate native-host feed bridge. A bare Linux toolchain container is not a replacement for either path. + +## Windows Docker Desktop WSL2 disk is smaller than expected + +### Symptom + +`docker run --rm ghcr.io/catthehacker/ubuntu:full-latest df -h /` shows much less capacity than Windows Explorer. + +### Diagnosis and remediation + +Docker Desktop stores Linux container data in a WSL-backed virtual disk. Older WSL2 installations can have a smaller default VHD maximum than newer ones, but the reported container filesystem is the evidence that matters for image pulls and builds. Inspect Docker usage first, then change Docker Desktop/WSL storage using the product's supported settings. See [Microsoft WSL disk-space guidance](https://learn.microsoft.com/windows/wsl/disk-space) and [Docker Desktop WSL guidance](https://docs.docker.com/desktop/features/wsl/). -### Privileged Containers +## Docker Container startup fails -Docker-DinD requires the host Docker runtime to allow privileged Linux containers. Confirm the Docker host supports: +### Privileged containers + +Docker Container requires a host Docker runtime that permits privileged Linux containers: ```bash docker run --rm --privileged alpine:3.20 true ``` -### Nested Docker Storage Driver +### Nested Docker storage driver -If Docker-DinD starts but nested Docker operations fail with overlay mount errors, keep the default inner daemon storage driver: +If nested Docker operations fail with overlay-mount errors, retain the default inner storage driver: ```text EPAR_DOCKERD_STORAGE_DRIVER=vfs ``` -Use `overlay2` or `auto` in a derived image only after proving that storage driver works on the exact host runtime. +Use `overlay2` or `auto` only in a derived image after proving it works on the exact host runtime. + +## WSL provider image build fails early -## WSL Provider Image Build Fails Early +### Symptom -This section applies to Windows hosts using `provider.type: wsl`. +The WSL image build fails before import, during import with `0xffffffff`, or before systemd is ready. -If the default WSL image build fails before importing or starting the temporary distro, confirm Docker is reachable because the default WSL full image converts a Docker image into a rootfs tar: +### Diagnosis and remediation + +The default WSL build obtains a Docker source image before importing it into WSL. Verify Docker and WSL first: ```powershell docker version @@ -381,58 +330,19 @@ docker pull ghcr.io/catthehacker/ubuntu:full-latest wsl -l -v ``` -### WSL Import Exits With `0xffffffff` - -If the Docker export completes but the first temporary-distro import fails like this: - -```text -wsl.exe --import ... --version 2 failed: exit status 0xffffffff: -``` - -WSL may be in an unstable service or VM session. The error can also appear as -`Wsl/Service/CreateInstance/E_UNEXPECTED` or `Catastrophic failure` when -starting an existing distro. When the import itself fails, the advertised WSL -build and guest logs may be empty or absent because no guest was created yet. +For `Wsl/Service/CreateInstance/E_UNEXPECTED`, `Catastrophic failure`, or import exit `0xffffffff`, stop EPAR, save work in other distros, then run `wsl --shutdown`. This stops every running WSL distro, including any Docker backend using WSL. Restart the affected Docker host runtime, verify a normal distro command returns `0`, then rerun `./start`; a matching cached source rootfs is reused. If it persists, update WSL, shut it down again, reboot, and consult [Microsoft's WSL troubleshooting guidance](https://learn.microsoft.com/windows/wsl/troubleshooting#error-code-0x8000ffff-unexpected-failure). -Reset the WSL session before deleting or rebuilding a completed source rootfs: +If a guest exists but systemd does not become ready, inspect `work/logs/builds/.wsl-build.log` and `work/logs/builds/.guest.log`. Do not unregister a distro until you have identified the exact EPAR-owned target and accepted that unregistration is irreversible. -1. Stop EPAR and quit Docker Desktop cleanly from its tray menu. -2. Run: +## GitHub runner registration fails - ```powershell - wsl --shutdown - ``` - -3. Start Docker Desktop again and wait until it is ready. -4. Verify WSL and Docker. Replace `Ubuntu-24.04` if your installed distro has a - different name: - - ```powershell - wsl -d Ubuntu-24.04 --user root --exec /bin/true - $LASTEXITCODE - docker version - ``` - -5. When the WSL command returns `0` and Docker is ready, rerun `./start` or - `.\start`. EPAR reuses a matching cached - `work/images/*.source.rootfs.tar`, avoiding another large Docker export. - -`wsl --shutdown` stops every running WSL distro, including Docker Desktop's WSL -backend. Save work in other distros first. If the failure persists after the -reset, run `wsl --update`, shut WSL down again, reboot Windows, and retry once. -For persistent `0x8000FFFF` or `E_UNEXPECTED` failures, follow -[Microsoft's WSL troubleshooting guidance](https://learn.microsoft.com/windows/wsl/troubleshooting#error-code-0x8000ffff-unexpected-failure). - -If the WSL image build fails after import but before systemd is ready, inspect: +### Symptom -```text -work/logs/builds/.wsl-build.log -work/logs/builds/.guest.log -``` +EPAR cannot request a registration token, add a runner to a group, or observe the runner online. -## GitHub Runner Registration Fails +### Diagnosis and remediation -Confirm the GitHub App has organization self-hosted runner read/write permission and that the private key path in the config is readable from the EPAR process: +Verify GitHub App organization self-hosted-runner read/write permission and a readable private key: ```yaml github: @@ -441,10 +351,12 @@ github: privateKeyPath: .local/github-app.pem ``` -If stale runner records remain after an interrupted run: +Then inspect runner-group policy and the first registration error. A strict policy can intentionally block a group that is default, overly broad, or public-repository enabled. See [Runner Group Security](runner-groups.md). + +For a confirmed stale EPAR resource, run the configured cleanup command: ```bash go run ./cmd/ephemeral-action-runner cleanup ``` -Cleanup only targets runner names matching `pool.namePrefix`, so keep that prefix unique per machine/config within the GitHub organization. +Cleanup is bounded by the configured pool and durable exact lifecycle identities; it does not authorize a broad prefix deletion, wildcard, Docker prune, or removal of unknown/shared resources. Keep `pool.namePrefix` unique per controller and organization. diff --git a/docs/usage.md b/docs/usage.md index 15df358..95577a6 100644 --- a/docs/usage.md +++ b/docs/usage.md @@ -1,357 +1,151 @@ # Usage -This page is the operational walkthrough. Start with the supported host you already have: - -- Docker-DinD on a Docker-capable host -- WSL2 on Windows -- Tart on Apple Silicon macOS +Use this page for normal EPAR tasks. Start with the host and provider you already have; the [documentation hub](README.md) links to the provider-specific guides. ## Prerequisites -Install the host tools you need: - -| Required for | Tool | +| Task | Required tool or access | | --- | --- | -| Source archive quick start | Go 1.25 or newer, or Docker (see [no-Go-install](advanced/no-go-install.md)) | -| Updating the pinned `actions/runner-images` checkout | Git | -| macOS provider | Tart | -| Windows provider | WSL2 | -| Windows WSL2 default image build | Docker Desktop, Docker Engine, or another working Docker daemon for the one-time Docker image export | -| Docker-DinD provider | Docker Engine, OrbStack, or Docker Desktop with privileged container support | -| Optional Docker registry mirrors | A running mirror service on the host, LAN, intranet, or cloud registry cache | -| Runner registration | GitHub App with organization self-hosted runner read/write permission | - -Packer, GitHub CLI, and sshpass are not required. - -Set up the GitHub App before registering runners. The image build command can run without GitHub credentials, but `pool verify --register-only`, `pool up`, `status`, and GitHub cleanup need the app settings. See [GitHub App Setup](github-app.md). - -## Get The Source - -For normal use, open the [EPAR Releases page](https://github.com/solutionforest/ephemeral-action-runner/releases), select the release you want, and download GitHub's automatically generated **Source code (zip)** or **Source code (tar.gz)**. Extract the source archive and open a terminal in the extracted folder: - -```bash -cd path/to/ephemeral-action-runner- -go run ./cmd/ephemeral-action-runner version -``` +| Run a source archive | Go 1.25 or newer, or Docker for the no-Go controller builder | +| Register or inspect GitHub runners | A GitHub App with organization self-hosted runner read/write permission | +| Docker Container | Docker with privileged Linux-container support | +| Docker Sandboxes | Docker and the `sbx` CLI with at least one diagnostic pass and zero failures; the wizard builds and imports the selected EPAR template | +| WSL | Native Windows, WSL2, and Docker when preparing the default WSL image | +| Tart | Native Apple Silicon macOS and Tart | -The examples below use `go run ./cmd/ephemeral-action-runner` for the public source-first path. +Get the source from the [EPAR releases page](https://github.com/solutionforest/ephemeral-action-runner/releases), extract the source archive, and work from that folder. You do not need Packer, GitHub CLI, or `sshpass`. -Don't want to install Go at all? See [Running EPAR Without Installing Go](advanced/no-go-install.md) for running the source archive in a container. +EPAR works with any Docker installation that supports the selected provider. -## One-Command Start +## Start a pool -For the default Docker-DinD setup, run EPAR from the source folder. On macOS, Linux, WSL, or Git Bash, use the `./start` wrapper; on native Windows PowerShell/cmd, use `.\start.ps1` or `start.cmd`. Either uses Go if installed, and otherwise runs EPAR from source with a containerized Go toolchain automatically, without creating a standalone EPAR executable (see [Running EPAR Without Installing Go](advanced/no-go-install.md)): +On macOS, Linux, WSL, or Git Bash, run: ```bash ./start ``` -Equivalent without the wrapper: - -```bash -go run ./cmd/ephemeral-action-runner -``` - -If no config exists, EPAR starts the initializer, asks for the GitHub App ID, organization, and private key path, then writes `.local/config.yml`. Docker-DinD is the default. For a new Docker-DinD config, the wizard asks whether to inherit the controller host's trusted TLS roots and defaults to yes; existing configs remain disabled unless they explicitly set `image.hostTrustMode: overlay`. On native Windows, when `wsl.exe --status` successfully confirms default version 2, the wizard also offers a WSL2 config. On macOS, when `tart --version` succeeds, it offers an experimental Tart config. Press Enter to retain Docker-DinD. The Docker preflight applies to Docker-DinD and the default WSL image, which uses Docker for its one-time rootfs export, but not to Tart. EPAR then checks the configured image, builds or replaces it when the image is missing or no longer matches the config, and starts the configured number of runners. The default config uses `pool.instances: 1`. - -Pass flags through `./start` to choose a config or runner count: - -```bash -./start --config .local/config.yml --instances 2 -``` - -Equivalent without the wrapper: - -```bash -go run ./cmd/ephemeral-action-runner start --config .local/config.yml --instances 2 -``` - -On Windows PowerShell: +On native Windows PowerShell, run: ```powershell -.\start.ps1 --config .local\wsl.yml --instances 2 +.\start.ps1 ``` -Equivalent without the wrapper: +The wrapper uses local Go when available, otherwise it uses Docker to build and cache a native controller under `.local/bin`. See [Running EPAR Without Installing Go](advanced/no-go-install.md) for the fallback details. The equivalent direct source command is: -```powershell -go run ./cmd/ephemeral-action-runner start --config .local\wsl.yml --instances 2 +```bash +go run ./cmd/ephemeral-action-runner start ``` -Stop the foreground process with Ctrl-C. Cleanup is enabled by default. +When `.local/config.yml` is absent and the terminal is interactive, `./start` launches the same first-run wizard as `init`. It asks for the GitHub App and an explicit runner group. The runner-group list orders GitHub's Default group first, hides blocked groups and policy details initially, and lets you reveal either from the menu. The provider list shows every provider with its prerequisite status and refuses unavailable selections. Storage does not make a provider unavailable. Docker Container, Docker Sandboxes, and WSL share one Catthehacker image/profile and custom-script flow, followed by an informational physical-growth estimate and confirmation. The wizard writes the desired configuration first; direct `init` then exits, while embedded `./start` continues through the ordinary image/template provisioning and pool startup path. When `sbx` is installed, the wizard runs `sbx daemon start --detach` before Docker Sandboxes diagnostics so a stopped daemon does not require a manual retry. -If `--instances` is omitted, `start`, `pool up`, and `pool verify` use `pool.instances` from the config. Passing `--instances N` overrides the config for that run. +See [Docker Sandboxes](providers/docker-sandboxes.md) for source profiles, capacity, local receipts, and platform validation status. -## Configure Only +## Create or choose configuration -Use `init` when you only want to create a config without building an image or starting runners. It creates Docker-DinD by default, with the same conditional native-Windows WSL2 and macOS Tart choices described above: +Create configuration without starting runners: ```bash go run ./cmd/ephemeral-action-runner init ``` -On Windows PowerShell: - -```powershell -go run ./cmd/ephemeral-action-runner init -``` - -For other WSL or Tart variants, or for custom labels, copy one example config into `.local/config.yml`, then edit the GitHub App fields and any labels you want to expose to workflows. - -| Host and image | Example config | -| --- | --- | -| macOS Tart, experimental basic Ubuntu ARM64 image | `configs/tart.example.yml` | -| macOS Tart, web/E2E with Rosetta amd64 Docker support | `configs/tart.web-e2e.example.yml` | -| Windows WSL2, default full Catthehacker runner image | `configs/wsl.example.yml` | -| Windows WSL2, lean runner-only tar | `configs/wsl.lean.example.yml` | -| Windows WSL2, lean web/E2E tar | `configs/wsl.web-e2e.example.yml` | -| Docker-DinD, default full Catthehacker runner image | `configs/docker-dind.example.yml` | -| Docker-DinD, Docker-focused Catthehacker Act image | `configs/docker-dind.act.example.yml` | -| Docker-DinD, smaller web/E2E custom image | `configs/docker-dind.web-e2e.example.yml` | - -Tart is experimental. Its default image is a basic Ubuntu ARM64 OS image with the EPAR runner lifecycle, not the dependency-rich environment described by [`actions/runner-images`](https://github.com/actions/runner-images). If your workflows depend on that environment, adapt the upstream build scripts to create and maintain your own bootable Tart image and point `image.sourceImage` at it; EPAR does not automatically create one. - -macOS: +Pass a config path and an instance count through the wrapper: ```bash -mkdir -p .local -cp configs/tart.example.yml .local/config.yml -``` - -Windows: - -```powershell -New-Item -ItemType Directory -Force .local -Copy-Item configs/wsl.example.yml .local/config.yml +./start --config .local/ci.yml --instances 2 ``` -Default Docker-DinD manually: +Equivalent direct command: ```bash -mkdir -p .local -cp configs/docker-dind.example.yml .local/config.yml +go run ./cmd/ephemeral-action-runner start --config .local/ci.yml --instances 2 ``` -EPAR looks for config in this order: - -1. `--config ` -2. `EPAR_CONFIG` -3. `./.local/config.yml` -4. `~/.config/ephemeral-action-runner/config.yml` - -Tracked configs are examples only. Keep real app IDs and private key paths in an ignored config file. - -## Optional Docker Registry Mirrors - -If repeated jobs spend time pulling the same Docker Hub images into fresh runner Docker daemons, configure mirrors in your ignored local config: - -```yaml -docker: - registryMirrors: - - http://host.docker.internal:5050 -``` - -This is optional. Without it, EPAR behaves normally and pulls directly from registries. Mirror benefits vary by workflow and mainly affect Docker image pull time; they do not make application startup, volume sync, health checks, browser tests, or CPU-bound work faster. - -EPAR only configures runner-side Docker daemons; it does not run or secure the mirror service. Docker Engine, Docker Desktop, or OrbStack can run a local `registry:2` pull-through cache on the EPAR host, or you can use a mirror reachable on the LAN/intranet. For private images, keep using `docker login` inside the workflow unless your mirror is deliberately configured and secured with upstream credentials. See [Docker Registry Mirrors](advanced/docker-registry-mirrors.md). - -## Prepare A WSL Source - -Skip this section for Tart and Docker-DinD. - -The default WSL config starts from `ghcr.io/catthehacker/ubuntu:full-latest`. During `image build`, EPAR runs Docker on the Windows host to pull that image, create a temporary container, export its filesystem into a rootfs tar, and then import that tar into WSL for EPAR's normal runner bootstrap. Docker is needed for this preparation step. Running WSL runner instances afterward does not require Docker Desktop unless your jobs need it. - -If you use `configs/wsl.lean.example.yml`, `configs/wsl.web-e2e.example.yml`, or another `image.sourceType: rootfs-tar` config, create the clean Ubuntu 24.04 source tar once: +On Windows PowerShell, use backslash paths when that is clearer: ```powershell -New-Item -ItemType Directory -Force work/images -wsl --install -d Ubuntu-24.04 --no-launch -wsl --export Ubuntu-24.04 work/images/ubuntu-24.04-clean.rootfs.tar -``` - -After that, EPAR imports disposable temporary distros for image builds and pool instances. - -## Build The Runner Image Manually - -The `start` command builds or replaces the configured image automatically. Use this section when developing from source, debugging image builds, or intentionally separating image preparation from runner startup. - -Default WSL and Docker-DinD builds and runner-only Tart builds do not need the upstream `actions/runner-images` checkout: - -```bash -go run ./cmd/ephemeral-action-runner image build --replace +.\start.ps1 --config .local\ci.yml --instances 2 +go run ./cmd/ephemeral-action-runner start --config .local\ci.yml --instances 2 ``` -If `image.customInstallScripts` includes EPAR's Docker/browser or web/E2E scripts, update the pinned upstream checkout first: +If `--instances` is omitted, `start`, `pool up`, and `pool verify` use `pool.instances` from the selected config. EPAR resolves configuration from `--config`, `EPAR_CONFIG`, `.local/config.yml`, then `~/.config/ephemeral-action-runner/config.yml`. Tracked files in `configs/` are examples; keep App values and key paths in an ignored local file. See [Configuration](configuration.md) for every setting and [Runner Group Security](runner-groups.md) before broadening repository access. -```bash -go run ./cmd/ephemeral-action-runner image update-upstream -go run ./cmd/ephemeral-action-runner image build --replace -``` +Multiple configs from the same checkout may run concurrently when they use different canonical config paths, unique `pool.namePrefix` values, unique workflow-routing labels, and preferably separate log directories. EPAR rejects a second controller for the same config path or prefix before provisioning or cleanup can mutate provider state. Config-scoped BuildKit builders and transient workspaces keep divergent registry, trust, and cache settings isolated. -The Tart web/E2E example sets `provider.rosettaTag: rosetta`. Tart builds with that option start with `tart run --rosetta rosetta`, install Rosetta guest support, and validate that Docker can run a `linux/amd64` Alpine container returning `x86_64`. - -Tart output is a local Tart image name, such as `epar-ubuntu-24-arm64`. Confirm it with: - -```bash -tart list -``` +Storage-consuming commands fail before their provider side effects when an authoritative physical surface cannot retain `storage.minimumFree`. The one-invocation `--allow-insufficient-storage` option keeps all probes and warnings but permits only storage admission to continue; provider diagnostics, GitHub policy, ownership, lifecycle, and cleanup protections remain enforced. The option is available on `start`, `pool up`, `pool verify`, `image update`, `image build`, and `image update-upstream`, including the equivalent `./start ...` wrapper forms. -The default WSL output is a rootfs tar path: +Each normal start also reconciles interrupted exact-owned work and retires unreferenced superseded artifacts after replacement readback. Use `./start storage status` to inspect the result. `./start storage prune --legacy` previews prefix-era resources, which remain manual and require the displayed plan hash before execution. -```text -work/images/epar-wsl-catthehacker-ubuntu.tar -``` +Press `Ctrl-C` once to stop a foreground pool, then wait for cleanup to finish before closing the terminal. Use `--keep-on-exit` only to retain owned resources for deliberate debugging. -When the WSL source is a Docker image, EPAR also writes an intermediate source rootfs tar and env cache next to the output image, for example `work/images/epar-wsl-catthehacker-ubuntu.source.rootfs.tar` and `.env`. Later builds reuse that source cache; delete those files when you intentionally want to reconvert the Docker image. +## Update runner artifacts -EPAR also writes image manifests so `start` can tell whether the local image still matches the config. Docker-DinD stores the manifest hash as a Docker image label and stores the manifest at `/opt/epar/image-manifest.json`. WSL stores `/opt/epar/image-manifest.json` inside the exported image and writes a sidecar next to the tar. +By default, EPAR checks mutable source-image tags and `runnerVersion: latest` weekly at 07:00 local time. The wizard can select daily, weekly, every two weeks, monthly, or manual checks. Local image settings, script or certificate content, platform, EPAR assets, and missing or corrupt artifacts always apply on the next start without waiting for the schedule. -Docker-DinD output is a Docker image tag, such as `epar-docker-dind-catthehacker-ubuntu`. Confirm it with: +Force an immediate remote check without forcing a rebuild: ```bash -docker image ls epar-docker-dind-catthehacker-ubuntu -``` - -Build logs are written under `work/logs/builds` by default. Run `ephemeral-action-runner logs path` to resolve a customized logging root and see [Logging](logging.md) for rotation and retention. - -## Customize The Image - -WSL and Docker-DinD use the full Catthehacker runner image by default. For Docker-focused jobs, `configs/docker-dind.act.example.yml` uses the smaller Catthehacker Act image, which includes Node and the Docker Engine/CLI/Compose/Buildx stack EPAR needs. It does not guarantee browser dependencies; use `configs/docker-dind.web-e2e.example.yml` for Playwright or other browser tests. Tart and the WSL lean examples are runner-only. Use `image.customInstallScripts` when you want a different image shape, such as the smaller WSL or Docker-DinD web/E2E examples: - -```yaml -image: - customInstallScripts: - - scripts/guest/ubuntu/install-web-e2e.sh - - examples/custom-install/install-extra-apt-tools.sh +./start image update ``` -Scripts run as root during image build, after the GitHub Actions runner is installed and before validation/finalization. See [Image Build](image-build.md) for the full layering model and custom script guidance. +Manual policy means this command triggers remote checks. `./start image build` remains the force-build path. A running ephemeral pool checks when due, drains only after busy jobs finish, activates the verified replacement, and restores pool capacity; persistent runners record the update for the next process start. -## Verify Runners +## Verify before sending jobs -For a local runtime check without GitHub registration: +Verify one disposable runner without GitHub registration: ```bash go run ./cmd/ephemeral-action-runner pool verify --instances 1 --cleanup ``` -For a full registration check: +Verify registration and online/idle state: ```bash go run ./cmd/ephemeral-action-runner pool verify --instances 2 --register-only --cleanup ``` -Healthy output should show each generated instance name moving through: - -1. clone -2. start -3. runtime validation -4. GitHub online/idle, when registration is enabled -5. cleanup - -Runtime validation always checks the base runner files and runner user. Images with optional feature markers also validate those features: - -- Docker/browser images validate Docker, Compose v2, Buildx, `hello-world`, and a headless browser. -- Default WSL full images validate Docker, Compose v2, Buildx, and `hello-world`. -- Docker-DinD images validate the private inner Docker daemon inside each runner container. -- Tart Rosetta images validate `docker run --platform linux/amd64 alpine:3.20` and expect `uname -m` to return `x86_64`. -- Web/E2E images also validate `node`, `npm`, `zip`, `unzip`, `tar`, `rsync`, and `mysql`. - -When `docker.registryMirrors` is configured, EPAR applies the mirror configuration before runtime validation. - -If a Docker-DinD workflow depends on amd64-only images while the host is ARM64, validate host emulation inside a running EPAR instance: - -```bash -docker exec docker run --rm --platform linux/amd64 alpine:3.20 uname -m -``` +`--cleanup` removes verification resources after the check. Docker Sandboxes uses its exact ownership records; legacy providers use the configured pool-name boundary. Use [Operations](operations.md) for the distinction and recovery guidance. -The expected output is `x86_64`. +## Run, inspect, and clean up -## Run A Foreground Pool Manually +`start` is the normal command because it checks the reusable image or template first. `pool up` is for a pool you have deliberately prepared: ```bash go run ./cmd/ephemeral-action-runner pool up --instances 2 -``` - -`start` is the recommended public command because it also checks the image before starting runners. `pool up` is the lower-level supervisor command for users who already prepared the image. - -`pool up` keeps the requested number of runners online. Each GitHub ephemeral runner exits after one job. EPAR then retires that instance and creates a fresh replacement. The requested count is a strict physical local-instance cap: provisioning, ready, draining, quarantined, and cleanup-pending resources all consume a slot, including old runners during host-trust rotation. - -When GitHub registration or readiness has a transient network, `429`, or `5xx` failure during supervised replacement, EPAR pauses allocation and retries with the configured exponential backoff while continuing monitoring and cleanup. It does not create extra candidates while remote state is uncertain; see [Configuration](configuration.md#common-edits) and [Operations](operations.md#capacity-reconciliation-and-outage-recovery) for retry settings and recovery steps. - -Stop the supervisor with Ctrl-C. By default, EPAR cleans up active instances and matching GitHub runner records before it exits. - -For startup after login, see [Windows Startup](advanced/windows-startup.md) or [macOS Startup](advanced/macos-startup.md). - -Use these flags only for debugging: - -- `--keep-on-exit`: leave instances running when the supervisor exits. -- `--replace-completed=false`: do not create replacements after completed jobs. - -## Status And Cleanup - -```bash go run ./cmd/ephemeral-action-runner status go run ./cmd/ephemeral-action-runner cleanup ``` -Cleanup only touches local instances and GitHub runners whose names match `pool.namePrefix`. - -## Runner Labels +Use `status --no-github` or `cleanup --no-github` when you intentionally need to skip GitHub runner status or deletion. `pool down` is an alias for cleanup. -By default, EPAR appends an `epar-host-` label to the configured labels. The machine name is lowercased, unsafe characters are replaced with `-`, and the final label is kept within GitHub's 256-character label limit. Set `runner.includeHostLabel: false` to disable it. +For a command-construction preview on compatible providers, add `--dry-run`: -Use provider-specific labels in workflows. For the Tart web/E2E Rosetta image, target the existing web/E2E label plus the Rosetta label when the job needs amd64 Docker images: - -```yaml -runs-on: [self-hosted, linux, ARM64, epar-tart-ubuntu-24.04-web-e2e, epar-tart-rosetta-amd64] +```bash +go run ./cmd/ephemeral-action-runner pool verify --dry-run --instances 1 ``` -For the default WSL image, target the default WSL label: +Docker Sandboxes intentionally does not support dry-run instance creation because EPAR must read back the exact active template-cache identity. Use its admission and template checks instead. -```yaml -runs-on: [self-hosted, linux, X64, epar-wsl-catthehacker-ubuntu] -``` +## Target the right runner -For the default Docker-DinD image, target the default Docker-DinD label: +GitHub matches every value in `runs-on` against a runner's labels. The smallest workflow selector is: ```yaml -runs-on: [self-hosted, linux, epar-docker-dind-catthehacker-ubuntu] +runs-on: [self-hosted] ``` -For the Docker-focused Act image, target its dedicated label: +Add a provider or workload label to avoid routing work to the wrong environment: ```yaml -runs-on: [self-hosted, linux, epar-docker-dind-catthehacker-act] +runs-on: [self-hosted, linux, epar-docker-container-catthehacker-ubuntu] ``` -For Docker-DinD web/E2E images, target the custom web/E2E label: - -```yaml -runs-on: [self-hosted, linux, epar-docker-dind-catthehacker-ubuntu-web-e2e] -``` - -When that Docker-DinD runner is used for amd64-only runtime images, keep the workflow's Docker platform explicit, for example `DOCKER_PLATFORM=linux/amd64` or the equivalent variable used by your compose scripts, and verify the host runtime supports amd64 emulation as described above. - -Do not use `ubuntu-latest` for these self-hosted runners. - -## Dry Run - -Use `--dry-run` to inspect provider command construction without mutating local instances: - -```bash -go run ./cmd/ephemeral-action-runner pool verify --dry-run --instances 2 -``` - -## Maintainer Source-Only Releases - -Releases are manually dispatched from GitHub Actions and contain no uploaded assets. GitHub automatically provides **Source code (zip)** and **Source code (tar.gz)** downloads for each release tag. - -```bash -git tag -a v0.1.0-beta.1 -m "v0.1.0-beta.1" -git push origin v0.1.0-beta.1 -``` +EPAR adds an `epar-host-` label by default. Use it only when a job must target one specific host. Give each independent pool in the same organization a unique `pool.namePrefix`; this is also its cleanup boundary. -Before dispatching, a repository administrator must enable immutable releases in the repository. In **Actions → Release → Run workflow**, supply an existing remote tag that matches `[v]MAJOR.MINOR.PATCH` or `[v]MAJOR.MINOR.PATCH-(alpha|beta|rc).N`, then type `publish source-only release` exactly. The workflow requires annotated tags, verifies the remote tag, confirms its commit is reachable from `origin/main`, checks out and tests that exact commit, and refuses to overwrite an existing release. +## Common next tasks -Alpha, beta, and RC tags are published as prereleases and are not marked latest. To promote a prerelease to a stable tag without changing the commit, provide the existing prerelease tag in `promotion_from`; the workflow verifies that the tag and its GitHub Release exist, that its normalized `MAJOR.MINOR.PATCH` core matches the stable tag, and that both tags point to the same commit, then creates stable promotion notes instead of a generated delta. +- [Customize a runner image](image-build.md). +- [Configure Docker registry mirrors](advanced/docker-registry-mirrors.md). +- [Start EPAR after login on Windows](advanced/windows-startup.md) or [macOS](advanced/macos-startup.md). +- [Inspect logs, capacity, cleanup, and recovery](operations.md). +- [Diagnose a symptom](troubleshooting.md). diff --git a/internal/config/config.go b/internal/config/config.go index 71c1a1a..dab690f 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -10,18 +10,22 @@ import ( "path/filepath" "strconv" "strings" + "time" ) type Config struct { - GitHub GitHubConfig - Image ImageConfig - Pool PoolConfig - Logging LoggingConfig - Runner RunnerConfig - Provider ProviderConfig - Docker DockerConfig - Timeouts TimeoutConfig - warnings []string + GitHub GitHubConfig + Image ImageConfig + Pool PoolConfig + Storage StorageConfig + Logging LoggingConfig + Runner RunnerConfig + Security SecurityConfig + Provider ProviderConfig + Docker DockerConfig + DockerSandboxes DockerSandboxesConfig + Timeouts TimeoutConfig + warnings []string } // Warnings returns non-fatal configuration migration notices discovered while loading. @@ -45,6 +49,8 @@ type ImageConfig struct { UpstreamDir string UpstreamLock string RunnerVersion string + UpdateFrequency string + UpdateTime string CustomInstallScripts []string TrustedCACertificatePaths []string HostTrustMode string @@ -57,6 +63,13 @@ const ( HostTrustScopeSystem = "system" HostTrustScopeUser = "user" + + ImageUpdateFrequencyDaily = "daily" + ImageUpdateFrequencyWeekly = "weekly" + ImageUpdateFrequencyBiweekly = "biweekly" + ImageUpdateFrequencyMonthly = "monthly" + ImageUpdateFrequencyManual = "manual" + DefaultImageUpdateTime = "07:00" ) type PoolConfig struct { @@ -68,6 +81,23 @@ type PoolConfig struct { ReplacementRetryJitterPercent int } +// StorageConfig controls provider-neutral capacity admission and conservative +// retention. String byte sizes keep the YAML readable and are validated before +// any provider or cleanup operation is constructed. +type StorageConfig struct { + MinimumFree string + GracePeriod string + KeepPrevious int + AutomaticHousekeeping string + BuildCacheLimit string + GoCacheLimit string +} + +const ( + StorageHousekeepingConservative = "conservative" + StorageHousekeepingDisabled = "disabled" +) + type LoggingConfig struct { Directory string ManagerSinks []string @@ -98,6 +128,27 @@ type RunnerConfig struct { NoDefaultLabels bool } +type SecurityConfig struct { + RunnerGroup RunnerGroupSecurityConfig +} + +type RunnerGroupSecurityConfig struct { + Enforcement string + RequireExplicitGroup bool + RequireNonDefaultGroup bool + RequiredRepositoryAccess string + RequirePublicRepositoriesDisabled bool +} + +const ( + RunnerGroupEnforcementEnforce = "enforce" + RunnerGroupEnforcementWarn = "warn" + + RunnerGroupRepositoryAccessSelected = "selected" + RunnerGroupRepositoryAccessPrivate = "private" + RunnerGroupRepositoryAccessAll = "all" +) + type ProviderConfig struct { Type string SourceImage string @@ -114,6 +165,50 @@ type DockerConfig struct { NoProxy string } +// DockerSandboxesConfig configures host/runtime behavior for the +// docker-sandboxes provider. The desired source belongs to ImageConfig, while +// the exact imported template identity is stored in EPAR's local artifact +// receipt rather than user configuration. +type DockerSandboxesConfig struct { + PolicyGeneration string + NetworkBaseline string + AdditionalAllow []string + AdditionalDeny []string + StagingRoot string + CPUs int + Memory string + RootDisk string + DockerDisk string + MaxConcurrentCreates int +} + +const ( + DockerSandboxesNetworkBaselineOpen = "open" + DockerSandboxesNetworkBaselineBalanced = "balanced" +) + +var dockerSandboxesOpenDefaultDenyResources = []string{ + "host.docker.internal", + "gateway.docker.internal", + "kubernetes.docker.internal", + "host.containers.internal", +} + +// DockerSandboxesOpenDefaultDenyResources returns host aliases that EPAR denies +// in every sandbox-scoped Open policy. Docker Sandboxes can proxy an allowed +// host.docker.internal request to a native-host loopback service, so +// public egress must not implicitly enable these host-service aliases. +func DockerSandboxesOpenDefaultDenyResources() []string { + return append([]string(nil), dockerSandboxesOpenDefaultDenyResources...) +} + +const ( + DockerSandboxesAutomaticRootDisk = "auto" + DockerSandboxesMinimumRootDiskBytes int64 = 20 << 30 + DockerSandboxesMinimumDockerDiskBytes int64 = 1 << 30 + DockerSandboxesDefaultDockerDisk = "50GiB" +) + type TimeoutConfig struct { BootSeconds int GitHubOnlineSeconds int @@ -134,12 +229,14 @@ func Default() Config { WebBaseURL: "https://github.com", }, Image: ImageConfig{ - SourceImage: "ghcr.io/cirruslabs/ubuntu:latest", - OutputImage: "epar-ubuntu-24-arm64", - UpstreamDir: "third_party/runner-images", - UpstreamLock: "third_party/runner-images.lock", - RunnerVersion: "latest", - HostTrustMode: HostTrustModeDisabled, + SourceImage: "ghcr.io/cirruslabs/ubuntu:latest", + OutputImage: "epar-ubuntu-24-arm64", + UpstreamDir: "third_party/runner-images", + UpstreamLock: "third_party/runner-images.lock", + RunnerVersion: "latest", + UpdateFrequency: ImageUpdateFrequencyWeekly, + UpdateTime: DefaultImageUpdateTime, + HostTrustMode: HostTrustModeDisabled, HostTrustScopes: []string{ HostTrustScopeSystem, }, @@ -152,6 +249,14 @@ func Default() Config { ReplacementRetryMultiplier: 2, ReplacementRetryJitterPercent: 20, }, + Storage: StorageConfig{ + MinimumFree: "1GiB", + GracePeriod: "168h", + KeepPrevious: 0, + AutomaticHousekeeping: StorageHousekeepingConservative, + BuildCacheLimit: "20GiB", + GoCacheLimit: "10GiB", + }, Logging: LoggingConfig{ Directory: "work/logs", ManagerSinks: []string{"console"}, @@ -176,12 +281,30 @@ func Default() Config { IncludeHostLabel: true, Ephemeral: true, }, + Security: SecurityConfig{ + RunnerGroup: RunnerGroupSecurityConfig{ + Enforcement: RunnerGroupEnforcementWarn, + RequireExplicitGroup: true, + RequireNonDefaultGroup: true, + RequiredRepositoryAccess: RunnerGroupRepositoryAccessSelected, + RequirePublicRepositoriesDisabled: true, + }, + }, Provider: ProviderConfig{ Type: "tart", SourceImage: "epar-ubuntu-24-arm64", Network: "default", InstallRoot: "work/wsl", }, + DockerSandboxes: DockerSandboxesConfig{ + NetworkBaseline: DockerSandboxesNetworkBaselineOpen, + StagingRoot: ".local/docker-sandboxes-staging", + CPUs: 4, + Memory: "8GiB", + RootDisk: DockerSandboxesAutomaticRootDisk, + DockerDisk: DockerSandboxesDefaultDockerDisk, + MaxConcurrentCreates: 2, + }, Timeouts: TimeoutConfig{ BootSeconds: 180, GitHubOnlineSeconds: 180, @@ -203,6 +326,8 @@ func Load(path string) (Config, error) { defer file.Close() section := "" + subsection := "" + subsectionIndent := 0 scanner := bufio.NewScanner(file) lineNo := 0 var pendingList *pendingListKey @@ -241,6 +366,8 @@ func Load(path string) (Config, error) { return cfg, fmt.Errorf("%s:%d: unknown section %q", path, lineNo, key) } section = key + subsection = "" + subsectionIndent = 0 continue } if indent == 0 { @@ -249,24 +376,42 @@ func Load(path string) (Config, error) { if section == "" { return cfg, fmt.Errorf("%s:%d: key %q must be under a section", path, lineNo, key) } - if section == "pool" && key == "logDir" { + if section == "security" && value == "" { + if key != "runnerGroup" { + return cfg, fmt.Errorf("%s:%d: unknown subsection security.%s", path, lineNo, key) + } + subsection = key + subsectionIndent = indent + continue + } + effectiveSection := section + if section == "security" { + if subsection == "" { + return cfg, fmt.Errorf("%s:%d: key %q must be under security.runnerGroup", path, lineNo, key) + } + if indent <= subsectionIndent { + return cfg, fmt.Errorf("%s:%d: key %q must be nested under security.%s", path, lineNo, key, subsection) + } + effectiveSection = section + "." + subsection + } + if effectiveSection == "pool" && key == "logDir" { legacyLogDir = trimQuotes(value) legacyLogDirLine = lineNo explicit["pool.logDir"] = true continue } - if value == "" && isListKey(section, key) { - if err := setListValue(&cfg, section, key, nil); err != nil { + if value == "" && isListKey(effectiveSection, key) { + if err := setListValue(&cfg, effectiveSection, key, nil); err != nil { return cfg, fmt.Errorf("%s:%d: %w", path, lineNo, err) } - explicit[section+"."+key] = true - pendingList = &pendingListKey{section: section, key: key, indent: indent} + explicit[effectiveSection+"."+key] = true + pendingList = &pendingListKey{section: effectiveSection, key: key, indent: indent} continue } - if err := apply(&cfg, section, key, value); err != nil { + if err := apply(&cfg, effectiveSection, key, value); err != nil { return cfg, fmt.Errorf("%s:%d: %w", path, lineNo, err) } - explicit[section+"."+key] = true + explicit[effectiveSection+"."+key] = true } if err := scanner.Err(); err != nil { return cfg, err @@ -278,6 +423,16 @@ func Load(path string) (Config, error) { cfg.Logging.Directory = legacyLogDir cfg.warnings = append(cfg.warnings, fmt.Sprintf("%s:%d: pool.logDir is deprecated; using its value as logging.directory (move it to the top-level logging section)", path, legacyLogDirLine)) } + if !explicit["security.runnerGroup.enforcement"] && + !explicit["security.runnerGroup.requireExplicitGroup"] && + !explicit["security.runnerGroup.requireNonDefaultGroup"] && + !explicit["security.runnerGroup.requiredRepositoryAccess"] && + !explicit["security.runnerGroup.requirePublicRepositoriesDisabled"] { + cfg.warnings = append(cfg.warnings, fmt.Sprintf("%s: runner-group security policy is not configured; using strict recommended checks in warn mode (add security.runnerGroup.enforcement: enforce after reviewing the policy)", path)) + } + if !explicit["image.updateFrequency"] && !explicit["image.updateTime"] { + cfg.warnings = append(cfg.warnings, fmt.Sprintf("%s: image update policy is not configured; using weekly checks at 07:00 local time", path)) + } applyProviderDefaults(&cfg, explicit) applyRunnerHostLabel(&cfg) cfg.GitHub.PrivateKeyPath = expandHome(cfg.GitHub.PrivateKeyPath) @@ -331,6 +486,10 @@ func apply(cfg *Config, section, key, value string) error { cfg.Image.UpstreamLock = value case "runnerVersion": cfg.Image.RunnerVersion = value + case "updateFrequency": + cfg.Image.UpdateFrequency = strings.ToLower(value) + case "updateTime": + cfg.Image.UpdateTime = value case "profile": return fmt.Errorf("image.profile is not supported; use image.customInstallScripts") case "customInstallScripts": @@ -451,6 +610,27 @@ func apply(cfg *Config, section, key, value string) error { default: return unknownKey(section, key) } + case "storage": + switch key { + case "minimumFree": + cfg.Storage.MinimumFree = value + case "gracePeriod": + cfg.Storage.GracePeriod = value + case "keepPrevious": + v, err := strconv.Atoi(value) + if err != nil { + return fmt.Errorf("invalid storage.keepPrevious: %w", err) + } + cfg.Storage.KeepPrevious = v + case "automaticHousekeeping": + cfg.Storage.AutomaticHousekeeping = strings.ToLower(value) + case "buildCacheLimit": + cfg.Storage.BuildCacheLimit = value + case "goCacheLimit": + cfg.Storage.GoCacheLimit = value + default: + return unknownKey(section, key) + } case "runner": switch key { case "labels": @@ -478,6 +658,33 @@ func apply(cfg *Config, section, key, value string) error { default: return unknownKey(section, key) } + case "security.runnerGroup": + switch key { + case "enforcement": + cfg.Security.RunnerGroup.Enforcement = strings.ToLower(value) + case "requireExplicitGroup": + v, err := strconv.ParseBool(value) + if err != nil { + return fmt.Errorf("invalid security.runnerGroup.requireExplicitGroup: %w", err) + } + cfg.Security.RunnerGroup.RequireExplicitGroup = v + case "requireNonDefaultGroup": + v, err := strconv.ParseBool(value) + if err != nil { + return fmt.Errorf("invalid security.runnerGroup.requireNonDefaultGroup: %w", err) + } + cfg.Security.RunnerGroup.RequireNonDefaultGroup = v + case "requiredRepositoryAccess": + cfg.Security.RunnerGroup.RequiredRepositoryAccess = strings.ToLower(value) + case "requirePublicRepositoriesDisabled": + v, err := strconv.ParseBool(value) + if err != nil { + return fmt.Errorf("invalid security.runnerGroup.requirePublicRepositoriesDisabled: %w", err) + } + cfg.Security.RunnerGroup.RequirePublicRepositoriesDisabled = v + default: + return unknownKey(section, key) + } case "provider": switch key { case "type": @@ -508,6 +715,46 @@ func apply(cfg *Config, section, key, value string) error { default: return unknownKey(section, key) } + case "dockerSandboxes": + switch key { + case "template": + return fmt.Errorf("dockerSandboxes.template is no longer supported; remove generated template identities and rerun ./start to provision a Docker Sandboxes runner template") + case "templateDigest": + return fmt.Errorf("dockerSandboxes.templateDigest is no longer supported; remove generated template identities and rerun ./start to provision a Docker Sandboxes runner template") + case "policyGeneration": + cfg.DockerSandboxes.PolicyGeneration = value + case "networkBaseline": + cfg.DockerSandboxes.NetworkBaseline = strings.ToLower(value) + case "additionalAllow", "additionalDeny": + return setListValue(cfg, section, key, parseList(value)) + case "stagingRoot": + cfg.DockerSandboxes.StagingRoot = value + case "cpus": + v, err := strconv.Atoi(value) + if err != nil { + return fmt.Errorf("invalid dockerSandboxes.cpus: %w", err) + } + cfg.DockerSandboxes.CPUs = v + case "memory", "rootDisk", "dockerDisk": + switch key { + case "memory": + cfg.DockerSandboxes.Memory = value + case "rootDisk": + cfg.DockerSandboxes.RootDisk = value + case "dockerDisk": + cfg.DockerSandboxes.DockerDisk = value + } + case "minHostFreeSpace": + return fmt.Errorf("dockerSandboxes.minHostFreeSpace is no longer supported; remove it and configure the provider-neutral storage.minimumFree value instead") + case "maxConcurrentCreates": + v, err := strconv.Atoi(value) + if err != nil { + return fmt.Errorf("invalid dockerSandboxes.maxConcurrentCreates: %w", err) + } + cfg.DockerSandboxes.MaxConcurrentCreates = v + default: + return unknownKey(section, key) + } case "timeouts": v, err := strconv.Atoi(value) if err != nil { @@ -535,7 +782,7 @@ func unknownKey(section, key string) error { func isKnownSection(section string) bool { switch section { - case "github", "image", "pool", "logging", "runner", "provider", "docker", "timeouts": + case "github", "image", "pool", "storage", "logging", "runner", "security", "provider", "docker", "dockerSandboxes", "timeouts": return true default: return false @@ -583,7 +830,7 @@ func applyProviderDefaults(cfg *Config, explicit map[string]bool) { if !explicit["pool.namePrefix"] && !explicit["pool.vmPrefix"] { cfg.Pool.NamePrefix = "epar-wsl" } - case "docker-dind": + case "docker-container": if !explicit["image.sourceType"] { cfg.Image.SourceType = ImageSourceDockerImage } @@ -591,16 +838,45 @@ func applyProviderDefaults(cfg *Config, explicit map[string]bool) { cfg.Image.SourceImage = "ghcr.io/catthehacker/ubuntu:full-latest" } if !explicit["image.outputImage"] { - cfg.Image.OutputImage = "epar-docker-dind-catthehacker-ubuntu" + cfg.Image.OutputImage = "epar-docker-container-catthehacker-ubuntu" } if !explicit["provider.sourceImage"] { cfg.Provider.SourceImage = cfg.Image.OutputImage } if !explicit["runner.labels"] { - cfg.Runner.Labels = []string{"self-hosted", "linux", "epar-docker-dind-catthehacker-ubuntu"} + cfg.Runner.Labels = []string{"self-hosted", "linux", "epar-docker-container-catthehacker-ubuntu"} } if !explicit["pool.namePrefix"] && !explicit["pool.vmPrefix"] { - cfg.Pool.NamePrefix = "epar-dind" + cfg.Pool.NamePrefix = "epar-docker-container" + } + case "docker-sandboxes": + if !explicit["provider.sourceImage"] { + cfg.Provider.SourceImage = "" + } + if !explicit["provider.platform"] { + cfg.Provider.Platform = "linux/amd64" + } + if !explicit["image.sourceType"] { + cfg.Image.SourceType = ImageSourceDockerImage + } + if !explicit["image.sourceImage"] { + cfg.Image.SourceImage = "ghcr.io/catthehacker/ubuntu:full-latest" + } + if !explicit["image.sourcePlatform"] { + cfg.Image.SourcePlatform = cfg.Provider.Platform + } + if !explicit["image.outputImage"] { + cfg.Image.OutputImage = "" + } + if !explicit["runner.labels"] { + architecture := "X64" + if cfg.Provider.Platform == "linux/arm64" { + architecture = "ARM64" + } + cfg.Runner.Labels = []string{"self-hosted", "linux", architecture, "epar-docker-sandboxes"} + } + if !explicit["pool.namePrefix"] && !explicit["pool.vmPrefix"] { + cfg.Pool.NamePrefix = "epar-docker-sandboxes" } } } @@ -701,6 +977,8 @@ func isListKey(section, key string) bool { return key == "labels" case "docker": return key == "registryMirrors" + case "dockerSandboxes": + return key == "additionalAllow" || key == "additionalDeny" case "logging": return key == "managerSinks" || key == "transcriptSinks" default: @@ -732,6 +1010,15 @@ func setListValue(cfg *Config, section, key string, values []string) error { cfg.Docker.RegistryMirrors = values return nil } + case "dockerSandboxes": + switch key { + case "additionalAllow": + cfg.DockerSandboxes.AdditionalAllow = values + return nil + case "additionalDeny": + cfg.DockerSandboxes.AdditionalDeny = values + return nil + } case "logging": switch key { case "managerSinks": @@ -773,6 +1060,15 @@ func appendListValue(cfg *Config, section, key, value string) error { cfg.Docker.RegistryMirrors = append(cfg.Docker.RegistryMirrors, item) return nil } + case "dockerSandboxes": + switch key { + case "additionalAllow": + cfg.DockerSandboxes.AdditionalAllow = append(cfg.DockerSandboxes.AdditionalAllow, item) + return nil + case "additionalDeny": + cfg.DockerSandboxes.AdditionalDeny = append(cfg.DockerSandboxes.AdditionalDeny, item) + return nil + } case "logging": switch key { case "managerSinks": @@ -787,22 +1083,54 @@ func appendListValue(cfg *Config, section, key, value string) error { } func Validate(cfg Config) error { + if err := ValidateRunnerGroupSecurity(cfg.Security.RunnerGroup); err != nil { + return err + } if err := ValidateLogging(cfg.Logging); err != nil { return err } + if err := ValidateStorage(cfg.Storage); err != nil { + return err + } if cfg.Provider.Type == "" { return fmt.Errorf("provider.type is required") } switch cfg.Provider.Type { - case "tart", "wsl", "docker-dind": + case "tart", "wsl", "docker-container", "docker-sandboxes": case "docker-socket": - return fmt.Errorf("provider.type docker-socket is intentionally unsupported; use provider.type=docker-dind for a private Docker daemon") + return fmt.Errorf("provider.type docker-socket is intentionally unsupported; use provider.type=docker-container for a private Docker daemon") default: return fmt.Errorf("unsupported provider.type %q", cfg.Provider.Type) } - if cfg.Provider.SourceImage == "" { + if cfg.Provider.Type != "docker-sandboxes" && cfg.Provider.SourceImage == "" { return fmt.Errorf("provider.sourceImage is required") } + if cfg.Provider.Type == "docker-sandboxes" { + if cfg.Provider.SourceImage != "" { + return fmt.Errorf("provider.sourceImage is not supported with provider.type=docker-sandboxes; use image.sourceImage") + } + if cfg.Provider.Platform != "linux/amd64" && cfg.Provider.Platform != "linux/arm64" { + return fmt.Errorf("provider.platform must be linux/amd64 or linux/arm64 with provider.type=docker-sandboxes") + } + if cfg.Image.SourceType != "" && cfg.Image.SourceType != ImageSourceDockerImage { + return fmt.Errorf("image.sourceType must be docker-image with provider.type=docker-sandboxes") + } + if cfg.Image.SourceType != "" && !validDockerSandboxesSourceImage(cfg.Image.SourceImage) { + return fmt.Errorf("image.sourceImage with provider.type=docker-sandboxes must be an exact ghcr.io/catthehacker/ubuntu: reference") + } + if cfg.Image.SourcePlatform != "" && cfg.Image.SourcePlatform != cfg.Provider.Platform { + return fmt.Errorf("image.sourcePlatform must match provider.platform with provider.type=docker-sandboxes") + } + if !cfg.Runner.Ephemeral { + return fmt.Errorf("runner.ephemeral must be true with provider.type=docker-sandboxes") + } + if cfg.Security.RunnerGroup.Enforcement != RunnerGroupEnforcementEnforce { + return fmt.Errorf("security.runnerGroup.enforcement must be enforce with provider.type=docker-sandboxes") + } + if err := ValidateDockerSandboxes(cfg.DockerSandboxes); err != nil { + return err + } + } if cfg.Provider.RosettaTag != "" { if cfg.Provider.Type != "tart" { return fmt.Errorf("provider.rosettaTag is only supported with provider.type=tart") @@ -812,8 +1140,8 @@ func Validate(cfg Config) error { } } if cfg.Provider.Platform != "" { - if cfg.Provider.Type != "docker-dind" { - return fmt.Errorf("provider.platform is only supported with provider.type=docker-dind") + if cfg.Provider.Type != "docker-container" && cfg.Provider.Type != "docker-sandboxes" { + return fmt.Errorf("provider.platform is only supported with provider.type=docker-container or docker-sandboxes") } if err := ValidateDockerPlatform(cfg.Provider.Platform); err != nil { return err @@ -832,6 +1160,9 @@ func Validate(cfg Config) error { if err := ValidateHostTrust(cfg.Image, cfg.Provider, cfg.Runner); err != nil { return err } + if err := ValidateImageUpdatePolicy(cfg.Image); err != nil { + return err + } switch cfg.Image.SourceType { case "", ImageSourceDockerImage, ImageSourceRootFSTar: default: @@ -863,6 +1194,13 @@ func Validate(cfg Config) error { if err := ValidatePrefix(cfg.Pool.NamePrefix); err != nil { return err } + if cfg.Provider.Type == "docker-sandboxes" { + for _, r := range cfg.Pool.NamePrefix { + if !((r >= 'a' && r <= 'z') || (r >= '0' && r <= '9') || r == '-' || r == '.') { + return fmt.Errorf("pool.namePrefix for docker-sandboxes must contain only lowercase letters, digits, hyphens, and periods") + } + } + } if len(cfg.Runner.Labels) == 0 { return fmt.Errorf("runner.labels must not be empty") } @@ -888,6 +1226,292 @@ func Validate(cfg Config) error { return nil } +// ValidateImageUpdatePolicy keeps remote freshness scheduling predictable while +// allowing manual mode to retain an otherwise valid update time for later use. +func ValidateImageUpdatePolicy(image ImageConfig) error { + switch image.UpdateFrequency { + case ImageUpdateFrequencyDaily, ImageUpdateFrequencyWeekly, ImageUpdateFrequencyBiweekly, ImageUpdateFrequencyMonthly, ImageUpdateFrequencyManual: + default: + return fmt.Errorf("unsupported image.updateFrequency %q; supported values are daily, weekly, biweekly, monthly, and manual", image.UpdateFrequency) + } + if image.UpdateFrequency == ImageUpdateFrequencyManual { + return nil + } + if _, err := time.Parse("15:04", image.UpdateTime); err != nil { + return fmt.Errorf("invalid image.updateTime %q; use 24-hour HH:MM local time", image.UpdateTime) + } + return nil +} + +// ValidateStorage rejects policies that would silently disable capacity or +// leave a supposedly bounded cache without a usable limit. +func ValidateStorage(storage StorageConfig) error { + for key, value := range map[string]string{ + "minimumFree": storage.MinimumFree, + "buildCacheLimit": storage.BuildCacheLimit, + "goCacheLimit": storage.GoCacheLimit, + } { + if _, err := ParseByteSize(value); err != nil { + return fmt.Errorf("invalid storage.%s: %w", key, err) + } + } + grace, err := time.ParseDuration(storage.GracePeriod) + if err != nil { + return fmt.Errorf("invalid storage.gracePeriod: %w", err) + } + if grace <= 0 { + return fmt.Errorf("storage.gracePeriod must be greater than zero") + } + if storage.KeepPrevious < 0 { + return fmt.Errorf("storage.keepPrevious must be zero or greater") + } + switch storage.AutomaticHousekeeping { + case StorageHousekeepingConservative, StorageHousekeepingDisabled: + default: + return fmt.Errorf("unsupported storage.automaticHousekeeping %q; supported values are conservative and disabled", storage.AutomaticHousekeeping) + } + return nil +} + +// ValidateDockerSandboxes validates provider settings independently so +// callers constructing Config values programmatically receive the same checks as +// YAML-loaded configuration. +func ValidateDockerSandboxes(sandboxes DockerSandboxesConfig) error { + if err := validateSHA256Fingerprint("dockerSandboxes.policyGeneration", sandboxes.PolicyGeneration); err != nil { + return err + } + if sandboxes.NetworkBaseline != DockerSandboxesNetworkBaselineOpen && sandboxes.NetworkBaseline != DockerSandboxesNetworkBaselineBalanced { + return fmt.Errorf("unsupported dockerSandboxes.networkBaseline %q; supported values are open and balanced", sandboxes.NetworkBaseline) + } + if err := validateDockerSandboxHostnameList("additionalAllow", sandboxes.AdditionalAllow); err != nil { + return err + } + if err := validateDockerSandboxHostnameList("additionalDeny", sandboxes.AdditionalDeny); err != nil { + return err + } + allowSet := make(map[string]struct{}, len(sandboxes.AdditionalAllow)) + for _, resource := range sandboxes.AdditionalAllow { + allowSet[resource] = struct{}{} + } + if sandboxes.NetworkBaseline == DockerSandboxesNetworkBaselineOpen { + for _, resource := range DockerSandboxesOpenDefaultDenyResources() { + if _, exists := allowSet[resource]; exists { + return fmt.Errorf("dockerSandboxes.additionalAllow must not override the Open host-boundary deny for %q", resource) + } + } + } + for _, resource := range sandboxes.AdditionalDeny { + if _, exists := allowSet[resource]; exists { + return fmt.Errorf("dockerSandboxes.additionalAllow and dockerSandboxes.additionalDeny must not both contain %q", resource) + } + } + if err := validateDockerSandboxesStagingRoot(sandboxes.StagingRoot); err != nil { + return err + } + if sandboxes.CPUs <= 0 { + return fmt.Errorf("dockerSandboxes.cpus must be greater than zero") + } + parsedSizes := make(map[string]int64, 2) + for key, value := range map[string]string{ + "memory": sandboxes.Memory, + "dockerDisk": sandboxes.DockerDisk, + } { + parsed, err := ParseByteSize(value) + if err != nil { + return fmt.Errorf("invalid dockerSandboxes.%s: %w", key, err) + } + parsedSizes[key] = parsed + } + if sandboxes.RootDisk != DockerSandboxesAutomaticRootDisk { + rootDisk, err := ParseByteSize(sandboxes.RootDisk) + if err != nil { + return fmt.Errorf("invalid dockerSandboxes.rootDisk: %w", err) + } + if rootDisk < DockerSandboxesMinimumRootDiskBytes { + return fmt.Errorf("dockerSandboxes.rootDisk must be auto or at least 20GiB") + } + } + if parsedSizes["dockerDisk"] < DockerSandboxesMinimumDockerDiskBytes { + return fmt.Errorf("dockerSandboxes.dockerDisk must be at least 1GiB") + } + if sandboxes.MaxConcurrentCreates <= 0 { + return fmt.Errorf("dockerSandboxes.maxConcurrentCreates must be greater than zero") + } + return nil +} + +func validDockerSandboxesSourceImage(value string) bool { + const prefix = "ghcr.io/catthehacker/ubuntu:" + if !strings.HasPrefix(value, prefix) { + return false + } + tag := strings.TrimPrefix(value, prefix) + if tag == "" || len(tag) > 128 || strings.ContainsAny(tag, "/@\t\r\n ") { + return false + } + for _, character := range tag { + if (character >= 'a' && character <= 'z') || (character >= 'A' && character <= 'Z') || (character >= '0' && character <= '9') || character == '_' || character == '.' || character == '-' { + continue + } + return false + } + first := tag[0] + return (first >= 'a' && first <= 'z') || (first >= 'A' && first <= 'Z') || (first >= '0' && first <= '9') || first == '_' +} + +func validateDockerSandboxesStagingRoot(value string) error { + if value == "" || value != strings.TrimSpace(value) || strings.ContainsAny(value, "\x00\r\n:") { + return fmt.Errorf("dockerSandboxes.stagingRoot must be a canonical project-relative path under .local") + } + normalizedInput := strings.ReplaceAll(value, `\`, "/") + native := filepath.FromSlash(normalizedInput) + if filepath.IsAbs(native) || filepath.VolumeName(native) != "" { + return fmt.Errorf("dockerSandboxes.stagingRoot must not select an absolute host path") + } + clean := filepath.ToSlash(filepath.Clean(native)) + if clean != normalizedInput || clean == ".local" || !strings.HasPrefix(clean, ".local/") { + return fmt.Errorf("dockerSandboxes.stagingRoot must be a canonical project-relative path under .local") + } + for _, reserved := range []string{".local/bin", ".local/state"} { + if clean == reserved || strings.HasPrefix(clean, reserved+"/") { + return fmt.Errorf("dockerSandboxes.stagingRoot must not overlap reserved EPAR path %s", reserved) + } + } + return nil +} + +func validateSHA256Fingerprint(key, value string) error { + const prefix = "sha256:" + if len(value) != len(prefix)+64 || !strings.HasPrefix(value, prefix) { + return fmt.Errorf("%s must be a lowercase sha256:<64-hex> fingerprint", key) + } + for _, r := range value[len(prefix):] { + if !((r >= '0' && r <= '9') || (r >= 'a' && r <= 'f')) { + return fmt.Errorf("%s must be a lowercase sha256:<64-hex> fingerprint", key) + } + } + return nil +} + +func validateDockerSandboxHostnameList(key string, resources []string) error { + seen := make(map[string]struct{}, len(resources)) + for _, resource := range resources { + if err := ValidateDockerSandboxHostname(resource); err != nil { + return fmt.Errorf("invalid dockerSandboxes.%s value %q: %w", key, resource, err) + } + if _, exists := seen[resource]; exists { + return fmt.Errorf("dockerSandboxes.%s must not contain duplicate value %q", key, resource) + } + seen[resource] = struct{}{} + } + return nil +} + +// ValidateDockerSandboxHostname accepts an exact DNS hostname or a wildcard +// constrained to the left-most label (for example, *.githubusercontent.com), +// with an optional numeric port. +func ValidateDockerSandboxHostname(resource string) error { + if resource == "" || resource != strings.TrimSpace(resource) || strings.ContainsAny(resource, "\\/@?#[]") || strings.ContainsAny(resource, "\t\r\n") { + return fmt.Errorf("must be an exact hostname or *.domain with an optional port") + } + host := resource + if strings.Contains(resource, ":") { + parsedHost, port, err := net.SplitHostPort(resource) + if err != nil || parsedHost == "" || port == "" { + return fmt.Errorf("must be an exact hostname or *.domain with an optional port") + } + portNumber, err := strconv.Atoi(port) + if err != nil || portNumber < 1 || portNumber > 65535 { + return fmt.Errorf("port must be between 1 and 65535") + } + host = parsedHost + } + if strings.HasPrefix(host, "*.") { + host = strings.TrimPrefix(host, "*.") + if !strings.Contains(host, ".") { + return fmt.Errorf("wildcards must be followed by a domain") + } + } else if strings.Contains(host, "*") { + return fmt.Errorf("wildcards are only supported as the left-most label") + } + if net.ParseIP(host) != nil || len(host) > 253 || host == "" { + return fmt.Errorf("must be an exact hostname or *.domain with an optional port") + } + for _, label := range strings.Split(host, ".") { + if label == "" || len(label) > 63 || strings.HasPrefix(label, "-") || strings.HasSuffix(label, "-") { + return fmt.Errorf("must be an exact hostname or *.domain with an optional port") + } + for _, r := range label { + if !((r >= 'a' && r <= 'z') || (r >= '0' && r <= '9') || r == '-') { + return fmt.Errorf("must be an exact hostname or *.domain with an optional port") + } + } + } + return nil +} + +// ParseByteSize parses a positive byte count with an explicit binary unit. The +// original configuration string is retained because the sandbox command surface +// consumes these values verbatim. +func ParseByteSize(value string) (int64, error) { + if value == "" || value != strings.TrimSpace(value) { + return 0, fmt.Errorf("must be a positive byte size such as 4GiB") + } + units := []struct { + suffix string + multiplier int64 + }{ + {suffix: "TiB", multiplier: 1 << 40}, + {suffix: "GiB", multiplier: 1 << 30}, + {suffix: "MiB", multiplier: 1 << 20}, + {suffix: "KiB", multiplier: 1 << 10}, + {suffix: "B", multiplier: 1}, + } + for _, unit := range units { + if !strings.HasSuffix(value, unit.suffix) { + continue + } + number := strings.TrimSuffix(value, unit.suffix) + if number == "" { + break + } + parsed, err := strconv.ParseInt(number, 10, 64) + if err != nil || parsed <= 0 || parsed > math.MaxInt64/unit.multiplier { + return 0, fmt.Errorf("must be a positive byte size such as 4GiB") + } + return parsed * unit.multiplier, nil + } + return 0, fmt.Errorf("must be a positive byte size such as 4GiB") +} + +// EffectiveMinimumFreeBytes returns the single provider-neutral free-space +// reserve used by every storage-consuming provider operation. +func EffectiveMinimumFreeBytes(cfg Config) (uint64, error) { + value := cfg.Storage.MinimumFree + if value == "" { + value = Default().Storage.MinimumFree + } + minimum, err := ParseByteSize(value) + if err != nil { + return 0, fmt.Errorf("parse storage minimum free reserve: %w", err) + } + return uint64(minimum), nil +} + +func ValidateRunnerGroupSecurity(policy RunnerGroupSecurityConfig) error { + switch policy.Enforcement { + case RunnerGroupEnforcementEnforce, RunnerGroupEnforcementWarn: + default: + return fmt.Errorf("unsupported security.runnerGroup.enforcement %q; supported values are enforce and warn", policy.Enforcement) + } + switch policy.RequiredRepositoryAccess { + case RunnerGroupRepositoryAccessSelected, RunnerGroupRepositoryAccessPrivate, RunnerGroupRepositoryAccessAll: + default: + return fmt.Errorf("unsupported security.runnerGroup.requiredRepositoryAccess %q; supported values are selected, private, and all", policy.RequiredRepositoryAccess) + } + return nil +} + func ValidateLogging(logging LoggingConfig) error { if strings.TrimSpace(logging.Directory) == "" { return fmt.Errorf("logging.directory is required") @@ -1007,17 +1631,14 @@ func validateConsoleTextFormat(key, template, outputFormat string, allowed []str return nil } -// ValidateHostTrust keeps host trust inheritance deliberately limited to the -// ephemeral Docker-in-Docker image path. Other providers do not have a -// portable, unambiguous host trust boundary. -func ValidateHostTrust(image ImageConfig, provider ProviderConfig, runner RunnerConfig) error { +// ValidateHostTrust applies the provider-neutral ephemeral-runner trust +// contract. The common pool installs and validates the resolved trust snapshot +// in an unregistered guest before requesting a registration token. +func ValidateHostTrust(image ImageConfig, _ ProviderConfig, runner RunnerConfig) error { switch image.HostTrustMode { case "", HostTrustModeDisabled: return nil case HostTrustModeOverlay: - if provider.Type != "docker-dind" { - return fmt.Errorf("image.hostTrustMode %q is only supported with provider.type=docker-dind", HostTrustModeOverlay) - } if !runner.Ephemeral { return fmt.Errorf("image.hostTrustMode %q requires runner.ephemeral=true", HostTrustModeOverlay) } @@ -1164,6 +1785,20 @@ func ValidateDockerNoProxy(value string) error { return nil } +func DockerSandboxesGuestPlatform(hostOS, hostArch string) (string, error) { + if strings.TrimSpace(hostOS) == "" { + return "", fmt.Errorf("docker-sandboxes controller host operating system is empty") + } + switch hostArch { + case "amd64": + return "linux/amd64", nil + case "arm64": + return "linux/arm64", nil + default: + return "", fmt.Errorf("docker-sandboxes has no EPAR template for controller architecture %s on %s", hostArch, hostOS) + } +} + func ValidateDockerPlatform(platform string) error { if strings.TrimSpace(platform) != platform || platform == "" { return fmt.Errorf("provider.platform must be a non-empty Docker platform") @@ -1284,7 +1919,12 @@ func stripComment(s string) string { func trimQuotes(s string) string { s = strings.TrimSpace(s) if len(s) >= 2 { - if (s[0] == '"' && s[len(s)-1] == '"') || (s[0] == '\'' && s[len(s)-1] == '\'') { + if s[0] == '"' && s[len(s)-1] == '"' { + if value, err := strconv.Unquote(s); err == nil { + return value + } + } + if s[0] == '\'' && s[len(s)-1] == '\'' { return s[1 : len(s)-1] } } diff --git a/internal/config/config_test.go b/internal/config/config_test.go index f1c2407..f0be9aa 100644 --- a/internal/config/config_test.go +++ b/internal/config/config_test.go @@ -102,6 +102,90 @@ image: } } +func TestImageUpdatePolicyDefaults(t *testing.T) { + cfg := Default() + if got, want := cfg.Image.UpdateFrequency, ImageUpdateFrequencyWeekly; got != want { + t.Fatalf("Image.UpdateFrequency = %q, want %q", got, want) + } + if got, want := cfg.Image.UpdateTime, DefaultImageUpdateTime; got != want { + t.Fatalf("Image.UpdateTime = %q, want %q", got, want) + } + if err := ValidateImageUpdatePolicy(cfg.Image); err != nil { + t.Fatal(err) + } +} + +func TestLoadImageUpdatePolicy(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "config.yml") + content := "image:\n updateFrequency: biweekly\n updateTime: \"06:30\"\n" + if err := os.WriteFile(path, []byte(content), 0o600); err != nil { + t.Fatal(err) + } + cfg, err := Load(path) + if err != nil { + t.Fatal(err) + } + if got, want := cfg.Image.UpdateFrequency, ImageUpdateFrequencyBiweekly; got != want { + t.Fatalf("Image.UpdateFrequency = %q, want %q", got, want) + } + if got, want := cfg.Image.UpdateTime, "06:30"; got != want { + t.Fatalf("Image.UpdateTime = %q, want %q", got, want) + } + if slices.ContainsFunc(cfg.Warnings(), func(warning string) bool { + return strings.Contains(warning, "image update policy is not configured") + }) { + t.Fatalf("Warnings() = %#v, want no image-policy default notice", cfg.Warnings()) + } +} + +func TestLoadOmittedImageUpdatePolicyAddsNotice(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "config.yml") + if err := os.WriteFile(path, []byte("image:\n runnerVersion: latest\n"), 0o600); err != nil { + t.Fatal(err) + } + cfg, err := Load(path) + if err != nil { + t.Fatal(err) + } + if !slices.ContainsFunc(cfg.Warnings(), func(warning string) bool { + return strings.Contains(warning, "using weekly checks at 07:00 local time") + }) { + t.Fatalf("Warnings() = %#v, want image-policy default notice", cfg.Warnings()) + } +} + +func TestValidateImageUpdatePolicy(t *testing.T) { + for _, frequency := range []string{ + ImageUpdateFrequencyDaily, + ImageUpdateFrequencyWeekly, + ImageUpdateFrequencyBiweekly, + ImageUpdateFrequencyMonthly, + ImageUpdateFrequencyManual, + } { + image := Default().Image + image.UpdateFrequency = frequency + if err := ValidateImageUpdatePolicy(image); err != nil { + t.Fatalf("ValidateImageUpdatePolicy(%q): %v", frequency, err) + } + } + image := Default().Image + image.UpdateFrequency = "hourly" + if err := ValidateImageUpdatePolicy(image); err == nil || !strings.Contains(err.Error(), "unsupported image.updateFrequency") { + t.Fatalf("invalid frequency error = %v", err) + } + image = Default().Image + image.UpdateTime = "7am" + if err := ValidateImageUpdatePolicy(image); err == nil || !strings.Contains(err.Error(), "24-hour HH:MM") { + t.Fatalf("invalid time error = %v", err) + } + image.UpdateFrequency = ImageUpdateFrequencyManual + if err := ValidateImageUpdatePolicy(image); err != nil { + t.Fatalf("manual mode should ignore image.updateTime: %v", err) + } +} + func TestRunnerRegistrationControlsDefaultToDisabled(t *testing.T) { cfg := Default() if cfg.Runner.Group != "" { @@ -112,6 +196,152 @@ func TestRunnerRegistrationControlsDefaultToDisabled(t *testing.T) { } } +func TestStorageDefaultsAreBoundedAndConservative(t *testing.T) { + storage := Default().Storage + if storage.MinimumFree != "1GiB" || + storage.GracePeriod != "168h" || + storage.KeepPrevious != 0 || + storage.AutomaticHousekeeping != StorageHousekeepingConservative || + storage.BuildCacheLimit != "20GiB" || + storage.GoCacheLimit != "10GiB" { + t.Fatalf("unexpected storage defaults: %+v", storage) + } + if err := ValidateStorage(storage); err != nil { + t.Fatal(err) + } +} + +func TestLoadStoragePolicy(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "config.yml") + content := "storage:\n minimumFree: 30GiB\n gracePeriod: 72h\n keepPrevious: 1\n automaticHousekeeping: disabled\n buildCacheLimit: 80GiB\n goCacheLimit: 12GiB\n" + if err := os.WriteFile(path, []byte(content), 0644); err != nil { + t.Fatal(err) + } + cfg, err := Load(path) + if err != nil { + t.Fatal(err) + } + if got, want := cfg.Storage.MinimumFree, "30GiB"; got != want { + t.Fatalf("storage.minimumFree = %q, want %q", got, want) + } + if got, want := cfg.Storage.GracePeriod, "72h"; got != want { + t.Fatalf("storage.gracePeriod = %q, want %q", got, want) + } + if cfg.Storage.KeepPrevious != 1 || cfg.Storage.AutomaticHousekeeping != StorageHousekeepingDisabled { + t.Fatalf("unexpected loaded storage policy: %+v", cfg.Storage) + } + if err := ValidateStorage(cfg.Storage); err != nil { + t.Fatal(err) + } +} + +func TestValidateStorageRejectsUnboundedOrUnknownPolicy(t *testing.T) { + tests := []func(*StorageConfig){ + func(storage *StorageConfig) { storage.MinimumFree = "0GiB" }, + func(storage *StorageConfig) { storage.GracePeriod = "0h" }, + func(storage *StorageConfig) { storage.KeepPrevious = -1 }, + func(storage *StorageConfig) { storage.AutomaticHousekeeping = "aggressive" }, + func(storage *StorageConfig) { storage.BuildCacheLimit = "64GB" }, + func(storage *StorageConfig) { storage.GoCacheLimit = "" }, + } + for _, mutate := range tests { + storage := Default().Storage + mutate(&storage) + if err := ValidateStorage(storage); err == nil { + t.Fatalf("ValidateStorage accepted invalid policy: %+v", storage) + } + } +} + +func TestRunnerGroupSecurityDefaultsToStrictWarnings(t *testing.T) { + policy := Default().Security.RunnerGroup + if policy.Enforcement != RunnerGroupEnforcementWarn || + !policy.RequireExplicitGroup || + !policy.RequireNonDefaultGroup || + policy.RequiredRepositoryAccess != RunnerGroupRepositoryAccessSelected || + !policy.RequirePublicRepositoriesDisabled { + t.Fatalf("unexpected runner-group security defaults: %+v", policy) + } +} + +func TestLoadNestedRunnerGroupSecurity(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "config.yml") + content := `security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: false + requireNonDefaultGroup: false + requiredRepositoryAccess: all + requirePublicRepositoriesDisabled: false +` + if err := os.WriteFile(path, []byte(content), 0644); err != nil { + t.Fatal(err) + } + cfg, err := Load(path) + if err != nil { + t.Fatal(err) + } + policy := cfg.Security.RunnerGroup + if policy.Enforcement != RunnerGroupEnforcementEnforce || policy.RequireExplicitGroup || policy.RequireNonDefaultGroup || policy.RequiredRepositoryAccess != RunnerGroupRepositoryAccessAll || policy.RequirePublicRepositoriesDisabled { + t.Fatalf("unexpected parsed runner-group policy: %+v", policy) + } + for _, warning := range cfg.Warnings() { + if strings.Contains(warning, "runner-group security policy is not configured") { + t.Fatalf("unexpected migration warning for configured policy: %q", warning) + } + } +} + +func TestLoadLegacyConfigWarnsAboutRunnerGroupSecurity(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "config.yml") + if err := os.WriteFile(path, []byte("runner:\n group: existing-group\n"), 0644); err != nil { + t.Fatal(err) + } + cfg, err := Load(path) + if err != nil { + t.Fatal(err) + } + if !slices.ContainsFunc(cfg.Warnings(), func(warning string) bool { + return strings.Contains(warning, "runner-group security policy is not configured") && strings.Contains(warning, "warn mode") + }) { + t.Fatalf("Warnings() = %#v, want runner-group migration warning", cfg.Warnings()) + } +} + +func TestLoadRejectsInvalidRunnerGroupSecurityNesting(t *testing.T) { + for _, content := range []string{ + "security:\n unknown:\n enforcement: enforce\n", + "security:\n enforcement: enforce\n", + "security:\n runnerGroup:\n enforcement: enforce\n", + "security:\n runnerGroup:\n unknown: true\n", + } { + dir := t.TempDir() + path := filepath.Join(dir, "config.yml") + if err := os.WriteFile(path, []byte(content), 0644); err != nil { + t.Fatal(err) + } + if _, err := Load(path); err == nil { + t.Fatalf("Load(%q) succeeded, want nesting/key error", content) + } + } +} + +func TestValidateRunnerGroupSecurityRejectsInvalidValues(t *testing.T) { + for _, mutate := range []func(*RunnerGroupSecurityConfig){ + func(policy *RunnerGroupSecurityConfig) { policy.Enforcement = "off" }, + func(policy *RunnerGroupSecurityConfig) { policy.RequiredRepositoryAccess = "repositories" }, + } { + cfg := Default() + mutate(&cfg.Security.RunnerGroup) + if err := Validate(cfg); err == nil || !strings.Contains(err.Error(), "security.runnerGroup") { + t.Fatalf("Validate() error = %v, want runner-group policy error", err) + } + } +} + func TestLoggingDefaults(t *testing.T) { got := Default().Logging if got.Directory != "work/logs" || !slices.Equal(got.ManagerSinks, []string{"console"}) || got.ManagerConsoleFormat != "text" || got.ManagerFileFormat != "json" || !slices.Equal(got.TranscriptSinks, []string{"file"}) || got.TranscriptConsoleFormat != "text" { @@ -228,7 +458,9 @@ func TestLoadMigratesPoolLogDirInMemoryWithWarning(t *testing.T) { t.Fatalf("logging.directory = %q, want legacy value %q", got, want) } warnings := cfg.Warnings() - if len(warnings) != 1 || !strings.Contains(warnings[0], "pool.logDir is deprecated") || !strings.Contains(warnings[0], "logging.directory") { + if !slices.ContainsFunc(warnings, func(warning string) bool { + return strings.Contains(warning, "pool.logDir is deprecated") && strings.Contains(warning, "logging.directory") + }) { t.Fatalf("Warnings() = %#v, want migration warning", warnings) } if err := Validate(cfg); err != nil { @@ -254,8 +486,10 @@ func TestLoadRejectsUnknownSectionsAndKeys(t *testing.T) { "github:\n unknown: value\n", "image:\n unknown: value\n", "pool:\n unknown: value\n", + "storage:\n unknown: value\n", "logging:\n unknown: value\n", "runner:\n unknown: value\n", + "security:\n runnerGroup:\n unknown: value\n", "provider:\n unknown: value\n", "docker:\n unknown: value\n", "timeouts:\n unknown: 1\n", @@ -334,7 +568,7 @@ image: hostTrustMode: overlay hostTrustScopes: [system, user] provider: - type: docker-dind + type: docker-container `), 0644); err != nil { t.Fatal(err) } @@ -361,12 +595,11 @@ func TestValidateRejectsInvalidHostTrustConfigurations(t *testing.T) { provider string ephemeral bool }{ - {name: "unknown mode", mode: "mirror", scopes: []string{HostTrustScopeSystem}, provider: "docker-dind", ephemeral: true}, - {name: "wrong provider", mode: HostTrustModeOverlay, scopes: []string{HostTrustScopeSystem}, provider: "wsl", ephemeral: true}, - {name: "non-ephemeral", mode: HostTrustModeOverlay, scopes: []string{HostTrustScopeSystem}, provider: "docker-dind"}, - {name: "empty scopes", mode: HostTrustModeOverlay, provider: "docker-dind", ephemeral: true}, - {name: "unknown scope", mode: HostTrustModeOverlay, scopes: []string{"global"}, provider: "docker-dind", ephemeral: true}, - {name: "duplicate scope", mode: HostTrustModeOverlay, scopes: []string{HostTrustScopeSystem, HostTrustScopeSystem}, provider: "docker-dind", ephemeral: true}, + {name: "unknown mode", mode: "mirror", scopes: []string{HostTrustScopeSystem}, provider: "docker-container", ephemeral: true}, + {name: "non-ephemeral", mode: HostTrustModeOverlay, scopes: []string{HostTrustScopeSystem}, provider: "docker-container"}, + {name: "empty scopes", mode: HostTrustModeOverlay, provider: "docker-container", ephemeral: true}, + {name: "unknown scope", mode: HostTrustModeOverlay, scopes: []string{"global"}, provider: "docker-container", ephemeral: true}, + {name: "duplicate scope", mode: HostTrustModeOverlay, scopes: []string{HostTrustScopeSystem, HostTrustScopeSystem}, provider: "docker-container", ephemeral: true}, } { t.Run(test.name, func(t *testing.T) { cfg := Default() @@ -382,6 +615,28 @@ func TestValidateRejectsInvalidHostTrustConfigurations(t *testing.T) { } } +func TestValidateHostTrustAllowsDockerSandboxesOverlay(t *testing.T) { + cfg := validDockerSandboxesConfig() + cfg.Image.HostTrustMode = HostTrustModeOverlay + cfg.Image.HostTrustScopes = []string{HostTrustScopeSystem} + if err := Validate(cfg); err != nil { + t.Fatalf("Docker Sandboxes host trust overlay rejected: %v", err) + } +} + +func TestValidateHostTrustAllowsWSLOverlay(t *testing.T) { + cfg := Default() + cfg.Provider.Type = "wsl" + cfg.Provider.SourceImage = "runner-image.tar" + cfg.Provider.InstallRoot = "work/wsl" + cfg.Runner.Ephemeral = true + cfg.Image.HostTrustMode = HostTrustModeOverlay + cfg.Image.HostTrustScopes = []string{HostTrustScopeSystem} + if err := Validate(cfg); err != nil { + t.Fatalf("WSL host trust overlay rejected: %v", err) + } +} + func TestRunnerHostLabelDefaultsToEnabled(t *testing.T) { oldHostname := osHostname osHostname = func() (string, error) { return "Build Box_01.example", nil } @@ -392,7 +647,7 @@ func TestRunnerHostLabelDefaultsToEnabled(t *testing.T) { path := filepath.Join(dir, "config.yml") if err := os.WriteFile(path, []byte(` provider: - type: docker-dind + type: docker-container `), 0644); err != nil { t.Fatal(err) } @@ -415,7 +670,7 @@ func TestRunnerHostLabelPrefersHostNameEnv(t *testing.T) { path := filepath.Join(dir, "config.yml") if err := os.WriteFile(path, []byte(` provider: - type: docker-dind + type: docker-container `), 0644); err != nil { t.Fatal(err) } @@ -440,7 +695,7 @@ func TestRunnerHostLabelCanBeDisabled(t *testing.T) { runner: includeHostLabel: false provider: - type: docker-dind + type: docker-container `), 0644); err != nil { t.Fatal(err) } @@ -467,7 +722,7 @@ func TestRunnerHostLabelDoesNotDuplicateExistingLabel(t *testing.T) { runner: labels: [self-hosted, linux, epar-host-build-box] provider: - type: docker-dind + type: docker-container `), 0644); err != nil { t.Fatal(err) } @@ -517,18 +772,18 @@ func TestSanitizeNamePart(t *testing.T) { } } -func TestLoadDockerDindPlatform(t *testing.T) { +func TestLoadDockerContainerPlatform(t *testing.T) { dir := t.TempDir() path := filepath.Join(dir, "config.yml") if err := os.WriteFile(path, []byte(` pool: instances: 1 - namePrefix: epar-dind + namePrefix: epar-docker-container runner: - labels: [self-hosted, linux, ARM64, epar-docker-dind] + labels: [self-hosted, linux, ARM64, epar-docker-container] provider: - type: docker-dind - sourceImage: epar-docker-dind-ubuntu-24 + type: docker-container + sourceImage: epar-docker-container-ubuntu-24 platform: linux/arm64 docker: registryMirrors: @@ -562,10 +817,10 @@ docker: t.Fatal(err) } if !DockerRegistryMirrorsNeedHostGateway(cfg.Docker.RegistryMirrors) { - t.Fatal("host.docker.internal mirror should request docker-dind host gateway") + t.Fatal("host.docker.internal mirror should request docker-container host gateway") } if !DockerConfigNeedsHostGateway(cfg.Docker) { - t.Fatal("host.docker.internal Docker config should request docker-dind host gateway") + t.Fatal("host.docker.internal Docker config should request docker-container host gateway") } } @@ -702,12 +957,12 @@ provider: } } -func TestProviderDefaultsForMinimalDockerDindConfig(t *testing.T) { +func TestProviderDefaultsForMinimalDockerContainerConfig(t *testing.T) { dir := t.TempDir() path := filepath.Join(dir, "config.yml") if err := os.WriteFile(path, []byte(` provider: - type: docker-dind + type: docker-container `), 0644); err != nil { t.Fatal(err) } @@ -721,16 +976,16 @@ provider: if got, want := cfg.Image.SourceImage, "ghcr.io/catthehacker/ubuntu:full-latest"; got != want { t.Fatalf("image.sourceImage = %q, want %q", got, want) } - if got, want := cfg.Image.OutputImage, "epar-docker-dind-catthehacker-ubuntu"; got != want { + if got, want := cfg.Image.OutputImage, "epar-docker-container-catthehacker-ubuntu"; got != want { t.Fatalf("image.outputImage = %q, want %q", got, want) } if got, want := cfg.Provider.SourceImage, cfg.Image.OutputImage; got != want { t.Fatalf("provider.sourceImage = %q, want %q", got, want) } - if got, want := cfg.Pool.NamePrefix, "epar-dind"; got != want { + if got, want := cfg.Pool.NamePrefix, "epar-docker-container"; got != want { t.Fatalf("pool.namePrefix = %q, want %q", got, want) } - if got, want := cfg.Runner.Labels[2], "epar-docker-dind-catthehacker-ubuntu"; got != want { + if got, want := cfg.Runner.Labels[2], "epar-docker-container-catthehacker-ubuntu"; got != want { t.Fatalf("runner label = %q, want %q", got, want) } if err := Validate(cfg); err != nil { @@ -790,7 +1045,7 @@ func TestValidateRosettaTag(t *testing.T) { func TestValidateDockerPlatform(t *testing.T) { cfg := Default() - cfg.Provider.Type = "docker-dind" + cfg.Provider.Type = "docker-container" cfg.Provider.SourceImage = "runner-image" cfg.Provider.Platform = "linux/amd64" if err := Validate(cfg); err != nil { @@ -799,7 +1054,7 @@ func TestValidateDockerPlatform(t *testing.T) { for _, platform := range []string{"bad platform", "-linux/amd64", "linux/$bad"} { cfg := Default() - cfg.Provider.Type = "docker-dind" + cfg.Provider.Type = "docker-container" cfg.Provider.SourceImage = "runner-image" cfg.Provider.Platform = platform if err := Validate(cfg); err == nil { @@ -874,7 +1129,7 @@ func TestValidateDockerDaemonProxy(t *testing.T) { func TestDockerConfigNeedsHostGatewayForProxy(t *testing.T) { if !DockerConfigNeedsHostGateway(DockerConfig{HTTPSProxy: "http://host.docker.internal:3128"}) { - t.Fatal("host.docker.internal proxy should request docker-dind host gateway") + t.Fatal("host.docker.internal proxy should request docker-container host gateway") } if DockerConfigNeedsHostGateway(DockerConfig{HTTPSProxy: "http://http.docker.internal:3128"}) { t.Fatal("http.docker.internal proxy should not request host.docker.internal mapping") @@ -889,7 +1144,7 @@ func TestValidateRejectsDockerSocketProvider(t *testing.T) { if err == nil { t.Fatal("docker-socket provider accepted") } - if got := err.Error(); got != "provider.type docker-socket is intentionally unsupported; use provider.type=docker-dind for a private Docker daemon" { + if got := err.Error(); got != "provider.type docker-socket is intentionally unsupported; use provider.type=docker-container for a private Docker daemon" { t.Fatalf("error = %q", got) } } @@ -1047,3 +1302,297 @@ image: t.Fatal("Load accepted image.profile") } } + +func TestDockerSandboxesRejectsNamePrefixOutsideProviderGrammar(t *testing.T) { + cfg := Default() + cfg.Provider.Type = "docker-sandboxes" + cfg.Provider.Platform = "linux/amd64" + cfg.Provider.SourceImage = "" + cfg.Pool.NamePrefix = "EPAR_sandbox" + cfg.Runner.Ephemeral = true + cfg.Security.RunnerGroup.Enforcement = RunnerGroupEnforcementEnforce + cfg.DockerSandboxes.PolicyGeneration = "sha256:" + strings.Repeat("b", 64) + cfg.DockerSandboxes.RootDisk = "120GiB" + cfg.DockerSandboxes.DockerDisk = "50GiB" + if err := Validate(cfg); err == nil || !strings.Contains(err.Error(), "lowercase") { + t.Fatalf("Validate() error = %v, want Docker Sandboxes prefix grammar rejection", err) + } +} + +func TestLoadDockerSandboxesConfig(t *testing.T) { + t.Setenv(HostNameEnv, "sandbox-preview-host") + dir := t.TempDir() + path := filepath.Join(dir, "docker-sandboxes.yml") + if err := os.WriteFile(path, []byte(` +provider: + type: docker-sandboxes +image: + sourceType: docker-image + sourceImage: ghcr.io/catthehacker/ubuntu:full-latest + sourcePlatform: linux/amd64 +runner: + ephemeral: true +security: + runnerGroup: + enforcement: enforce +dockerSandboxes: + policyGeneration: sha256:abcdef0123456789abcdef0123456789abcdef0123456789abcdef0123456789 + networkBaseline: balanced + additionalAllow: [api.github.com, '*.githubusercontent.com:443'] + additionalDeny: + - telemetry.example.invalid + stagingRoot: .local/docker-sandboxes + cpus: 2 + memory: 4GiB + rootDisk: auto + dockerDisk: 50GiB + maxConcurrentCreates: 1 +`), 0644); err != nil { + t.Fatal(err) + } + cfg, err := Load(path) + if err != nil { + t.Fatal(err) + } + if got, want := cfg.Provider.SourceImage, ""; got != want { + t.Fatalf("provider.sourceImage = %q, want empty for docker-sandboxes", got) + } + if got, want := cfg.Provider.Platform, "linux/amd64"; got != want { + t.Fatalf("provider.platform = %q, want %q", got, want) + } + if got, want := cfg.Pool.NamePrefix, "epar-docker-sandboxes"; got != want { + t.Fatalf("pool.namePrefix = %q, want %q", got, want) + } + if got, want := cfg.Runner.Labels, []string{"self-hosted", "linux", "X64", "epar-docker-sandboxes", "epar-host-sandbox-preview-host"}; !slices.Equal(got, want) { + t.Fatalf("runner.labels = %#v, want %#v", got, want) + } + if got, want := cfg.DockerSandboxes.Memory, "4GiB"; got != want { + t.Fatalf("dockerSandboxes.memory = %q, want %q", got, want) + } + if got, want := cfg.DockerSandboxes.AdditionalAllow, []string{"api.github.com", "*.githubusercontent.com:443"}; !slices.Equal(got, want) { + t.Fatalf("dockerSandboxes.additionalAllow = %#v, want %#v", got, want) + } + if err := Validate(cfg); err != nil { + t.Fatal(err) + } +} + +func TestLoadDockerSandboxesRejectsRemovedHostReserve(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "docker-sandboxes.yml") + if err := os.WriteFile(path, []byte("provider:\n type: docker-sandboxes\ndockerSandboxes:\n minHostFreeSpace: 50GiB\n"), 0644); err != nil { + t.Fatal(err) + } + _, err := Load(path) + if err == nil || !strings.Contains(err.Error(), "remove it and configure the provider-neutral storage.minimumFree value instead") { + t.Fatalf("Load() error = %v, want exact regeneration guidance", err) + } +} + +func TestValidateDockerSandboxesRejectsInvalidPreviewConfiguration(t *testing.T) { + tests := []struct { + name string + mutate func(*Config) + }{ + { + name: "source image is unsupported", + mutate: func(cfg *Config) { cfg.Provider.SourceImage = "runner-image" }, + }, + { + name: "platform is not a supported sandbox guest", + mutate: func(cfg *Config) { cfg.Provider.Platform = "linux/s390x" }, + }, + { + name: "runner is persistent", + mutate: func(cfg *Config) { cfg.Runner.Ephemeral = false }, + }, + { + name: "runner group enforcement is not fail closed", + mutate: func(cfg *Config) { cfg.Security.RunnerGroup.Enforcement = RunnerGroupEnforcementWarn }, + }, + { + name: "policy generation is not content addressed", + mutate: func(cfg *Config) { cfg.DockerSandboxes.PolicyGeneration = "verified-balanced-policy-fingerprint" }, + }, + { + name: "network baseline is unsupported", + mutate: func(cfg *Config) { cfg.DockerSandboxes.NetworkBaseline = "locked-down" }, + }, + { + name: "allowlist wildcard is unsafe", + mutate: func(cfg *Config) { cfg.DockerSandboxes.AdditionalAllow = []string{"**.example.test"} }, + }, + { + name: "allowlist contains a URL", + mutate: func(cfg *Config) { cfg.DockerSandboxes.AdditionalAllow = []string{"https://example.test"} }, + }, + { + name: "allowlist overlaps denylist", + mutate: func(cfg *Config) { cfg.DockerSandboxes.AdditionalDeny = []string{"api.github.com"} }, + }, + { + name: "open allowlist overrides host boundary", + mutate: func(cfg *Config) { cfg.DockerSandboxes.AdditionalAllow = []string{"host.docker.internal"} }, + }, + { + name: "cpus are not positive", + mutate: func(cfg *Config) { cfg.DockerSandboxes.CPUs = 0 }, + }, + { + name: "memory is not a positive byte size", + mutate: func(cfg *Config) { cfg.DockerSandboxes.Memory = "0GiB" }, + }, + { + name: "root disk is below the hard minimum", + mutate: func(cfg *Config) { cfg.DockerSandboxes.RootDisk = "19GiB" }, + }, + { + name: "docker disk is below the hard minimum", + mutate: func(cfg *Config) { cfg.DockerSandboxes.DockerDisk = "512MiB" }, + }, + { + name: "concurrency is not positive", + mutate: func(cfg *Config) { cfg.DockerSandboxes.MaxConcurrentCreates = 0 }, + }, + { + name: "staging root is an absolute host path", + mutate: func(cfg *Config) { cfg.DockerSandboxes.StagingRoot = `C:\epar-staging` }, + }, + { + name: "staging root is an absolute Unix host path", + mutate: func(cfg *Config) { cfg.DockerSandboxes.StagingRoot = "/tmp/epar-staging" }, + }, + { + name: "staging root escapes project local state", + mutate: func(cfg *Config) { cfg.DockerSandboxes.StagingRoot = "../epar-staging" }, + }, + { + name: "staging root overlaps ledger", + mutate: func(cfg *Config) { cfg.DockerSandboxes.StagingRoot = ".local/state/docker-sandboxes" }, + }, + { + name: "staging root overlaps native binary cache", + mutate: func(cfg *Config) { cfg.DockerSandboxes.StagingRoot = ".local/bin/staging" }, + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + cfg := validDockerSandboxesConfig() + test.mutate(&cfg) + if err := Validate(cfg); err == nil { + t.Fatal("Validate accepted invalid docker-sandboxes configuration") + } + }) + } +} + +func TestValidateDockerSandboxesAcceptsARM64Configuration(t *testing.T) { + cfg := validDockerSandboxesConfig() + cfg.Provider.Platform = "linux/arm64" + if err := Validate(cfg); err != nil { + t.Fatalf("Validate() rejected linux/arm64 Docker Sandboxes configuration: %v", err) + } +} + +func TestDockerSandboxesGuestPlatform(t *testing.T) { + tests := []struct { + hostOS string + hostArch string + want string + wantErr bool + }{ + {hostOS: "windows", hostArch: "amd64", want: "linux/amd64"}, + {hostOS: "linux", hostArch: "amd64", want: "linux/amd64"}, + {hostOS: "darwin", hostArch: "arm64", want: "linux/arm64"}, + {hostOS: "windows", hostArch: "arm64", want: "linux/arm64"}, + {hostOS: "linux", hostArch: "arm64", want: "linux/arm64"}, + {hostOS: "darwin", hostArch: "amd64", want: "linux/amd64"}, + {hostOS: "futureos", hostArch: "amd64", want: "linux/amd64"}, + {hostOS: "futureos", hostArch: "arm64", want: "linux/arm64"}, + {hostOS: "futureos", hostArch: "386", wantErr: true}, + {hostOS: "", hostArch: "amd64", wantErr: true}, + } + for _, test := range tests { + t.Run(test.hostOS+"_"+test.hostArch, func(t *testing.T) { + got, err := DockerSandboxesGuestPlatform(test.hostOS, test.hostArch) + if test.wantErr { + if err == nil { + t.Fatalf("DockerSandboxesGuestPlatform(%q, %q) = %q, nil; want error", test.hostOS, test.hostArch, got) + } + return + } + if err != nil { + t.Fatalf("DockerSandboxesGuestPlatform(%q, %q) error = %v", test.hostOS, test.hostArch, err) + } + if got != test.want { + t.Fatalf("DockerSandboxesGuestPlatform(%q, %q) = %q, want %q", test.hostOS, test.hostArch, got, test.want) + } + }) + } +} + +func TestValidateDockerSandboxHostname(t *testing.T) { + for _, hostname := range []string{"api.github.com", "api.github.com:443", "*.githubusercontent.com", "*.githubusercontent.com:443"} { + if err := ValidateDockerSandboxHostname(hostname); err != nil { + t.Fatalf("hostname %q rejected: %v", hostname, err) + } + } + for _, hostname := range []string{"**.githubusercontent.com", "api.*.github.com", "https://api.github.com", "api.github.com/path", "api.github.com:0", "api.github.com:65536", "127.0.0.1", "[::1]:443"} { + if err := ValidateDockerSandboxHostname(hostname); err == nil { + t.Fatalf("hostname %q accepted", hostname) + } + } +} + +func TestParseByteSize(t *testing.T) { + if got, err := ParseByteSize("4GiB"); err != nil || got != 4*(1<<30) { + t.Fatalf("ParseByteSize(4GiB) = %d, %v", got, err) + } + for _, value := range []string{"", " 4GiB", "4GB", "0GiB", "-1GiB", "999999999999999999999TiB"} { + if _, err := ParseByteSize(value); err == nil { + t.Fatalf("ParseByteSize(%q) accepted invalid value", value) + } + } +} + +func TestEffectiveMinimumFreeBytesUsesCommonReserve(t *testing.T) { + cfg := Default() + cfg.Provider.Type = "docker-sandboxes" + cfg.Storage.MinimumFree = "20GiB" + + got, err := EffectiveMinimumFreeBytes(cfg) + if err != nil { + t.Fatal(err) + } + if want := uint64(20 << 30); got != want { + t.Fatalf("EffectiveMinimumFreeBytes() = %d, want %d", got, want) + } +} + +func TestEffectiveMinimumFreeBytesKeepsStricterCommonReserve(t *testing.T) { + cfg := Default() + cfg.Provider.Type = "docker-sandboxes" + cfg.Storage.MinimumFree = "60GiB" + + got, err := EffectiveMinimumFreeBytes(cfg) + if err != nil { + t.Fatal(err) + } + if want := uint64(60 << 30); got != want { + t.Fatalf("EffectiveMinimumFreeBytes() = %d, want %d", got, want) + } +} + +func validDockerSandboxesConfig() Config { + cfg := Default() + cfg.Provider.Type = "docker-sandboxes" + cfg.Provider.SourceImage = "" + cfg.Provider.Platform = "linux/amd64" + cfg.Runner.Ephemeral = true + cfg.Security.RunnerGroup.Enforcement = RunnerGroupEnforcementEnforce + cfg.DockerSandboxes.PolicyGeneration = "sha256:abcdef0123456789abcdef0123456789abcdef0123456789abcdef0123456789" + cfg.DockerSandboxes.AdditionalAllow = []string{"api.github.com"} + cfg.DockerSandboxes.RootDisk = "120GiB" + cfg.DockerSandboxes.DockerDisk = "50GiB" + return cfg +} diff --git a/internal/filelock/filelock.go b/internal/filelock/filelock.go index 40bd1b2..0b9c58e 100644 --- a/internal/filelock/filelock.go +++ b/internal/filelock/filelock.go @@ -4,6 +4,7 @@ package filelock import ( "errors" "fmt" + "io" "os" "sync" ) @@ -38,6 +39,29 @@ func Acquire(path string) (*Lock, error) { return &Lock{file: file}, nil } +// ReplaceContent atomically with respect to this lock replaces the lock-file +// payload while retaining ownership of the platform lock. This matters on +// Windows, where writing the locked byte range through a second file handle +// can fail even when both handles belong to the same process. +func (lock *Lock) ReplaceContent(content []byte) error { + if lock == nil || lock.file == nil { + return errors.New("file lock is not open") + } + if err := lock.file.Truncate(0); err != nil { + return fmt.Errorf("truncate lock file: %w", err) + } + if _, err := lock.file.Seek(0, io.SeekStart); err != nil { + return fmt.Errorf("seek lock file: %w", err) + } + if _, err := lock.file.Write(content); err != nil { + return fmt.Errorf("write lock file: %w", err) + } + if err := lock.file.Sync(); err != nil { + return fmt.Errorf("sync lock file: %w", err) + } + return nil +} + // Close releases the lock. func (lock *Lock) Close() error { if lock == nil { diff --git a/internal/filelock/filelock_test.go b/internal/filelock/filelock_test.go index 94c4efb..b6a0e4f 100644 --- a/internal/filelock/filelock_test.go +++ b/internal/filelock/filelock_test.go @@ -57,3 +57,26 @@ func TestCloseAllowsReacquireAndIsIdempotent(t *testing.T) { t.Fatalf("Close second: %v", err) } } + +func TestReplaceContentUsesHeldLockFile(t *testing.T) { + path := filepath.Join(t.TempDir(), "metadata.lock") + lock, err := Acquire(path) + if err != nil { + t.Fatalf("Acquire: %v", err) + } + defer lock.Close() + + if err := lock.ReplaceContent([]byte("first payload that is longer\n")); err != nil { + t.Fatalf("ReplaceContent first: %v", err) + } + if err := lock.ReplaceContent([]byte("short\n")); err != nil { + t.Fatalf("ReplaceContent second: %v", err) + } + content, err := os.ReadFile(path) + if err != nil { + t.Fatalf("ReadFile: %v", err) + } + if got, want := string(content), "short\n"; got != want { + t.Fatalf("lock metadata = %q, want %q", got, want) + } +} diff --git a/internal/filelock/filelock_windows.go b/internal/filelock/filelock_windows.go index e9db7b8..d55c165 100644 --- a/internal/filelock/filelock_windows.go +++ b/internal/filelock/filelock_windows.go @@ -12,6 +12,7 @@ import ( const ( lockfileFailImmediately = 0x00000001 lockfileExclusiveLock = 0x00000002 + lockfileOffsetHigh = 1 ) var ( @@ -22,7 +23,10 @@ var ( ) func lockFile(file *os.File) error { - var overlapped syscall.Overlapped + // Keep the lock range outside the metadata payload. Windows enforces byte + // range locks for ordinary reads, so locking byte zero would prevent another + // process from reading owner diagnostics while the lock is held. + overlapped := syscall.Overlapped{OffsetHigh: lockfileOffsetHigh} result, _, callErr := procLockFileEx.Call(file.Fd(), lockfileExclusiveLock|lockfileFailImmediately, 0, 1, 0, uintptr(unsafe.Pointer(&overlapped))) if result != 0 { return nil @@ -34,7 +38,7 @@ func lockFile(file *os.File) error { } func unlockFile(file *os.File) error { - var overlapped syscall.Overlapped + overlapped := syscall.Overlapped{OffsetHigh: lockfileOffsetHigh} result, _, callErr := procUnlockFileEx.Call(file.Fd(), 0, 1, 0, uintptr(unsafe.Pointer(&overlapped))) if result != 0 { return nil diff --git a/internal/github/client.go b/internal/github/client.go index 6ccc2dd..d2f3b06 100644 --- a/internal/github/client.go +++ b/internal/github/client.go @@ -10,6 +10,7 @@ import ( "net/http" "net/url" "strings" + "sync" "time" "github.com/solutionforest/ephemeral-action-runner/internal/config" @@ -18,6 +19,7 @@ import ( type Client struct { cfg config.GitHubConfig httpClient *http.Client + tokenMu sync.Mutex token string tokenExpires time.Time } @@ -175,6 +177,9 @@ func (c *Client) installationRequest(ctx context.Context, method, path string, b } func (c *Client) installationToken(ctx context.Context) (string, error) { + c.tokenMu.Lock() + defer c.tokenMu.Unlock() + if c.token != "" && time.Now().Before(c.tokenExpires.Add(-2*time.Minute)) { return c.token, nil } diff --git a/internal/github/client_test.go b/internal/github/client_test.go index b0c87d6..c40045e 100644 --- a/internal/github/client_test.go +++ b/internal/github/client_test.go @@ -6,10 +6,13 @@ import ( "crypto/rsa" "crypto/x509" "encoding/pem" + "fmt" "io" "net/http" "os" "strings" + "sync" + "sync/atomic" "testing" "time" @@ -88,6 +91,65 @@ func TestListRunnersUsesInstallationToken(t *testing.T) { } } +func TestInstallationTokenCacheIsConcurrentAndSingleFlight(t *testing.T) { + keyPath := writeKey(t) + client := New(config.GitHubConfig{ + AppID: 123, + Organization: "example", + PrivateKeyPath: keyPath, + APIBaseURL: "https://api.github.test", + WebBaseURL: "https://github.test", + }) + var installationRequests atomic.Int32 + var tokenRequests atomic.Int32 + client.httpClient = &http.Client{Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) { + var body string + switch r.URL.Path { + case "/orgs/example/installation": + installationRequests.Add(1) + body = `{"id":42}` + case "/app/installations/42/access_tokens": + tokenRequests.Add(1) + body = `{"token":"installation-token","expires_at":"2099-01-01T00:00:00Z"}` + default: + return nil, fmt.Errorf("unexpected path %s", r.URL.Path) + } + return &http.Response{ + StatusCode: http.StatusOK, + Header: make(http.Header), + Body: io.NopCloser(strings.NewReader(body)), + }, nil + })} + + const callers = 32 + errorsCh := make(chan error, callers) + var wait sync.WaitGroup + wait.Add(callers) + for index := 0; index < callers; index++ { + go func() { + defer wait.Done() + token, err := client.installationToken(context.Background()) + if err == nil && token != "installation-token" { + err = fmt.Errorf("installationToken() = %q", token) + } + errorsCh <- err + }() + } + wait.Wait() + close(errorsCh) + for err := range errorsCh { + if err != nil { + t.Fatal(err) + } + } + if got := installationRequests.Load(); got != 1 { + t.Fatalf("installation lookup requests = %d, want 1", got) + } + if got := tokenRequests.Load(); got != 1 { + t.Fatalf("access-token requests = %d, want 1", got) + } +} + func TestWaitRunnerOnlineAcceptsBusyRunner(t *testing.T) { keyPath := writeKey(t) client := New(config.GitHubConfig{ diff --git a/internal/github/runner_groups.go b/internal/github/runner_groups.go new file mode 100644 index 0000000..59f3953 --- /dev/null +++ b/internal/github/runner_groups.go @@ -0,0 +1,175 @@ +package github + +import ( + "context" + "fmt" + "net/http" + "net/url" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" +) + +type RunnerGroup struct { + ID int64 `json:"id"` + Name string `json:"name"` + Visibility string `json:"visibility"` + Default bool `json:"default"` + Inherited bool `json:"inherited"` + AllowsPublicRepositories bool `json:"allows_public_repositories"` +} + +type RunnerGroupRepository struct { + ID int64 `json:"id"` + Name string `json:"name"` + FullName string `json:"full_name"` + Private bool `json:"private"` +} + +type RunnerGroupPolicyResult struct { + Group RunnerGroup + Resolved bool + Repositories []RunnerGroupRepository + Violations []string + Advisories []string +} + +func (r RunnerGroupPolicyResult) Allowed() bool { + return len(r.Violations) == 0 +} + +func (c *Client) ListRunnerGroups(ctx context.Context) ([]RunnerGroup, error) { + var all []RunnerGroup + for page := 1; ; page++ { + var response struct { + TotalCount int `json:"total_count"` + RunnerGroups []RunnerGroup `json:"runner_groups"` + } + path := fmt.Sprintf("/orgs/%s/actions/runner-groups?per_page=100&page=%d", url.PathEscape(c.cfg.Organization), page) + if err := c.installationRequest(ctx, http.MethodGet, path, nil, &response); err != nil { + return nil, fmt.Errorf("list organization runner groups: %w", err) + } + all = append(all, response.RunnerGroups...) + if len(response.RunnerGroups) < 100 { + return all, nil + } + } +} + +func (c *Client) ListRunnerGroupRepositories(ctx context.Context, groupID int64) ([]RunnerGroupRepository, error) { + var all []RunnerGroupRepository + for page := 1; ; page++ { + var response struct { + TotalCount int `json:"total_count"` + Repositories []RunnerGroupRepository `json:"repositories"` + } + path := fmt.Sprintf("/orgs/%s/actions/runner-groups/%d/repositories?per_page=100&page=%d", url.PathEscape(c.cfg.Organization), groupID, page) + if err := c.installationRequest(ctx, http.MethodGet, path, nil, &response); err != nil { + return nil, fmt.Errorf("list repositories for runner group id %d: %w", groupID, err) + } + all = append(all, response.Repositories...) + if len(response.Repositories) < 100 { + return all, nil + } + } +} + +func (c *Client) EvaluateRunnerGroupPolicy(ctx context.Context, configuredGroup string, policy config.RunnerGroupSecurityConfig) (RunnerGroupPolicyResult, error) { + groups, err := c.ListRunnerGroups(ctx) + if err != nil { + return RunnerGroupPolicyResult{}, err + } + result := ResolveRunnerGroup(configuredGroup, policy, groups) + if !result.Resolved || result.Group.Visibility != config.RunnerGroupRepositoryAccessSelected { + return result, nil + } + repositories, err := c.ListRunnerGroupRepositories(ctx, result.Group.ID) + if err != nil { + return RunnerGroupPolicyResult{}, err + } + return EvaluateRunnerGroupPolicy(result.Group, repositories, policy), nil +} + +func ResolveRunnerGroup(configuredGroup string, policy config.RunnerGroupSecurityConfig, groups []RunnerGroup) RunnerGroupPolicyResult { + name := configuredGroup + if strings.TrimSpace(name) == "" { + if policy.RequireExplicitGroup { + return RunnerGroupPolicyResult{Violations: []string{"runner.group is required by security.runnerGroup.requireExplicitGroup"}} + } + for _, group := range groups { + if group.Default { + result := EvaluateRunnerGroupPolicy(group, nil, policy) + result.Advisories = append(result.Advisories, "runner.group is empty; GitHub's default runner group will be used") + return result + } + } + return RunnerGroupPolicyResult{Violations: []string{"runner.group is empty and GitHub did not return a default runner group"}} + } + for _, group := range groups { + if group.Name == name { + return EvaluateRunnerGroupPolicy(group, nil, policy) + } + } + return RunnerGroupPolicyResult{Violations: []string{fmt.Sprintf("runner group %q was not found in the organization", name)}} +} + +func EvaluateRunnerGroupPolicy(group RunnerGroup, repositories []RunnerGroupRepository, policy config.RunnerGroupSecurityConfig) RunnerGroupPolicyResult { + result := RunnerGroupPolicyResult{ + Group: group, + Resolved: true, + Repositories: append([]RunnerGroupRepository(nil), repositories...), + } + if group.Default { + if policy.RequireNonDefaultGroup { + result.Violations = append(result.Violations, fmt.Sprintf("runner group %q is GitHub's default group but security.runnerGroup.requireNonDefaultGroup is true", group.Name)) + } else { + result.Advisories = append(result.Advisories, fmt.Sprintf("runner group %q is GitHub's default group; repository access may broaden as organization policy changes", group.Name)) + } + } + if group.Inherited { + result.Advisories = append(result.Advisories, fmt.Sprintf("runner group %q is inherited from the enterprise; its policy must be managed at enterprise level", group.Name)) + } + if !repositoryAccessAllows(policy.RequiredRepositoryAccess, group.Visibility) { + result.Violations = append(result.Violations, fmt.Sprintf("runner group %q has repository access %q, broader than security.runnerGroup.requiredRepositoryAccess %q", group.Name, group.Visibility, policy.RequiredRepositoryAccess)) + } + switch group.Visibility { + case config.RunnerGroupRepositoryAccessPrivate: + result.Advisories = append(result.Advisories, fmt.Sprintf("runner group %q is available to all private repositories in the organization", group.Name)) + case config.RunnerGroupRepositoryAccessAll: + result.Advisories = append(result.Advisories, fmt.Sprintf("runner group %q is available to all repositories allowed by its public-repository setting", group.Name)) + case config.RunnerGroupRepositoryAccessSelected: + if len(repositories) == 0 { + result.Advisories = append(result.Advisories, fmt.Sprintf("runner group %q has no selected repositories and cannot receive jobs", group.Name)) + } + default: + result.Violations = append(result.Violations, fmt.Sprintf("runner group %q returned unsupported repository access %q", group.Name, group.Visibility)) + } + if policy.RequirePublicRepositoriesDisabled { + if group.AllowsPublicRepositories { + result.Violations = append(result.Violations, fmt.Sprintf("runner group %q allows public repositories but security.runnerGroup.requirePublicRepositoriesDisabled is true", group.Name)) + } + for _, repository := range repositories { + if !repository.Private { + name := repository.FullName + if name == "" { + name = repository.Name + } + result.Violations = append(result.Violations, fmt.Sprintf("runner group %q includes public repository %q while public repositories are required to be disabled", group.Name, name)) + } + } + } else if group.AllowsPublicRepositories { + result.Advisories = append(result.Advisories, fmt.Sprintf("runner group %q allows public repositories; untrusted public or fork workflows may reach self-hosted runners", group.Name)) + } + return result +} + +func repositoryAccessAllows(maximum, actual string) bool { + rank := map[string]int{ + config.RunnerGroupRepositoryAccessSelected: 0, + config.RunnerGroupRepositoryAccessPrivate: 1, + config.RunnerGroupRepositoryAccessAll: 2, + } + maximumRank, maximumKnown := rank[maximum] + actualRank, actualKnown := rank[actual] + return maximumKnown && actualKnown && actualRank <= maximumRank +} diff --git a/internal/github/runner_groups_test.go b/internal/github/runner_groups_test.go new file mode 100644 index 0000000..9a0d9cd --- /dev/null +++ b/internal/github/runner_groups_test.go @@ -0,0 +1,279 @@ +package github + +import ( + "context" + "fmt" + "io" + "net/http" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" +) + +func TestRunnerGroupRepositoryAccessUsesMaximumBreadth(t *testing.T) { + for _, test := range []struct { + maximum string + actual string + allowed bool + }{ + {maximum: "selected", actual: "selected", allowed: true}, + {maximum: "selected", actual: "private", allowed: false}, + {maximum: "selected", actual: "all", allowed: false}, + {maximum: "private", actual: "selected", allowed: true}, + {maximum: "private", actual: "private", allowed: true}, + {maximum: "private", actual: "all", allowed: false}, + {maximum: "all", actual: "selected", allowed: true}, + {maximum: "all", actual: "private", allowed: true}, + {maximum: "all", actual: "all", allowed: true}, + } { + t.Run(test.maximum+"_allows_"+test.actual, func(t *testing.T) { + if got := repositoryAccessAllows(test.maximum, test.actual); got != test.allowed { + t.Fatalf("repositoryAccessAllows(%q, %q) = %t, want %t", test.maximum, test.actual, got, test.allowed) + } + }) + } +} + +func TestEvaluateRunnerGroupPolicy(t *testing.T) { + strict := config.Default().Security.RunnerGroup + for _, test := range []struct { + name string + group RunnerGroup + repositories []RunnerGroupRepository + mutate func(*config.RunnerGroupSecurityConfig) + wantAllowed bool + wantViolation string + wantAdvisory string + }{ + { + name: "strict selected private repositories", + group: RunnerGroup{ID: 1, Name: "restricted", Visibility: "selected"}, + repositories: []RunnerGroupRepository{{FullName: "example/private", Private: true}}, + wantAllowed: true, + }, + { + name: "default rejected when required", + group: RunnerGroup{ID: 1, Name: "Default", Visibility: "selected", Default: true}, + wantViolation: "requireNonDefaultGroup is true", + }, + { + name: "default allowed with advisory", + group: RunnerGroup{ID: 1, Name: "Default", Visibility: "all", Default: true}, + mutate: func(policy *config.RunnerGroupSecurityConfig) { + policy.RequireNonDefaultGroup = false + policy.RequiredRepositoryAccess = "all" + }, + wantAllowed: true, + wantAdvisory: "default group", + }, + { + name: "broader access rejected", + group: RunnerGroup{ID: 1, Name: "broad", Visibility: "private"}, + wantViolation: "broader than", + }, + { + name: "public group rejected", + group: RunnerGroup{ID: 1, Name: "public", Visibility: "selected", AllowsPublicRepositories: true}, + wantViolation: "allows public repositories", + }, + { + name: "public repository defensively rejected", + group: RunnerGroup{ID: 1, Name: "inconsistent", Visibility: "selected"}, + repositories: []RunnerGroupRepository{{FullName: "example/public", Private: false}}, + wantViolation: "includes public repository", + }, + { + name: "public exception remains advisory", + group: RunnerGroup{ID: 1, Name: "public", Visibility: "selected", AllowsPublicRepositories: true}, + mutate: func(policy *config.RunnerGroupSecurityConfig) { policy.RequirePublicRepositoriesDisabled = false }, + wantAllowed: true, + wantAdvisory: "untrusted public or fork workflows", + }, + { + name: "inherited group allowed with advisory", + group: RunnerGroup{ID: 1, Name: "enterprise", Visibility: "selected", Inherited: true}, + wantAllowed: true, + wantAdvisory: "enterprise level", + }, + } { + t.Run(test.name, func(t *testing.T) { + policy := strict + if test.mutate != nil { + test.mutate(&policy) + } + result := EvaluateRunnerGroupPolicy(test.group, test.repositories, policy) + if result.Allowed() != test.wantAllowed { + t.Fatalf("Allowed() = %t, violations=%#v", result.Allowed(), result.Violations) + } + if test.wantViolation != "" && !containsText(result.Violations, test.wantViolation) { + t.Fatalf("violations = %#v, want text %q", result.Violations, test.wantViolation) + } + if test.wantAdvisory != "" && !containsText(result.Advisories, test.wantAdvisory) { + t.Fatalf("advisories = %#v, want text %q", result.Advisories, test.wantAdvisory) + } + }) + } +} + +func TestResolveRunnerGroupRequiresExactOrDefault(t *testing.T) { + groups := []RunnerGroup{{ID: 1, Name: "Restricted", Visibility: "selected"}, {ID: 2, Name: "Default", Visibility: "all", Default: true}} + policy := config.Default().Security.RunnerGroup + if result := ResolveRunnerGroup("restricted", policy, groups); result.Allowed() || !containsText(result.Violations, "was not found") { + t.Fatalf("case-mismatched group result = %+v, want not found", result) + } + policy.RequireExplicitGroup = false + policy.RequireNonDefaultGroup = false + policy.RequiredRepositoryAccess = "all" + result := ResolveRunnerGroup("", policy, groups) + if !result.Allowed() || !result.Resolved || result.Group.Name != "Default" || !containsText(result.Advisories, "runner.group is empty") { + t.Fatalf("default result = %+v", result) + } +} + +func TestListRunnerGroupsAndRepositoriesPaginates(t *testing.T) { + keyPath := writeKey(t) + client := New(config.GitHubConfig{AppID: 123, Organization: "example", PrivateKeyPath: keyPath, APIBaseURL: "https://api.github.test"}) + client.httpClient = &http.Client{Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) { + body := "{}" + switch r.URL.Path { + case "/orgs/example/installation": + body = `{"id":42}` + case "/app/installations/42/access_tokens": + body = `{"token":"installation-token","expires_at":"2099-01-01T00:00:00Z"}` + case "/orgs/example/actions/runner-groups": + page := r.URL.Query().Get("page") + if page == "1" { + items := make([]string, 100) + for i := range items { + items[i] = fmt.Sprintf(`{"id":%d,"name":"group-%d","visibility":"selected"}`, i+1, i+1) + } + body = `{"total_count":101,"runner_groups":[` + strings.Join(items, ",") + `]}` + } else { + body = `{"total_count":101,"runner_groups":[{"id":101,"name":"group-101","visibility":"selected"}]}` + } + case "/orgs/example/actions/runner-groups/1/repositories": + page := r.URL.Query().Get("page") + if page == "1" { + items := make([]string, 100) + for i := range items { + items[i] = fmt.Sprintf(`{"id":%d,"name":"repo-%d","full_name":"example/repo-%d","private":true}`, i+1, i+1, i+1) + } + body = `{"total_count":101,"repositories":[` + strings.Join(items, ",") + `]}` + } else { + body = `{"total_count":101,"repositories":[{"id":101,"name":"repo-101","full_name":"example/repo-101","private":true}]}` + } + default: + t.Fatalf("unexpected path %s", r.URL.Path) + } + return &http.Response{StatusCode: http.StatusOK, Header: make(http.Header), Body: io.NopCloser(strings.NewReader(body))}, nil + })} + groups, err := client.ListRunnerGroups(context.Background()) + if err != nil { + t.Fatal(err) + } + if len(groups) != 101 { + t.Fatalf("groups = %d, want 101", len(groups)) + } + repositories, err := client.ListRunnerGroupRepositories(context.Background(), 1) + if err != nil { + t.Fatal(err) + } + if len(repositories) != 101 { + t.Fatalf("repositories = %d, want 101", len(repositories)) + } +} + +func TestLiveRunnerGroupPolicy(t *testing.T) { + configPath := os.Getenv("EPAR_LIVE_CONFIG") + safeGroup := os.Getenv("EPAR_LIVE_SAFE_GROUP") + publicGroup := os.Getenv("EPAR_LIVE_PUBLIC_GROUP") + if configPath == "" || safeGroup == "" || publicGroup == "" { + t.Skip("set EPAR_LIVE_CONFIG, EPAR_LIVE_SAFE_GROUP, and EPAR_LIVE_PUBLIC_GROUP to run live GitHub policy validation") + } + repositoryRoot, err := filepath.Abs(filepath.Join("..", "..")) + if err != nil { + t.Fatal(err) + } + if !filepath.IsAbs(configPath) { + configPath = config.ProjectPath(repositoryRoot, configPath) + } + cfg, err := config.Load(configPath) + if err != nil { + t.Fatal(err) + } + if !filepath.IsAbs(cfg.GitHub.PrivateKeyPath) { + cfg.GitHub.PrivateKeyPath = config.ProjectPath(repositoryRoot, cfg.GitHub.PrivateKeyPath) + } + client := New(cfg.GitHub) + strict := config.Default().Security.RunnerGroup + strict.Enforcement = config.RunnerGroupEnforcementEnforce + + safeResult, err := client.EvaluateRunnerGroupPolicy(context.Background(), safeGroup, strict) + if err != nil { + t.Fatal(err) + } + if !safeResult.Allowed() { + t.Fatalf("safe group violations = %#v", safeResult.Violations) + } + + groups, err := client.ListRunnerGroups(context.Background()) + if err != nil { + t.Fatal(err) + } + var defaultName string + for _, group := range groups { + if group.Default { + defaultName = group.Name + break + } + } + if defaultName == "" { + t.Fatal("GitHub returned no default runner group") + } + defaultStrict, err := client.EvaluateRunnerGroupPolicy(context.Background(), defaultName, strict) + if err != nil { + t.Fatal(err) + } + if defaultStrict.Allowed() { + t.Fatalf("default group unexpectedly passed strict policy: %+v", defaultStrict) + } + defaultPolicy := strict + defaultPolicy.RequireNonDefaultGroup = false + defaultPolicy.RequiredRepositoryAccess = config.RunnerGroupRepositoryAccessAll + defaultAllowed, err := client.EvaluateRunnerGroupPolicy(context.Background(), defaultName, defaultPolicy) + if err != nil { + t.Fatal(err) + } + if !defaultAllowed.Allowed() || !containsText(defaultAllowed.Advisories, "default group") { + t.Fatalf("default group deliberate policy result = %+v", defaultAllowed) + } + + publicStrict, err := client.EvaluateRunnerGroupPolicy(context.Background(), publicGroup, strict) + if err != nil { + t.Fatal(err) + } + if publicStrict.Allowed() || !containsText(publicStrict.Violations, "allows public repositories") { + t.Fatalf("public group strict result = %+v", publicStrict) + } + publicPolicy := strict + publicPolicy.RequirePublicRepositoriesDisabled = false + publicAllowed, err := client.EvaluateRunnerGroupPolicy(context.Background(), publicGroup, publicPolicy) + if err != nil { + t.Fatal(err) + } + if !publicAllowed.Allowed() || !containsText(publicAllowed.Advisories, "untrusted public or fork workflows") { + t.Fatalf("public group deliberate policy result = %+v", publicAllowed) + } +} + +func containsText(values []string, text string) bool { + for _, value := range values { + if strings.Contains(value, text) { + return true + } + } + return false +} diff --git a/internal/hosttrust/hosttrust_test.go b/internal/hosttrust/hosttrust_test.go index 50a3017..e4b1293 100644 --- a/internal/hosttrust/hosttrust_test.go +++ b/internal/hosttrust/hosttrust_test.go @@ -96,7 +96,7 @@ func TestParseFeedVerifiesCertificateHashAndFreshness(t *testing.T) { SchemaVersion: feedSchemaVersion, HostOS: "darwin", Scopes: []string{ScopeSystem, ScopeUser}, - GeneratedAt: now.Add(-29 * time.Second), + GeneratedAt: now.Add(-maxFeedAge + time.Second), ExpiresAt: now.Add(time.Minute), Certificates: []FeedCertificate{{SHA256: certificates[0].SHA256, PEM: string(certificates[0].PEM)}}, } @@ -123,7 +123,7 @@ func TestParseFeedVerifiesCertificateHashAndFreshness(t *testing.T) { t.Fatalf("bad per-certificate hash error = %v", err) } feed.Certificates[0].SHA256 = certificates[0].SHA256 - feed.GeneratedAt = now.Add(-31 * time.Second) + feed.GeneratedAt = now.Add(-maxFeedAge - time.Second) content, err = json.Marshal(feed) if err != nil { t.Fatal(err) diff --git a/internal/image/README.md b/internal/image/README.md index 9cd73ad..f19021b 100644 --- a/internal/image/README.md +++ b/internal/image/README.md @@ -1,3 +1,3 @@ # Image Package -Provider-neutral image build abstractions can move here if the image build surface grows beyond the current manager-owned flow. Tart, WSL, and Docker-DinD have provider-specific build steps, while shared Ubuntu guest provisioning and image install scripts remain coordinated from the pool manager. +Provider-neutral image build abstractions can move here if the image build surface grows beyond the current manager-owned flow. Tart, WSL, and Docker Container have provider-specific build steps, while shared Ubuntu guest provisioning and image install scripts remain coordinated from the pool manager. diff --git a/internal/image/acquisition.go b/internal/image/acquisition.go new file mode 100644 index 0000000..1f615ff --- /dev/null +++ b/internal/image/acquisition.go @@ -0,0 +1,179 @@ +package image + +import ( + "context" + "errors" + "fmt" + "io" + "strings" + "time" + + "github.com/moby/moby/client" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +const dockerPullProgressInterval = 250 * time.Millisecond + +type DockerSourcePullOptions struct { + Image string + Platform string + LogPath string + AnnounceRemoteSize bool +} + +func (m *Coordinator) PullDockerSource(ctx context.Context, opts DockerSourcePullOptions) error { + result, err := PullDockerImage(ctx, DockerPullOptions{ + Image: opts.Image, + Platform: opts.Platform, + FallbackPlatform: m.Config.Provider.Platform, + QueryRemoteSize: opts.AnnounceRemoteSize, + }) + var enginePullErr *DockerEnginePullError + if err != nil && !errors.As(err, &enginePullErr) { + return m.pullDockerSourceWithCLI(ctx, opts, err) + } + if opts.AnnounceRemoteSize { + if result.RemoteCompressedError != nil { + m.WriteDockerPullNotice(opts.LogPath, "warning: could not determine remote compressed layer size: "+sanitizeImageError(result.RemoteCompressedError)) + } else { + m.WriteDockerPullNotice(opts.LogPath, fmt.Sprintf("Remote compressed layers: %s; actual transfer may be lower when Docker reuses layers.", FormatDockerPullBytes(result.RemoteCompressedSize))) + } + } + if result.RegistryAuthError != nil { + m.WriteDockerPullNotice(opts.LogPath, "warning: could not load Docker registry credentials; continuing without explicit credentials: "+sanitizeImageError(result.RegistryAuthError)) + } + if err != nil { + return err + } + if err := m.renderDockerPullProgress(ctx, result.Response, opts.LogPath); err != nil { + return fmt.Errorf("Docker Engine pull %s: %w", opts.Image, err) + } + m.WriteDockerPullNotice(opts.LogPath, "Docker source pull complete: "+opts.Image) + return nil +} + +func (m *Coordinator) pullDockerSourceWithCLI(ctx context.Context, opts DockerSourcePullOptions, apiErr error) error { + m.WriteDockerPullNotice(opts.LogPath, "warning: "+sanitizeImageError(apiErr)+"; falling back to docker pull CLI") + args := []string{"pull"} + if opts.Platform != "" { + args = append(args, "--platform", opts.Platform) + } + args = append(args, opts.Image) + return m.runHostLogged(ctx, opts.LogPath, "docker", args...) +} + +func (m *Coordinator) renderDockerPullProgress(ctx context.Context, response client.ImagePullResponse, logPath string) error { + transcript, err := m.transcript(logPath, "", "docker-pull") + if err != nil { + return err + } + layers := map[string]DockerPullProgress{} + lastRender := time.Time{} + rendered := false + err = ConsumeDockerPullProgress(ctx, response, func(message DockerPullEvent) error { + writeDockerPullEvent(transcript.Stdout, message) + if message.Error != nil { + return nil + } + if message.ID != "" { + layer := layers[message.ID] + if message.Progress != nil { + layer.Current = message.Progress.Current + if message.Progress.Total > 0 { + layer.Total = message.Progress.Total + } + } + if IsDockerPullLayerComplete(message.Status) { + layer.Completed = true + if layer.Total > 0 { + layer.Current = layer.Total + } + } + layers[message.ID] = layer + } + if time.Since(lastRender) >= dockerPullProgressInterval { + m.writeDockerPullProgress(logPath, layers) + lastRender = time.Now() + rendered = true + } + return nil + }) + if err != nil { + return err + } + if rendered { + m.writeDockerPullProgress(logPath, layers) + if m.dockerPullProgressIsInteractive() { + fmt.Fprintln(m.environment.ProgressConsole()) + } + } + return nil +} + +func (m *Coordinator) WriteDockerPullNotice(logPath, message string) { + attributes := []any{"provider", m.Config.Provider.Type, "operation", "docker-pull", "logPath", logPath} + if strings.HasPrefix(message, "warning:") { + m.environment.LogWarn(strings.TrimSpace(strings.TrimPrefix(message, "warning:")), attributes...) + } else { + m.environment.LogInfo(message, attributes...) + } + transcript, err := m.transcript(logPath, "", "docker-pull") + if err != nil { + m.environment.LogWarn("docker pull transcript unavailable", "operation", "docker-pull", "logPath", logPath, "error", err) + return + } + _, _ = fmt.Fprintf(transcript.Stdout, "%s\n", message) +} + +func writeDockerPullEvent(logFile io.Writer, event DockerPullEvent) { + if logFile == nil { + return + } + parts := make([]string, 0, 4) + if event.ID != "" { + parts = append(parts, event.ID) + } + if event.Status != "" { + parts = append(parts, event.Status) + } + if event.Progress != nil { + parts = append(parts, fmt.Sprintf("progress=%d/%d", event.Progress.Current, event.Progress.Total)) + } + if event.Stream != "" { + parts = append(parts, strings.TrimSpace(event.Stream)) + } + if event.Error != nil { + parts = append(parts, "error="+sanitizeImageError(event.Error)) + } + fmt.Fprintf(logFile, "%s %s\n", time.Now().UTC().Format(time.RFC3339Nano), strings.Join(parts, " ")) +} + +func (m *Coordinator) writeDockerPullProgress(logPath string, layers map[string]DockerPullProgress) { + line := DockerPullProgressSummary(layers) + if m.dockerPullProgressIsInteractive() { + _, _ = fmt.Fprintf(m.environment.ProgressConsole(), "\r\033[2K%s", line) + return + } + m.environment.LogInfo(line, "provider", m.Config.Provider.Type, "operation", "docker-pull", "logPath", logPath) +} + +func (m *Coordinator) dockerPullProgressIsInteractive() bool { + return m.environment.ProgressTerminal() && containsString(m.Config.Logging.ManagerSinks, "console") && m.Config.Logging.ManagerConsoleFormat == "text" +} + +func containsString(values []string, wanted string) bool { + for _, value := range values { + if value == wanted { + return true + } + } + return false +} + +func sanitizeImageError(err error) string { + text := provider.RedactText(strings.Join(strings.Fields(err.Error()), " ")) + if len(text) > 500 { + return text[:500] + "..." + } + return text +} diff --git a/internal/image/artifact_lifecycle.go b/internal/image/artifact_lifecycle.go new file mode 100644 index 0000000..9430b02 --- /dev/null +++ b/internal/image/artifact_lifecycle.go @@ -0,0 +1,710 @@ +package image + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "sort" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +const ( + imageManifestSchemaVersion = ManifestSchemaVersion + imageManifestGuestPath = ManifestGuestPath + imageManifestLabel = ManifestLabel +) + +type ImageManifest = Manifest +type fileDigest = FileDigest +type sourceCacheManifest = SourceCacheManifest + +func hostTrustMetadata(snapshot hosttrust.Snapshot) *HostTrustMetadata { + if snapshot.Generation == "" { + return nil + } + return &HostTrustMetadata{ + Mode: hosttrust.ModeOverlay, + HostOS: snapshot.HostOS, + Scopes: append([]string(nil), snapshot.Scopes...), + Generation: snapshot.Generation, + CertificateCount: len(snapshot.Certificates), + } +} + +func (m *Coordinator) EnsureImage(ctx context.Context) error { + return m.ensureImage(ctx, false) +} + +// UpdateImage forces an immediate remote freshness observation and rebuilds +// only when the resolved immutable inputs differ from the active artifact. +func (m *Coordinator) UpdateImage(ctx context.Context) error { + return m.ensureImage(ctx, true) +} + +func (m *Coordinator) ensureImage(ctx context.Context, forceRemote bool) error { + if err := m.cleanupSupersededCatalog(ctx); err != nil { + return fmt.Errorf("reconcile EPAR storage before image provisioning: %w", err) + } + m.logEffectiveUpdatePolicy(forceRemote) + if m.Config.Provider.Type == "docker-sandboxes" { + return m.ensureDockerSandboxesTemplateWithPolicy(ctx, forceRemote) + } + if artifactManager, ok := m.Lifecycle.(provider.ArtifactManager); ok { + handled, err := artifactManager.EnsureArtifacts(ctx, m.DryRun) + if handled || err != nil { + return err + } + } + localManifest, err := m.desiredLocalImageManifest(ctx) + if err != nil { + return err + } + localHash, err := imageManifestHash(localManifest) + if err != nil { + return err + } + if m.DryRun { + m.infof("[dry-run] would ensure image %s has local-input manifest %s\n", m.Config.Image.OutputImage, localHash) + manifest, err := m.resolveRemoteImageManifest(ctx, localManifest) + if err != nil { + return err + } + return m.BuildImage(ctx, ImageBuildOptions{Replace: true, Manifest: &manifest}) + } + + now := m.now() + updateState, err := m.readUpdatePolicyState() + if err != nil { + m.warnf("ignoring stale image update state and performing an immediate check: %v\n", err) + updateState = UpdatePolicyState{SchemaVersion: updatePolicyStateSchemaVersion} + } + if m.Config.Provider.Type == "wsl" { + outputPath := config.ProjectPath(m.ProjectRoot, m.Config.Image.OutputImage) + sidecarPath := wslImageManifestSidecarPath(outputPath) + if stored, readErr := readStoredImageManifest(sidecarPath); readErr == nil { + if info, statErr := os.Stat(sidecarPath); statErr == nil { + bootstrapped, bootstrapErr := bootstrapUpdatePolicyState(&updateState, m.Config.Image, localManifest, stored.Manifest, nil, info.ModTime(), now.Location()) + if bootstrapErr != nil { + return bootstrapErr + } + if bootstrapped { + if err := m.writeUpdatePolicyState(updateState); err != nil { + return err + } + m.infof("initialized image update schedule from the verified active WSL artifact\n") + } + } + } + } + if recalculateScheduleForTimeZone(&updateState, m.Config.Image, now.Location()) { + if err := m.writeUpdatePolicyState(updateState); err != nil { + return err + } + } + + localChanged := updateState.LocalInputHash != "" && updateState.LocalInputHash != localHash + var ( + currentState imageState + currentHash string + haveResolved bool + ) + if updateState.LocalInputHash == localHash && updateState.LastResolvedManifest != nil { + currentHash, err = imageManifestHash(*updateState.LastResolvedManifest) + if err != nil { + return err + } + currentState, err = m.currentImageState(ctx, currentHash) + if err != nil { + return err + } + haveResolved = true + } + if updateState.LocalInputHash == localHash && pendingUpdateReady(updateState, now) { + if err := m.ApplyPendingUpdate(ctx, now); err != nil { + if !forceRemote && haveResolved && currentState == imageStateCurrent { + status, _ := m.UpdatePolicyStatus() + m.warnf("pending scheduled image update failed; continuing with the previous verified artifact and retrying after %s: %v\n", formatUpdateTime(status.NextRetryAt), err) + return nil + } + return err + } + m.infof("pending image update activated\n") + return nil + } + + remoteDue := updateCheckDue(updateState, m.Config.Image, now) + needsRemote := forceRemote || !haveResolved || localChanged || currentState != imageStateCurrent || remoteDue + if !needsRemote { + m.infof("image is current: %s; next remote check %s\n", m.Config.Image.OutputImage, formatUpdateTime(updateState.NextEligibleAt)) + if err := m.recordCurrentArtifact(ctx, currentHash); err != nil { + return fmt.Errorf("record current EPAR artifact ownership: %w", err) + } + return m.cleanupSupersededCatalog(ctx) + } + + updateState.LastAttemptAt = now.UTC() + manifest, resolveErr := m.resolveRemoteImageManifest(ctx, localManifest) + if resolveErr != nil { + scheduleUpdateFailure(&updateState, now, resolveErr) + _ = m.writeUpdatePolicyState(updateState) + if !forceRemote && !localChanged && haveResolved && currentState == imageStateCurrent { + m.warnf("scheduled image update check failed; continuing with the last verified artifact and retrying after %s: %v\n", formatUpdateTime(updateState.NextRetryAt), resolveErr) + return nil + } + return resolveErr + } + resolvedHash, err := imageManifestHash(manifest) + if err != nil { + return err + } + resolvedState, err := m.currentImageState(ctx, resolvedHash) + if err != nil { + return err + } + if resolvedState == imageStateCurrent { + updateState.LocalInputHash = localHash + updateState.LastResolvedManifest = &manifest + updateState.PendingManifest = nil + if err := scheduleNextSuccess(&updateState, m.Config.Image, now); err != nil { + return err + } + if err := m.writeUpdatePolicyState(updateState); err != nil { + return err + } + m.infof("image is current: %s; next remote check %s\n", m.Config.Image.OutputImage, formatUpdateTime(updateState.NextEligibleAt)) + if err := m.recordCurrentArtifact(ctx, resolvedHash); err != nil { + return fmt.Errorf("record current EPAR artifact ownership: %w", err) + } + return m.cleanupSupersededCatalog(ctx) + } + if resolvedState == imageStateMissing { + m.infof("image is missing; building %s\n", m.Config.Image.OutputImage) + } else { + m.infof("image inputs changed; rebuilding %s\n", m.Config.Image.OutputImage) + } + updateState.LocalInputHash = localHash + updateState.PendingManifest = &manifest + updateState.DeferredReason = "artifact build and activation pending" + if err := m.writeUpdatePolicyState(updateState); err != nil { + return err + } + if err := m.buildResolvedImage(ctx, manifest); err != nil { + scheduleUpdateFailure(&updateState, now, err) + _ = m.writeUpdatePolicyState(updateState) + if !forceRemote && !localChanged && haveResolved && currentState == imageStateCurrent { + m.warnf("scheduled image update failed; restoring the last verified artifact and retrying after %s: %v\n", formatUpdateTime(updateState.NextRetryAt), err) + return nil + } + return err + } + if err := m.recordCurrentArtifact(ctx, resolvedHash); err != nil { + return fmt.Errorf("record current EPAR artifact ownership: %w", err) + } + updateState.LastResolvedManifest = &manifest + updateState.PendingManifest = nil + if err := scheduleNextSuccess(&updateState, m.Config.Image, m.now()); err != nil { + return err + } + if err := m.writeUpdatePolicyState(updateState); err != nil { + return err + } + return m.cleanupSupersededCatalog(ctx) +} + +func (m *Coordinator) logEffectiveUpdatePolicy(forceRemote bool) { + if forceRemote { + m.infof("image update policy: immediate manual check requested\n") + return + } + if m.Config.Image.UpdateFrequency == config.ImageUpdateFrequencyManual { + m.infof("image update policy: manual; remote image and Actions runner checks run only when requested\n") + return + } + m.infof("image update policy: %s at %s local time\n", m.Config.Image.UpdateFrequency, m.Config.Image.UpdateTime) +} + +func (m *Coordinator) buildResolvedImage(ctx context.Context, manifest Manifest) error { + originalSource := m.Config.Image.SourceImage + if manifest.SourceType == config.ImageSourceDockerImage && strings.Contains(manifest.SourceDigest, "@sha256:") { + m.Config.Image.SourceImage = manifest.SourceDigest + } + defer func() { + m.Config.Image.SourceImage = originalSource + }() + return m.BuildImage(ctx, ImageBuildOptions{Replace: true, Manifest: &manifest}) +} + +type imageState int + +const ( + imageStateMissing imageState = iota + imageStateOutdated + imageStateCurrent +) + +func (m *Coordinator) currentImageState(ctx context.Context, wantHash string) (imageState, error) { + switch m.Config.Provider.Type { + case "docker-container": + got, exists, err := m.currentDockerContainerManifestHash(ctx) + if err != nil { + return imageStateMissing, err + } + if !exists { + return imageStateMissing, nil + } + if got != wantHash { + return imageStateOutdated, nil + } + return imageStateCurrent, nil + case "wsl": + got, exists, err := m.currentWSLManifestHash() + if err != nil { + return imageStateMissing, err + } + if !exists { + return imageStateMissing, nil + } + if got != wantHash { + return imageStateOutdated, nil + } + return imageStateCurrent, nil + case "tart": + return m.currentTartImageState(ctx, wantHash) + default: + return imageStateMissing, fmt.Errorf("unsupported provider.type %q", m.Config.Provider.Type) + } +} + +func (m *Coordinator) currentDockerContainerManifestHash(ctx context.Context) (string, bool, error) { + output := strings.TrimSpace(m.Config.Image.OutputImage) + if output == "" { + return "", false, fmt.Errorf("image.outputImage is required") + } + out, err := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{json .Config.Labels}}", output) + if err != nil { + if dockerInspectMeansMissing(err) { + return "", false, nil + } + return "", false, err + } + labels := map[string]string{} + if err := json.Unmarshal([]byte(strings.TrimSpace(out)), &labels); err != nil { + return "", true, fmt.Errorf("parse Docker image labels for %s: %w", output, err) + } + return labels[imageManifestLabel], true, nil +} + +func (m *Coordinator) currentWSLManifestHash() (string, bool, error) { + outputPath := config.ProjectPath(m.ProjectRoot, m.Config.Image.OutputImage) + if _, err := os.Stat(outputPath); err != nil { + if errors.Is(err, os.ErrNotExist) { + return "", false, nil + } + return "", false, err + } + stored, err := readStoredImageManifest(wslImageManifestSidecarPath(outputPath)) + if err != nil { + if errors.Is(err, os.ErrNotExist) { + return "", true, nil + } + return "", true, err + } + return stored.Hash, true, nil +} + +func (m *Coordinator) currentTartImageState(ctx context.Context, wantHash string) (imageState, error) { + instances, err := m.Provider.List(ctx) + if err != nil { + return imageStateMissing, err + } + var current provider.Instance + for _, instance := range instances { + if instance.Name == m.Config.Image.OutputImage || instance.Source == m.Config.Image.OutputImage { + current = instance + break + } + } + if current.Name == "" { + return imageStateMissing, nil + } + if current.ProviderID == "" { + return imageStateOutdated, nil + } + store, err := m.hostCatalog() + if err != nil { + return imageStateMissing, err + } + value, err := store.Load(time.Now().UTC()) + if err != nil { + return imageStateMissing, err + } + configID, err := storagecatalog.ConfigID(m.ProjectRoot, m.effectiveConfigPath()) + if err != nil { + return imageStateMissing, err + } + if tartCatalogReferenceMatches(value, configID, strings.TrimSpace(m.Config.Image.OutputImage), current.ProviderID, wantHash) { + return imageStateCurrent, nil + } + return imageStateOutdated, nil +} + +func tartCatalogReferenceMatches(value storagecatalog.Catalog, configID, locator, identity, manifestHash string) bool { + for _, resource := range value.Resources { + if resource.Kind != catalogTartImageKind || resource.Locator != locator || resource.Identity != identity { + continue + } + for _, reference := range resource.References { + if reference.ConfigID == configID && reference.Role == "provider-artifact" && reference.ManifestHash == manifestHash { + return true + } + } + } + return false +} + +func (m *Coordinator) desiredImageManifest(ctx context.Context) (ImageManifest, error) { + snapshot, err := m.resolveHostTrust(ctx) + if err != nil { + return ImageManifest{}, err + } + return m.desiredImageManifestWithHostTrust(ctx, snapshot) +} + +func (m *Coordinator) desiredImageManifestWithHostTrust(ctx context.Context, snapshot hosttrust.Snapshot) (ImageManifest, error) { + manifest, err := m.desiredLocalImageManifestWithHostTrust(snapshot) + if err != nil { + return ImageManifest{}, err + } + return m.resolveRemoteImageManifest(ctx, manifest) +} + +func (m *Coordinator) desiredLocalImageManifest(ctx context.Context) (ImageManifest, error) { + snapshot, err := m.resolveHostTrust(ctx) + if err != nil { + return ImageManifest{}, err + } + return m.desiredLocalImageManifestWithHostTrust(snapshot) +} + +func (m *Coordinator) desiredLocalImageManifestWithHostTrust(snapshot hosttrust.Snapshot) (ImageManifest, error) { + sourceType := m.Config.Image.SourceType + if sourceType == "" { + sourceType = config.ImageSourceRootFSTar + if m.Config.Provider.Type == "docker-container" { + sourceType = config.ImageSourceDockerImage + } + } + manifest := ImageManifest{ + SchemaVersion: imageManifestSchemaVersion, + ProviderType: m.Config.Provider.Type, + ProviderPlatform: m.Config.Provider.Platform, + ProviderRosettaTag: m.Config.Provider.RosettaTag, + SourceType: sourceType, + SourceImage: m.Config.Image.SourceImage, + SourcePlatform: m.Config.Image.SourcePlatform, + OutputImage: m.Config.Image.OutputImage, + RunnerSelector: normalizedRunnerSelector(m.Config.Image.RunnerVersion), + HostTrust: hostTrustMetadata(snapshot), + } + switch sourceType { + case config.ImageSourceDockerImage: + case config.ImageSourceRootFSTar: + if m.Config.Provider.Type == "wsl" { + digest, err := m.fileSHA256(config.ProjectPath(m.ProjectRoot, m.Config.Image.SourceImage)) + if err != nil { + return manifest, err + } + manifest.SourceDigest = digest + } + case "": + default: + return manifest, fmt.Errorf("unsupported image.sourceType %q", sourceType) + } + scripts, err := m.eparScriptDigests() + if err != nil { + return manifest, err + } + manifest.EPARScripts = scripts + customScripts, err := m.customInstallScriptDigests() + if err != nil { + return manifest, err + } + manifest.CustomInstallScripts = customScripts + trustedCACertificates, err := m.trustedCACertificateDigests() + if err != nil { + return manifest, err + } + manifest.TrustedCACertificates = trustedCACertificates + if m.runnerImagesCopyMode() != runnerImagesCopyNone { + commit, err := m.runnerImagesCommit() + if err != nil { + return manifest, err + } + manifest.UpstreamCommit = commit + } + return manifest, nil +} + +func (m *Coordinator) resolveRemoteImageManifest(ctx context.Context, manifest ImageManifest) (ImageManifest, error) { + if manifest.SourceType == config.ImageSourceDockerImage { + switch m.Config.Provider.Type { + case "docker-container", "wsl": + source, err := m.resolveDockerSandboxesSource(ctx) + if err != nil { + return manifest, err + } + manifest.SourceImage = source.Reference + manifest.SourcePlatform = source.Platform + manifest.SourceDigest = source.ImmutableReference + manifest.SourcePlatformDigest = source.PlatformDigest + case "tart": + source, err := m.resolveTartOCIReference(ctx, manifest.SourceImage) + if err != nil { + return manifest, err + } + manifest.SourceDigest = source + default: + return manifest, fmt.Errorf("unsupported provider.type %q for Docker image source resolution", m.Config.Provider.Type) + } + } + return m.resolveActionsRunner(ctx, manifest) +} + +func (m *Coordinator) refreshDockerSourceDigest(ctx context.Context) (string, error) { + var digest string + err := m.timeStartupStage("source_image_pull", func() error { + var err error + digest, err = m.refreshDockerSourceDigestUntimed(ctx) + return err + }) + return digest, err +} + +func (m *Coordinator) refreshDockerSourceDigestUntimed(ctx context.Context) (string, error) { + if m.DryRun { + return "dry-run", nil + } + image := strings.TrimSpace(m.Config.Image.SourceImage) + if image == "" { + return "", fmt.Errorf("image.sourceImage is required when image.sourceType=docker-image") + } + platform := strings.TrimSpace(m.Config.Image.SourcePlatform) + logPath := m.buildLogPath(imageLogStem(m.Config.Image.OutputImage) + ".source.log") + defer m.releaseTranscript(logPath) + m.infof("refreshing Docker source image %s\n", image) + backendID, releaseBackend, err := m.acquireDockerBackendLock(ctx) + if err != nil { + return "", err + } + defer releaseBackend() + previousID := "" + if value, inspectErr := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", image); inspectErr == nil { + previousID = strings.TrimSpace(value) + } + if err := m.beginDockerRoleAcquisition(backendID, "build-source", image, previousID, time.Now().UTC()); err != nil { + return "", fmt.Errorf("journal Docker source acquisition: %w", err) + } + if err := m.pullDockerSource(ctx, DockerSourcePullOptions{ + Image: image, + Platform: platform, + LogPath: logPath, + AnnounceRemoteSize: true, + }); err != nil { + return "", fmt.Errorf("refresh Docker source image %s: %w", image, err) + } + currentID, err := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", image) + if err != nil { + return "", err + } + currentID = strings.TrimSpace(currentID) + if err := m.recordDockerSourceAcquisition(ctx, image, previousID, currentID, time.Now().UTC()); err != nil { + return "", fmt.Errorf("record Docker source acquisition: %w", err) + } + digestsJSON, err := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{json .RepoDigests}}", image) + if err != nil { + return "", err + } + var digests []string + if err := json.Unmarshal([]byte(strings.TrimSpace(digestsJSON)), &digests); err != nil { + return "", fmt.Errorf("parse Docker source RepoDigests for %s: %w", image, err) + } + sort.Strings(digests) + if len(digests) > 0 { + digest := digests[0] + m.WriteDockerPullNotice(logPath, "Docker source image digest: "+digest) + return digest, nil + } + imageID, err := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", image) + if err != nil { + return "", err + } + digest := strings.TrimSpace(imageID) + m.WriteDockerPullNotice(logPath, "Docker source image ID: "+digest) + return digest, nil +} + +func (m *Coordinator) eparScriptDigests() ([]fileDigest, error) { + var roots []string + switch m.Config.Provider.Type { + case "docker-container": + roots = []string{ + filepath.Join(m.ProjectRoot, "scripts", "guest", "ubuntu"), + filepath.Join(m.ProjectRoot, "scripts", "container", "ubuntu"), + } + case "wsl", "tart": + roots = []string{filepath.Join(m.ProjectRoot, "scripts", "guest", "ubuntu")} + default: + return nil, nil + } + var out []fileDigest + for _, root := range roots { + digests, err := m.fileDigestsUnder(root) + if err != nil { + return nil, err + } + out = append(out, digests...) + } + sortFileDigests(out) + return out, nil +} + +func (m *Coordinator) customInstallScriptDigests() ([]fileDigest, error) { + var out []fileDigest + for _, script := range m.Config.Image.CustomInstallScripts { + path, err := m.customInstallScriptHostPath(script) + if err != nil { + return nil, err + } + digest, err := m.fileDigest(path) + if err != nil { + return nil, err + } + out = append(out, digest) + } + sortFileDigests(out) + return out, nil +} + +func (m *Coordinator) trustedCACertificateDigests() ([]fileDigest, error) { + if err := m.validateTrustedCACertificates(); err != nil { + return nil, err + } + var out []fileDigest + for _, configuredPath := range m.Config.Image.TrustedCACertificatePaths { + path := config.ProjectPath(m.ProjectRoot, strings.TrimSpace(configuredPath)) + digest, err := m.fileDigest(path) + if err != nil { + return nil, err + } + out = append(out, digest) + } + sortFileDigests(out) + return out, nil +} + +func (m *Coordinator) fileDigestsUnder(root string) ([]fileDigest, error) { + var out []fileDigest + if err := filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error { + if err != nil { + return err + } + if d.IsDir() || !strings.HasSuffix(d.Name(), ".sh") { + return nil + } + digest, err := m.fileDigest(path) + if err != nil { + return err + } + out = append(out, digest) + return nil + }); err != nil { + return nil, err + } + sortFileDigests(out) + return out, nil +} + +func (m *Coordinator) fileDigest(path string) (fileDigest, error) { + sha, err := m.fileSHA256(path) + if err != nil { + return fileDigest{}, err + } + rel, err := filepath.Rel(m.ProjectRoot, path) + if err != nil || strings.HasPrefix(rel, ".."+string(filepath.Separator)) || rel == ".." || filepath.IsAbs(rel) { + rel = path + } + return fileDigest{Path: filepath.ToSlash(filepath.Clean(rel)), SHA256: sha}, nil +} + +func (m *Coordinator) fileSHA256(path string) (string, error) { + if m.DryRun { + return "dry-run", nil + } + content, err := os.ReadFile(path) + if err != nil { + return "", err + } + sum := sha256.Sum256(content) + return hex.EncodeToString(sum[:]), nil +} + +func sortFileDigests(values []fileDigest) { + sort.Slice(values, func(i, j int) bool { + return values[i].Path < values[j].Path + }) +} + +func imageManifestHash(manifest ImageManifest) (string, error) { + return ManifestHash(manifest) +} + +func storedImageManifestContent(manifest ImageManifest) (string, string, error) { + return StoredManifestContent(manifest) +} + +func readStoredImageManifest(path string) (StoredManifest, error) { + return ReadStoredManifest(path) +} + +func writeStoredImageManifest(path string, manifest ImageManifest) error { + return WriteStoredManifest(path, manifest) +} + +func (m *Coordinator) installImageManifest(ctx context.Context, vmName string, manifest ImageManifest) error { + content, _, err := storedImageManifestContent(manifest) + if err != nil { + return err + } + return provider.CopyText(ctx, m.Provider, vmName, imageManifestGuestPath, "0644", content) +} + +func wslImageManifestSidecarPath(outputPath string) string { + return WSLImageManifestPath(outputPath) +} + +func sourceCacheManifestPath(rootfsPath string) string { + return SourceCacheManifestPath(rootfsPath) +} + +func sourceCacheMatches(path string, want sourceCacheManifest) bool { + return SourceCacheMatches(path, want) +} + +func writeSourceCacheManifest(path string, manifest sourceCacheManifest) error { + return WriteSourceCacheManifest(path, manifest) +} + +func dockerInspectMeansMissing(err error) bool { + return DockerInspectMeansMissing(err) +} diff --git a/internal/pool/image.go b/internal/image/build.go similarity index 67% rename from internal/pool/image.go rename to internal/image/build.go index 4c49f8e..22f57c4 100644 --- a/internal/pool/image.go +++ b/internal/image/build.go @@ -1,8 +1,9 @@ -package pool +package image import ( "context" "encoding/json" + "errors" "fmt" "io" "os" @@ -33,14 +34,10 @@ type wslExporter interface { Export(ctx context.Context, name, outputPath string) error } -var ( - runHostCommand = runHost - runHostLoggedCommand = runHostLogged - runHostOutputCommand = runHostOutput - runHostQuietCommand = runHostQuiet -) - -func (m *Manager) UpdateUpstream(ctx context.Context) error { +func (m *Coordinator) UpdateUpstream(ctx context.Context) error { + if err := m.preflightStorage("source-update", sourceUpdateExpansionBytes); err != nil { + return err + } dir := config.ProjectPath(m.ProjectRoot, m.Config.Image.UpstreamDir) logPath := m.buildLogPath("runner-images.source.log") defer m.releaseTranscript(logPath) @@ -77,8 +74,21 @@ func (m *Manager) UpdateUpstream(ctx context.Context) error { return nil } -func (m *Manager) BuildImage(ctx context.Context, opts ImageBuildOptions) error { - if _, err := m.trustedCACertificates(); err != nil { +func (m *Coordinator) BuildImage(ctx context.Context, opts ImageBuildOptions) error { + if err := m.cleanupSupersededCatalog(ctx); err != nil { + return fmt.Errorf("reconcile EPAR storage before image build: %w", err) + } + if m.Config.Provider.Type == "docker-sandboxes" { + return m.ensureDockerSandboxesTemplate(ctx, true) + } + if err := m.validateTrustedCACertificates(); err != nil { + return err + } + plan, err := m.configuredArtifactStoragePlan(ctx, false) + if err != nil { + return err + } + if err := m.preflightStorage("image-build", plan.EstimatedIncrementalPeak); err != nil { return err } upstreamDir := config.ProjectPath(m.ProjectRoot, m.Config.Image.UpstreamDir) @@ -95,20 +105,20 @@ func (m *Manager) BuildImage(ctx context.Context, opts ImageBuildOptions) error return m.buildTartImage(ctx, opts, upstreamDir) case "wsl": return m.buildWSLImage(ctx, opts, upstreamDir) - case "docker-dind": - return m.buildDockerDindImage(ctx, opts, upstreamDir) + case "docker-container": + return m.buildDockerContainerImage(ctx, opts, upstreamDir) default: return fmt.Errorf("unsupported provider.type %q", m.Config.Provider.Type) } } -func (m *Manager) buildDockerDindImage(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { - return m.timeStartupStage("dind_image_build", func() error { - return m.buildDockerDindImageUntimed(ctx, opts, upstreamDir) +func (m *Coordinator) buildDockerContainerImage(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { + return m.timeStartupStage("docker_container_image_build", func() error { + return m.buildDockerContainerImageUntimed(ctx, opts, upstreamDir) }) } -func (m *Manager) buildDockerDindImageUntimed(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { +func (m *Coordinator) buildDockerContainerImageUntimed(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { buildLogPath := m.buildLogPath(imageLogStem(m.Config.Image.OutputImage) + ".docker-build.log") defer m.releaseTranscript(buildLogPath) if err := resetLogs(buildLogPath); err != nil { @@ -119,6 +129,10 @@ func (m *Manager) buildDockerDindImageUntimed(ctx context.Context, opts ImageBui return fmt.Errorf("docker image %s already exists; rerun with --replace", m.Config.Image.OutputImage) } } + builder, err := m.ensureBuildxBuilder(ctx, []string{m.Config.Image.SourceImage, "docker.io/docker/dockerfile:1"}) + if err != nil { + return err + } attempts := 1 if m.hostTrustEnabled() { attempts = 3 @@ -129,66 +143,136 @@ func (m *Manager) buildDockerDindImageUntimed(ctx context.Context, opts ImageBui return err } manifest := opts.Manifest - if m.hostTrustEnabled() || manifest == nil { + if manifest == nil { value, err := m.desiredImageManifestWithHostTrust(ctx, snapshot) if err != nil { return err } manifest = &value + } else if m.hostTrustEnabled() { + value := *manifest + value.HostTrust = hostTrustMetadata(snapshot) + manifest = &value + } + manifestHash, err := ManifestHash(*manifest) + if err != nil { + return err + } + var releaseOutputClaim func() error + claimContext := ctx + if !m.DryRun { + claimContext, releaseOutputClaim, err = m.claimDockerOutputTag(ctx, m.Config.Image.OutputImage, manifestHash) + if err != nil { + return err + } + } + releaseClaim := func() error { + if releaseOutputClaim == nil { + return nil + } + err := releaseOutputClaim() + releaseOutputClaim = nil + return err } targetImage := m.Config.Image.OutputImage if m.hostTrustEnabled() { targetImage = temporaryDockerImageTag(targetImage, snapshot.Generation, attempt) } - if err := m.buildDockerDindImageAttempt(ctx, upstreamDir, buildLogPath, targetImage, *manifest, snapshot); err != nil { - return err + if err := m.buildDockerContainerImageAttempt(claimContext, upstreamDir, buildLogPath, builder, targetImage, *manifest, snapshot); err != nil { + return errors.Join(err, releaseClaim()) + } + if cause := context.Cause(claimContext); cause != nil { + return errors.Join(cause, releaseClaim()) } if m.DryRun || !m.hostTrustEnabled() { + if !m.DryRun { + if err := m.recordCurrentArtifact(claimContext, manifestHash); err != nil { + return errors.Join(fmt.Errorf("record current Docker Container artifact ownership: %w", err), releaseClaim()) + } + } + if err := releaseClaim(); err != nil { + return fmt.Errorf("release Docker output image tag claim: %w", err) + } return nil } - current, err := m.resolveHostTrust(ctx) + current, err := m.resolveHostTrust(claimContext) if err != nil { - _ = runHostQuiet(context.Background(), "docker", "image", "rm", "-f", targetImage) - return err + _ = m.runHostQuiet(context.Background(), "docker", "image", "rm", "-f", targetImage) + return errors.Join(err, releaseClaim()) } if current.Generation != snapshot.Generation { m.infof("host trust changed during image build (%s -> %s); discarding attempt %d/%d\n", snapshot.Generation, current.Generation, attempt, attempts) - _ = runHostQuiet(context.Background(), "docker", "image", "rm", "-f", targetImage) + _ = m.runHostQuiet(context.Background(), "docker", "image", "rm", "-f", targetImage) + if err := releaseClaim(); err != nil { + return fmt.Errorf("release Docker output image tag claim: %w", err) + } continue } - if err := runHostCommand(ctx, "docker", "image", "tag", targetImage, m.Config.Image.OutputImage); err != nil { - _ = runHostQuiet(context.Background(), "docker", "image", "rm", "-f", targetImage) - return err + if cause := context.Cause(claimContext); cause != nil { + _ = m.runHostQuiet(context.Background(), "docker", "image", "rm", "-f", targetImage) + return errors.Join(cause, releaseClaim()) + } + if err := m.runHost(claimContext, "docker", "image", "tag", targetImage, m.Config.Image.OutputImage); err != nil { + _ = m.runHostQuiet(context.Background(), "docker", "image", "rm", "-f", targetImage) + return errors.Join(err, releaseClaim()) + } + _ = m.runHostQuiet(context.Background(), "docker", "image", "rm", "-f", targetImage) + if err := m.recordCurrentArtifact(claimContext, manifestHash); err != nil { + return errors.Join(fmt.Errorf("record current Docker Container artifact ownership: %w", err), releaseClaim()) + } + if err := releaseClaim(); err != nil { + return fmt.Errorf("release Docker output image tag claim: %w", err) } - _ = runHostQuiet(context.Background(), "docker", "image", "rm", "-f", targetImage) m.infof("image build complete: %s is available in `docker image ls`\n", m.Config.Image.OutputImage) return nil } return fmt.Errorf("host trust changed during all %d image build attempts; retry after the host trust store stabilizes", attempts) } -func (m *Manager) buildDockerDindImageAttempt(ctx context.Context, upstreamDir, buildLogPath, targetImage string, manifest ImageManifest, snapshot hosttrust.Snapshot) error { +func (m *Coordinator) buildDockerContainerImageAttempt(ctx context.Context, upstreamDir, buildLogPath, builder, targetImage string, manifest ImageManifest, snapshot hosttrust.Snapshot) error { manifestContent, manifestHash, err := storedImageManifestContent(manifest) if err != nil { return err } - buildCtx, err := os.MkdirTemp("", "epar-docker-dind-build-*") + buildCtx, err := os.MkdirTemp("", "epar-docker-container-build-*") if err != nil { return err } defer os.RemoveAll(buildCtx) - if err := m.prepareDockerDindBuildContextWithHostTrust(buildCtx, upstreamDir, manifestContent, snapshot); err != nil { + if err := m.prepareDockerContainerBuildContextWithHostTrust(buildCtx, upstreamDir, manifestContent, snapshot); err != nil { return err } - m.infof("building Docker-DinD image %s from %s\n", targetImage, m.Config.Image.SourceImage) + if !m.DryRun { + runnerPackage, err := m.acquireActionsRunner(ctx, manifest) + if err != nil { + return err + } + if err := os.MkdirAll(filepath.Join(buildCtx, "inputs"), 0o755); err != nil { + return err + } + if err := copyFile(runnerPackage, filepath.Join(buildCtx, "inputs", "actions-runner.tar.gz"), 0o600); err != nil { + return err + } + } + m.infof("building Docker Container image %s from %s\n", targetImage, m.Config.Image.SourceImage) m.infof("log: %s\n", buildLogPath) - args := []string{"build", "-t", targetImage} + args := []string{"buildx", "build", "--builder", builder, "--load", "-t", targetImage} + installationID, err := m.catalogInstallationID(time.Now().UTC()) + if err != nil { + return fmt.Errorf("resolve EPAR image ownership identity: %w", err) + } if m.Config.Provider.Platform != "" { args = append(args, "--platform", m.Config.Provider.Platform) } args = append(args, + "--label", "io.solutionforest.epar.schema=1", + "--label", "io.solutionforest.epar.installation="+installationID, + "--label", "io.solutionforest.epar.provider=docker-container", + "--label", "io.solutionforest.epar.role=runtime-image", + "--label", "io.solutionforest.epar.manifest="+manifestHash, "--build-arg", "BASE_IMAGE="+m.Config.Image.SourceImage, - "--build-arg", "RUNNER_VERSION="+m.Config.Image.RunnerVersion, + "--build-arg", "RUNNER_VERSION="+manifest.RunnerVersion, + "--build-arg", "RUNNER_SHA256="+manifest.RunnerAssetDigest, "--build-arg", "EPAR_IMAGE_MANIFEST_SHA256="+manifestHash, buildCtx, ) @@ -214,7 +298,13 @@ func temporaryDockerImageTag(output, generation string, attempt int) string { return fmt.Sprintf("%s-epar-build-%s-%d", output, short, attempt) } -func (m *Manager) buildTartImage(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { +func (m *Coordinator) buildTartImage(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { + return m.withTartBackendLock(ctx, func() error { + return m.buildTartImageLocked(ctx, opts, upstreamDir) + }) +} + +func (m *Coordinator) buildTartImageLocked(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { if opts.Manifest == nil { manifest, err := m.desiredImageManifest(ctx) if err != nil { @@ -222,6 +312,23 @@ func (m *Manager) buildTartImage(ctx context.Context, opts ImageBuildOptions, up } opts.Manifest = &manifest } + manifestHash, err := imageManifestHash(*opts.Manifest) + if err != nil { + return err + } + outputName := strings.TrimSpace(m.Config.Image.OutputImage) + buildName := tartBuildName(outputName, manifestHash) + existing, err := m.Provider.List(ctx) + if err != nil { + return err + } + output, outputExists := findTartImage(existing, outputName) + if outputExists && !opts.Replace { + return fmt.Errorf("Tart image %s already exists; rerun with --replace", outputName) + } + if _, exists := findTartImage(existing, buildName); exists { + return fmt.Errorf("Tart build candidate %q already exists from an interrupted operation; inspect it with storage status and remove it through an exact approved cleanup before retrying", buildName) + } buildLogPath := m.buildLogPath(imageLogStem(m.Config.Image.OutputImage) + ".build.log") guestLogPath := m.buildLogPath(imageLogStem(m.Config.Image.OutputImage) + ".guest.log") defer m.releaseTranscript(buildLogPath) @@ -229,86 +336,229 @@ func (m *Manager) buildTartImage(ctx context.Context, opts ImageBuildOptions, up if err := resetLogs(buildLogPath, guestLogPath); err != nil { return err } - m.infof("building Tart image %s from %s\n", m.Config.Image.OutputImage, m.Config.Image.SourceImage) + m.infof("building Tart image candidate %s from %s\n", buildName, m.Config.Image.SourceImage) m.infof("logs: %s, %s\n", buildLogPath, guestLogPath) - if opts.Replace { - m.infof("replacing existing Tart image %s if present\n", m.Config.Image.OutputImage) - _ = m.Provider.Stop(ctx, m.Config.Image.OutputImage) - _ = m.Provider.Delete(ctx, m.Config.Image.OutputImage) - } m.infof("cloning source image\n") - if err := m.Provider.Clone(ctx, m.Config.Image.SourceImage, m.Config.Image.OutputImage); err != nil { + if err := m.Provider.Clone(ctx, m.Config.Image.SourceImage, buildName); err != nil { return err } + if err := m.recordTartStagingImage(ctx, buildName, "build-candidate"); err != nil { + _ = m.Provider.Delete(ctx, buildName) + return fmt.Errorf("record Tart build candidate ownership: %w", err) + } + buildComplete := false + defer func() { + if buildComplete { + return + } + stopContext, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + _ = m.Provider.Stop(stopContext, buildName) + }() m.infof("starting image with Tart network mode %q\n", m.Config.Provider.Network) - startOptions, err := m.startOptions(buildLogPath, m.Config.Image.OutputImage) + startOptions, err := m.startOptions(buildLogPath, buildName) if err != nil { return err } - if _, err := m.Provider.Start(ctx, m.Config.Image.OutputImage, startOptions); err != nil { + if _, err := m.Provider.Start(ctx, buildName, startOptions); err != nil { return err } - ip, err := m.Provider.IP(ctx, m.Config.Image.OutputImage, m.Config.Timeouts.BootSeconds) + ip, err := m.Provider.IP(ctx, buildName, m.Config.Timeouts.BootSeconds) if err != nil { return err } m.infof("guest reachable at %s\n", ip) m.infof("copying guest scripts\n") - if err := m.installGuestScripts(ctx, m.Config.Image.OutputImage); err != nil { + if err := m.installGuestScripts(ctx, buildName); err != nil { return err } - if err := m.installTrustedCACertificates(ctx, m.Config.Image.OutputImage); err != nil { + if err := m.installTrustedCACertificates(ctx, buildName); err != nil { return err } - if err := m.installImageManifest(ctx, m.Config.Image.OutputImage, *opts.Manifest); err != nil { + if err := m.installImageManifest(ctx, buildName, *opts.Manifest); err != nil { return err } switch m.runnerImagesCopyMode() { case runnerImagesCopySubset: m.infof("copying runner-images script subset\n") - if err := m.copyRunnerImagesSubset(ctx, m.Config.Image.OutputImage, upstreamDir); err != nil { + if err := m.copyRunnerImagesSubset(ctx, buildName, upstreamDir); err != nil { return err } case runnerImagesCopyNone: m.infof("skipping runner-images script subset; no selected install script requires it\n") } m.infof("installing base runner runtime\n") - if _, err := m.execBuildGuest(ctx, m.Config.Image.OutputImage, []string{"sudo", "bash", "/opt/epar/install-base.sh", "/opt/epar/upstream/runner-images"}, provider.ExecOptions{}); err != nil { + if _, err := m.execBuildGuest(ctx, buildName, []string{"sudo", "bash", "/opt/epar/install-base.sh", "/opt/epar/upstream/runner-images"}, provider.ExecOptions{}); err != nil { return err } m.infof("installing GitHub Actions runner\n") - if _, err := m.execBuildGuest(ctx, m.Config.Image.OutputImage, []string{"sudo", "bash", "/opt/epar/install-runner.sh", m.Config.Image.RunnerVersion}, provider.ExecOptions{}); err != nil { + if err := m.installActionsRunnerPackage(ctx, buildName, *opts.Manifest); err != nil { return err } - if err := m.installRosettaSupport(ctx, m.Config.Image.OutputImage); err != nil { + if err := m.installRosettaSupport(ctx, buildName); err != nil { return err } - if err := m.installCustomInstallScripts(ctx, m.Config.Image.OutputImage); err != nil { + if err := m.installCustomInstallScripts(ctx, buildName); err != nil { return err } m.infof("validating runner runtime inside the instance\n") - if err := m.validateRuntime(ctx, m.Config.Image.OutputImage); err != nil { + if err := m.validateRuntime(ctx, buildName); err != nil { return err } m.infof("finalizing image for clean Tart clones\n") - if _, err := m.execBuildGuest(ctx, m.Config.Image.OutputImage, []string{"sudo", "bash", "/opt/epar/finalize-image.sh"}, provider.ExecOptions{}); err != nil { + if _, err := m.execBuildGuest(ctx, buildName, []string{"sudo", "bash", "/opt/epar/finalize-image.sh"}, provider.ExecOptions{}); err != nil { return err } m.infof("stopping image\n") - if err := m.Provider.Stop(ctx, m.Config.Image.OutputImage); err != nil { + if err := m.Provider.Stop(ctx, buildName); err != nil { + return err + } + if err := m.activateTartImage(ctx, output, outputExists, buildName, outputName); err != nil { + return err + } + buildComplete = true + m.infof("image build complete: %s is available in `tart list`\n", outputName) + return nil +} + +func tartBuildName(outputName, manifestHash string) string { + short := manifestHash + if len(short) > 12 { + short = short[:12] + } + return outputName + "-epar-build-" + short +} + +func tartBackupName(outputName, providerID string) string { + replacer := strings.NewReplacer(":", "", "-", "") + identity := replacer.Replace(strings.ToLower(providerID)) + if len(identity) > 12 { + identity = identity[len(identity)-12:] + } + return outputName + "-epar-previous-" + identity +} + +func findTartImage(instances []provider.Instance, name string) (provider.Instance, bool) { + for _, instance := range instances { + if instance.Name == name { + return instance, true + } + } + return provider.Instance{}, false +} + +func (m *Coordinator) activateTartImage(ctx context.Context, previous provider.Instance, previousExists bool, buildName, outputName string) error { + instances, err := m.Provider.List(ctx) + if err != nil { + return err + } + candidate, candidateExists := findTartImage(instances, buildName) + if !candidateExists || candidate.ProviderID == "" { + return fmt.Errorf("Tart build candidate %q has no exact immutable identity at activation", buildName) + } + if current, exists := findTartImage(instances, outputName); exists { + if !previousExists || previous.ProviderID == "" || current.ProviderID != previous.ProviderID { + return fmt.Errorf("Tart output image %q changed during replacement; refusing to delete it", outputName) + } + } else if previousExists { + return fmt.Errorf("Tart output image %q disappeared during replacement", outputName) + } + + backupName := "" + if previousExists { + backupName = tartBackupName(outputName, previous.ProviderID) + if _, exists := findTartImage(instances, backupName); exists { + return fmt.Errorf("Tart rollback image %q already exists; refusing an ambiguous replacement", backupName) + } + if err := m.Provider.Clone(ctx, outputName, backupName); err != nil { + return fmt.Errorf("create Tart rollback image: %w", err) + } + if err := m.verifyTartImageIdentity(ctx, backupName); err != nil { + return fmt.Errorf("read back Tart rollback image: %w", err) + } + if err := m.recordTartStagingImage(ctx, backupName, "activation-rollback"); err != nil { + _ = m.Provider.Delete(ctx, backupName) + return fmt.Errorf("record Tart rollback image ownership: %w", err) + } + if strings.EqualFold(previous.State, "running") { + if err := m.Provider.Stop(ctx, outputName); err != nil { + return fmt.Errorf("stop previous Tart output image: %w", err) + } + } + if err := m.Provider.Delete(ctx, outputName); err != nil { + return fmt.Errorf("remove previous Tart output name after rollback copy: %w", err) + } + } + + activationErr := m.Provider.Clone(ctx, buildName, outputName) + if activationErr == nil { + activationErr = m.verifyTartImageIdentity(ctx, outputName) + } + if activationErr != nil { + if !previousExists { + return fmt.Errorf("activate Tart image: %w; the verified build candidate remains available as %q", activationErr, buildName) + } + restoreErr := m.restoreTartImage(ctx, outputName, backupName) + if restoreErr != nil { + return fmt.Errorf("activate Tart image: %v; rollback also failed: %w", activationErr, restoreErr) + } + return fmt.Errorf("activate Tart image: %w; previous image was restored", activationErr) + } + + if backupName != "" { + if err := m.Provider.Delete(ctx, backupName); err != nil { + m.warnf("EPAR Tart rollback image cleanup deferred for %s: %v\n", backupName, err) + } + } + if err := m.Provider.Delete(ctx, buildName); err != nil { + m.warnf("EPAR Tart build candidate cleanup deferred for %s: %v\n", buildName, err) + } + return nil +} + +func (m *Coordinator) verifyTartImageIdentity(ctx context.Context, name string) error { + instances, err := m.Provider.List(ctx) + if err != nil { return err } - m.infof("image build complete: %s is available in `tart list`\n", m.Config.Image.OutputImage) + instance, exists := findTartImage(instances, name) + if !exists || instance.ProviderID == "" { + return fmt.Errorf("Tart image %q did not pass immutable identity readback", name) + } return nil } -func (m *Manager) buildWSLImage(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { +func (m *Coordinator) restoreTartImage(ctx context.Context, outputName, backupName string) error { + if backupName == "" { + return errors.New("no previous Tart image exists to restore") + } + instances, err := m.Provider.List(ctx) + if err != nil { + return err + } + if current, exists := findTartImage(instances, outputName); exists { + if strings.EqualFold(current.State, "running") { + if err := m.Provider.Stop(ctx, outputName); err != nil { + return err + } + } + if err := m.Provider.Delete(ctx, outputName); err != nil { + return err + } + } + if err := m.Provider.Clone(ctx, backupName, outputName); err != nil { + return err + } + return m.verifyTartImageIdentity(ctx, outputName) +} + +func (m *Coordinator) buildWSLImage(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { return m.timeStartupStage("wsl_image_build", func() error { return m.buildWSLImageUntimed(ctx, opts, upstreamDir) }) } -func (m *Manager) buildWSLImageUntimed(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { +func (m *Coordinator) buildWSLImageUntimed(ctx context.Context, opts ImageBuildOptions, upstreamDir string) error { exporter, ok := m.Provider.(wslExporter) if !ok { return fmt.Errorf("provider.type=wsl requires provider export support") @@ -318,7 +568,7 @@ func (m *Manager) buildWSLImageUntimed(ctx context.Context, opts ImageBuildOptio if sourceType == "" { sourceType = config.ImageSourceRootFSTar } - buildName := RunnerName(m.Config.Pool.NamePrefix+"-image", 1, time.Now()) + buildName := m.runnerName(m.Config.Pool.NamePrefix+"-image", 1, time.Now()) buildLogPath := m.buildLogPath(imageLogStem(m.Config.Image.OutputImage) + ".wsl-build.log") guestLogPath := m.buildLogPath(buildName + ".guest.log") defer m.releaseTranscript(buildLogPath) @@ -326,21 +576,6 @@ func (m *Manager) buildWSLImageUntimed(ctx context.Context, opts ImageBuildOptio if err := resetLogs(buildLogPath, guestLogPath); err != nil { return err } - if !m.DryRun { - if _, err := os.Stat(outputPath); err == nil && !opts.Replace { - return fmt.Errorf("wsl output image %s already exists; rerun with --replace", outputPath) - } else if err != nil && !os.IsNotExist(err) { - return err - } - if opts.Replace { - if err := os.Remove(outputPath); err != nil && !os.IsNotExist(err) { - return err - } - if err := os.Remove(wslImageManifestSidecarPath(outputPath)); err != nil && !os.IsNotExist(err) { - return err - } - } - } if opts.Manifest == nil { manifest, err := m.desiredImageManifest(ctx) if err != nil { @@ -348,6 +583,31 @@ func (m *Manager) buildWSLImageUntimed(ctx context.Context, opts ImageBuildOptio } opts.Manifest = &manifest } + manifestHash, err := imageManifestHash(*opts.Manifest) + if err != nil { + return err + } + candidateOutputPath := outputPath + ".epar-candidate-" + manifestHash[:16] + if !m.DryRun { + if err := recoverWSLArtifactSwap(outputPath); err != nil { + return fmt.Errorf("recover interrupted WSL artifact activation: %w", err) + } + if err := removeRegularFileIfPresent(candidateOutputPath); err != nil { + return err + } + if err := removeRegularFileIfPresent(wslImageManifestSidecarPath(candidateOutputPath)); err != nil { + return err + } + if _, err := os.Stat(outputPath); err == nil && !opts.Replace { + return fmt.Errorf("wsl output image %s already exists; rerun with --replace", outputPath) + } else if err != nil && !os.IsNotExist(err) { + return err + } + defer func() { + _ = removeRegularFileIfPresent(candidateOutputPath) + _ = removeRegularFileIfPresent(wslImageManifestSidecarPath(candidateOutputPath)) + }() + } sourceForClone := m.Config.Image.SourceImage sourcePath := config.ProjectPath(m.ProjectRoot, sourceForClone) @@ -442,7 +702,7 @@ func (m *Manager) buildWSLImageUntimed(ctx context.Context, opts ImageBuildOptio return err } m.infof("installing GitHub Actions runner\n") - if _, err := m.execBuildGuest(ctx, buildName, []string{"sudo", "bash", "/opt/epar/install-runner.sh", m.Config.Image.RunnerVersion}, provider.ExecOptions{}); err != nil { + if err := m.installActionsRunnerPackage(ctx, buildName, *opts.Manifest); err != nil { return err } if err := m.installWSLDockerEngine(ctx, buildName); err != nil { @@ -462,20 +722,31 @@ func (m *Manager) buildWSLImageUntimed(ctx context.Context, opts ImageBuildOptio if err := m.Provider.Stop(ctx, buildName); err != nil { return err } - m.infof("exporting reusable WSL image to %s\n", outputPath) - if err := exporter.Export(ctx, buildName, m.Config.Image.OutputImage); err != nil { + m.infof("exporting reusable WSL image candidate to %s\n", candidateOutputPath) + if err := exporter.Export(ctx, buildName, candidateOutputPath); err != nil { return err } if !m.DryRun { - if err := writeStoredImageManifest(wslImageManifestSidecarPath(outputPath), *opts.Manifest); err != nil { + candidateSidecarPath := wslImageManifestSidecarPath(candidateOutputPath) + if err := writeStoredImageManifest(candidateSidecarPath, *opts.Manifest); err != nil { return err } + stored, err := readStoredImageManifest(candidateSidecarPath) + if err != nil { + return fmt.Errorf("read back WSL image candidate manifest: %w", err) + } + if stored.Hash != manifestHash { + return fmt.Errorf("WSL image candidate manifest mismatch: got %s, want %s", stored.Hash, manifestHash) + } + if err := activateWSLArtifact(candidateOutputPath, outputPath); err != nil { + return fmt.Errorf("activate WSL image %s: %w", outputPath, err) + } } m.infof("image build complete: %s is available for WSL imports\n", outputPath) return nil } -func (m *Manager) prepareWSLDockerSourceRootfs(ctx context.Context, outputPath, buildLogPath string, manifest ImageManifest) (string, string, error) { +func (m *Coordinator) prepareWSLDockerSourceRootfs(ctx context.Context, outputPath, buildLogPath string, manifest ImageManifest) (string, string, error) { image := strings.TrimSpace(m.Config.Image.SourceImage) if image == "" { return "", "", fmt.Errorf("image.sourceImage is required when image.sourceType=docker-image") @@ -544,12 +815,12 @@ func (m *Manager) prepareWSLDockerSourceRootfs(ctx context.Context, outputPath, return "", "", err } m.infof("preparing WSL source rootfs from Docker image %s\n", image) - if err := pullDockerSourceCommand(m, ctx, dockerSourcePullOptions{ + if err := m.pullDockerSource(ctx, DockerSourcePullOptions{ Image: image, Platform: platform, LogPath: buildLogPath, }); err != nil { - return "", "", fmt.Errorf("wsl image.sourceType=docker-image requires Docker Desktop, Docker Engine, or another reachable Docker daemon; alternatively set image.sourceType=rootfs-tar and provide a prepared rootfs tar: %w", err) + return "", "", fmt.Errorf("could not pull the WSL Docker image source; alternatively set image.sourceType=rootfs-tar and provide a prepared rootfs tar: %w", err) } if err := m.runHostLogged(ctx, buildLogPath, "docker", createArgs...); err != nil { return "", "", err @@ -557,9 +828,9 @@ func (m *Manager) prepareWSLDockerSourceRootfs(ctx context.Context, outputPath, defer func() { cleanupCtx, cancel := context.WithTimeout(context.Background(), time.Minute) defer cancel() - _ = runHostQuietCommand(cleanupCtx, "docker", "rm", "-f", containerName) + _ = m.runHostQuiet(cleanupCtx, "docker", "rm", "-f", containerName) }() - envJSON, err := runHostOutputCommand(ctx, "docker", "container", "inspect", "--format", "{{json .Config.Env}}", containerName) + envJSON, err := m.runHostOutput(ctx, "docker", "container", "inspect", "--format", "{{json .Config.Env}}", containerName) if err != nil { return "", "", err } @@ -587,8 +858,8 @@ func (m *Manager) prepareWSLDockerSourceRootfs(ctx context.Context, outputPath, return rootfsPath, envContent, nil } -func (m *Manager) dockerImageEnvContent(ctx context.Context, image string) (string, error) { - envJSON, err := runHostOutputCommand(ctx, "docker", "image", "inspect", "--format", "{{json .Config.Env}}", image) +func (m *Coordinator) dockerImageEnvContent(ctx context.Context, image string) (string, error) { + envJSON, err := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{json .Config.Env}}", image) if err != nil { return "", err } @@ -600,30 +871,21 @@ func (m *Manager) dockerImageEnvContent(ctx context.Context, image string) (stri } func wslDockerSourceRootfsPath(outputPath string) string { - switch { - case strings.HasSuffix(outputPath, ".tar.gz"): - return strings.TrimSuffix(outputPath, ".tar.gz") + ".source.rootfs.tar" - case strings.HasSuffix(outputPath, ".tgz"): - return strings.TrimSuffix(outputPath, ".tgz") + ".source.rootfs.tar" - case strings.HasSuffix(outputPath, ".tar"): - return strings.TrimSuffix(outputPath, ".tar") + ".source.rootfs.tar" - default: - return outputPath + ".source.rootfs.tar" - } + return WSLSourceRootfsPath(outputPath) } func wslDockerSourceContainerName() string { return fmt.Sprintf("epar-wsl-source-%d-%d", os.Getpid(), time.Now().UnixNano()) } -func (m *Manager) installSourceImageEnv(ctx context.Context, vmName, content string) error { +func (m *Coordinator) installSourceImageEnv(ctx context.Context, vmName, content string) error { if strings.TrimSpace(content) == "" { return nil } return provider.CopyText(ctx, m.Provider, vmName, "/opt/epar/source-image.env", "0644", content) } -func (m *Manager) prepareWSLDockerSourceGuest(ctx context.Context, vmName string) error { +func (m *Coordinator) prepareWSLDockerSourceGuest(ctx context.Context, vmName string) error { script := `set -euo pipefail cat >/etc/fstab <<'FSTAB' # EPAR: Docker image rootfs prepared for WSL imports. @@ -659,7 +921,7 @@ done return err } -func (m *Manager) installWSLDockerEngine(ctx context.Context, vmName string) error { +func (m *Coordinator) installWSLDockerEngine(ctx context.Context, vmName string) error { if m.Config.Provider.Type != "wsl" || m.Config.Image.SourceType != config.ImageSourceDockerImage { return nil } @@ -702,20 +964,20 @@ func validShellEnvName(name string) bool { return true } -func (m *Manager) RefreshScripts(ctx context.Context) error { +func (m *Coordinator) RefreshScripts(ctx context.Context) error { switch m.Config.Provider.Type { case "tart": return m.refreshTartScripts(ctx) case "wsl": return m.refreshWSLScripts(ctx) - case "docker-dind": - return m.buildDockerDindImage(ctx, ImageBuildOptions{Replace: true}, config.ProjectPath(m.ProjectRoot, m.Config.Image.UpstreamDir)) + case "docker-container": + return m.buildDockerContainerImage(ctx, ImageBuildOptions{Replace: true}, config.ProjectPath(m.ProjectRoot, m.Config.Image.UpstreamDir)) default: return fmt.Errorf("unsupported provider.type %q", m.Config.Provider.Type) } } -func (m *Manager) refreshTartScripts(ctx context.Context) error { +func (m *Coordinator) refreshTartScripts(ctx context.Context) error { logPath := m.buildLogPath(imageLogStem(m.Config.Image.OutputImage) + ".refresh.log") defer m.releaseTranscript(logPath) defer m.releaseTranscript(m.buildLogPath(imageLogStem(m.Config.Image.OutputImage) + ".guest.log")) @@ -756,7 +1018,7 @@ func (m *Manager) refreshTartScripts(ctx context.Context) error { return nil } -func (m *Manager) refreshWSLScripts(ctx context.Context) error { +func (m *Coordinator) refreshWSLScripts(ctx context.Context) error { exporter, ok := m.Provider.(wslExporter) if !ok { return fmt.Errorf("provider.type=wsl requires provider export support") @@ -767,7 +1029,7 @@ func (m *Manager) refreshWSLScripts(ctx context.Context) error { return fmt.Errorf("wsl image %s: %w", imagePath, err) } } - name := RunnerName(m.Config.Pool.NamePrefix+"-refresh", 1, time.Now()) + name := m.runnerName(m.Config.Pool.NamePrefix+"-refresh", 1, time.Now()) logPath := m.buildLogPath(imageLogStem(m.Config.Image.OutputImage) + ".wsl-refresh.log") defer m.releaseTranscript(logPath) defer m.releaseTranscript(m.buildLogPath(name + ".guest.log")) @@ -811,7 +1073,7 @@ func (m *Manager) refreshWSLScripts(ctx context.Context) error { return nil } -func (m *Manager) installGuestScripts(ctx context.Context, vmName string) error { +func (m *Coordinator) installGuestScripts(ctx context.Context, vmName string) error { scriptDir := filepath.Join(m.ProjectRoot, "scripts", "guest", "ubuntu") entries, err := os.ReadDir(scriptDir) if err != nil { @@ -836,8 +1098,8 @@ func (m *Manager) installGuestScripts(ctx context.Context, vmName string) error return nil } -func (m *Manager) startOptions(logPath, instance string) (provider.StartOptions, error) { - transcript, err := m.transcript(logPath, instance, transcriptComponent(logPath)) +func (m *Coordinator) startOptions(logPath, instance string) (provider.StartOptions, error) { + transcript, err := m.transcript(logPath, instance, m.transcriptComponent(logPath)) if err != nil { return provider.StartOptions{}, err } @@ -850,14 +1112,14 @@ func (m *Manager) startOptions(logPath, instance string) (provider.StartOptions, }, nil } -func (m *Manager) execBuildGuest(ctx context.Context, name string, command []string, opts provider.ExecOptions) (provider.ExecResult, error) { +func (m *Coordinator) execBuildGuest(ctx context.Context, name string, command []string, opts provider.ExecOptions) (provider.ExecResult, error) { if opts.LogPath == "" { opts.LogPath = m.buildLogPath(imageLogStem(name) + ".guest.log") } return m.execGuest(ctx, name, command, opts) } -func (m *Manager) installRosettaSupport(ctx context.Context, vmName string) error { +func (m *Coordinator) installRosettaSupport(ctx context.Context, vmName string) error { if m.Config.Provider.Type != "tart" || strings.TrimSpace(m.Config.Provider.RosettaTag) == "" { return nil } @@ -868,7 +1130,7 @@ func (m *Manager) installRosettaSupport(ctx context.Context, vmName string) erro return err } -func (m *Manager) copyRunnerImagesSubset(ctx context.Context, vmName, upstreamDir string) error { +func (m *Coordinator) copyRunnerImagesSubset(ctx context.Context, vmName, upstreamDir string) error { type copyRoot struct { host string guest string @@ -945,7 +1207,7 @@ func (m *Manager) copyRunnerImagesSubset(ctx context.Context, vmName, upstreamDi return nil } -func (m *Manager) copyRunnerImagesCommitToGuest(ctx context.Context, vmName string) error { +func (m *Coordinator) copyRunnerImagesCommitToGuest(ctx context.Context, vmName string) error { commit, err := m.runnerImagesCommit() if err != nil { return err @@ -956,11 +1218,11 @@ func (m *Manager) copyRunnerImagesCommitToGuest(ctx context.Context, vmName stri return provider.CopyText(ctx, m.Provider, vmName, "/opt/epar/upstream/runner-images/epar-commit", "0644", commit+"\n") } -func (m *Manager) runnerImageBuildScripts() []string { +func (m *Coordinator) runnerImageBuildScripts() []string { return []string{"install-docker.sh", "install-google-chrome.sh", "install-nodejs.sh"} } -func (m *Manager) runnerImagesCopyMode() runnerImagesCopyMode { +func (m *Coordinator) runnerImagesCopyMode() runnerImagesCopyMode { for _, script := range m.Config.Image.CustomInstallScripts { normalized := m.normalizedCustomInstallScript(script) switch normalized { @@ -972,11 +1234,11 @@ func (m *Manager) runnerImagesCopyMode() runnerImagesCopyMode { return runnerImagesCopyNone } -func (m *Manager) prepareDockerDindBuildContext(buildCtx, upstreamDir, manifestContent string) error { - return m.prepareDockerDindBuildContextWithHostTrust(buildCtx, upstreamDir, manifestContent, hosttrust.Snapshot{}) +func (m *Coordinator) prepareDockerContainerBuildContext(buildCtx, upstreamDir, manifestContent string) error { + return m.prepareDockerContainerBuildContextWithHostTrust(buildCtx, upstreamDir, manifestContent, hosttrust.Snapshot{}) } -func (m *Manager) prepareDockerDindBuildContextWithHostTrust(buildCtx, upstreamDir, manifestContent string, snapshot hosttrust.Snapshot) error { +func (m *Coordinator) prepareDockerContainerBuildContextWithHostTrust(buildCtx, upstreamDir, manifestContent string, snapshot hosttrust.Snapshot) error { if err := copyDir(filepath.Join(m.ProjectRoot, "scripts", "guest", "ubuntu"), filepath.Join(buildCtx, "scripts", "guest", "ubuntu")); err != nil { return err } @@ -989,7 +1251,7 @@ func (m *Manager) prepareDockerDindBuildContextWithHostTrust(buildCtx, upstreamD } switch m.runnerImagesCopyMode() { case runnerImagesCopySubset: - m.infof("preparing Docker-DinD build context with runner-images script subset\n") + m.infof("preparing Docker Container build context with runner-images script subset\n") if err := copyRunnerImagesSubsetToDir(upstreamDir, upstreamDest, m.runnerImageBuildScripts()); err != nil { return err } @@ -997,7 +1259,7 @@ func (m *Manager) prepareDockerDindBuildContextWithHostTrust(buildCtx, upstreamD return err } case runnerImagesCopyNone: - m.infof("preparing Docker-DinD build context without runner-images resources\n") + m.infof("preparing Docker Container build context without runner-images resources\n") } customDir := filepath.Join(buildCtx, "custom-install") if err := os.MkdirAll(customDir, 0755); err != nil { @@ -1027,10 +1289,11 @@ func (m *Manager) prepareDockerDindBuildContextWithHostTrust(buildCtx, upstreamD dockerfile := fmt.Sprintf(`ARG BASE_IMAGE=ghcr.io/catthehacker/ubuntu:full-latest FROM ${BASE_IMAGE} USER root -ARG RUNNER_VERSION=latest +ARG RUNNER_VERSION +ARG RUNNER_SHA256 ARG EPAR_IMAGE_MANIFEST_SHA256 ARG OCI_SOURCE=https://github.com/solutionforest/ephemeral-action-runner -ARG OCI_DESCRIPTION="EPAR Docker-DinD runner image" +ARG OCI_DESCRIPTION="EPAR Docker Container runner image" ARG OCI_LICENSES=MIT LABEL org.opencontainers.image.source="${OCI_SOURCE}" LABEL org.opencontainers.image.description="${OCI_DESCRIPTION}" @@ -1049,10 +1312,11 @@ COPY trusted-ca-certificates/ `+trustedCAGuestDir+`/ COPY host-trust-certificates/ `+hostTrustGuestDir+`/ COPY host-trust-metadata/ /opt/epar/ COPY image-manifest.json /opt/epar/image-manifest.json +COPY inputs/actions-runner.tar.gz /opt/epar/actions-runner.tar.gz RUN chmod 0755 /opt/epar/*.sh /opt/epar/container-entrypoint.sh /opt/epar/custom-install/*.sh 2>/dev/null || true RUN bash /opt/epar/install-trusted-ca-certificates.sh RUN bash /opt/epar/install-base.sh /opt/epar/upstream/runner-images -RUN bash /opt/epar/install-runner.sh "${RUNNER_VERSION}" +RUN bash /opt/epar/install-runner.sh "${RUNNER_VERSION}" /opt/epar/actions-runner.tar.gz "${RUNNER_SHA256}" RUN EPAR_CONTAINER_IMAGE_BUILD=true bash /opt/epar/install-docker-engine.sh /opt/epar/upstream/runner-images %sRUN EPAR_CONTAINER_IMAGE_BUILD=true bash /opt/epar/validate-runtime.sh RUN bash /opt/epar/finalize-image.sh @@ -1061,7 +1325,7 @@ ENTRYPOINT ["/opt/epar/container-entrypoint.sh"] return os.WriteFile(filepath.Join(buildCtx, "Dockerfile"), []byte(dockerfile), 0644) } -func (m *Manager) normalizedCustomInstallScript(script string) string { +func (m *Coordinator) normalizedCustomInstallScript(script string) string { script = strings.TrimSpace(script) if script == "" { return "" @@ -1078,7 +1342,7 @@ func (m *Manager) normalizedCustomInstallScript(script string) string { return filepath.ToSlash(filepath.Clean(script)) } -func (m *Manager) installCustomInstallScripts(ctx context.Context, vmName string) error { +func (m *Coordinator) installCustomInstallScripts(ctx context.Context, vmName string) error { scripts := m.Config.Image.CustomInstallScripts if len(scripts) == 0 { return nil @@ -1108,7 +1372,7 @@ func (m *Manager) installCustomInstallScripts(ctx context.Context, vmName string return nil } -func (m *Manager) customInstallScriptHostPath(script string) (string, error) { +func (m *Coordinator) customInstallScriptHostPath(script string) (string, error) { script = strings.TrimSpace(script) if script == "" { return "", fmt.Errorf("custom install script path is empty") @@ -1159,13 +1423,13 @@ func guestScriptName(name string) string { return b.String() } -func (m *Manager) enableWSLSystemd(ctx context.Context, name string) error { +func (m *Coordinator) enableWSLSystemd(ctx context.Context, name string) error { content := "[boot]\nsystemd=true\n\n[interop]\nappendWindowsPath=false\n\n[user]\ndefault=root\n" _, err := m.execBuildGuest(ctx, name, provider.ShellCommand("mkdir -p /etc && cat >/etc/wsl.conf"), provider.ExecOptions{Stdin: content, LogPath: m.buildLogPath(name + ".guest.log")}) return err } -func (m *Manager) waitForSystemd(ctx context.Context, name string) error { +func (m *Coordinator) waitForSystemd(ctx context.Context, name string) error { waitSeconds := m.Config.Timeouts.BootSeconds if waitSeconds <= 0 { waitSeconds = 180 @@ -1266,7 +1530,7 @@ func copyRunnerImagesSubsetToDir(upstreamDir, dest string, buildScripts []string return nil } -func (m *Manager) runnerImagesCommit() (string, error) { +func (m *Coordinator) runnerImagesCommit() (string, error) { lockPath := config.ProjectPath(m.ProjectRoot, m.Config.Image.UpstreamLock) content, err := os.ReadFile(lockPath) if err != nil { @@ -1278,7 +1542,7 @@ func (m *Manager) runnerImagesCommit() (string, error) { return strings.TrimSpace(string(content)), nil } -func (m *Manager) writeRunnerImagesCommitFile(dest string) error { +func (m *Coordinator) writeRunnerImagesCommitFile(dest string) error { commit, err := m.runnerImagesCommit() if err != nil { return err diff --git a/internal/image/buildx.go b/internal/image/buildx.go new file mode 100644 index 0000000..614df44 --- /dev/null +++ b/internal/image/buildx.go @@ -0,0 +1,552 @@ +package image + +import ( + "bytes" + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "runtime" + "sort" + "strconv" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/filelock" + "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +const ( + buildxMetadataSchemaVersion = 4 + legacyBuildxMaxSchemaVersion = 3 + buildkitImageReference = "moby/buildkit:buildx-stable-1" +) + +type BuildxMetadata struct { + SchemaVersion int `json:"schemaVersion"` + Builder string `json:"builder"` + Driver string `json:"driver"` + ProjectRoot string `json:"projectRoot"` + ConfigID string `json:"configId,omitempty"` + EPARConfigPath string `json:"eparConfigPath,omitempty"` + CacheLimit string `json:"cacheLimit"` + ConfigPath string `json:"configPath"` + ConfigSHA256 string `json:"configSha256,omitempty"` + TrustGeneration string `json:"trustGeneration,omitempty"` + CertificateBundle string `json:"certificateBundle,omitempty"` + CertificateSHA256 string `json:"certificateSha256,omitempty"` + RegistryHosts []string `json:"registryHosts,omitempty"` + BuildKitImageID string `json:"buildkitImageId,omitempty"` + CreatedAt time.Time `json:"createdAt"` + LastReconciledAt time.Time `json:"lastReconciledAt,omitempty"` +} + +// BuildxMetadataPath returns the dedicated Buildx metadata path for the +// project's default configuration. Callers that accept --config must use +// BuildxMetadataPathForConfig so independent controllers never share mutable +// BuildKit state. +func BuildxMetadataPath(projectRoot string) string { + path, err := BuildxMetadataPathForConfig(projectRoot, filepath.Join(projectRoot, ".local", "config.yml")) + if err != nil { + return filepath.Join(projectRoot, ".local", "storage", "buildx", "invalid", "metadata.json") + } + return path +} + +func BuildxMetadataPathForConfig(projectRoot, configPath string) (string, error) { + scope, err := resolveBuildxScope(projectRoot, configPath) + if err != nil { + return "", err + } + return scope.metadataPath, nil +} + +// LegacyBuildxMetadataPath identifies the project-scoped metadata used before +// Buildx resources became config-scoped. It remains discoverable for exact +// inventory and operator-directed cleanup; current controllers never mutate +// or adopt the legacy builder implicitly. +func LegacyBuildxMetadataPath(projectRoot string) string { + return filepath.Join(projectRoot, ".local", "storage", "buildx.json") +} + +func LoadLegacyBuildxMetadata(projectRoot string) (BuildxMetadata, error) { + content, err := os.ReadFile(LegacyBuildxMetadataPath(projectRoot)) + if err != nil { + return BuildxMetadata{}, err + } + var metadata BuildxMetadata + if err := json.Unmarshal(content, &metadata); err != nil { + return BuildxMetadata{}, err + } + canonicalRoot, err := storagecatalog.CanonicalPath(projectRoot) + if err != nil { + return BuildxMetadata{}, err + } + metadataRoot, err := storagecatalog.CanonicalPath(metadata.ProjectRoot) + if err != nil { + return BuildxMetadata{}, fmt.Errorf("invalid legacy EPAR Buildx ownership metadata") + } + if metadata.SchemaVersion < 1 || metadata.SchemaVersion > legacyBuildxMaxSchemaVersion || metadata.Builder != legacyBuildxBuilderName(metadata.ProjectRoot) || metadata.Driver != "docker-container" || metadataRoot != canonicalRoot || strings.TrimSpace(metadata.ConfigPath) == "" { + return BuildxMetadata{}, fmt.Errorf("invalid legacy EPAR Buildx ownership metadata") + } + return metadata, nil +} + +func LoadBuildxMetadata(projectRoot string) (BuildxMetadata, error) { + return LoadBuildxMetadataForConfig(projectRoot, filepath.Join(projectRoot, ".local", "config.yml")) +} + +func LoadBuildxMetadataForConfig(projectRoot, configPath string) (BuildxMetadata, error) { + scope, err := resolveBuildxScope(projectRoot, configPath) + if err != nil { + return BuildxMetadata{}, err + } + content, err := os.ReadFile(scope.metadataPath) + if err != nil { + return BuildxMetadata{}, err + } + var metadata BuildxMetadata + if err := json.Unmarshal(content, &metadata); err != nil { + return BuildxMetadata{}, err + } + if metadata.SchemaVersion != buildxMetadataSchemaVersion || metadata.Builder != "epar-"+scope.configID || metadata.Driver != "docker-container" || filepath.Clean(metadata.ProjectRoot) != scope.projectRoot || metadata.ConfigID != scope.configID || filepath.Clean(metadata.EPARConfigPath) != scope.configPath || filepath.Clean(metadata.ConfigPath) != scope.buildkitConfig { + return BuildxMetadata{}, fmt.Errorf("invalid EPAR Buildx ownership metadata") + } + return metadata, nil +} + +func buildxBuilderName(projectRoot string) string { + builder, err := buildxBuilderNameForConfig(projectRoot, filepath.Join(projectRoot, ".local", "config.yml")) + if err != nil { + return "epar-invalid" + } + return builder +} + +func buildxBuilderNameForConfig(projectRoot, configPath string) (string, error) { + scope, err := resolveBuildxScope(projectRoot, configPath) + if err != nil { + return "", err + } + return "epar-" + scope.configID, nil +} + +func legacyBuildxBuilderName(projectRoot string) string { + canonical := filepath.Clean(projectRoot) + if runtime.GOOS == "windows" { + canonical = strings.ToLower(canonical) + } + sum := sha256.Sum256([]byte(canonical)) + return "epar-" + hex.EncodeToString(sum[:6]) +} + +type buildxScope struct { + configID string + projectRoot string + configPath string + metadataPath string + lockPath string + buildkitConfig string + certificateDir string +} + +func resolveBuildxScope(projectRoot, configPath string) (buildxScope, error) { + if strings.TrimSpace(configPath) == "" { + configPath = filepath.Join(projectRoot, ".local", "config.yml") + } + configID, err := storagecatalog.ConfigID(projectRoot, configPath) + if err != nil { + return buildxScope{}, err + } + canonicalRoot, err := canonicalBuildxPath(projectRoot) + if err != nil { + return buildxScope{}, err + } + canonicalConfig, err := canonicalBuildxPath(configPath) + if err != nil { + return buildxScope{}, err + } + storageRoot := filepath.Join(canonicalRoot, ".local", "storage") + buildxRoot := filepath.Join(storageRoot, "buildx", configID) + return buildxScope{ + configID: configID, + projectRoot: canonicalRoot, + configPath: canonicalConfig, + metadataPath: filepath.Join(buildxRoot, "metadata.json"), + lockPath: filepath.Join(buildxRoot, "reconcile.lock"), + buildkitConfig: filepath.Join(storageRoot, "buildkit", configID, "buildkitd.toml"), + certificateDir: filepath.Join(storageRoot, "buildkit-certs", configID), + }, nil +} + +func canonicalBuildxPath(path string) (string, error) { + return storagecatalog.CanonicalPath(path) +} + +func (m *Coordinator) ensureBuildxBuilder(ctx context.Context, registryReferences []string) (string, error) { + scope, err := resolveBuildxScope(m.ProjectRoot, m.effectiveConfigPath()) + if err != nil { + return "", fmt.Errorf("resolve Buildx configuration scope: %w", err) + } + builder := "epar-" + scope.configID + if legacy, legacyErr := LoadLegacyBuildxMetadata(m.ProjectRoot); legacyErr == nil { + m.warnf("Legacy project-scoped EPAR Buildx builder %q remains recorded at %s; it is not reused by config-scoped controllers and remains visible to storage inventory for explicit cleanup.\n", legacy.Builder, LegacyBuildxMetadataPath(m.ProjectRoot)) + } else if !os.IsNotExist(legacyErr) { + m.warnf("Legacy EPAR Buildx metadata at %s is invalid and was left untouched: %v\n", LegacyBuildxMetadataPath(m.ProjectRoot), legacyErr) + } + cacheLimit := strings.TrimSpace(m.Config.Storage.BuildCacheLimit) + if cacheLimit == "" { + cacheLimit = "20GiB" + } + limitBytes, err := config.ParseByteSize(cacheLimit) + if err != nil { + return "", fmt.Errorf("parse storage.buildCacheLimit: %w", err) + } + registryHosts, err := buildRegistryHosts(registryReferences) + if err != nil { + return "", err + } + if m.DryRun { + m.infof("[dry-run] ensure EPAR-owned Buildx builder %s with cache limit %s for registries %s\n", builder, cacheLimit, strings.Join(registryHosts, ", ")) + return builder, nil + } + var buildKitImageID string + if err := func() error { + backendID, releaseBackend, acquireErr := m.acquireDockerBackendLock(ctx) + if acquireErr != nil { + return acquireErr + } + defer releaseBackend() + previousBuildKitImageID := "" + if output, inspectErr := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", buildkitImageReference); inspectErr == nil { + previousBuildKitImageID = strings.TrimSpace(output) + } + if journalErr := m.beginDockerRoleAcquisition(backendID, "buildkit-image", buildkitImageReference, previousBuildKitImageID, time.Now().UTC()); journalErr != nil { + return fmt.Errorf("journal EPAR BuildKit image acquisition: %w", journalErr) + } + if pullErr := m.runHost(ctx, "docker", "pull", buildkitImageReference); pullErr != nil { + return fmt.Errorf("resolve EPAR BuildKit image %s: %w", buildkitImageReference, pullErr) + } + output, inspectErr := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", buildkitImageReference) + if inspectErr != nil { + return fmt.Errorf("read EPAR BuildKit image identity: %w", inspectErr) + } + buildKitImageID = strings.TrimSpace(output) + if recordErr := m.recordDockerRoleAcquisition(ctx, "buildkit-image", buildkitImageReference, previousBuildKitImageID, buildKitImageID, time.Now().UTC()); recordErr != nil { + return fmt.Errorf("record EPAR BuildKit image acquisition: %w", recordErr) + } + return nil + }(); err != nil { + return "", err + } + trust, err := m.resolveBuildTrust(ctx) + if err != nil { + return "", err + } + if len(trust.Certificates) == 0 { + return "", fmt.Errorf("operational BuildKit trust resolved no system CA certificates") + } + bundle, err := buildTrustBundle(trust) + if err != nil { + return "", err + } + bundleSHA := sha256.Sum256(bundle) + certificatePath := filepath.Join(scope.certificateDir, trust.Generation, "ca.pem") + configPath := scope.buildkitConfig + configContent := buildkitConfig(uint64(limitBytes), trust.Generation, certificatePath, registryHosts) + configSHA := sha256.Sum256(configContent) + expected := BuildxMetadata{ + SchemaVersion: buildxMetadataSchemaVersion, + Builder: builder, + Driver: "docker-container", + ProjectRoot: scope.projectRoot, + ConfigID: scope.configID, + EPARConfigPath: scope.configPath, + CacheLimit: cacheLimit, + ConfigPath: configPath, + ConfigSHA256: hex.EncodeToString(configSHA[:]), + TrustGeneration: trust.Generation, + CertificateBundle: certificatePath, + CertificateSHA256: hex.EncodeToString(bundleSHA[:]), + RegistryHosts: registryHosts, + BuildKitImageID: buildKitImageID, + } + storageDirectory := filepath.Join(scope.projectRoot, ".local", "storage") + if err := validateRegularParent(storageDirectory, scope.projectRoot); err != nil { + return "", fmt.Errorf("validate EPAR storage directory: %w", err) + } + if err := validateRegularParent(filepath.Dir(scope.lockPath), storageDirectory); err != nil { + return "", fmt.Errorf("validate EPAR Buildx lock directory: %w", err) + } + if err := validateRegularParent(filepath.Dir(configPath), storageDirectory); err != nil { + return "", fmt.Errorf("validate EPAR BuildKit configuration directory: %w", err) + } + reconcileLock, err := filelock.Acquire(scope.lockPath) + if err != nil { + if errors.Is(err, filelock.ErrLocked) { + return "", fmt.Errorf("another EPAR process is reconciling the project Buildx builder") + } + return "", err + } + defer reconcileLock.Close() + + if err := validateRegularParent(filepath.Dir(certificatePath), storageDirectory); err != nil { + return "", fmt.Errorf("validate BuildKit certificate generation directory: %w", err) + } + if err := writeAtomicFile(certificatePath, bundle, 0o600); err != nil { + return "", fmt.Errorf("write operational BuildKit CA bundle: %w", err) + } + if err := writeAtomicFile(configPath, configContent, 0o600); err != nil { + return "", fmt.Errorf("write EPAR BuildKit configuration: %w", err) + } + + metadata, metadataErr := LoadBuildxMetadataForConfig(m.ProjectRoot, m.effectiveConfigPath()) + _, inspectErr := m.runHostOutput(ctx, "docker", "buildx", "inspect", builder) + if inspectErr == nil { + if metadataErr != nil { + return "", fmt.Errorf("Buildx builder %q already exists without valid EPAR ownership metadata; refusing to adopt it: %w", builder, metadataErr) + } + if !buildxOwnershipMatches(metadata, expected) { + return "", fmt.Errorf("Buildx builder %q ownership metadata does not match this project; refusing to remove or adopt it", builder) + } + if !buildxMetadataMatches(metadata, expected) { + m.infof("upgrading EPAR-owned Buildx builder %s for operational trust generation %s while preserving its cache state\n", builder, trust.Generation) + if err := m.runHost(ctx, "docker", "buildx", "rm", "--keep-state", "--force", builder); err != nil { + return "", fmt.Errorf("remove outdated EPAR Buildx builder %q while retaining state: %w", builder, err) + } + inspectErr = fmt.Errorf("builder removed for owned upgrade") + } + } else if metadataErr == nil && !buildxOwnershipMatches(metadata, expected) { + return "", fmt.Errorf("EPAR Buildx metadata does not match this project; refusing to reuse it") + } + + if inspectErr != nil { + if err := m.runHost(ctx, "docker", "buildx", "create", "--name", builder, "--driver", "docker-container", "--driver-opt", "image="+buildkitImageReference, "--buildkitd-config", configPath); err != nil { + return "", fmt.Errorf("create EPAR Buildx builder %q: %w", builder, err) + } + } + if err := m.runHost(ctx, "docker", "buildx", "inspect", "--bootstrap", builder); err != nil { + return "", fmt.Errorf("bootstrap EPAR Buildx builder %q: %w", builder, err) + } + if err := m.verifyBuildxConfiguration(ctx, builder, expected); err != nil { + return "", err + } + now := time.Now().UTC() + if metadataErr == nil && !metadata.CreatedAt.IsZero() { + expected.CreatedAt = metadata.CreatedAt + } else { + expected.CreatedAt = now + } + expected.LastReconciledAt = now + content, err := json.MarshalIndent(expected, "", " ") + if err != nil { + return "", err + } + if err := writeAtomicFile(scope.metadataPath, append(content, '\n'), 0o600); err != nil { + return "", fmt.Errorf("publish EPAR Buildx ownership metadata: %w", err) + } + return builder, nil +} + +func buildxOwnershipMatches(actual, expected BuildxMetadata) bool { + return actual.Builder == expected.Builder && + actual.Driver == expected.Driver && + filepath.Clean(actual.ProjectRoot) == expected.ProjectRoot && + actual.ConfigID == expected.ConfigID && + filepath.Clean(actual.EPARConfigPath) == expected.EPARConfigPath && + filepath.Clean(actual.ConfigPath) == expected.ConfigPath +} + +func buildxMetadataMatches(actual, expected BuildxMetadata) bool { + return actual.SchemaVersion == expected.SchemaVersion && + buildxOwnershipMatches(actual, expected) && + actual.CacheLimit == expected.CacheLimit && + actual.ConfigSHA256 == expected.ConfigSHA256 && + actual.TrustGeneration == expected.TrustGeneration && + filepath.Clean(actual.CertificateBundle) == expected.CertificateBundle && + actual.CertificateSHA256 == expected.CertificateSHA256 && + actual.BuildKitImageID == expected.BuildKitImageID && + sameOrderedStrings(actual.RegistryHosts, expected.RegistryHosts) +} + +func buildRegistryHosts(references []string) ([]string, error) { + seen := make(map[string]struct{}) + for _, reference := range references { + reference = strings.TrimSpace(reference) + if reference == "" { + continue + } + parts := strings.SplitN(reference, "/", 2) + first := parts[0] + host := "docker.io" + if len(parts) == 2 && (strings.Contains(first, ".") || strings.Contains(first, ":") || first == "localhost") { + host = strings.ToLower(first) + } + if host == "index.docker.io" || host == "registry-1.docker.io" { + host = "docker.io" + } + if strings.ContainsAny(host, "\"'[] \t\r\n") { + return nil, fmt.Errorf("invalid registry host derived from %q", reference) + } + seen[host] = struct{}{} + } + hosts := make([]string, 0, len(seen)) + for host := range seen { + hosts = append(hosts, host) + } + sort.Strings(hosts) + return hosts, nil +} + +func buildTrustBundle(snapshot hosttrust.Snapshot) ([]byte, error) { + canonical, err := hosttrust.Canonicalize(snapshot) + if err != nil { + return nil, err + } + var bundle bytes.Buffer + for _, certificate := range canonical.Certificates { + bundle.Write(certificate.PEM) + } + return bundle.Bytes(), nil +} + +func buildkitConfig(cacheLimit uint64, generation, certificatePath string, registryHosts []string) []byte { + var content strings.Builder + fmt.Fprintf(&content, "# epar-build-trust-generation=%s\n", generation) + content.WriteString("[worker.oci]\n") + content.WriteString(" gc = true\n") + fmt.Fprintf(&content, " reservedSpace = %s\n", strconv.Quote("2GiB")) + fmt.Fprintf(&content, " maxUsedSpace = %s\n", strconv.Quote(strconv.FormatUint(cacheLimit, 10)+"B")) + fmt.Fprintf(&content, " minFreeSpace = %s\n", strconv.Quote("1GiB")) + certificatePath = strings.ReplaceAll(filepath.Clean(certificatePath), `\`, "/") + for _, host := range registryHosts { + fmt.Fprintf(&content, "\n[registry.%s]\n", strconv.Quote(host)) + fmt.Fprintf(&content, " ca = [%s]\n", strconv.Quote(certificatePath)) + } + return []byte(content.String()) +} + +func (m *Coordinator) verifyBuildxConfiguration(ctx context.Context, builder string, expected BuildxMetadata) error { + container := buildxControlContainer(builder) + imageID, err := m.runHostOutput(ctx, "docker", "inspect", "--format", "{{.Image}}", container) + if err != nil { + return fmt.Errorf("verify EPAR Buildx builder %q image identity: %w", builder, err) + } + if strings.TrimSpace(imageID) != expected.BuildKitImageID { + return fmt.Errorf("EPAR Buildx builder %q uses image %s, expected %s", builder, strings.TrimSpace(imageID), expected.BuildKitImageID) + } + content, err := m.runHostOutput(ctx, "docker", "exec", container, "cat", "/etc/buildkit/buildkitd.toml") + if err != nil { + return fmt.Errorf("verify EPAR Buildx builder %q configuration readback: %w", builder, err) + } + for _, host := range expected.RegistryHosts { + doubleQuoted := "[registry." + strconv.Quote(host) + "]" + singleQuoted := "[registry.'" + host + "']" + if !strings.Contains(content, doubleQuoted) && !strings.Contains(content, singleQuoted) { + return fmt.Errorf("EPAR Buildx builder %q did not install registry trust for %s", builder, host) + } + installedBundle, err := m.runHostOutput(ctx, "docker", "exec", container, "cat", "/etc/buildkit/certs/"+host+"/ca.pem") + if err != nil { + return fmt.Errorf("verify EPAR Buildx builder %q CA readback for %s: %w", builder, host, err) + } + sum := sha256.Sum256([]byte(installedBundle)) + if hex.EncodeToString(sum[:]) != expected.CertificateSHA256 { + return fmt.Errorf("EPAR Buildx builder %q CA readback digest for %s does not match trust generation %s", builder, host, expected.TrustGeneration) + } + } + return nil +} + +func buildxControlContainer(builder string) string { + return "buildx_buildkit_" + builder + "0" +} + +func validateRegularParent(path, allowedRoot string) error { + absoluteRoot, err := filepath.Abs(allowedRoot) + if err != nil { + return err + } + absolutePath, err := filepath.Abs(path) + if err != nil { + return err + } + relative, err := filepath.Rel(absoluteRoot, absolutePath) + if err != nil || relative == ".." || strings.HasPrefix(relative, ".."+string(filepath.Separator)) { + return fmt.Errorf("%s is outside %s", absolutePath, absoluteRoot) + } + current := absoluteRoot + for _, part := range strings.Split(relative, string(filepath.Separator)) { + if part == "" || part == "." { + continue + } + current = filepath.Join(current, part) + info, statErr := os.Lstat(current) + if os.IsNotExist(statErr) { + if err := os.Mkdir(current, 0o700); err != nil { + return err + } + continue + } + if statErr != nil { + return statErr + } + if info.Mode()&os.ModeSymlink != 0 || !info.IsDir() { + return fmt.Errorf("%s is not a real directory", current) + } + } + return nil +} + +func sameOrderedStrings(left, right []string) bool { + if len(left) != len(right) { + return false + } + for index := range left { + if left[index] != right[index] { + return false + } + } + return true +} + +func writeAtomicFile(path string, content []byte, mode os.FileMode) error { + if err := os.MkdirAll(filepath.Dir(path), 0755); err != nil { + return err + } + if existing, err := os.ReadFile(path); err == nil && bytes.Equal(existing, content) { + return nil + } + temporary, err := os.CreateTemp(filepath.Dir(path), "."+filepath.Base(path)+".*") + if err != nil { + return err + } + temporaryPath := temporary.Name() + defer os.Remove(temporaryPath) + if err := temporary.Chmod(mode); err != nil { + temporary.Close() + return err + } + if _, err := temporary.Write(content); err != nil { + temporary.Close() + return err + } + if err := temporary.Sync(); err != nil { + temporary.Close() + return err + } + if err := temporary.Close(); err != nil { + return err + } + if runtime.GOOS == "windows" { + if err := os.Remove(path); err != nil && !os.IsNotExist(err) { + return err + } + } + return os.Rename(temporaryPath, path) +} diff --git a/internal/image/buildx_progress.go b/internal/image/buildx_progress.go new file mode 100644 index 0000000..b0f06a2 --- /dev/null +++ b/internal/image/buildx_progress.go @@ -0,0 +1,242 @@ +package image + +import ( + "bytes" + "fmt" + "io" + "math" + "regexp" + "strconv" + "strings" + "sync" + "time" +) + +const buildxProgressReportInterval = 2 * time.Second + +var ( + buildxLayerProgressPattern = regexp.MustCompile(`^#([0-9]+)\s+(sha256:[0-9a-f]{64})\s+([0-9]+(?:\.[0-9]+)?)([kMGTPE]?i?B)\s+/\s+([0-9]+(?:\.[0-9]+)?)([kMGTPE]?i?B)(?:\s+[0-9]+(?:\.[0-9]+)?s)?(?:\s+(done))?\s*$`) + buildxStepPattern = regexp.MustCompile(`^#([0-9]+)(?:\s|$)`) +) + +type buildxLayerProgress struct { + current int64 + total int64 + complete bool +} + +// BuildxProgressSnapshot is a bounded summary of BuildKit plain progress. +type BuildxProgressSnapshot struct { + CurrentBytes int64 + TotalBytes int64 + CompletedLayers int + KnownLayers int + ActiveStep int + Elapsed time.Duration + ObservedProgress bool + ObservedByteTotal bool +} + +// BuildxProgressMonitor consumes one or more BuildKit plain-progress streams. +// Call NewStream separately for stdout and stderr so partial writes cannot +// corrupt each other's line framing. +type BuildxProgressMonitor struct { + mu sync.Mutex + started time.Time + now func() time.Time + interval time.Duration + lastReport time.Time + layers map[string]buildxLayerProgress + activeStep int + observed bool + report func(BuildxProgressSnapshot) +} + +// NewBuildxProgressMonitor creates a monitor suitable for a live Buildx build. +func NewBuildxProgressMonitor(report func(BuildxProgressSnapshot)) *BuildxProgressMonitor { + return newBuildxProgressMonitor(buildxProgressReportInterval, time.Now, report) +} + +func newBuildxProgressMonitor(interval time.Duration, now func() time.Time, report func(BuildxProgressSnapshot)) *BuildxProgressMonitor { + started := now() + return &BuildxProgressMonitor{ + started: started, + now: now, + interval: interval, + layers: make(map[string]buildxLayerProgress), + report: report, + } +} + +// NewStream returns an independent line-framing writer for one process stream. +func (monitor *BuildxProgressMonitor) NewStream() *BuildxProgressStream { + return &BuildxProgressStream{monitor: monitor} +} + +// Snapshot returns the current summary. +func (monitor *BuildxProgressMonitor) Snapshot() BuildxProgressSnapshot { + monitor.mu.Lock() + defer monitor.mu.Unlock() + return monitor.snapshotLocked(monitor.now()) +} + +func (monitor *BuildxProgressMonitor) consumeLine(line string) { + line = strings.TrimSpace(line) + if line == "" { + return + } + now := monitor.now() + monitor.mu.Lock() + changed := monitor.consumeLineLocked(line) + if !changed || (monitor.observed && !monitor.lastReport.IsZero() && now.Sub(monitor.lastReport) < monitor.interval) { + monitor.mu.Unlock() + return + } + monitor.observed = true + monitor.lastReport = now + snapshot := monitor.snapshotLocked(now) + report := monitor.report + monitor.mu.Unlock() + if report != nil { + report(snapshot) + } +} + +func (monitor *BuildxProgressMonitor) consumeLineLocked(line string) bool { + layerMatch := buildxLayerProgressPattern.FindStringSubmatch(line) + if layerMatch == nil && strings.Contains(line, "sha256:") && strings.Contains(line, " / ") { + return false + } + var current, total int64 + if layerMatch != nil { + var currentOK, totalOK bool + current, currentOK = parseBuildxBytes(layerMatch[3], layerMatch[4]) + total, totalOK = parseBuildxBytes(layerMatch[5], layerMatch[6]) + if !currentOK || !totalOK || total <= 0 { + return false + } + } + stepMatch := buildxStepPattern.FindStringSubmatch(line) + if stepMatch == nil { + return false + } + step, err := strconv.Atoi(stepMatch[1]) + if err != nil { + return false + } + changed := step != monitor.activeStep + monitor.activeStep = step + + if layerMatch == nil { + return changed + } + if current > total { + current = total + } + layer := buildxLayerProgress{ + current: current, + total: total, + complete: layerMatch[7] == "done" || current == total, + } + previous, exists := monitor.layers[layerMatch[2]] + if !exists || previous != layer { + monitor.layers[layerMatch[2]] = layer + changed = true + } + return changed +} + +func (monitor *BuildxProgressMonitor) snapshotLocked(now time.Time) BuildxProgressSnapshot { + snapshot := BuildxProgressSnapshot{ + ActiveStep: monitor.activeStep, + Elapsed: max(now.Sub(monitor.started), 0), + ObservedProgress: monitor.observed || monitor.activeStep > 0 || len(monitor.layers) > 0, + } + for _, layer := range monitor.layers { + snapshot.KnownLayers++ + snapshot.CurrentBytes += layer.current + snapshot.TotalBytes += layer.total + if layer.complete { + snapshot.CompletedLayers++ + } + } + snapshot.ObservedByteTotal = snapshot.TotalBytes > 0 + return snapshot +} + +func parseBuildxBytes(number, unit string) (int64, bool) { + value, err := strconv.ParseFloat(number, 64) + if err != nil || value < 0 { + return 0, false + } + multipliers := map[string]float64{ + "B": 1, + "kB": 1e3, "MB": 1e6, "GB": 1e9, "TB": 1e12, "PB": 1e15, "EB": 1e18, + "KiB": 1 << 10, "MiB": 1 << 20, "GiB": 1 << 30, "TiB": 1 << 40, "PiB": 1 << 50, "EiB": 1 << 60, + } + multiplier, ok := multipliers[unit] + if !ok || value > float64(math.MaxInt64)/multiplier { + return 0, false + } + return int64(value * multiplier), true +} + +// FormatBuildxProgress renders a concise console-safe summary. +func FormatBuildxProgress(prefix string, snapshot BuildxProgressSnapshot) string { + parts := make([]string, 0, 4) + if snapshot.ObservedByteTotal { + percent := float64(snapshot.CurrentBytes) * 100 / float64(snapshot.TotalBytes) + parts = append(parts, fmt.Sprintf("%s/%s (%.0f%%)", FormatDockerPullBytes(snapshot.CurrentBytes), FormatDockerPullBytes(snapshot.TotalBytes), percent)) + parts = append(parts, fmt.Sprintf("%d/%d layer downloads complete", snapshot.CompletedLayers, snapshot.KnownLayers)) + } + if snapshot.ActiveStep > 0 { + parts = append(parts, fmt.Sprintf("BuildKit step #%d", snapshot.ActiveStep)) + } + parts = append(parts, "elapsed "+formatBuildxElapsed(snapshot.Elapsed)) + return prefix + ": " + strings.Join(parts, "; ") +} + +func formatBuildxElapsed(duration time.Duration) string { + duration = duration.Round(time.Second) + if duration < time.Second { + return "0s" + } + return duration.String() +} + +// BuildxProgressStream frames arbitrary process writes into complete lines. +type BuildxProgressStream struct { + mu sync.Mutex + monitor *BuildxProgressMonitor + pending []byte +} + +func (stream *BuildxProgressStream) Write(content []byte) (int, error) { + stream.mu.Lock() + defer stream.mu.Unlock() + stream.pending = append(stream.pending, content...) + for { + index := bytes.IndexByte(stream.pending, '\n') + if index < 0 { + break + } + line := string(bytes.TrimSuffix(stream.pending[:index], []byte{'\r'})) + stream.pending = stream.pending[index+1:] + stream.monitor.consumeLine(line) + } + return len(content), nil +} + +// Flush consumes a final unterminated line. +func (stream *BuildxProgressStream) Flush() { + stream.mu.Lock() + defer stream.mu.Unlock() + if len(stream.pending) == 0 { + return + } + line := string(bytes.TrimSuffix(stream.pending, []byte{'\r'})) + stream.pending = nil + stream.monitor.consumeLine(line) +} + +var _ io.Writer = (*BuildxProgressStream)(nil) diff --git a/internal/image/buildx_progress_test.go b/internal/image/buildx_progress_test.go new file mode 100644 index 0000000..f8cb97f --- /dev/null +++ b/internal/image/buildx_progress_test.go @@ -0,0 +1,71 @@ +package image + +import ( + "strings" + "testing" + "time" +) + +func TestBuildxProgressMonitorFramesStreamsAndAggregatesLayers(t *testing.T) { + now := time.Date(2026, 7, 30, 0, 0, 0, 0, time.UTC) + var reports []BuildxProgressSnapshot + monitor := newBuildxProgressMonitor(2*time.Second, func() time.Time { return now }, func(snapshot BuildxProgressSnapshot) { + reports = append(reports, snapshot) + }) + stdout := monitor.NewStream() + stderr := monitor.NewStream() + firstDigest := "sha256:" + strings.Repeat("a", 64) + secondDigest := "sha256:" + strings.Repeat("b", 64) + + if _, err := stdout.Write([]byte("#12 " + firstDigest + " 1.00G")); err != nil { + t.Fatal(err) + } + if _, err := stderr.Write([]byte("#11 [source 1/1] resolve image config\n")); err != nil { + t.Fatal(err) + } + if _, err := stdout.Write([]byte("B / 2.00GB 10.0s\n")); err != nil { + t.Fatal(err) + } + now = now.Add(3 * time.Second) + if _, err := stderr.Write([]byte("#12 " + secondDigest + " 1.00GB / 1.00GB 13.0s done\n")); err != nil { + t.Fatal(err) + } + stdout.Flush() + stderr.Flush() + + if len(reports) != 2 { + t.Fatalf("reports = %d, want 2: %#v", len(reports), reports) + } + snapshot := reports[1] + if snapshot.CurrentBytes != 2_000_000_000 || snapshot.TotalBytes != 3_000_000_000 || snapshot.CompletedLayers != 1 || snapshot.KnownLayers != 2 || snapshot.ActiveStep != 12 { + t.Fatalf("unexpected aggregate snapshot: %+v", snapshot) + } + if got, want := FormatBuildxProgress("Docker Sandboxes template build", snapshot), "Docker Sandboxes template build: 1.9 GiB/2.8 GiB (67%); 1/2 layer downloads complete; BuildKit step #12; elapsed 3s"; got != want { + t.Fatalf("FormatBuildxProgress = %q, want %q", got, want) + } +} + +func TestBuildxProgressMonitorIgnoresUnrecognizedAndOverflowingOutput(t *testing.T) { + now := time.Date(2026, 7, 30, 0, 0, 0, 0, time.UTC) + reported := false + monitor := newBuildxProgressMonitor(0, func() time.Time { return now }, func(BuildxProgressSnapshot) { + reported = true + }) + stream := monitor.NewStream() + lines := []string{ + "ordinary command output", + "#x malformed", + "#12 sha256:" + strings.Repeat("c", 64) + " 999999999999999999999EB / 1.00GB 1.0s", + } + for _, line := range lines { + if _, err := stream.Write([]byte(line + "\n")); err != nil { + t.Fatal(err) + } + } + if reported { + t.Fatal("unrecognized Buildx output produced a progress report") + } + if snapshot := monitor.Snapshot(); snapshot.ObservedProgress { + t.Fatalf("unrecognized Buildx output changed snapshot: %+v", snapshot) + } +} diff --git a/internal/image/buildx_test.go b/internal/image/buildx_test.go new file mode 100644 index 0000000..d8be3aa --- /dev/null +++ b/internal/image/buildx_test.go @@ -0,0 +1,276 @@ +package image + +import ( + "encoding/json" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +func TestBuildxBuilderNameIsStableAndProjectScoped(t *testing.T) { + project := filepath.Join("one", "project") + config := filepath.Join(project, ".local", "config.yml") + first, err := buildxBuilderNameForConfig(project, config) + if err != nil { + t.Fatal(err) + } + second, err := buildxBuilderNameForConfig(project, config) + if err != nil { + t.Fatal(err) + } + other, err := buildxBuilderNameForConfig(filepath.Join("two", "project"), filepath.Join("two", "project", ".local", "config.yml")) + if err != nil { + t.Fatal(err) + } + if first != second { + t.Fatalf("builder names are not stable: %q != %q", first, second) + } + if first == other { + t.Fatalf("different projects share builder name %q", first) + } + if !strings.HasPrefix(first, "epar-") || len(first) != len("epar-")+24 { + t.Fatalf("builder name %q does not use the bounded EPAR identity", first) + } +} + +func TestBuildxScopeSeparatesConfigurationsInOneProject(t *testing.T) { + root := t.TempDir() + firstConfig := filepath.Join(root, ".local", "config.yml") + secondConfig := filepath.Join(root, ".local", "config.docker-container.yml") + first, err := resolveBuildxScope(root, firstConfig) + if err != nil { + t.Fatal(err) + } + second, err := resolveBuildxScope(root, secondConfig) + if err != nil { + t.Fatal(err) + } + firstAgain, err := resolveBuildxScope(root, firstConfig) + if err != nil { + t.Fatal(err) + } + if first.configID != firstAgain.configID || first.metadataPath != firstAgain.metadataPath || first.lockPath != firstAgain.lockPath { + t.Fatalf("same configuration scope is not stable: first=%+v again=%+v", first, firstAgain) + } + for _, pair := range [][2]string{{first.configID, second.configID}, {first.metadataPath, second.metadataPath}, {first.lockPath, second.lockPath}, {first.buildkitConfig, second.buildkitConfig}, {first.certificateDir, second.certificateDir}} { + if pair[0] == pair[1] { + t.Fatalf("distinct configuration scopes unexpectedly share %q", pair[0]) + } + } +} + +func TestBuildxScopeUsesCanonicalConfigurationIdentity(t *testing.T) { + root := t.TempDir() + realConfig := filepath.Join(root, ".local", "config.yml") + linkConfig := filepath.Join(root, ".local", "config-link.yml") + if err := os.MkdirAll(filepath.Dir(realConfig), 0o700); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(realConfig, []byte("provider: {}\n"), 0o600); err != nil { + t.Fatal(err) + } + if err := os.Symlink(realConfig, linkConfig); err != nil { + t.Skipf("symlinks unavailable: %v", err) + } + realScope, err := resolveBuildxScope(root, realConfig) + if err != nil { + t.Fatal(err) + } + linkScope, err := resolveBuildxScope(root, linkConfig) + if err != nil { + t.Fatal(err) + } + if realScope != linkScope { + t.Fatalf("Buildx scope split through a config symlink: real=%+v link=%+v", realScope, linkScope) + } +} + +func TestLoadBuildxMetadataRequiresExactOwnershipFields(t *testing.T) { + root := t.TempDir() + configPath := filepath.Join(root, ".local", "config.yml") + path, err := BuildxMetadataPathForConfig(root, configPath) + if err != nil { + t.Fatal(err) + } + if err := os.MkdirAll(filepath.Dir(path), 0755); err != nil { + t.Fatal(err) + } + scope, err := resolveBuildxScope(root, configPath) + if err != nil { + t.Fatal(err) + } + builder, err := buildxBuilderNameForConfig(root, configPath) + if err != nil { + t.Fatal(err) + } + metadata := BuildxMetadata{ + SchemaVersion: buildxMetadataSchemaVersion, + Builder: builder, + Driver: "docker-container", + ProjectRoot: scope.projectRoot, + ConfigID: scope.configID, + EPARConfigPath: scope.configPath, + CacheLimit: "64GiB", + ConfigPath: scope.buildkitConfig, + CreatedAt: time.Now().UTC(), + } + content, err := json.Marshal(metadata) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, content, 0644); err != nil { + t.Fatal(err) + } + loaded, err := LoadBuildxMetadataForConfig(root, configPath) + if err != nil { + t.Fatal(err) + } + if loaded.Builder != metadata.Builder || loaded.CacheLimit != "64GiB" { + t.Fatalf("loaded metadata = %+v, want %+v", loaded, metadata) + } + + metadata.Driver = "docker" + content, _ = json.Marshal(metadata) + if err := os.WriteFile(path, content, 0644); err != nil { + t.Fatal(err) + } + if _, err := LoadBuildxMetadataForConfig(root, configPath); err == nil { + t.Fatal("LoadBuildxMetadata accepted a shared Docker driver") + } +} + +func TestLoadLegacyBuildxMetadataRetainsExactInventoryEvidence(t *testing.T) { + root := t.TempDir() + metadata := BuildxMetadata{ + SchemaVersion: legacyBuildxMaxSchemaVersion, + Builder: legacyBuildxBuilderName(root), + Driver: "docker-container", + ProjectRoot: root, + CacheLimit: "20GiB", + ConfigPath: filepath.Join(root, ".local", "storage", "buildkitd.toml"), + CreatedAt: time.Now().UTC(), + } + content, err := json.Marshal(metadata) + if err != nil { + t.Fatal(err) + } + path := LegacyBuildxMetadataPath(root) + if err := os.MkdirAll(filepath.Dir(path), 0o700); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, content, 0o600); err != nil { + t.Fatal(err) + } + loaded, err := LoadLegacyBuildxMetadata(root) + if err != nil { + t.Fatal(err) + } + if loaded.Builder != metadata.Builder || loaded.ProjectRoot != root { + t.Fatalf("legacy metadata = %+v, want %+v", loaded, metadata) + } + metadata.Builder = "epar-unowned" + content, _ = json.Marshal(metadata) + if err := os.WriteFile(path, content, 0o600); err != nil { + t.Fatal(err) + } + if _, err := LoadLegacyBuildxMetadata(root); err == nil { + t.Fatal("legacy metadata accepted a builder not derived from its project root") + } +} + +func TestBuildRegistryHostsUsesDockerHubForUnqualifiedImages(t *testing.T) { + got, err := buildRegistryHosts([]string{ + "source:latest", + "docker.io/docker/dockerfile:1", + "ghcr.io/catthehacker/ubuntu@sha256:abc", + }) + if err != nil { + t.Fatal(err) + } + want := []string{"docker.io", "ghcr.io"} + if strings.Join(got, ",") != strings.Join(want, ",") { + t.Fatalf("registry hosts = %v, want %v", got, want) + } +} + +func TestBuildkitConfigIsDeterministicAndEscapesPaths(t *testing.T) { + first := buildkitConfig(20<<30, "generation", `C:\repo path\.local\ca.pem`, []string{"docker.io", "ghcr.io"}) + second := buildkitConfig(20<<30, "generation", `C:\repo path\.local\ca.pem`, []string{"docker.io", "ghcr.io"}) + if string(first) != string(second) { + t.Fatal("BuildKit configuration is not deterministic") + } + text := string(first) + for _, want := range []string{ + "# epar-build-trust-generation=generation", + `[registry."docker.io"]`, + `[registry."ghcr.io"]`, + `"C:/repo path/.local/ca.pem"`, + `maxUsedSpace = "21474836480B"`, + } { + if !strings.Contains(text, want) { + t.Fatalf("BuildKit configuration omitted %q:\n%s", want, text) + } + } +} + +func TestParseBuildxUsageBytesUsesSummaryOrSumsRecords(t *testing.T) { + for _, test := range []struct { + content string + want uint64 + }{ + {content: `[{"ID":"a","Size":1024},{"ID":"b","Size":2048}]`, want: 3072}, + {content: "{\"ID\":\"a\",\"Size\":1024}\n{\"Total\":4096}\n", want: 4096}, + {content: `[{"ID":"a","Size":"4.128kB"},{"ID":"b","Size":"310.8MB"},{"ID":"c","Size":"15.22GB"}]`, want: 15_530_804_128}, + } { + got, err := parseBuildxUsageBytes([]byte(test.content)) + if err != nil { + t.Fatal(err) + } + if got != test.want { + t.Fatalf("parseBuildxUsageBytes() = %d, want %d", got, test.want) + } + } +} + +func TestBuildxSchemaOneMetadataIsOwnedButRequiresUpgrade(t *testing.T) { + root := t.TempDir() + configPath := filepath.Join(root, ".local", "config.yml") + scope, err := resolveBuildxScope(root, configPath) + if err != nil { + t.Fatal(err) + } + builder, err := buildxBuilderNameForConfig(root, configPath) + if err != nil { + t.Fatal(err) + } + expected := BuildxMetadata{ + SchemaVersion: buildxMetadataSchemaVersion, + Builder: builder, + Driver: "docker-container", + ProjectRoot: scope.projectRoot, + ConfigID: scope.configID, + EPARConfigPath: scope.configPath, + CacheLimit: "64GiB", + ConfigPath: scope.buildkitConfig, + ConfigSHA256: strings.Repeat("a", 64), + TrustGeneration: strings.Repeat("b", 64), + CertificateBundle: filepath.Join(root, ".local", "storage", "buildkit-certs", "g", "ca.pem"), + CertificateSHA256: strings.Repeat("c", 64), + RegistryHosts: []string{"docker.io"}, + } + legacy := expected + legacy.SchemaVersion = buildxMetadataSchemaVersion - 1 + legacy.ConfigSHA256 = "" + legacy.TrustGeneration = "" + legacy.CertificateBundle = "" + legacy.CertificateSHA256 = "" + legacy.RegistryHosts = nil + if !buildxOwnershipMatches(legacy, expected) { + t.Fatal("previous EPAR metadata lost exact ownership") + } + if buildxMetadataMatches(legacy, expected) { + t.Fatal("schema-one EPAR metadata did not require an owned builder upgrade") + } +} diff --git a/internal/image/compat.go b/internal/image/compat.go new file mode 100644 index 0000000..d5f2a9d --- /dev/null +++ b/internal/image/compat.go @@ -0,0 +1,96 @@ +package image + +import ( + "context" + "io" + "os" + + "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" +) + +// ImageState is the reusable-artifact state returned by CurrentImageState. +type ImageState = imageState +type RunnerImagesCopyMode = runnerImagesCopyMode + +const ( + ImageStateMissing = imageStateMissing + ImageStateCurrent = imageStateCurrent + ImageStateOutdated = imageStateOutdated + RunnerImagesCopyNone = runnerImagesCopyNone + RunnerImagesCopySubset = runnerImagesCopySubset +) + +func (m *Coordinator) DesiredImageManifest(ctx context.Context) (Manifest, error) { + return m.desiredImageManifest(ctx) +} + +func (m *Coordinator) DesiredLocalImageManifest(ctx context.Context) (Manifest, error) { + if m.Config.Provider.Type == "docker-sandboxes" { + return m.dockerSandboxesLocalManifest(ctx) + } + return m.desiredLocalImageManifest(ctx) +} + +func (m *Coordinator) CurrentImageState(ctx context.Context, wantedHash string) (ImageState, error) { + return m.currentImageState(ctx, wantedHash) +} + +func (m *Coordinator) PrepareDockerContainerBuildContext(buildContext, upstreamDirectory, manifestContent string) error { + return m.prepareDockerContainerBuildContext(buildContext, upstreamDirectory, manifestContent) +} + +func (m *Coordinator) PrepareDockerContainerBuildContextWithHostTrust(buildContext, upstreamDirectory, manifestContent string, snapshot hosttrust.Snapshot) error { + return m.prepareDockerContainerBuildContextWithHostTrust(buildContext, upstreamDirectory, manifestContent, snapshot) +} + +func (m *Coordinator) BuildDockerContainerImage(ctx context.Context, options ImageBuildOptions, upstreamDirectory string) error { + return m.buildDockerContainerImage(ctx, options, upstreamDirectory) +} + +func (m *Coordinator) PrepareWSLDockerSourceRootfs(ctx context.Context, outputPath, buildLogPath string, manifest Manifest) (string, string, error) { + return m.prepareWSLDockerSourceRootfs(ctx, outputPath, buildLogPath, manifest) +} + +func (m *Coordinator) WriteDockerPullProgress(logPath string, layers map[string]DockerPullProgress) { + m.writeDockerPullProgress(logPath, layers) +} + +func WriteDockerPullEvent(writer io.Writer, event DockerPullEvent) { + writeDockerPullEvent(writer, event) +} + +func ImageManifestHash(manifest Manifest) (string, error) { + return ManifestHash(manifest) +} + +func SourceImageEnvContent(environment []string) string { + return sourceImageEnvContent(environment) +} + +func CopyFile(source, destination string, mode os.FileMode) error { + return copyFile(source, destination, mode) +} + +func (m *Coordinator) RunnerImageBuildScripts() []string { + return m.runnerImageBuildScripts() +} + +func (m *Coordinator) RunnerImagesCopyMode() RunnerImagesCopyMode { + return m.runnerImagesCopyMode() +} + +func (m *Coordinator) PrepareWSLDockerSourceGuest(ctx context.Context, instance string) error { + return m.prepareWSLDockerSourceGuest(ctx, instance) +} + +func (m *Coordinator) InstallCustomInstallScripts(ctx context.Context, instance string) error { + return m.installCustomInstallScripts(ctx, instance) +} + +func (m *Coordinator) CustomInstallScriptHostPath(script string) (string, error) { + return m.customInstallScriptHostPath(script) +} + +func (m *Coordinator) EnableWSLSystemd(ctx context.Context, instance string) error { + return m.enableWSLSystemd(ctx, instance) +} diff --git a/internal/image/coordinator.go b/internal/image/coordinator.go new file mode 100644 index 0000000..cb445be --- /dev/null +++ b/internal/image/coordinator.go @@ -0,0 +1,193 @@ +package image + +import ( + "context" + "fmt" + "io" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + "github.com/solutionforest/ephemeral-action-runner/internal/logging" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +const ( + sourceUpdateExpansionBytes = 5 * storage.GiB + hostTrustGuestDir = "/usr/local/share/ca-certificates/epar-host" + hostTrustMarkerGuest = "/opt/epar/host-trust-generation.json" +) + +// Environment exposes the provider-neutral host and guest operations needed +// while creating reusable artifacts. The pool supplies these operations but +// does not own image policy or provider-specific build selection. +type Environment interface { + PreflightStorage(operation string, peakBytes uint64) error + BuildLogPath(name string) string + ReleaseTranscript(path string) error + Infof(format string, args ...any) + Warnf(format string, args ...any) + RunHostLogged(ctx context.Context, logPath, name string, args ...string) error + RunHostBuildxLogged(ctx context.Context, logPath, name string, args ...string) error + RunHost(ctx context.Context, name string, args ...string) error + RunHostOutput(ctx context.Context, name string, args ...string) (string, error) + RunHostOutputTo(ctx context.Context, output io.Writer, name string, args ...string) error + RunHostQuiet(ctx context.Context, name string, args ...string) error + TimeStartupStage(stage string, fn func() error) error + HostTrustEnabled() bool + ResolveHostTrust(ctx context.Context) (hosttrust.Snapshot, error) + ResolveBuildTrust(ctx context.Context) (hosttrust.Snapshot, error) + WriteHostTrustBuildInputs(buildContext string, snapshot hosttrust.Snapshot) error + ValidateRuntime(ctx context.Context, instance string) error + ExecGuest(ctx context.Context, instance string, command []string, opts provider.ExecOptions) (provider.ExecResult, error) + Transcript(path, instance, component string) (*logging.Transcript, error) + PullDockerSource(ctx context.Context, options DockerSourcePullOptions) error + LogInfo(message string, args ...any) + LogWarn(message string, args ...any) + ProgressTerminal() bool + ProgressConsole() io.Writer + RunnerName(prefix string, sequence int, now time.Time) string + TranscriptComponent(path string) string +} + +type Coordinator struct { + Config config.Config + Provider provider.Provider + Lifecycle provider.Lifecycle + ProjectRoot string + ConfigPath string + DryRun bool + Clock func() time.Time + environment Environment +} + +func NewCoordinator(cfg config.Config, legacy provider.Provider, lifecycle provider.Lifecycle, projectRoot string, dryRun bool, environment Environment) *Coordinator { + return &Coordinator{ + Config: cfg, + Provider: legacy, + Lifecycle: lifecycle, + ProjectRoot: projectRoot, + DryRun: dryRun, + environment: environment, + } +} + +func (m *Coordinator) now() time.Time { + if m.Clock != nil { + return m.Clock() + } + return time.Now() +} + +func (m *Coordinator) preflightStorage(operation string, peakBytes uint64) error { + return m.environment.PreflightStorage(operation, peakBytes) +} + +func (m *Coordinator) buildLogPath(name string) string { + return m.environment.BuildLogPath(name) +} + +func (m *Coordinator) releaseTranscript(path string) error { + return m.environment.ReleaseTranscript(path) +} + +func (m *Coordinator) infof(format string, args ...any) { + m.environment.Infof(format, args...) +} + +func (m *Coordinator) warnf(format string, args ...any) { + m.environment.Warnf(format, args...) +} + +func (m *Coordinator) HousekeepStorage(ctx context.Context) error { + return m.cleanupSupersededCatalog(ctx) +} + +func (m *Coordinator) runHostLogged(ctx context.Context, logPath, name string, args ...string) error { + return m.environment.RunHostLogged(ctx, logPath, name, args...) +} + +func (m *Coordinator) runHostBuildxLogged(ctx context.Context, logPath, name string, args ...string) error { + return m.environment.RunHostBuildxLogged(ctx, logPath, name, args...) +} + +func (m *Coordinator) runHost(ctx context.Context, name string, args ...string) error { + return m.environment.RunHost(ctx, name, args...) +} + +func (m *Coordinator) runHostOutput(ctx context.Context, name string, args ...string) (string, error) { + return m.environment.RunHostOutput(ctx, name, args...) +} + +func (m *Coordinator) runHostOutputTo(ctx context.Context, output io.Writer, name string, args ...string) error { + return m.environment.RunHostOutputTo(ctx, output, name, args...) +} + +func (m *Coordinator) runHostQuiet(ctx context.Context, name string, args ...string) error { + return m.environment.RunHostQuiet(ctx, name, args...) +} + +func (m *Coordinator) timeStartupStage(stage string, fn func() error) error { + return m.environment.TimeStartupStage(stage, fn) +} + +func (m *Coordinator) hostTrustEnabled() bool { + return m.environment.HostTrustEnabled() +} + +func (m *Coordinator) resolveHostTrust(ctx context.Context) (hosttrust.Snapshot, error) { + return m.environment.ResolveHostTrust(ctx) +} + +func (m *Coordinator) resolveBuildTrust(ctx context.Context) (hosttrust.Snapshot, error) { + snapshot, err := m.environment.ResolveBuildTrust(ctx) + if err != nil { + return hosttrust.Snapshot{}, err + } + explicit, err := m.trustedCACertificates() + if err != nil { + return hosttrust.Snapshot{}, err + } + for _, certificate := range explicit { + parsed, err := hosttrust.CertificatesFromBytes(certificate.PEM) + if err != nil { + return hosttrust.Snapshot{}, fmt.Errorf("canonicalize explicit build CA %s: %w", certificate.DestinationName, err) + } + snapshot.Certificates = append(snapshot.Certificates, parsed...) + } + return hosttrust.Canonicalize(snapshot) +} + +func (m *Coordinator) writeHostTrustBuildInputs(buildContext string, snapshot hosttrust.Snapshot) error { + return m.environment.WriteHostTrustBuildInputs(buildContext, snapshot) +} + +func (m *Coordinator) validateTrustedCACertificates() error { + _, err := m.trustedCACertificates() + return err +} + +func (m *Coordinator) validateRuntime(ctx context.Context, instance string) error { + return m.environment.ValidateRuntime(ctx, instance) +} + +func (m *Coordinator) execGuest(ctx context.Context, instance string, command []string, opts provider.ExecOptions) (provider.ExecResult, error) { + return m.environment.ExecGuest(ctx, instance, command, opts) +} + +func (m *Coordinator) transcript(path, instance, component string) (*logging.Transcript, error) { + return m.environment.Transcript(path, instance, component) +} + +func (m *Coordinator) pullDockerSource(ctx context.Context, options DockerSourcePullOptions) error { + return m.environment.PullDockerSource(ctx, options) +} + +func (m *Coordinator) runnerName(prefix string, sequence int, now time.Time) string { + return m.environment.RunnerName(prefix, sequence, now) +} + +func (m *Coordinator) transcriptComponent(path string) string { + return m.environment.TranscriptComponent(path) +} diff --git a/internal/image/docker_archive.go b/internal/image/docker_archive.go new file mode 100644 index 0000000..dab7791 --- /dev/null +++ b/internal/image/docker_archive.go @@ -0,0 +1,351 @@ +package image + +import ( + "archive/tar" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "io" + "os" + "path" + "path/filepath" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +const ( + maxTemplateArchiveEntries = 100000 + maxTemplateControlFileSize = 16 << 20 + maxTemplateControlBytes = 64 << 20 +) + +type dockerArchiveVerification struct { + ImageDigest string + ArchiveSHA256 string + ArchiveBytes uint64 +} + +type archivedFile struct { + digest string + size uint64 + data []byte +} + +type dockerSaveManifestEntry struct { + Config string `json:"Config"` + RepoTags []string `json:"RepoTags"` + Layers []string `json:"Layers"` +} + +type dockerImageConfig struct { + Architecture string `json:"architecture"` + OS string `json:"os"` + Config struct { + Labels map[string]string `json:"Labels"` + } `json:"config"` + RootFS struct { + DiffIDs []string `json:"diff_ids"` + } `json:"rootfs"` +} + +type ociDescriptor struct { + MediaType string `json:"mediaType"` + Digest string `json:"digest"` + Size uint64 `json:"size"` + Annotations map[string]string `json:"annotations"` + Platform *struct { + OS string `json:"os"` + Architecture string `json:"architecture"` + } `json:"platform,omitempty"` +} + +type ociIndex struct { + SchemaVersion int `json:"schemaVersion"` + Manifests []ociDescriptor `json:"manifests"` +} + +type ociManifest struct { + SchemaVersion int `json:"schemaVersion"` + Config ociDescriptor `json:"config"` + Layers []ociDescriptor `json:"layers"` +} + +func verifyDockerSandboxesArchive(archivePath, expectedTag, expectedPlatform, expectedBuildDigest string, expectedLabels map[string]string) (dockerArchiveVerification, error) { + target, err := storage.SnapshotFilesystemTarget(archivePath) + if err != nil { + return dockerArchiveVerification{}, fmt.Errorf("inspect Docker Sandboxes template archive: %w", err) + } + if target.Kind != storage.TargetFile { + return dockerArchiveVerification{}, errors.New("Docker Sandboxes template archive is not a regular file") + } + file, err := os.Open(archivePath) + if err != nil { + return dockerArchiveVerification{}, err + } + files, err := scanTemplateArchive(file) + closeErr := file.Close() + if err != nil { + return dockerArchiveVerification{}, err + } + if closeErr != nil { + return dockerArchiveVerification{}, closeErr + } + + var imageDigest string + switch { + case files["index.json"].data != nil: + imageDigest, err = verifyOCIArchive(files, expectedTag, expectedPlatform, expectedLabels) + case files["manifest.json"].data != nil: + imageDigest, err = verifyDockerSaveArchive(files, expectedTag, expectedPlatform, expectedLabels) + default: + err = errors.New("template archive contains neither OCI index.json nor Docker manifest.json") + } + if err != nil { + return dockerArchiveVerification{}, err + } + if imageDigest != expectedBuildDigest { + return dockerArchiveVerification{}, fmt.Errorf("template archive identity %s does not match Buildx metadata identity %s", imageDigest, expectedBuildDigest) + } + archiveSHA, archiveBytes, err := hashFile(archivePath) + if err != nil { + return dockerArchiveVerification{}, err + } + return dockerArchiveVerification{ImageDigest: imageDigest, ArchiveSHA256: archiveSHA, ArchiveBytes: archiveBytes}, nil +} + +func scanTemplateArchive(reader io.Reader) (map[string]archivedFile, error) { + result := make(map[string]archivedFile) + tarReader := tar.NewReader(reader) + var controlBytes uint64 + for entries := 0; ; entries++ { + if entries >= maxTemplateArchiveEntries { + return nil, errors.New("template archive contains too many entries") + } + header, err := tarReader.Next() + if errors.Is(err, io.EOF) { + break + } + if err != nil { + return nil, fmt.Errorf("read template archive: %w", err) + } + name, err := safeArchiveName(header.Name) + if err != nil { + return nil, err + } + if _, duplicate := result[name]; duplicate { + return nil, fmt.Errorf("template archive contains duplicate entry %q", name) + } + switch header.Typeflag { + case tar.TypeDir: + result[name] = archivedFile{} + continue + case tar.TypeReg, tar.TypeRegA: + default: + return nil, fmt.Errorf("template archive entry %q has unsafe type %d", name, header.Typeflag) + } + if header.Size < 0 { + return nil, fmt.Errorf("template archive entry %q has a negative size", name) + } + hasher := sha256.New() + var capture []byte + if shouldCaptureArchiveControl(name, header.Size) { + if uint64(header.Size) > maxTemplateControlFileSize || controlBytes > maxTemplateControlBytes-uint64(header.Size) { + return nil, errors.New("template archive control data exceeds the bounded verification limit") + } + capture = make([]byte, header.Size) + if _, err := io.ReadFull(io.TeeReader(tarReader, hasher), capture); err != nil { + return nil, fmt.Errorf("read template archive entry %q: %w", name, err) + } + controlBytes += uint64(header.Size) + } else { + if _, err := io.Copy(hasher, tarReader); err != nil { + return nil, fmt.Errorf("hash template archive entry %q: %w", name, err) + } + } + result[name] = archivedFile{ + digest: "sha256:" + hex.EncodeToString(hasher.Sum(nil)), + size: uint64(header.Size), + data: capture, + } + } + return result, nil +} + +func safeArchiveName(value string) (string, error) { + if value == "" || strings.ContainsRune(value, 0) || strings.Contains(value, "\\") || strings.HasPrefix(value, "/") { + return "", fmt.Errorf("template archive contains unsafe path %q", value) + } + normalized := strings.TrimSuffix(value, "/") + clean := path.Clean(normalized) + if clean == "." || clean == ".." || strings.HasPrefix(clean, "../") || clean != normalized { + return "", fmt.Errorf("template archive contains unsafe path %q", value) + } + return clean, nil +} + +func shouldCaptureArchiveControl(name string, size int64) bool { + if size > maxTemplateControlFileSize { + return false + } + return name == "manifest.json" || name == "index.json" || name == "oci-layout" || strings.HasSuffix(name, ".json") || strings.HasPrefix(name, "blobs/sha256/") +} + +func verifyDockerSaveArchive(files map[string]archivedFile, expectedTag, expectedPlatform string, expectedLabels map[string]string) (string, error) { + var entries []dockerSaveManifestEntry + if err := json.Unmarshal(files["manifest.json"].data, &entries); err != nil { + return "", fmt.Errorf("decode Docker archive manifest: %w", err) + } + if len(entries) != 1 { + return "", fmt.Errorf("Docker archive must contain exactly one image, found %d", len(entries)) + } + entry := entries[0] + if len(entry.RepoTags) != 1 || !sameDockerReference(entry.RepoTags[0], expectedTag) { + return "", fmt.Errorf("Docker archive tag %q does not match expected tag %q", strings.Join(entry.RepoTags, ", "), expectedTag) + } + configFile, ok := files[entry.Config] + if !ok || configFile.data == nil { + return "", fmt.Errorf("Docker archive is missing configuration %q", entry.Config) + } + configDigest := configFile.digest + configName := strings.TrimSuffix(filepath.Base(entry.Config), ".json") + if len(configName) != 64 || configDigest != "sha256:"+configName { + return "", errors.New("Docker archive configuration filename does not match its recomputed digest") + } + var config dockerImageConfig + if err := json.Unmarshal(configFile.data, &config); err != nil { + return "", fmt.Errorf("decode Docker archive configuration: %w", err) + } + if err := verifyArchivePlatformAndLabels(config.OS, config.Architecture, config.Config.Labels, expectedPlatform, expectedLabels); err != nil { + return "", err + } + if len(entry.Layers) == 0 || len(entry.Layers) != len(config.RootFS.DiffIDs) { + return "", errors.New("Docker archive layer list does not match configuration rootfs") + } + seen := make(map[string]bool, len(entry.Layers)) + for index, layerName := range entry.Layers { + if seen[layerName] { + return "", fmt.Errorf("Docker archive references layer %q more than once", layerName) + } + seen[layerName] = true + layer, ok := files[layerName] + if !ok { + return "", fmt.Errorf("Docker archive is missing layer %q", layerName) + } + if layer.digest != config.RootFS.DiffIDs[index] { + return "", fmt.Errorf("Docker archive layer %q digest does not match configuration diff ID", layerName) + } + } + return configDigest, nil +} + +func verifyOCIArchive(files map[string]archivedFile, expectedTag, expectedPlatform string, expectedLabels map[string]string) (string, error) { + if layout, ok := files["oci-layout"]; !ok || layout.data == nil { + return "", errors.New("OCI archive is missing oci-layout") + } else { + var version struct { + ImageLayoutVersion string `json:"imageLayoutVersion"` + } + if err := json.Unmarshal(layout.data, &version); err != nil || version.ImageLayoutVersion != "1.0.0" { + return "", errors.New("OCI archive has an unsupported image layout version") + } + } + var index ociIndex + if err := json.Unmarshal(files["index.json"].data, &index); err != nil { + return "", fmt.Errorf("decode OCI archive index: %w", err) + } + if index.SchemaVersion != 2 { + return "", errors.New("OCI archive index has an unsupported schema version") + } + var selected *ociDescriptor + for i := range index.Manifests { + descriptor := &index.Manifests[i] + fullName := descriptor.Annotations["io.containerd.image.name"] + shortTag := descriptor.Annotations["org.opencontainers.image.ref.name"] + _, expectedSuffix, _ := strings.Cut(expectedTag, ":") + matches := fullName != "" && sameDockerReference(fullName, expectedTag) + if fullName == "" { + matches = shortTag != "" && shortTag == expectedSuffix + } + if matches { + if selected != nil { + return "", fmt.Errorf("OCI archive contains duplicate tag %q", expectedTag) + } + selected = descriptor + } + } + if selected == nil { + return "", fmt.Errorf("OCI archive does not contain exact tag %q", expectedTag) + } + manifestBytes, err := verifyOCIDescriptor(files, *selected) + if err != nil { + return "", err + } + var manifest ociManifest + if err := json.Unmarshal(manifestBytes, &manifest); err != nil || manifest.SchemaVersion != 2 { + return "", errors.New("OCI archive contains an invalid image manifest") + } + configBytes, err := verifyOCIDescriptor(files, manifest.Config) + if err != nil { + return "", err + } + var config dockerImageConfig + if err := json.Unmarshal(configBytes, &config); err != nil { + return "", fmt.Errorf("decode OCI image configuration: %w", err) + } + if err := verifyArchivePlatformAndLabels(config.OS, config.Architecture, config.Config.Labels, expectedPlatform, expectedLabels); err != nil { + return "", err + } + if selected.Platform != nil && expectedPlatform != selected.Platform.OS+"/"+selected.Platform.Architecture { + return "", fmt.Errorf("OCI index platform %s/%s does not match expected platform %s", selected.Platform.OS, selected.Platform.Architecture, expectedPlatform) + } + for _, layer := range manifest.Layers { + if _, err := verifyOCIDescriptor(files, layer); err != nil { + return "", err + } + } + return selected.Digest, nil +} + +func verifyOCIDescriptor(files map[string]archivedFile, descriptor ociDescriptor) ([]byte, error) { + if !validSHA256(descriptor.Digest) { + return nil, fmt.Errorf("OCI archive contains invalid descriptor digest %q", descriptor.Digest) + } + name := "blobs/sha256/" + strings.TrimPrefix(descriptor.Digest, "sha256:") + file, ok := files[name] + if !ok { + return nil, fmt.Errorf("OCI archive is missing blob %s", descriptor.Digest) + } + if file.digest != descriptor.Digest || file.size != descriptor.Size { + return nil, fmt.Errorf("OCI archive blob %s does not match its descriptor", descriptor.Digest) + } + return file.data, nil +} + +func verifyArchivePlatformAndLabels(osName, architecture string, labels map[string]string, expectedPlatform string, expectedLabels map[string]string) error { + if osName+"/"+architecture != expectedPlatform { + return fmt.Errorf("template archive platform %s/%s does not match expected platform %s", osName, architecture, expectedPlatform) + } + for name, expected := range expectedLabels { + if expected == "*" && labels[name] == "" { + return fmt.Errorf("template archive label %s is missing", name) + } + if expected != "*" && labels[name] != expected { + return fmt.Errorf("template archive label %s=%q does not match expected value %q", name, labels[name], expected) + } + } + return nil +} + +func sameDockerReference(left, right string) bool { + normalize := func(value string) string { + value = strings.TrimPrefix(value, "docker.io/") + if !strings.Contains(strings.SplitN(value, ":", 2)[0], "/") { + value = "library/" + value + } + return value + } + return normalize(left) == normalize(right) +} diff --git a/internal/image/docker_archive_test.go b/internal/image/docker_archive_test.go new file mode 100644 index 0000000..f2c5db9 --- /dev/null +++ b/internal/image/docker_archive_test.go @@ -0,0 +1,204 @@ +package image + +import ( + "archive/tar" + "bytes" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "os" + "path/filepath" + "strings" + "testing" +) + +func TestVerifyDockerSandboxesArchiveAcceptsExactDockerExport(t *testing.T) { + path, digest, labels := writeDockerArchiveFixture(t, false, false) + verified, err := verifyDockerSandboxesArchive(path, "docker.io/library/epar-template:test-amd64", "linux/amd64", digest, labels) + if err != nil { + t.Fatal(err) + } + if verified.ImageDigest != digest || verified.ArchiveBytes == 0 || !validSHA256(verified.ArchiveSHA256) { + t.Fatalf("verification = %+v", verified) + } +} + +func TestVerifyDockerSandboxesArchiveAgainstOptionalBuildxFixture(t *testing.T) { + archivePath := os.Getenv("EPAR_TEST_DOCKER_ARCHIVE") + metadataPath := os.Getenv("EPAR_TEST_DOCKER_ARCHIVE_METADATA") + if archivePath == "" || metadataPath == "" { + t.Skip("set EPAR_TEST_DOCKER_ARCHIVE and EPAR_TEST_DOCKER_ARCHIVE_METADATA to validate a real Buildx export") + } + var metadata dockerSandboxesBuildMetadata + if err := readJSONFile(metadataPath, &metadata); err != nil { + t.Fatal(err) + } + if _, err := verifyDockerSandboxesArchive(archivePath, "docker.io/library/epar-template:probe", "linux/amd64", metadata.ImageDigest, map[string]string{ + "io.solutionforest.epar.schema": "1", + "io.solutionforest.epar.installation": "probe", + "io.solutionforest.epar.provider": "docker-sandboxes", + "io.solutionforest.epar.role": "template-staging", + "io.solutionforest.epar.manifest": strings.Repeat("a", 64), + }); err != nil { + t.Fatal(err) + } +} + +func TestVerifyDockerSandboxesArchiveRejectsTamperingAndUnsafeEntries(t *testing.T) { + t.Run("digest tampering", func(t *testing.T) { + path, digest, labels := writeDockerArchiveFixture(t, true, false) + if _, err := verifyDockerSandboxesArchive(path, "epar-template:test-amd64", "linux/amd64", digest, labels); err == nil { + t.Fatal("tampered layer was accepted") + } + }) + t.Run("duplicate control entry", func(t *testing.T) { + path, digest, labels := writeDockerArchiveFixture(t, false, true) + if _, err := verifyDockerSandboxesArchive(path, "epar-template:test-amd64", "linux/amd64", digest, labels); err == nil || !strings.Contains(err.Error(), "duplicate") { + t.Fatalf("err = %v", err) + } + }) + t.Run("unsafe link", func(t *testing.T) { + path := filepath.Join(t.TempDir(), "unsafe.tar") + writeTarFixture(t, path, []tarFixtureEntry{{name: "manifest.json", content: []byte("[]")}, {name: "escape", typeflag: tar.TypeSymlink, linkname: "../outside"}}) + if _, err := verifyDockerSandboxesArchive(path, "epar-template:test-amd64", "linux/amd64", "sha256:"+strings.Repeat("a", 64), nil); err == nil || !strings.Contains(err.Error(), "unsafe type") { + t.Fatalf("err = %v", err) + } + }) + t.Run("truncated", func(t *testing.T) { + path, digest, labels := writeDockerArchiveFixture(t, false, false) + content, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, content[:len(content)/2], 0o600); err != nil { + t.Fatal(err) + } + if _, err := verifyDockerSandboxesArchive(path, "epar-template:test-amd64", "linux/amd64", digest, labels); err == nil { + t.Fatal("truncated archive was accepted") + } + }) +} + +func TestVerifyDockerSandboxesArchiveAcceptsExactOCIIndex(t *testing.T) { + labels := archiveFixtureLabels() + config := dockerImageConfig{Architecture: "arm64", OS: "linux"} + config.Config.Labels = labels + configBytes, _ := json.Marshal(config) + configDigest := digestBytes(configBytes) + layer := []byte("layer") + layerDigest := digestBytes(layer) + manifest := ociManifest{ + SchemaVersion: 2, + Config: ociDescriptor{MediaType: "application/vnd.oci.image.config.v1+json", Digest: configDigest, Size: uint64(len(configBytes))}, + Layers: []ociDescriptor{{MediaType: "application/vnd.oci.image.layer.v1.tar", Digest: layerDigest, Size: uint64(len(layer))}}, + } + manifestBytes, _ := json.Marshal(manifest) + manifestDigest := digestBytes(manifestBytes) + platform := struct { + OS string `json:"os"` + Architecture string `json:"architecture"` + }{OS: "linux", Architecture: "arm64"} + index := ociIndex{SchemaVersion: 2, Manifests: []ociDescriptor{{ + MediaType: "application/vnd.oci.image.manifest.v1+json", + Digest: manifestDigest, + Size: uint64(len(manifestBytes)), + Annotations: map[string]string{ + "io.containerd.image.name": "docker.io/library/epar-template:test-arm64", + "org.opencontainers.image.ref.name": "test-arm64", + }, + Platform: &platform, + }}} + indexBytes, _ := json.Marshal(index) + path := filepath.Join(t.TempDir(), "oci.tar") + writeTarFixture(t, path, []tarFixtureEntry{ + {name: "oci-layout", content: []byte(`{"imageLayoutVersion":"1.0.0"}`)}, + {name: "index.json", content: indexBytes}, + {name: "blobs/sha256/" + strings.TrimPrefix(manifestDigest, "sha256:"), content: manifestBytes}, + {name: "blobs/sha256/" + strings.TrimPrefix(configDigest, "sha256:"), content: configBytes}, + {name: "blobs/sha256/" + strings.TrimPrefix(layerDigest, "sha256:"), content: layer}, + }) + if _, err := verifyDockerSandboxesArchive(path, "docker.io/library/epar-template:test-arm64", "linux/arm64", manifestDigest, labels); err != nil { + t.Fatal(err) + } +} + +type tarFixtureEntry struct { + name string + content []byte + typeflag byte + linkname string +} + +func writeDockerArchiveFixture(t *testing.T, tamperLayer, duplicateManifest bool) (string, string, map[string]string) { + t.Helper() + labels := archiveFixtureLabels() + layer := []byte("verified layer content") + layerDigest := digestBytes(layer) + config := dockerImageConfig{Architecture: "amd64", OS: "linux"} + config.Config.Labels = labels + config.RootFS.DiffIDs = []string{layerDigest} + configBytes, _ := json.Marshal(config) + configDigest := digestBytes(configBytes) + configName := strings.TrimPrefix(configDigest, "sha256:") + ".json" + manifestBytes, _ := json.Marshal([]dockerSaveManifestEntry{{ + Config: configName, + RepoTags: []string{"epar-template:test-amd64"}, + Layers: []string{"layer/layer.tar"}, + }}) + if tamperLayer { + layer = []byte("tampered") + } + entries := []tarFixtureEntry{ + {name: "manifest.json", content: manifestBytes}, + {name: configName, content: configBytes}, + {name: "layer/layer.tar", content: layer}, + } + if duplicateManifest { + entries = append(entries, tarFixtureEntry{name: "manifest.json", content: manifestBytes}) + } + path := filepath.Join(t.TempDir(), "template.tar") + writeTarFixture(t, path, entries) + return path, configDigest, labels +} + +func archiveFixtureLabels() map[string]string { + return map[string]string{ + "io.solutionforest.epar.schema": "1", + "io.solutionforest.epar.installation": "installation", + "io.solutionforest.epar.provider": "docker-sandboxes", + "io.solutionforest.epar.role": "template-staging", + "io.solutionforest.epar.manifest": strings.Repeat("a", 64), + } +} + +func writeTarFixture(t *testing.T, path string, entries []tarFixtureEntry) { + t.Helper() + var content bytes.Buffer + writer := tar.NewWriter(&content) + for _, entry := range entries { + typeflag := entry.typeflag + if typeflag == 0 { + typeflag = tar.TypeReg + } + header := &tar.Header{Name: entry.name, Mode: 0o600, Size: int64(len(entry.content)), Typeflag: typeflag, Linkname: entry.linkname} + if err := writer.WriteHeader(header); err != nil { + t.Fatal(err) + } + if typeflag == tar.TypeReg { + if _, err := writer.Write(entry.content); err != nil { + t.Fatal(err) + } + } + } + if err := writer.Close(); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, content.Bytes(), 0o600); err != nil { + t.Fatal(err) + } +} + +func digestBytes(content []byte) string { + sum := sha256.Sum256(content) + return "sha256:" + hex.EncodeToString(sum[:]) +} diff --git a/internal/image/docker_output_tag_claim_test.go b/internal/image/docker_output_tag_claim_test.go new file mode 100644 index 0000000..9113695 --- /dev/null +++ b/internal/image/docker_output_tag_claim_test.go @@ -0,0 +1,193 @@ +package image + +import ( + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +func TestDockerOutputTagClaimAllowsSameManifestAndRejectsDivergence(t *testing.T) { + t.Setenv("EPAR_STATE_HOME", filepath.Join(t.TempDir(), "host-state")) + project := t.TempDir() + firstPath := filepath.Join(project, ".local", "config.yml") + secondPath := filepath.Join(project, ".local", "config.docker-container.yml") + if err := os.MkdirAll(filepath.Dir(firstPath), 0o700); err != nil { + t.Fatal(err) + } + for _, path := range []string{firstPath, secondPath} { + if err := os.WriteFile(path, []byte("provider:\n type: docker-container\n"), 0o600); err != nil { + t.Fatal(err) + } + } + first := Coordinator{Config: config.Default(), ProjectRoot: project, ConfigPath: firstPath} + second := Coordinator{Config: config.Default(), ProjectRoot: project, ConfigPath: secondPath} + now := time.Date(2026, 7, 31, 2, 3, 4, 0, time.UTC) + const backendID = "docker:test-daemon" + const tag = "example/runner:current" + const firstManifest = "manifest-one" + if err := first.claimDockerOutputTagLocked(backendID, tag, firstManifest, now); err != nil { + t.Fatal(err) + } + if err := first.claimDockerOutputTagLocked(backendID, tag, firstManifest, now.Add(time.Second)); err != nil { + t.Fatalf("same configuration could not renew its stable tag claim: %v", err) + } + if err := second.claimDockerOutputTagLocked(backendID, tag, firstManifest, now.Add(2*time.Second)); err != nil { + t.Fatalf("identical manifests could not share output tag intent: %v", err) + } + err := second.claimDockerOutputTagLocked(backendID, tag, "manifest-two", now.Add(3*time.Second)) + if err == nil { + t.Fatal("divergent configuration claimed a shared mutable Docker output tag") + } + if !strings.Contains(err.Error(), secondPath) && !strings.Contains(err.Error(), firstPath) { + t.Fatalf("conflict error omitted actionable configuration path: %v", err) + } + if !strings.Contains(err.Error(), tag) || !strings.Contains(err.Error(), firstManifest) || !strings.Contains(err.Error(), "manifest-two") { + t.Fatalf("conflict error omitted tag or manifest evidence: %v", err) + } + + store, err := storagecatalog.Open("") + if err != nil { + t.Fatal(err) + } + value, err := store.Load(now.Add(4 * time.Second)) + if err != nil { + t.Fatal(err) + } + var claim storagecatalog.Resource + for _, resource := range value.Resources { + if resource.Kind == catalogDockerOutputTagClaimKind { + claim = resource + break + } + } + if claim.Key == "" || len(claim.References) != 2 || claim.ManifestHash != firstManifest { + t.Fatalf("shared exact tag claim = %#v, want one stable claim with both references", claim) + } + if err := first.releaseDockerOutputTagClaim(backendID, tag, now.Add(5*time.Second)); err != nil { + t.Fatal(err) + } + if err := second.releaseDockerOutputTagClaim(backendID, tag, now.Add(6*time.Second)); err != nil { + t.Fatal(err) + } + value, err = store.Load(now.Add(7 * time.Second)) + if err != nil { + t.Fatal(err) + } + claim = storagecatalog.Resource{} + for _, resource := range value.Resources { + if resource.Kind == catalogDockerOutputTagClaimKind { + claim = resource + break + } + } + if claim.State != storagecatalog.StateSuperseded || len(claim.References) != 0 { + t.Fatalf("released tag claim was not retained only as exact superseded cleanup evidence: %#v", claim) + } +} + +func TestDockerOutputTagClaimRejectsOtherActiveArtifactManifest(t *testing.T) { + t.Setenv("EPAR_STATE_HOME", filepath.Join(t.TempDir(), "host-state")) + project := t.TempDir() + firstPath := filepath.Join(project, ".local", "config.yml") + secondPath := filepath.Join(project, ".local", "config.docker-container.yml") + if err := os.MkdirAll(filepath.Dir(firstPath), 0o700); err != nil { + t.Fatal(err) + } + for _, path := range []string{firstPath, secondPath} { + if err := os.WriteFile(path, []byte("provider:\n type: docker-container\n"), 0o600); err != nil { + t.Fatal(err) + } + } + second := Coordinator{Config: config.Default(), ProjectRoot: project, ConfigPath: secondPath} + now := time.Date(2026, 7, 31, 3, 4, 5, 0, time.UTC) + const backendID = "docker:test-daemon" + const tag = "example/runner:current" + store, err := storagecatalog.Open("") + if err != nil { + t.Fatal(err) + } + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + record, err := storagecatalog.RegisterConfig(value, project, firstPath, now) + if err != nil { + return err + } + resource := storagecatalog.Resource{ + BackendID: backendID, Kind: catalogDockerImageKind, Provider: "docker-container", Role: "runtime-image", + Locator: tag, Identity: "sha256:active", Custody: storagecatalog.CustodyGenerated, ManifestHash: "manifest-one", + State: storagecatalog.StateCurrent, CreatedAt: now, LastSeenAt: now, + } + if err := storagecatalog.UpsertResource(value, resource); err != nil { + return err + } + storagecatalog.ReplaceConfigRoleReferences(value, record.ID, "provider-artifact", map[string]storagecatalog.Reference{ + storagecatalog.ResourceKey(backendID, catalogDockerImageKind, resource.Identity): {ManifestHash: "manifest-one"}, + }, now) + return nil + }) + if err != nil { + t.Fatal(err) + } + err = second.claimDockerOutputTagLocked(backendID, tag, "manifest-two", now.Add(time.Second)) + if err == nil || !strings.Contains(err.Error(), firstPath) { + t.Fatalf("active artifact conflict = %v, want first configuration path", err) + } +} + +func TestDockerOutputTagConflictFallsBackToCanonicalPath(t *testing.T) { + value := &storagecatalog.Catalog{Configs: []storagecatalog.Config{{ID: "legacy-config", Path: "canonical-config.yml"}}} + err := dockerOutputTagConflict(value, "legacy-config", "docker.io/example/runner:current", "manifest-one", "manifest-two") + if !strings.Contains(err.Error(), "canonical-config.yml") { + t.Fatalf("legacy catalog conflict omitted canonical configuration path: %v", err) + } +} + +func TestDockerOutputTagClaimExpiresAfterPublisherCrash(t *testing.T) { + t.Setenv("EPAR_STATE_HOME", filepath.Join(t.TempDir(), "host-state")) + project := t.TempDir() + firstPath := filepath.Join(project, ".local", "config.yml") + secondPath := filepath.Join(project, ".local", "config.other.yml") + if err := os.MkdirAll(filepath.Dir(firstPath), 0o700); err != nil { + t.Fatal(err) + } + for _, path := range []string{firstPath, secondPath} { + if err := os.WriteFile(path, []byte("provider:\n type: docker-container\n"), 0o600); err != nil { + t.Fatal(err) + } + } + first := Coordinator{Config: config.Default(), ProjectRoot: project, ConfigPath: firstPath} + second := Coordinator{Config: config.Default(), ProjectRoot: project, ConfigPath: secondPath} + now := time.Date(2026, 7, 31, 4, 5, 6, 0, time.UTC) + const backendID = "docker:test-daemon" + const tag = "example/runner:current" + if err := first.claimDockerOutputTagLocked(backendID, tag, "manifest-one", now); err != nil { + t.Fatal(err) + } + if err := second.claimDockerOutputTagLocked(backendID, tag, "manifest-two", now.Add(dockerOutputTagClaimLifetime-time.Second)); err == nil { + t.Fatal("fresh claim did not block a divergent manifest") + } + if err := second.claimDockerOutputTagLocked(backendID, tag, "manifest-two", now.Add(dockerOutputTagClaimLifetime)); err != nil { + t.Fatalf("expired claim still blocked a divergent manifest after publisher crash: %v", err) + } +} + +func TestNormalizedDockerTagCollapsesEquivalentNames(t *testing.T) { + tests := map[string]string{ + "epar-runner": "docker.io/library/epar-runner:latest", + "epar-runner:latest": "docker.io/library/epar-runner:latest", + "solutionforest/epar-runner": "docker.io/solutionforest/epar-runner:latest", + "index.docker.io/epar-runner:Current": "docker.io/library/epar-runner:Current", + "registry-1.docker.io/epar-runner:edge": "docker.io/library/epar-runner:edge", + "ghcr.io/SolutionForest/EPAR:edge": "ghcr.io/solutionforest/epar:edge", + "localhost:5000/epar-runner": "localhost:5000/epar-runner:latest", + } + for input, want := range tests { + if got := normalizedDockerTag(input); got != want { + t.Fatalf("normalizedDockerTag(%q) = %q, want %q", input, got, want) + } + } +} diff --git a/internal/image/docker_pull.go b/internal/image/docker_pull.go new file mode 100644 index 0000000..a243a81 --- /dev/null +++ b/internal/image/docker_pull.go @@ -0,0 +1,316 @@ +// Package image contains provider-neutral image acquisition primitives. +package image + +import ( + "context" + "encoding/base64" + "encoding/json" + "errors" + "fmt" + "reflect" + "runtime" + "strings" + + "github.com/google/go-containerregistry/pkg/authn" + "github.com/google/go-containerregistry/pkg/name" + gcrv1 "github.com/google/go-containerregistry/pkg/v1" + "github.com/google/go-containerregistry/pkg/v1/remote" + "github.com/moby/moby/api/types/jsonstream" + "github.com/moby/moby/api/types/registry" + "github.com/moby/moby/client" + ocispec "github.com/opencontainers/image-spec/specs-go/v1" +) + +// DockerEnginePullError means Docker Engine accepted the connection but rejected the image pull. +// Callers may distinguish it from connection and platform errors that can safely use a CLI fallback. +type DockerEnginePullError struct { + Image string + Err error +} + +func (err *DockerEnginePullError) Error() string { + return fmt.Sprintf("Docker Engine pull %s: %v", err.Image, err.Err) +} + +func (err *DockerEnginePullError) Unwrap() error { return err.Err } + +// DockerPullOptions defines a Docker Engine acquisition without any pool or provider logging policy. +type DockerPullOptions struct { + Image string + Platform string + FallbackPlatform string + QueryRemoteSize bool +} + +// DockerPullResult holds the pulled stream and optional non-fatal lookup failures. +type DockerPullResult struct { + Response client.ImagePullResponse + Platform ocispec.Platform + RemoteCompressedSize int64 + RemoteCompressedError error + RegistryAuthError error +} + +// DockerPullProgress is the shared per-layer state used to summarize progress. +type DockerPullProgress struct { + Current int64 + Total int64 + Completed bool +} + +// DockerPullEvent is one decoded Docker Engine pull event. +type DockerPullEvent struct { + ID string + Status string + Progress *jsonstream.Progress + Stream string + Error error +} + +// PullDockerImage opens the Docker Engine, resolves the requested platform, optionally queries registry layer sizes, and starts a pull. +// Remote metadata and explicit registry-auth failures are non-fatal and are returned on the result for the caller to report. +func PullDockerImage(ctx context.Context, opts DockerPullOptions) (DockerPullResult, error) { + cli, err := client.New(client.FromEnv) + if err != nil { + return DockerPullResult{}, fmt.Errorf("initialize Docker Engine client: %w", err) + } + if _, err := cli.Ping(ctx, client.PingOptions{}); err != nil { + return DockerPullResult{}, fmt.Errorf("connect to Docker Engine: %w", err) + } + + platform, err := ResolveDockerPlatform(ctx, cli, opts.Platform, opts.FallbackPlatform) + if err != nil { + return DockerPullResult{}, err + } + result := DockerPullResult{Platform: platform} + if opts.QueryRemoteSize { + result.RemoteCompressedSize, result.RemoteCompressedError = RemoteCompressedLayerSize(opts.Image, platform) + } + + registryAuth, err := DockerRegistryAuth(opts.Image) + if err != nil { + result.RegistryAuthError = err + } + response, err := cli.ImagePull(ctx, opts.Image, client.ImagePullOptions{ + RegistryAuth: registryAuth, + Platforms: []ocispec.Platform{platform}, + }) + if err != nil && !nilLikeError(err) { + return result, &DockerEnginePullError{Image: opts.Image, Err: err} + } + result.Response = response + return result, nil +} + +// nilLikeError protects the Engine API boundary from an error interface that +// contains a typed nil pointer. Such a value represents no failure but compares +// non-nil as an interface and otherwise renders as the misleading text "". +func nilLikeError(err error) bool { + for err != nil { + if strings.TrimSpace(err.Error()) == "" { + return true + } + value := reflect.ValueOf(err) + switch value.Kind() { + case reflect.Chan, reflect.Func, reflect.Interface, reflect.Map, reflect.Pointer, reflect.Slice: + if value.IsNil() { + return true + } + } + next := errors.Unwrap(err) + if next == nil { + return false + } + err = next + } + return true +} + +// ResolveDockerPlatform chooses an explicit platform, provider fallback platform, engine platform, then the local runtime platform. +func ResolveDockerPlatform(ctx context.Context, cli *client.Client, configured, fallback string) (ocispec.Platform, error) { + if platform, ok := NormalizedDockerPlatform(configured, ""); ok { + return platform, nil + } + if platform, ok := NormalizedDockerPlatform(fallback, ""); ok { + return platform, nil + } + info, err := cli.Info(ctx, client.InfoOptions{}) + if err != nil { + return ocispec.Platform{}, fmt.Errorf("inspect Docker Engine platform: %w", err) + } + if platform, ok := NormalizedDockerPlatform(info.Info.OSType+"/"+info.Info.Architecture, ""); ok { + return platform, nil + } + if platform, ok := NormalizedDockerPlatform(runtime.GOOS+"/"+runtime.GOARCH, ""); ok { + return platform, nil + } + return ocispec.Platform{}, fmt.Errorf("Docker Engine did not report a usable platform") +} + +// NormalizedDockerPlatform parses a Docker platform and normalizes common architecture aliases. +func NormalizedDockerPlatform(value, fallbackOS string) (ocispec.Platform, bool) { + parts := strings.Split(strings.Trim(strings.ToLower(value), "/"), "/") + if len(parts) == 0 || len(parts) > 3 || parts[0] == "" { + return ocispec.Platform{}, false + } + platform := ocispec.Platform{OS: fallbackOS} + if len(parts) == 1 { + platform.Architecture = normalizeDockerArchitecture(parts[0]) + } else { + platform.OS = parts[0] + platform.Architecture = normalizeDockerArchitecture(parts[1]) + if len(parts) == 3 { + platform.Variant = parts[2] + } + } + if platform.OS == "" { + platform.OS = "linux" + } + if platform.Architecture == "" { + return ocispec.Platform{}, false + } + return platform, true +} + +func normalizeDockerArchitecture(architecture string) string { + switch architecture { + case "x86_64", "x64": + return "amd64" + case "aarch64": + return "arm64" + default: + return architecture + } +} + +// RemoteCompressedLayerSize returns the total compressed size of the selected remote image layers. +func RemoteCompressedLayerSize(image string, platform ocispec.Platform) (int64, error) { + ref, authenticator, err := DockerImageReferenceAndAuth(image) + if err != nil { + return 0, err + } + remoteImage, err := remote.Image(ref, remote.WithAuth(authenticator), remote.WithPlatform(gcrv1.Platform{ + OS: platform.OS, + Architecture: platform.Architecture, + Variant: platform.Variant, + })) + if err != nil { + return 0, err + } + layers, err := remoteImage.Layers() + if err != nil { + return 0, err + } + var total int64 + for _, layer := range layers { + size, err := layer.Size() + if err != nil { + return 0, err + } + total += size + } + return total, nil +} + +// DockerRegistryAuth returns Docker Engine's base64-encoded registry auth payload. +func DockerRegistryAuth(image string) (string, error) { + ref, authenticator, err := DockerImageReferenceAndAuth(image) + if err != nil { + return "", err + } + credentials, err := authenticator.Authorization() + if err != nil { + return "", err + } + content, err := json.Marshal(registry.AuthConfig{ + Username: credentials.Username, + Password: credentials.Password, + Auth: credentials.Auth, + ServerAddress: ref.Context().RegistryStr(), + IdentityToken: credentials.IdentityToken, + RegistryToken: credentials.RegistryToken, + }) + if err != nil { + return "", err + } + return base64.RawURLEncoding.EncodeToString(content), nil +} + +// DockerImageReferenceAndAuth resolves an image reference and default credential helper. +func DockerImageReferenceAndAuth(image string) (name.Reference, authn.Authenticator, error) { + ref, err := name.ParseReference(image) + if err != nil { + return nil, nil, err + } + authenticator, err := authn.DefaultKeychain.Resolve(ref.Context().Registry) + if err != nil { + return nil, nil, err + } + return ref, authenticator, nil +} + +// ConsumeDockerPullProgress decodes an Engine pull response and delivers each event in order. +func ConsumeDockerPullProgress(ctx context.Context, response client.ImagePullResponse, handle func(DockerPullEvent) error) error { + for message, streamErr := range response.JSONMessages(ctx) { + if streamErr != nil && !nilLikeError(streamErr) { + return streamErr + } + event := DockerPullEvent{ID: message.ID, Status: message.Status, Progress: message.Progress, Stream: message.Stream, Error: message.Error} + if nilLikeError(event.Error) { + event.Error = nil + } + if err := handle(event); err != nil { + return err + } + if event.Error != nil { + return event.Error + } + } + return nil +} + +// IsDockerPullLayerComplete reports whether an Engine status marks a layer complete. +func IsDockerPullLayerComplete(status string) bool { + status = strings.ToLower(strings.TrimSpace(status)) + return status == "pull complete" || status == "already exists" || status == "exists" +} + +// DockerPullProgressSummary renders one concise progress line. +func DockerPullProgressSummary(layers map[string]DockerPullProgress) string { + var complete, known int + var currentBytes, totalBytes int64 + for _, layer := range layers { + if layer.Completed { + complete++ + } + if layer.Total > 0 { + known++ + totalBytes += layer.Total + currentBytes += min(layer.Current, layer.Total) + } + } + line := fmt.Sprintf("Docker source pull: %d/%d layers complete; %s/%s", complete, len(layers), FormatDockerPullBytes(currentBytes), FormatDockerPullBytes(totalBytes)) + if totalBytes > 0 { + line += fmt.Sprintf(" (%.0f%%)", float64(currentBytes)*100/float64(totalBytes)) + } + if known < len(layers) { + line += fmt.Sprintf("; %d layer(s) size pending", len(layers)-known) + } + return line +} + +// FormatDockerPullBytes renders byte counts used by pull progress and remote-size notices. +func FormatDockerPullBytes(value int64) string { + const unit = 1024 + if value < unit { + return fmt.Sprintf("%d B", value) + } + units := []string{"KiB", "MiB", "GiB", "TiB"} + size := float64(value) + index := -1 + for size >= unit && index+1 < len(units) { + size /= unit + index++ + } + return fmt.Sprintf("%.1f %s", size, units[index]) +} diff --git a/internal/image/docker_pull_test.go b/internal/image/docker_pull_test.go new file mode 100644 index 0000000..5b9ee7d --- /dev/null +++ b/internal/image/docker_pull_test.go @@ -0,0 +1,81 @@ +package image + +import ( + "errors" + "fmt" + "testing" +) + +type typedNilPullError struct{} + +func (*typedNilPullError) Error() string { return "pull failed" } + +type opaqueNilPullError struct{} + +func (opaqueNilPullError) Error() string { return "" } + +func TestNilLikeError(t *testing.T) { + var typedNil *typedNilPullError + if !nilLikeError(typedNil) { + t.Fatal("typed nil error was not recognized") + } + if !nilLikeError(fmt.Errorf("wrapped: %w", typedNil)) { + t.Fatal("wrapped typed nil error was not recognized") + } + if !nilLikeError(opaqueNilPullError{}) { + t.Fatal("opaque nil-rendering error was not recognized") + } + if nilLikeError(errors.New("real failure")) { + t.Fatal("real error was recognized as nil") + } +} + +func TestNormalizedDockerPlatform(t *testing.T) { + tests := []struct { + name string + input string + want string + valid bool + }{ + {name: "explicit amd64", input: "linux/x86_64", want: "linux/amd64", valid: true}, + {name: "arm variant", input: "linux/aarch64/v8", want: "linux/arm64/v8", valid: true}, + {name: "architecture only", input: "amd64", want: "linux/amd64", valid: true}, + {name: "empty", input: "", valid: false}, + {name: "too many segments", input: "linux/amd64/v8/extra", valid: false}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + got, ok := NormalizedDockerPlatform(test.input, "") + if ok != test.valid { + t.Fatalf("valid = %t, want %t", ok, test.valid) + } + if ok && got.OS+"/"+got.Architecture+platformVariant(got.Variant) != test.want { + t.Fatalf("platform = %s/%s%s, want %s", got.OS, got.Architecture, platformVariant(got.Variant), test.want) + } + }) + } +} + +func TestDockerPullProgressSummary(t *testing.T) { + layers := map[string]DockerPullProgress{ + "completed": {Current: 1024, Total: 1024, Completed: true}, + "partial": {Current: 512, Total: 1024}, + "unknown": {Completed: true}, + } + if got, want := DockerPullProgressSummary(layers), "Docker source pull: 2/3 layers complete; 1.5 KiB/2.0 KiB (75%); 1 layer(s) size pending"; got != want { + t.Fatalf("summary = %q, want %q", got, want) + } +} + +func TestDockerImageReferenceAndAuthRejectsInvalidReference(t *testing.T) { + if _, _, err := DockerImageReferenceAndAuth("not a valid image reference"); err == nil { + t.Fatal("invalid image reference was accepted") + } +} + +func platformVariant(variant string) string { + if variant == "" { + return "" + } + return "/" + variant +} diff --git a/internal/image/docker_sandboxes.go b/internal/image/docker_sandboxes.go new file mode 100644 index 0000000..92374c1 --- /dev/null +++ b/internal/image/docker_sandboxes.go @@ -0,0 +1,1700 @@ +package image + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "io" + "os" + "os/exec" + "path/filepath" + "regexp" + "runtime" + "sort" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +const ( + dockerSandboxesReceiptSchema = 3 + dockerSandboxesMetadataSchema = 5 +) + +var dockerTagPattern = regexp.MustCompile(`^[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}$`) + +type ResolvedDockerSource struct { + Reference string `json:"reference"` + ImmutableReference string `json:"immutableReference"` + IndexDigest string `json:"indexDigest"` + PlatformDigest string `json:"platformDigest"` + Platform string `json:"platform"` + CompressedLayerBytes uint64 `json:"compressedLayerBytes"` +} + +type dockerManifestDescriptor struct { + Digest string `json:"digest"` + Size uint64 `json:"size"` + Platform struct { + OS string `json:"os"` + Architecture string `json:"architecture"` + } `json:"platform"` +} + +type dockerManifestDocument struct { + MediaType string `json:"mediaType"` + Manifests []dockerManifestDescriptor `json:"manifests"` + Layers []dockerManifestDescriptor `json:"layers"` +} + +type dockerSandboxesReceipt struct { + SchemaVersion int `json:"schemaVersion"` + ManifestHash string `json:"manifestHash"` + Manifest Manifest `json:"manifest"` + Source ResolvedDockerSource `json:"source"` + Artifact provider.TemplateArtifact `json:"artifact"` + MetadataSHA256 string `json:"metadataSha256"` + ArchiveSHA256 string `json:"archiveSha256"` + ArchiveBytes uint64 `json:"archiveBytes"` + Evidence map[string]artifactEvidence `json:"evidence"` + ActivatedAt time.Time `json:"activatedAt"` +} + +type dockerSandboxesSourceLock struct { + SchemaVersion int `json:"schemaVersion"` + DockerfileFrontend struct { + Reference string `json:"reference"` + } `json:"dockerfileFrontend"` + SBOMGenerator struct { + InspectionReference string `json:"inspectionReference"` + } `json:"sbomGenerator"` + GoBuilder struct { + Version string `json:"version"` + IndexDigest string `json:"indexDigest"` + } `json:"goBuilder"` + HookLauncher struct { + SHA256 string `json:"sha256"` + } `json:"hookLauncher"` + Tini struct { + Version string `json:"version"` + } `json:"tini"` + Platforms map[string]struct { + GoBuilderReference string `json:"goBuilderReference"` + GoBuilderManifestDigest string `json:"goBuilderManifestDigest"` + SBOMGeneratorReference string `json:"sbomGeneratorReference"` + SBOMGeneratorManifestDigest string `json:"sbomGeneratorManifestDigest"` + DockerfileFrontendManifestDigest string `json:"dockerfileFrontendManifestDigest"` + Tini struct { + URL string `json:"url"` + SHA256 string `json:"sha256"` + } `json:"tini"` + } `json:"platforms"` +} + +type dockerSandboxesBuildMetadata struct { + ImageDigest string `json:"containerimage.digest"` + Provenance json.RawMessage `json:"buildx.build.provenance"` + BuildRef string `json:"buildx.build.ref"` +} + +type dockerSandboxesTemplateMetadata struct { + SchemaVersion int `json:"schemaVersion"` + Profile string `json:"profile"` + Platform string `json:"platform"` + ManifestHash string `json:"manifestHash"` + Template struct { + Tag string `json:"tag"` + Digest string `json:"digest"` + CacheID string `json:"cacheID"` + RootDisk string `json:"rootDisk"` + Archive string `json:"archive"` + ArchiveSHA256 string `json:"archiveSha256"` + ArchiveBytes uint64 `json:"archiveBytes"` + } `json:"template"` + Source ResolvedDockerSource `json:"source"` + Compatibility struct { + TemplateSchemaVersion int `json:"templateSchemaVersion"` + RunnerExecution string `json:"runnerExecution"` + DockerDaemonOwner string `json:"dockerDaemonOwner"` + ExpectedDockerDaemonCount int `json:"expectedDockerDaemonCount"` + } `json:"compatibility"` + Artifacts map[string]artifactEvidence `json:"artifacts"` +} + +type artifactEvidence struct { + Path string `json:"path"` + SHA256 string `json:"sha256"` + SourceDigest string `json:"sourceDigest,omitempty"` +} + +// NormalizeCatthehackerSource accepts the user-facing family shorthand or one +// exact tag while keeping the supported source repository explicit. +func NormalizeCatthehackerSource(input string) (string, error) { + value := strings.TrimSpace(input) + if value == "" { + value = "full" + } + switch value { + case "full", "act", "dotnet", "js": + value += "-latest" + } + const repository = "ghcr.io/catthehacker/ubuntu" + if strings.HasPrefix(value, repository+":") { + value = strings.TrimPrefix(value, repository+":") + } + if !dockerTagPattern.MatchString(value) || strings.Contains(value, "/") || strings.Contains(value, "@") { + return "", fmt.Errorf("Docker Sandboxes source must be a catthehacker/ubuntu tag such as full, act, dotnet, js, or go-24.04") + } + return repository + ":" + value, nil +} + +// ResolveCatthehackerSource resolves a mutable tag to the exact OCI index and +// platform manifest identities that will be recorded in the artifact receipt. +func ResolveCatthehackerSource(ctx context.Context, input, platform string) (ResolvedDockerSource, error) { + reference, err := NormalizeCatthehackerSource(input) + if err != nil { + return ResolvedDockerSource{}, err + } + indexRaw, err := exec.CommandContext(ctx, "docker", "buildx", "imagetools", "inspect", "--raw", reference).Output() + if err != nil { + return ResolvedDockerSource{}, fmt.Errorf("resolve source image %s: %w", reference, err) + } + var index dockerManifestDocument + if err := json.Unmarshal(indexRaw, &index); err != nil { + return ResolvedDockerSource{}, fmt.Errorf("parse source image index for %s: %w", reference, err) + } + parts := strings.Split(platform, "/") + var platformDigest string + if len(parts) == 2 { + for _, descriptor := range index.Manifests { + if descriptor.Platform.OS == parts[0] && descriptor.Platform.Architecture == parts[1] { + if platformDigest != "" { + return ResolvedDockerSource{}, fmt.Errorf("source image %s contains multiple %s manifests", reference, platform) + } + platformDigest = descriptor.Digest + } + } + } + if platformDigest == "" { + return ResolvedDockerSource{}, fmt.Errorf("source image %s does not provide %s", reference, platform) + } + tagSeparator := strings.LastIndex(reference, ":") + if tagSeparator < 0 { + return ResolvedDockerSource{}, fmt.Errorf("source image %s has no tag", reference) + } + platformRaw, err := exec.CommandContext(ctx, "docker", "buildx", "imagetools", "inspect", "--raw", reference[:tagSeparator]+"@"+platformDigest).Output() + if err != nil { + return ResolvedDockerSource{}, fmt.Errorf("resolve source image %s for %s: %w", reference, platform, err) + } + return parseResolvedDockerSource(reference, platform, indexRaw, platformRaw) +} + +func sourceProfile(reference string) string { + _, tag, found := strings.Cut(reference, ":") + if !found || tag == "" { + return "custom" + } + return strings.ToLower(tag) +} + +func parseResolvedDockerSource(reference, platform string, indexRaw, platformRaw []byte) (ResolvedDockerSource, error) { + indexRaw = trimManifestCommandNewline(indexRaw) + platformRaw = trimManifestCommandNewline(platformRaw) + var index dockerManifestDocument + if err := json.Unmarshal(indexRaw, &index); err != nil { + return ResolvedDockerSource{}, fmt.Errorf("parse source image manifest index: %w", err) + } + parts := strings.Split(platform, "/") + if len(parts) != 2 || parts[0] != "linux" || (parts[1] != "amd64" && parts[1] != "arm64") { + return ResolvedDockerSource{}, fmt.Errorf("unsupported Docker Sandboxes source platform %q", platform) + } + var matching []dockerManifestDescriptor + for _, descriptor := range index.Manifests { + if descriptor.Platform.OS == parts[0] && descriptor.Platform.Architecture == parts[1] { + matching = append(matching, descriptor) + } + } + if len(matching) != 1 { + return ResolvedDockerSource{}, fmt.Errorf("source image %s must contain exactly one %s manifest; found %d", reference, platform, len(matching)) + } + indexSum := sha256.Sum256(indexRaw) + indexDigest := "sha256:" + hex.EncodeToString(indexSum[:]) + if !validSHA256(matching[0].Digest) { + return ResolvedDockerSource{}, fmt.Errorf("source image %s omitted a valid %s manifest digest", reference, platform) + } + var manifest dockerManifestDocument + if err := json.Unmarshal(platformRaw, &manifest); err != nil { + return ResolvedDockerSource{}, fmt.Errorf("parse source image %s manifest: %w", platform, err) + } + var compressed uint64 + for _, layer := range manifest.Layers { + if layer.Size > ^uint64(0)-compressed { + return ResolvedDockerSource{}, fmt.Errorf("source image compressed layer size overflows") + } + compressed += layer.Size + } + repository, err := dockerRepository(reference) + if err != nil { + return ResolvedDockerSource{}, err + } + return ResolvedDockerSource{ + Reference: reference, + ImmutableReference: repository + "@" + indexDigest, + IndexDigest: indexDigest, + PlatformDigest: matching[0].Digest, + Platform: platform, + CompressedLayerBytes: compressed, + }, nil +} + +func trimManifestCommandNewline(content []byte) []byte { + if len(content) != 0 && content[len(content)-1] == '\n' { + content = content[:len(content)-1] + if len(content) != 0 && content[len(content)-1] == '\r' { + content = content[:len(content)-1] + } + } + return content +} + +func (m *Coordinator) resolveDockerSandboxesSource(ctx context.Context) (ResolvedDockerSource, error) { + reference := strings.TrimSpace(m.Config.Image.SourceImage) + if m.Config.Provider.Type == "docker-sandboxes" { + var err error + reference, err = NormalizeCatthehackerSource(reference) + if err != nil { + return ResolvedDockerSource{}, err + } + } else if reference == "" { + return ResolvedDockerSource{}, errors.New("image.sourceImage is required") + } + platform := strings.TrimSpace(m.Config.Image.SourcePlatform) + if platform == "" { + platform = strings.TrimSpace(m.Config.Provider.Platform) + } + if platform == "" { + architecture := runtime.GOARCH + if architecture == "386" { + architecture = "amd64" + } + if architecture != "amd64" && architecture != "arm64" { + return ResolvedDockerSource{}, fmt.Errorf("cannot infer a Linux source-image platform from host architecture %s; set image.sourcePlatform explicitly", runtime.GOARCH) + } + platform = "linux/" + architecture + } + indexRaw, err := m.runHostOutput(ctx, "docker", "buildx", "imagetools", "inspect", "--raw", reference) + if err != nil { + return ResolvedDockerSource{}, fmt.Errorf("resolve source image %s: %w", reference, err) + } + var index dockerManifestDocument + if err := json.Unmarshal([]byte(indexRaw), &index); err != nil { + return ResolvedDockerSource{}, fmt.Errorf("parse source image index for %s: %w", reference, err) + } + parts := strings.Split(platform, "/") + var platformDigest string + if len(parts) == 2 { + for _, descriptor := range index.Manifests { + if descriptor.Platform.OS == parts[0] && descriptor.Platform.Architecture == parts[1] { + if platformDigest != "" { + return ResolvedDockerSource{}, fmt.Errorf("source image %s contains multiple %s manifests", reference, platform) + } + platformDigest = descriptor.Digest + } + } + } + if platformDigest == "" { + return ResolvedDockerSource{}, fmt.Errorf("source image %s does not provide %s", reference, platform) + } + repository, err := dockerRepository(reference) + if err != nil { + return ResolvedDockerSource{}, err + } + platformRaw, err := m.runHostOutput(ctx, "docker", "buildx", "imagetools", "inspect", "--raw", repository+"@"+platformDigest) + if err != nil { + return ResolvedDockerSource{}, fmt.Errorf("resolve source platform manifest %s: %w", platform, err) + } + return parseResolvedDockerSource(reference, platform, []byte(indexRaw), []byte(platformRaw)) +} + +func dockerRepository(reference string) (string, error) { + if separator := strings.Index(reference, "@"); separator > 0 { + return reference[:separator], nil + } + tagSeparator := strings.LastIndex(reference, ":") + if tagSeparator <= strings.LastIndex(reference, "/") { + return "", fmt.Errorf("source image %s must include a tag or digest", reference) + } + return reference[:tagSeparator], nil +} + +func (m *Coordinator) dockerSandboxesDesiredManifest(ctx context.Context) (Manifest, ResolvedDockerSource, error) { + manifest, err := m.dockerSandboxesLocalManifest(ctx) + if err != nil { + return Manifest{}, ResolvedDockerSource{}, err + } + source, err := m.resolveDockerSandboxesSource(ctx) + if err != nil { + return Manifest{}, ResolvedDockerSource{}, err + } + manifest.ProviderPlatform = source.Platform + manifest.SourceImage = source.Reference + manifest.SourcePlatform = source.Platform + manifest.SourceDigest = source.IndexDigest + manifest.SourcePlatformDigest = source.PlatformDigest + manifest, err = m.resolveActionsRunner(ctx, manifest) + if err != nil { + return Manifest{}, ResolvedDockerSource{}, err + } + return manifest, source, nil +} + +func (m *Coordinator) dockerSandboxesLocalManifest(ctx context.Context) (Manifest, error) { + reference, err := NormalizeCatthehackerSource(strings.TrimSpace(m.Config.Image.SourceImage)) + if err != nil { + return Manifest{}, err + } + platform := strings.TrimSpace(m.Config.Image.SourcePlatform) + if platform == "" { + platform = strings.TrimSpace(m.Config.Provider.Platform) + } + configuredRunnerVersion := normalizedRunnerSelector(m.Config.Image.RunnerVersion) + snapshot, err := m.resolveHostTrust(ctx) + if err != nil { + return Manifest{}, err + } + customScripts, err := m.customInstallScriptDigests() + if err != nil { + return Manifest{}, err + } + trustedCertificates, err := m.trustedCACertificateDigests() + if err != nil { + return Manifest{}, err + } + templateInputs, err := fileDigestsRecursive(filepath.Join(m.ProjectRoot, "templates", "docker-sandboxes")) + if err != nil { + return Manifest{}, err + } + manifest := Manifest{ + SchemaVersion: ManifestSchemaVersion, + ProviderType: "docker-sandboxes", + ProviderPlatform: platform, + SourceType: config.ImageSourceDockerImage, + SourceImage: reference, + SourcePlatform: platform, + OutputImage: "docker-sandboxes-template", + RunnerSelector: configuredRunnerVersion, + TemplateInputs: templateInputs, + CustomInstallScripts: customScripts, + TrustedCACertificates: trustedCertificates, + HostTrust: hostTrustMetadata(snapshot), + } + return manifest, nil +} + +func fileDigestsRecursive(root string) ([]FileDigest, error) { + var result []FileDigest + err := filepath.WalkDir(root, func(path string, entry os.DirEntry, walkErr error) error { + if walkErr != nil { + return walkErr + } + if entry.IsDir() { + return nil + } + info, err := entry.Info() + if err != nil { + return err + } + if !info.Mode().IsRegular() { + return fmt.Errorf("template input %s is not a regular file", path) + } + content, err := os.ReadFile(path) + if err != nil { + return err + } + sum := sha256.Sum256(content) + relative, err := filepath.Rel(root, path) + if err != nil { + return err + } + result = append(result, FileDigest{Path: filepath.ToSlash(relative), SHA256: hex.EncodeToString(sum[:])}) + return nil + }) + sort.Slice(result, func(i, j int) bool { return result[i].Path < result[j].Path }) + return result, err +} + +func DockerSandboxesReceiptPath(projectRoot string) string { + return filepath.Join(projectRoot, ".local", "state", "image", "docker-sandboxes", "active.json") +} + +func DockerSandboxesReceiptPathForConfig(projectRoot, configPath string) (string, error) { + if strings.TrimSpace(configPath) == "" { + configPath = filepath.Join(projectRoot, ".local", "config.yml") + } + configID, err := storagecatalog.ConfigID(projectRoot, configPath) + if err != nil { + return "", err + } + return filepath.Join(projectRoot, ".local", "state", "image", configID, "docker-sandboxes", "active.json"), nil +} + +func (m *Coordinator) dockerSandboxesReceiptPath() (string, error) { + return DockerSandboxesReceiptPathForConfig(m.ProjectRoot, m.ConfigPath) +} + +func LoadDockerSandboxesReceipt(projectRoot string) (provider.TemplateArtifact, string, time.Time, error) { + receipt, err := readDockerSandboxesReceiptPath(DockerSandboxesReceiptPath(projectRoot)) + if err != nil { + return provider.TemplateArtifact{}, "", time.Time{}, err + } + return receipt.Artifact, receipt.MetadataSHA256, receipt.ActivatedAt, nil +} + +func LoadDockerSandboxesReceiptForConfig(projectRoot, configPath string) (provider.TemplateArtifact, string, time.Time, error) { + path, err := DockerSandboxesReceiptPathForConfig(projectRoot, configPath) + if err != nil { + return provider.TemplateArtifact{}, "", time.Time{}, err + } + receipt, err := readDockerSandboxesReceiptPath(path) + if err != nil { + return provider.TemplateArtifact{}, "", time.Time{}, err + } + return receipt.Artifact, receipt.MetadataSHA256, receipt.ActivatedAt, nil +} + +func (m *Coordinator) readDockerSandboxesReceipt() (dockerSandboxesReceipt, error) { + path, err := m.dockerSandboxesReceiptPath() + if err != nil { + return dockerSandboxesReceipt{}, err + } + return readDockerSandboxesReceiptPath(path) +} + +func readDockerSandboxesReceiptPath(path string) (dockerSandboxesReceipt, error) { + content, err := os.ReadFile(path) + if err != nil { + return dockerSandboxesReceipt{}, err + } + var receipt dockerSandboxesReceipt + if err := json.Unmarshal(content, &receipt); err != nil { + return dockerSandboxesReceipt{}, err + } + if receipt.SchemaVersion != dockerSandboxesReceiptSchema || receipt.ManifestHash == "" || receipt.Artifact.Reference == "" || receipt.Artifact.Digest == "" || receipt.Artifact.RootDisk == "" || receipt.MetadataSHA256 == "" || receipt.ArchiveSHA256 == "" || receipt.ArchiveBytes == 0 || len(receipt.Evidence) == 0 || receipt.ActivatedAt.IsZero() { + return dockerSandboxesReceipt{}, fmt.Errorf("invalid Docker Sandboxes active artifact receipt") + } + return receipt, nil +} + +func (m *Coordinator) ensureDockerSandboxesTemplate(ctx context.Context, force bool) error { + manifest, source, err := m.dockerSandboxesDesiredManifest(ctx) + if err != nil { + return err + } + return m.ensureDockerSandboxesTemplateResolved(ctx, force, manifest, source) +} + +func (m *Coordinator) ensureDockerSandboxesTemplateResolved(ctx context.Context, force bool, manifest Manifest, source ResolvedDockerSource) error { + runtime, ok := m.Lifecycle.(provider.TemplateArtifactRuntime) + if !ok { + return fmt.Errorf("docker-sandboxes provider is missing required template artifact integration") + } + manifestHash, err := ManifestHash(manifest) + if err != nil { + return err + } + rootDisk, err := m.effectiveDockerSandboxesRootDisk(source) + if err != nil { + return err + } + if !force { + receipt, receiptErr := m.readDockerSandboxesReceipt() + if receiptErr == nil && receipt.ManifestHash == manifestHash && receipt.Artifact.RootDisk == rootDisk { + if err := runtime.VerifyImportedTemplate(ctx, receipt.Artifact); err != nil { + if !errors.Is(err, provider.ErrTemplateNotFound) { + return fmt.Errorf("measure configured Docker Sandboxes artifact availability: %w", err) + } + m.warnf("recorded Docker Sandboxes template is absent from the authoritative Sandbox cache; rebuilding it\n") + } else { + if err := runtime.ActivateTemplate(receipt.Artifact); err != nil { + return err + } + if err := m.recordCurrentSandboxArtifact(ctx, receipt.Artifact, manifestHash, receipt.ActivatedAt); err != nil { + return fmt.Errorf("record current Docker Sandboxes template ownership: %w", err) + } + if err := m.cleanupSupersededCatalog(ctx); err != nil { + return err + } + m.infof("Docker Sandboxes runner template is current: %s@%s\n", receipt.Artifact.Reference, receipt.Artifact.Digest) + return nil + } + } + if receiptErr != nil && !errors.Is(receiptErr, os.ErrNotExist) { + m.warnf("ignoring stale unpublished Docker Sandboxes receipt and rebuilding: %v\n", receiptErr) + } + } + if m.DryRun { + m.infof("[dry-run] would build and import Docker Sandboxes runner template from %s for %s\n", source.Reference, source.Platform) + return nil + } + return m.buildDockerSandboxesTemplate(ctx, manifest, source, manifestHash, rootDisk, runtime) +} + +func (m *Coordinator) ensureDockerSandboxesTemplateWithPolicy(ctx context.Context, forceRemote bool) error { + localManifest, err := m.dockerSandboxesLocalManifest(ctx) + if err != nil { + return err + } + localHash, err := ManifestHash(localManifest) + if err != nil { + return err + } + if m.DryRun { + manifest, source, err := m.dockerSandboxesDesiredManifest(ctx) + if err != nil { + return err + } + return m.ensureDockerSandboxesTemplateResolved(ctx, false, manifest, source) + } + now := m.now() + state, err := m.readUpdatePolicyState() + if err != nil { + m.warnf("ignoring stale image update state and performing an immediate check: %v\n", err) + state = UpdatePolicyState{SchemaVersion: updatePolicyStateSchemaVersion} + } + if receipt, receiptErr := m.readDockerSandboxesReceipt(); receiptErr == nil { + runtime, ok := m.Lifecycle.(provider.TemplateArtifactRuntime) + if !ok { + return fmt.Errorf("docker-sandboxes provider is missing required template artifact integration") + } + if verifyErr := runtime.VerifyImportedTemplate(ctx, receipt.Artifact); verifyErr == nil { + bootstrapped, bootstrapErr := bootstrapUpdatePolicyState(&state, m.Config.Image, localManifest, receipt.Manifest, &receipt.Source, receipt.ActivatedAt, now.Location()) + if bootstrapErr != nil { + return bootstrapErr + } + if bootstrapped { + if err := m.writeUpdatePolicyState(state); err != nil { + return err + } + m.infof("initialized image update schedule from the verified active Docker Sandboxes template\n") + } + } else if !errors.Is(verifyErr, provider.ErrTemplateNotFound) { + return fmt.Errorf("measure configured Docker Sandboxes artifact availability: %w", verifyErr) + } + } + if recalculateScheduleForTimeZone(&state, m.Config.Image, now.Location()) { + if err := m.writeUpdatePolicyState(state); err != nil { + return err + } + } + localChanged := state.LocalInputHash != "" && state.LocalInputHash != localHash + currentVerified := false + if state.LocalInputHash == localHash && state.LastResolvedManifest != nil { + receipt, receiptErr := m.readDockerSandboxesReceipt() + if receiptErr == nil { + wantHash, hashErr := ManifestHash(*state.LastResolvedManifest) + if hashErr != nil { + return hashErr + } + runtime, ok := m.Lifecycle.(provider.TemplateArtifactRuntime) + if !ok { + return fmt.Errorf("docker-sandboxes provider is missing required template artifact integration") + } + if receipt.ManifestHash == wantHash { + if verifyErr := runtime.VerifyImportedTemplate(ctx, receipt.Artifact); verifyErr == nil { + currentVerified = true + } else if !errors.Is(verifyErr, provider.ErrTemplateNotFound) { + return fmt.Errorf("measure configured Docker Sandboxes artifact availability: %w", verifyErr) + } + } + } + } + if state.LocalInputHash == localHash && pendingUpdateReady(state, now) { + if err := m.ApplyPendingUpdate(ctx, now); err != nil { + if !forceRemote && currentVerified { + status, _ := m.UpdatePolicyStatus() + m.warnf("pending scheduled Docker Sandboxes update failed; continuing with the previous verified template and retrying after %s: %v\n", formatUpdateTime(status.NextRetryAt), err) + return m.ensureDockerSandboxesTemplateFromState(ctx, state) + } + return err + } + m.infof("pending Docker Sandboxes update activated\n") + return nil + } + if !forceRemote && currentVerified && !updateCheckDue(state, m.Config.Image, now) { + if err := m.ensureDockerSandboxesTemplateFromState(ctx, state); err != nil { + return err + } + m.infof("Docker Sandboxes runner template is current; next remote check %s\n", formatUpdateTime(state.NextEligibleAt)) + return nil + } + + state.LastAttemptAt = now.UTC() + manifest, source, resolveErr := m.dockerSandboxesDesiredManifest(ctx) + if resolveErr != nil { + scheduleUpdateFailure(&state, now, resolveErr) + _ = m.writeUpdatePolicyState(state) + if !forceRemote && !localChanged && currentVerified { + m.warnf("scheduled image update check failed; continuing with the last verified Docker Sandboxes template and retrying after %s: %v\n", formatUpdateTime(state.NextRetryAt), resolveErr) + return m.ensureDockerSandboxesTemplateFromState(ctx, state) + } + return resolveErr + } + state.LocalInputHash = localHash + state.PendingManifest = &manifest + state.PendingSource = &source + state.DeferredReason = "template build and activation pending" + if err := m.writeUpdatePolicyState(state); err != nil { + return err + } + if err := m.ensureDockerSandboxesTemplateResolved(ctx, false, manifest, source); err != nil { + scheduleUpdateFailure(&state, now, err) + _ = m.writeUpdatePolicyState(state) + if !forceRemote && !localChanged && currentVerified { + m.warnf("scheduled Docker Sandboxes update failed; restoring the last verified template and retrying after %s: %v\n", formatUpdateTime(state.NextRetryAt), err) + return m.ensureDockerSandboxesTemplateFromState(ctx, state) + } + return err + } + state.LastResolvedManifest = &manifest + state.LastResolvedSource = &source + state.PendingManifest = nil + state.PendingSource = nil + if err := scheduleNextSuccess(&state, m.Config.Image, m.now()); err != nil { + return err + } + return m.writeUpdatePolicyState(state) +} + +func (m *Coordinator) ensureDockerSandboxesTemplateFromState(ctx context.Context, state UpdatePolicyState) error { + if state.LastResolvedManifest == nil || state.LastResolvedSource == nil { + return fmt.Errorf("Docker Sandboxes update state is missing its resolved artifact inputs") + } + return m.ensureDockerSandboxesTemplateResolved(ctx, false, *state.LastResolvedManifest, *state.LastResolvedSource) +} + +func estimatedDockerSandboxesExpansion(source ResolvedDockerSource) uint64 { + estimate, err := EstimateSourceSize(source.CompressedLayerBytes, 0) + if err != nil { + return ^uint64(0) + } + dockerDisk, _ := config.ParseByteSize(config.DockerSandboxesDefaultDockerDisk) + plan, err := PlanArtifactStorage("docker-sandboxes", estimate, false, uint64(dockerDisk)) + if err != nil { + return ^uint64(0) + } + return plan.EstimatedIncrementalPeak +} + +func (m *Coordinator) effectiveDockerSandboxesRootDisk(source ResolvedDockerSource) (string, error) { + estimate, err := EstimateSourceSize(source.CompressedLayerBytes, 0) + if err != nil { + return "", err + } + required, err := AutomaticDockerSandboxesRootBytes(estimate.ExpandedBytes) + if err != nil { + return "", err + } + if m.Config.DockerSandboxes.RootDisk != config.DockerSandboxesAutomaticRootDisk { + configured, err := config.ParseByteSize(m.Config.DockerSandboxes.RootDisk) + if err != nil { + return "", err + } + if uint64(configured) < required { + return "", fmt.Errorf("dockerSandboxes.rootDisk %s is too small for %s; use rootDisk: auto or at least %dGiB", m.Config.DockerSandboxes.RootDisk, source.Reference, required/storage.GiB) + } + return m.Config.DockerSandboxes.RootDisk, nil + } + return fmt.Sprintf("%dGiB", required/storage.GiB), nil +} + +func (m *Coordinator) buildDockerSandboxesTemplate(ctx context.Context, manifest Manifest, source ResolvedDockerSource, manifestHash, rootDisk string, runtime provider.TemplateArtifactRuntime) error { + if err := m.preflightStorage("template-build", estimatedDockerSandboxesExpansion(source)); err != nil { + return err + } + if err := m.verifyDockerSandboxesNativeBuilder(ctx, source.Platform); err != nil { + return err + } + lock, err := loadDockerSandboxesSourceLock(m.ProjectRoot, source.Platform) + if err != nil { + return err + } + profile := sourceProfile(source.Reference) + architecture := strings.TrimPrefix(source.Platform, "linux/") + tagProfile := sanitizeTemplateTag(profile) + templateTag := fmt.Sprintf("epar-docker-sandboxes-catthehacker-%s:%s-%s", tagProfile, manifestHash[:16], architecture) + artifactRoot, err := m.dockerSandboxesArtifactRoot(manifestHash) + if err != nil { + return fmt.Errorf("resolve Docker Sandboxes template workspace: %w", err) + } + if err := os.MkdirAll(artifactRoot, 0o755); err != nil { + return err + } + if err := m.recordSandboxWorkspace(ctx, artifactRoot, manifestHash, storagecatalog.StateStaging, time.Now().UTC()); err != nil { + return fmt.Errorf("record Docker Sandboxes archive workspace ownership: %w", err) + } + platformLock := lock.Platforms[source.Platform] + builder, err := m.ensureBuildxBuilder(ctx, []string{ + source.ImmutableReference, + lock.DockerfileFrontend.Reference, + platformLock.GoBuilderReference, + platformLock.SBOMGeneratorReference, + }) + if err != nil { + return err + } + buildTrust, err := m.resolveBuildTrust(ctx) + if err != nil { + return err + } + downloadClient, err := buildTrustHTTPClient(buildTrust) + if err != nil { + return err + } + inputRoot := filepath.Join(artifactRoot, "inputs") + actionsRunnerPath := filepath.Join(inputRoot, "actions-runner.tar.gz") + tiniPath := filepath.Join(inputRoot, "tini") + actionsRunnerCachePath, err := m.acquireActionsRunner(ctx, manifest) + if err != nil { + return err + } + if err := copyFile(actionsRunnerCachePath, actionsRunnerPath, 0o600); err != nil { + return fmt.Errorf("stage GitHub Actions runner %s: %w", manifest.RunnerVersion, err) + } + if err := verifiedDownload(ctx, downloadClient, platformLock.Tini.URL, tiniPath, platformLock.Tini.SHA256, 0o700); err != nil { + return fmt.Errorf("acquire locked tini: %w", err) + } + buildMetadataPath := filepath.Join(artifactRoot, "build-metadata.json") + attestationMetadataPath := filepath.Join(artifactRoot, "attestation-metadata.json") + provenancePath := filepath.Join(artifactRoot, "provenance.json") + sbomPath := filepath.Join(artifactRoot, "sbom.intoto.json") + inventoryPath := filepath.Join(artifactRoot, "software-inventory.txt") + compatibilityEvidencePath := filepath.Join(artifactRoot, "compatibility.json") + archivePath := filepath.Join(artifactRoot, "runner-template.tar") + partialArchivePath := archivePath + ".partial" + metadataPath := filepath.Join(artifactRoot, "template-metadata.json") + resumed, err := m.resumeDockerSandboxesTemplate(ctx, manifest, source, manifestHash, rootDisk, artifactRoot, metadataPath, archivePath, runtime) + if err != nil { + return err + } + if resumed { + return nil + } + var buildMetadata dockerSandboxesBuildMetadata + var localDigest string + var contextRoot string + var buildArguments []string + var archiveSHA string + var archiveBytes uint64 + recoveredBuild := false + if recoveredBuild { + m.infof("reusing completed exact Docker Sandboxes Buildx result %s\n", localDigest) + if err := writeJSONFile(buildMetadataPath, buildMetadata); err != nil { + return err + } + if err := writeAtomicFile(provenancePath, append(buildMetadata.Provenance, '\n'), 0o644); err != nil { + return err + } + } else { + contextRoot, err = os.MkdirTemp(filepath.Join(m.ProjectRoot, ".local"), "docker-sandboxes-context-") + if err != nil { + return err + } + defer os.RemoveAll(contextRoot) + if err := copyDirectory(filepath.Join(m.ProjectRoot, "templates", "docker-sandboxes"), contextRoot); err != nil { + return err + } + compatibilityPath := filepath.Join(contextRoot, "profiles", "generated.compatibility.json") + compatibility := map[string]any{ + "schemaVersion": 2, + "templateSchemaVersion": 1, + "profile": profile, + "platform": source.Platform, + "runnerExecution": "direct-actions-listener", + "dockerDaemonOwner": "docker-sandboxes-runtime", + "expectedDockerDaemonCount": 1, + } + if err := writeJSONFile(compatibilityPath, compatibility); err != nil { + return err + } + if err := copyFile(compatibilityPath, compatibilityEvidencePath, 0o644); err != nil { + return err + } + if err := m.prepareDockerSandboxesCustomScripts(contextRoot); err != nil { + return err + } + runnerTrust, err := m.resolveHostTrust(ctx) + if err != nil { + return err + } + if err := m.prepareDockerSandboxesTrustPolicy(contextRoot, runnerTrust); err != nil { + return err + } + if err := os.MkdirAll(filepath.Join(contextRoot, "inputs"), 0o755); err != nil { + return err + } + if err := copyFile(actionsRunnerPath, filepath.Join(contextRoot, "inputs", "actions-runner.tar.gz"), 0o600); err != nil { + return err + } + if err := copyFile(tiniPath, filepath.Join(contextRoot, "inputs", "tini"), 0o755); err != nil { + return err + } + for _, path := range []string{buildMetadataPath, attestationMetadataPath, provenancePath, sbomPath, inventoryPath, archivePath, partialArchivePath, metadataPath} { + if err := os.Remove(path); err != nil && !os.IsNotExist(err) { + return err + } + } + installationID, err := m.catalogInstallationID(time.Now().UTC()) + if err != nil { + return fmt.Errorf("resolve EPAR template ownership identity: %w", err) + } + args := []string{ + "buildx", "build", "--builder", builder, "--platform", source.Platform, "--pull", "--progress", "plain", + "--target", "runner-template", "--output", "type=docker,dest=" + partialArchivePath, + "--provenance=false", "--sbom=false", + "--metadata-file", buildMetadataPath, "--tag", templateTag, + "--label", "io.solutionforest.epar.schema=1", + "--label", "io.solutionforest.epar.installation=" + installationID, + "--label", "io.solutionforest.epar.provider=docker-sandboxes", + "--label", "io.solutionforest.epar.role=template-staging", + "--label", "io.solutionforest.epar.manifest=" + manifestHash, + } + buildArguments = []string{ + "TEMPLATE_PLATFORM=" + source.Platform, + "SOURCE_IMAGE=" + source.ImmutableReference, + "GO_BUILDER_IMAGE=" + platformLock.GoBuilderReference, + "HOOK_LAUNCHER_SHA256=" + lock.HookLauncher.SHA256, + "SOURCE_PROFILE=" + profile, + "SOURCE_INDEX_DIGEST=" + source.IndexDigest, + "SOURCE_MANIFEST_DIGEST=" + source.PlatformDigest, + "SOURCE_REVISION=" + source.IndexDigest, + "TEMPLATE_VERSION=" + manifestHash[:16] + "-" + architecture, + "COMPATIBILITY_FILE=generated.compatibility.json", + "ACTIONS_RUNNER_VERSION=" + manifest.RunnerVersion, + "ACTIONS_RUNNER_SHA256=sha256:" + strings.TrimPrefix(manifest.RunnerAssetDigest, "sha256:"), + "TINI_SHA256=sha256:" + strings.TrimPrefix(platformLock.Tini.SHA256, "sha256:"), + } + for _, buildArg := range buildArguments { + args = append(args, "--build-arg", buildArg) + } + args = append(args, "--file", filepath.Join(contextRoot, "Dockerfile"), contextRoot) + buildLogPath := m.buildLogPath("docker-sandboxes-" + manifestHash[:16] + ".docker-build.log") + defer m.releaseTranscript(buildLogPath) + if err := resetLogs(buildLogPath); err != nil { + return err + } + m.infof("building Docker Sandboxes runner template from %s for %s\n", source.Reference, source.Platform) + m.infof("full Docker Sandboxes Buildx progress: %s\n", buildLogPath) + if err := m.runHostBuildxLogged(ctx, buildLogPath, "docker", args...); err != nil { + return fmt.Errorf("build Docker Sandboxes runner template: %w%s", err, boundedRedactedLogTail(buildLogPath, 32*1024)) + } + if err := readJSONFile(buildMetadataPath, &buildMetadata); err != nil { + return fmt.Errorf("read Docker Sandboxes Buildx metadata: %w", err) + } + expectedLabels := map[string]string{ + "io.solutionforest.epar.schema": "1", + "io.solutionforest.epar.installation": installationID, + "io.solutionforest.epar.provider": "docker-sandboxes", + "io.solutionforest.epar.role": "template-staging", + "io.solutionforest.epar.manifest": manifestHash, + } + archiveVerification, err := verifyDockerSandboxesArchive(partialArchivePath, "docker.io/library/"+templateTag, source.Platform, buildMetadata.ImageDigest, expectedLabels) + if err != nil { + return fmt.Errorf("verify directly exported Docker Sandboxes archive: %w", err) + } + localDigest = archiveVerification.ImageDigest + archiveSHA = archiveVerification.ArchiveSHA256 + archiveBytes = archiveVerification.ArchiveBytes + if err := os.Rename(partialArchivePath, archivePath); err != nil { + return fmt.Errorf("activate verified Docker Sandboxes archive: %w", err) + } + } + evidenceExportRoot := filepath.Join(artifactRoot, "evidence-export") + if err := os.RemoveAll(evidenceExportRoot); err != nil { + return err + } + attestationArgs := []string{ + "buildx", "build", "--builder", builder, "--platform", source.Platform, "--progress", "plain", + "--target", "software-inventory-export", "--output", "type=local,dest=" + evidenceExportRoot, + "--provenance", "mode=max", "--sbom", "generator=" + platformLock.SBOMGeneratorReference, + "--metadata-file", attestationMetadataPath, + } + for _, buildArg := range buildArguments { + attestationArgs = append(attestationArgs, "--build-arg", buildArg) + } + attestationArgs = append(attestationArgs, "--file", filepath.Join(contextRoot, "Dockerfile"), contextRoot) + attestationLogPath := m.buildLogPath("docker-sandboxes-" + manifestHash[:16] + "-attestation.docker-build.log") + defer m.releaseTranscript(attestationLogPath) + if err := resetLogs(attestationLogPath); err != nil { + return err + } + m.infof("full Docker Sandboxes provenance, SBOM, and software-inventory progress: %s\n", attestationLogPath) + if err := m.runHostBuildxLogged(ctx, attestationLogPath, "docker", attestationArgs...); err != nil { + return fmt.Errorf("generate Docker Sandboxes template evidence: %w%s", err, boundedRedactedLogTail(attestationLogPath, 16*1024)) + } + var attestationMetadata dockerSandboxesBuildMetadata + if err := readJSONFile(attestationMetadataPath, &attestationMetadata); err != nil { + return fmt.Errorf("read Docker Sandboxes attestation metadata: %w", err) + } + if len(attestationMetadata.Provenance) == 0 || string(attestationMetadata.Provenance) == "null" { + return fmt.Errorf("Docker Sandboxes attestation build omitted max-mode provenance") + } + if err := validateBuildxMaxProvenance(attestationMetadata.Provenance); err != nil { + return fmt.Errorf("validate Docker Sandboxes Buildx provenance metadata: %w", err) + } + exportedProvenancePath := filepath.Join(evidenceExportRoot, "provenance.json") + exportedProvenance, err := readVerifiedBuildEvidence(exportedProvenancePath, storage.GiB) + if err != nil { + return fmt.Errorf("read exported Docker Sandboxes provenance: %w", err) + } + if err := validateInTotoProvenance(exportedProvenance); err != nil { + return fmt.Errorf("validate exported Docker Sandboxes provenance: %w", err) + } + if err := writeAtomicFile(provenancePath, append(exportedProvenance, '\n'), 0o644); err != nil { + return err + } + exportedSBOMPath := filepath.Join(evidenceExportRoot, "sbom-runner-template.spdx.json") + exportedSBOM, err := readVerifiedBuildEvidence(exportedSBOMPath, storage.GiB) + if err != nil { + return fmt.Errorf("read exported Docker Sandboxes runner-template SBOM: %w", err) + } + if err := writeAtomicFile(sbomPath, exportedSBOM, 0o644); err != nil { + return err + } + if err := validateInTotoSPDX(sbomPath); err != nil { + return fmt.Errorf("validate exported Docker Sandboxes runner-template SBOM: %w", err) + } + sbomSourceDigest, _, err := hashFile(sbomPath) + if err != nil { + return err + } + inventory, err := readVerifiedBuildEvidence(filepath.Join(evidenceExportRoot, "software-inventory.txt"), 64*storage.MiB) + if err != nil { + return fmt.Errorf("read exported Docker Sandboxes software inventory: %w", err) + } + if len(strings.TrimSpace(string(inventory))) == 0 { + return errors.New("exported Docker Sandboxes software inventory is empty") + } + if err := writeAtomicFile(inventoryPath, inventory, 0o644); err != nil { + return err + } + artifact := provider.TemplateArtifact{ + Reference: "docker.io/library/" + templateTag, + Digest: localDigest, + CacheID: strings.TrimPrefix(localDigest, "sha256:")[:12], + Platform: source.Platform, + RootDisk: rootDisk, + } + metadata := dockerSandboxesTemplateMetadata{ + SchemaVersion: dockerSandboxesMetadataSchema, + Profile: profile, + Platform: source.Platform, + ManifestHash: manifestHash, + Source: source, + Artifacts: make(map[string]artifactEvidence), + } + metadata.Template.Tag = artifact.Reference + metadata.Template.Digest = localDigest + metadata.Template.CacheID = artifact.CacheID + metadata.Template.RootDisk = artifact.RootDisk + metadata.Template.Archive = filepath.Base(archivePath) + metadata.Template.ArchiveSHA256 = archiveSHA + metadata.Template.ArchiveBytes = archiveBytes + metadata.Compatibility.TemplateSchemaVersion = 1 + metadata.Compatibility.RunnerExecution = "direct-actions-listener" + metadata.Compatibility.DockerDaemonOwner = "docker-sandboxes-runtime" + metadata.Compatibility.ExpectedDockerDaemonCount = 1 + for name, path := range map[string]string{ + "buildMetadata": buildMetadataPath, + "attestationMetadata": attestationMetadataPath, + "provenance": provenancePath, + "sbom": sbomPath, + "softwareInventory": inventoryPath, + "compatibility": compatibilityEvidencePath, + } { + digest, _, err := hashFile(path) + if err != nil { + return err + } + evidence := artifactEvidence{Path: filepath.Base(path), SHA256: digest} + if name == "sbom" { + evidence.SourceDigest = sbomSourceDigest + } + metadata.Artifacts[name] = evidence + } + if err := writeJSONFile(metadataPath, metadata); err != nil { + return err + } + metadataSHA, _, err := hashFile(metadataPath) + if err != nil { + return err + } + if err := m.withSandboxBackendLock(ctx, func() error { + if err := runtime.VerifyImportedTemplate(ctx, artifact); err != nil { + if !errors.Is(err, provider.ErrTemplateNotFound) { + return err + } + label := fmt.Sprintf("Docker Sandboxes template-cache import for %s", artifact.Reference) + if err := m.runProgressOperation(label, nil, func() error { + return runtime.ImportTemplate(ctx, archivePath) + }); err != nil { + return err + } + } + if err := m.runProgressOperation("Docker Sandboxes imported-template verification", nil, func() error { + return runtime.VerifyImportedTemplate(ctx, artifact) + }); err != nil { + return fmt.Errorf("verify imported Docker Sandboxes runner template: %w", err) + } + return nil + }); err != nil { + return err + } + return m.activateDockerSandboxesTemplate(ctx, manifest, source, manifestHash, artifact, metadataPath, metadataSHA, archivePath, archiveSHA, runtime) +} + +func (m *Coordinator) dockerSandboxesArtifactRoot(manifestHash string) (string, error) { + configID, err := storagecatalog.ConfigID(m.ProjectRoot, m.effectiveConfigPath()) + if err != nil { + return "", err + } + return filepath.Join(m.ProjectRoot, "work", "template-builds", "docker-sandboxes", configID, manifestHash), nil +} + +func readVerifiedBuildEvidence(path string, maximumBytes uint64) ([]byte, error) { + if maximumBytes == 0 || maximumBytes > uint64(^uint(0)>>1) { + return nil, fmt.Errorf("invalid build-evidence size limit %d", maximumBytes) + } + before, err := storage.SnapshotFilesystemTarget(path) + if err != nil { + return nil, err + } + info, err := os.Lstat(path) + if err != nil { + return nil, err + } + if !info.Mode().IsRegular() || info.Size() <= 0 || uint64(info.Size()) > maximumBytes { + return nil, fmt.Errorf("build evidence %q has invalid size %d", path, info.Size()) + } + content, err := os.ReadFile(path) + if err != nil { + return nil, err + } + after, err := storage.SnapshotFilesystemTarget(path) + if err != nil { + return nil, err + } + if before.Identity != after.Identity || before.Fingerprint != after.Fingerprint || uint64(len(content)) != uint64(info.Size()) { + return nil, fmt.Errorf("build evidence %q changed during readback", path) + } + return content, nil +} + +func validateBuildxMaxProvenance(content []byte) error { + var provenance struct { + BuildType string `json:"buildType"` + Materials []json.RawMessage `json:"materials"` + Invocation struct { + Parameters json.RawMessage `json:"parameters"` + } `json:"invocation"` + } + if err := json.Unmarshal(content, &provenance); err != nil { + return err + } + if provenance.BuildType == "" || len(provenance.Materials) == 0 || len(provenance.Invocation.Parameters) == 0 || string(provenance.Invocation.Parameters) == "null" { + return errors.New("max-mode Buildx provenance omitted build type, materials, or invocation parameters") + } + return nil +} + +func validateInTotoProvenance(content []byte) error { + var statement struct { + Type string `json:"_type"` + PredicateType string `json:"predicateType"` + Subject []json.RawMessage `json:"subject"` + Predicate json.RawMessage `json:"predicate"` + } + if err := json.Unmarshal(content, &statement); err != nil { + return err + } + if statement.Type != "https://in-toto.io/Statement/v1" || statement.PredicateType != "https://slsa.dev/provenance/v1" || len(statement.Subject) == 0 || len(statement.Predicate) == 0 || string(statement.Predicate) == "null" { + return errors.New("exported provenance is not a complete in-toto SLSA v1 statement") + } + return nil +} + +func validateInTotoSPDX(path string) error { + file, err := os.Open(path) + if err != nil { + return err + } + defer file.Close() + decoder := json.NewDecoder(file) + open, err := decoder.Token() + if err != nil { + return err + } + if open != json.Delim('{') { + return fmt.Errorf("SBOM attachment is not an in-toto JSON object") + } + var statementType string + var predicateType string + var spdxID string + for decoder.More() { + key, err := decoder.Token() + if err != nil { + return err + } + switch key { + case "_type": + if err := decoder.Decode(&statementType); err != nil { + return err + } + case "predicateType": + if err := decoder.Decode(&predicateType); err != nil { + return err + } + case "predicate": + spdxID, err = scanSPDXPredicate(decoder) + if err != nil { + return err + } + default: + if err := skipJSONValue(decoder); err != nil { + return err + } + } + } + if _, err := decoder.Token(); err != nil { + return err + } + if statementType != "https://in-toto.io/Statement/v0.1" && statementType != "https://in-toto.io/Statement/v1" { + return fmt.Errorf("unsupported in-toto statement type %q", statementType) + } + if predicateType != "https://spdx.dev/Document" { + return fmt.Errorf("unexpected SBOM predicate type %q", predicateType) + } + if spdxID != "SPDXRef-DOCUMENT" { + return fmt.Errorf("unexpected SPDX document identity %q", spdxID) + } + if token, err := decoder.Token(); err != io.EOF { + if err == nil { + return fmt.Errorf("SBOM attachment contains trailing JSON token %v", token) + } + return err + } + return nil +} + +func scanSPDXPredicate(decoder *json.Decoder) (string, error) { + open, err := decoder.Token() + if err != nil { + return "", err + } + if open != json.Delim('{') { + return "", fmt.Errorf("SPDX predicate is not a JSON object") + } + var spdxID string + for decoder.More() { + key, err := decoder.Token() + if err != nil { + return "", err + } + if key == "SPDXID" { + if err := decoder.Decode(&spdxID); err != nil { + return "", err + } + continue + } + if err := skipJSONValue(decoder); err != nil { + return "", err + } + } + _, err = decoder.Token() + return spdxID, err +} + +func skipJSONValue(decoder *json.Decoder) error { + token, err := decoder.Token() + if err != nil { + return err + } + delim, compound := token.(json.Delim) + if !compound || (delim != '{' && delim != '[') { + return nil + } + for decoder.More() { + if delim == '{' { + if _, err := decoder.Token(); err != nil { + return err + } + } + if err := skipJSONValue(decoder); err != nil { + return err + } + } + _, err = decoder.Token() + return err +} + +func (m *Coordinator) verifyDockerSandboxesNativeBuilder(ctx context.Context, platform string) error { + reported, err := m.runHostOutput(ctx, "docker", "info", "--format", "{{.OSType}}/{{.Architecture}}") + if err != nil { + return fmt.Errorf("determine Docker server platform for Docker Sandboxes template build: %w", err) + } + native := strings.ToLower(strings.TrimSpace(reported)) + switch native { + case "linux/x86_64": + native = "linux/amd64" + case "linux/aarch64": + native = "linux/arm64" + } + if native != platform { + return fmt.Errorf("Docker Sandboxes template builds require a native %s Docker server; Docker reports %s and EPAR does not use emulation for this artifact", platform, native) + } + return nil +} + +func (m *Coordinator) resumeDockerSandboxesTemplate(ctx context.Context, manifest Manifest, source ResolvedDockerSource, manifestHash, rootDisk, artifactRoot, metadataPath, archivePath string, runtime provider.TemplateArtifactRuntime) (bool, error) { + metadata, artifact, metadataSHA, archiveSHA, valid, err := verifiedDockerSandboxesBuildArtifact(artifactRoot, metadataPath, archivePath, manifestHash, source) + if err != nil { + return false, err + } + if !valid { + return false, nil + } + if artifact.RootDisk != rootDisk { + return false, nil + } + if err := m.withSandboxBackendLock(ctx, func() error { + if err := runtime.VerifyImportedTemplate(ctx, artifact); err != nil { + if !errors.Is(err, provider.ErrTemplateNotFound) { + return err + } + if err := m.runProgressOperation("Docker Sandboxes resumed template-cache import", nil, func() error { + return runtime.ImportTemplate(ctx, archivePath) + }); err != nil { + return err + } + } + if err := m.runProgressOperation("Docker Sandboxes resumed imported-template verification", nil, func() error { + return runtime.VerifyImportedTemplate(ctx, artifact) + }); err != nil { + return fmt.Errorf("verify resumed Docker Sandboxes runner template: %w", err) + } + return nil + }); err != nil { + return false, err + } + if metadata.ManifestHash != manifestHash { + return false, fmt.Errorf("verified Docker Sandboxes build evidence changed during resume") + } + if err := m.activateDockerSandboxesTemplate(ctx, manifest, source, manifestHash, artifact, metadataPath, metadataSHA, archivePath, archiveSHA, runtime); err != nil { + return false, err + } + m.infof("resumed Docker Sandboxes runner template from verified interrupted build evidence\n") + return true, nil +} + +func verifiedDockerSandboxesBuildArtifact(artifactRoot, metadataPath, archivePath, manifestHash string, source ResolvedDockerSource) (dockerSandboxesTemplateMetadata, provider.TemplateArtifact, string, string, bool, error) { + var metadata dockerSandboxesTemplateMetadata + info, err := os.Lstat(metadataPath) + if errors.Is(err, os.ErrNotExist) { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + if err != nil { + return metadata, provider.TemplateArtifact{}, "", "", false, err + } + if !info.Mode().IsRegular() { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + if err := readJSONFile(metadataPath, &metadata); err != nil { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + if metadata.SchemaVersion != dockerSandboxesMetadataSchema || metadata.ManifestHash != manifestHash || metadata.Source != source || metadata.Platform != source.Platform || metadata.Profile != sourceProfile(source.Reference) { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + if !validSHA256(metadata.Template.Digest) || metadata.Template.CacheID != strings.TrimPrefix(metadata.Template.Digest, "sha256:")[:12] || metadata.Template.Tag == "" || metadata.Template.RootDisk == "" || metadata.Template.Archive != filepath.Base(archivePath) { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + if metadata.Compatibility.TemplateSchemaVersion != 1 || metadata.Compatibility.RunnerExecution != "direct-actions-listener" || metadata.Compatibility.DockerDaemonOwner != "docker-sandboxes-runtime" || metadata.Compatibility.ExpectedDockerDaemonCount != 1 { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + archiveInfo, err := os.Lstat(archivePath) + if err != nil || !archiveInfo.Mode().IsRegular() { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + archiveSHA, archiveBytes, err := hashFile(archivePath) + if err != nil { + return metadata, provider.TemplateArtifact{}, "", "", false, err + } + if archiveSHA != metadata.Template.ArchiveSHA256 || archiveBytes != metadata.Template.ArchiveBytes { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + var buildMetadata dockerSandboxesBuildMetadata + buildMetadataEvidence, found := metadata.Artifacts["buildMetadata"] + if !found || filepath.Base(buildMetadataEvidence.Path) != buildMetadataEvidence.Path { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + if err := readJSONFile(filepath.Join(artifactRoot, buildMetadataEvidence.Path), &buildMetadata); err != nil { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + archiveVerification, err := verifyDockerSandboxesArchive(archivePath, metadata.Template.Tag, source.Platform, buildMetadata.ImageDigest, map[string]string{ + "io.solutionforest.epar.schema": "1", + "io.solutionforest.epar.installation": "*", + "io.solutionforest.epar.provider": "docker-sandboxes", + "io.solutionforest.epar.role": "template-staging", + "io.solutionforest.epar.manifest": manifestHash, + }) + if err != nil || archiveVerification.ArchiveSHA256 != archiveSHA || archiveVerification.ArchiveBytes != archiveBytes { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + requiredEvidence := []string{"buildMetadata", "attestationMetadata", "provenance", "sbom", "softwareInventory", "compatibility"} + for _, name := range requiredEvidence { + evidence, found := metadata.Artifacts[name] + if !found || filepath.Base(evidence.Path) != evidence.Path || !validSHA256(evidence.SHA256) { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + evidencePath := filepath.Join(artifactRoot, evidence.Path) + evidenceInfo, err := os.Lstat(evidencePath) + if err != nil || !evidenceInfo.Mode().IsRegular() { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + digest, _, err := hashFile(evidencePath) + if err != nil { + return metadata, provider.TemplateArtifact{}, "", "", false, err + } + if digest != evidence.SHA256 { + return metadata, provider.TemplateArtifact{}, "", "", false, nil + } + } + metadataSHA, _, err := hashFile(metadataPath) + if err != nil { + return metadata, provider.TemplateArtifact{}, "", "", false, err + } + return metadata, provider.TemplateArtifact{ + Reference: metadata.Template.Tag, + Digest: metadata.Template.Digest, + CacheID: metadata.Template.CacheID, + Platform: metadata.Platform, + RootDisk: metadata.Template.RootDisk, + }, metadataSHA, archiveSHA, true, nil +} + +func (m *Coordinator) activateDockerSandboxesTemplate(ctx context.Context, manifest Manifest, source ResolvedDockerSource, manifestHash string, artifact provider.TemplateArtifact, metadataPath, metadataSHA, archivePath, archiveSHA string, runtime provider.TemplateArtifactRuntime) error { + if err := runtime.ActivateTemplate(artifact); err != nil { + return err + } + archiveInfo, err := os.Lstat(archivePath) + if err != nil { + return fmt.Errorf("inspect verified Docker Sandboxes archive before activation: %w", err) + } + if !archiveInfo.Mode().IsRegular() { + return errors.New("verified Docker Sandboxes archive is not a regular file") + } + evidence, err := m.persistDockerSandboxesCompactEvidence(manifestHash, filepath.Dir(metadataPath)) + if err != nil { + return err + } + receipt := dockerSandboxesReceipt{ + SchemaVersion: dockerSandboxesReceiptSchema, + ManifestHash: manifestHash, + Manifest: manifest, + Source: source, + Artifact: artifact, + MetadataSHA256: metadataSHA, + ArchiveSHA256: archiveSHA, + ArchiveBytes: uint64(archiveInfo.Size()), + Evidence: evidence, + ActivatedAt: time.Now().UTC(), + } + receiptPath, err := m.dockerSandboxesReceiptPath() + if err != nil { + return err + } + if err := writeJSONFile(receiptPath, receipt); err != nil { + return err + } + if err := m.recordCurrentSandboxArtifact(ctx, artifact, manifestHash, receipt.ActivatedAt); err != nil { + return fmt.Errorf("record current Docker Sandboxes template ownership: %w", err) + } + if err := m.recordSandboxWorkspace(ctx, filepath.Dir(archivePath), manifestHash, storagecatalog.StateSuperseded, receipt.ActivatedAt); err != nil { + return fmt.Errorf("record Docker Sandboxes staging ownership: %w", err) + } + if err := m.cleanupSupersededCatalog(ctx); err != nil { + return err + } + m.infof("activated Docker Sandboxes runner template %s@%s\n", artifact.Reference, artifact.Digest) + return nil +} + +func (m *Coordinator) persistDockerSandboxesCompactEvidence(manifestHash, artifactRoot string) (map[string]artifactEvidence, error) { + receiptPath, err := m.dockerSandboxesReceiptPath() + if err != nil { + return nil, err + } + evidenceRoot := filepath.Join(filepath.Dir(receiptPath), "evidence", manifestHash) + if err := os.MkdirAll(evidenceRoot, 0o700); err != nil { + return nil, err + } + result := make(map[string]artifactEvidence) + for name, filename := range map[string]string{ + "buildMetadata": "build-metadata.json", + "attestationMetadata": "attestation-metadata.json", + "provenance": "provenance.json", + "softwareInventory": "software-inventory.txt", + "compatibility": "compatibility.json", + "templateMetadata": "template-metadata.json", + } { + source := filepath.Join(artifactRoot, filename) + destination := filepath.Join(evidenceRoot, filename) + if err := copyFile(source, destination, 0o600); err != nil { + return nil, fmt.Errorf("retain Docker Sandboxes %s evidence: %w", name, err) + } + digest, _, err := hashFile(destination) + if err != nil { + return nil, err + } + result[name] = artifactEvidence{Path: filepath.ToSlash(filepath.Join("evidence", manifestHash, filename)), SHA256: digest} + } + sbomPath := filepath.Join(artifactRoot, "sbom.intoto.json") + sbomDigest, sbomBytes, err := hashFile(sbomPath) + if err != nil { + return nil, err + } + descriptorPath := filepath.Join(evidenceRoot, "sbom-descriptor.json") + if err := writeJSONFile(descriptorPath, map[string]any{ + "schemaVersion": 1, + "digest": sbomDigest, + "size": sbomBytes, + }); err != nil { + return nil, err + } + descriptorDigest, _, err := hashFile(descriptorPath) + if err != nil { + return nil, err + } + result["sbomDescriptor"] = artifactEvidence{Path: filepath.ToSlash(filepath.Join("evidence", manifestHash, "sbom-descriptor.json")), SHA256: descriptorDigest, SourceDigest: sbomDigest} + return result, nil +} + +func loadDockerSandboxesSourceLock(projectRoot, platform string) (dockerSandboxesSourceLock, error) { + var lock dockerSandboxesSourceLock + path := filepath.Join(projectRoot, "templates", "docker-sandboxes", "sources.lock.json") + if err := readJSONFile(path, &lock); err != nil { + return lock, fmt.Errorf("read Docker Sandboxes source lock: %w", err) + } + if lock.SchemaVersion != 2 { + return lock, fmt.Errorf("unsupported Docker Sandboxes source lock schema %d", lock.SchemaVersion) + } + if lock.DockerfileFrontend.Reference == "" || lock.SBOMGenerator.InspectionReference == "" || lock.GoBuilder.Version == "" || lock.GoBuilder.IndexDigest == "" || lock.HookLauncher.SHA256 == "" || lock.Tini.Version == "" { + return lock, errors.New("Docker Sandboxes source lock has incomplete shared build inputs") + } + platformLock, ok := lock.Platforms[platform] + if !ok || platformLock.GoBuilderReference == "" || platformLock.GoBuilderManifestDigest == "" || platformLock.SBOMGeneratorReference == "" || platformLock.SBOMGeneratorManifestDigest == "" || platformLock.DockerfileFrontendManifestDigest == "" || platformLock.Tini.URL == "" || platformLock.Tini.SHA256 == "" { + return lock, fmt.Errorf("Docker Sandboxes source lock has incomplete build inputs for %s", platform) + } + return lock, nil +} + +func sanitizeTemplateTag(value string) string { + value = strings.ToLower(value) + var builder strings.Builder + for _, character := range value { + if (character >= 'a' && character <= 'z') || (character >= '0' && character <= '9') || character == '.' || character == '_' || character == '-' { + builder.WriteRune(character) + } else { + builder.WriteByte('-') + } + } + result := strings.Trim(builder.String(), "-.") + if result == "" { + return "custom" + } + return result +} + +func (m *Coordinator) prepareDockerSandboxesCustomScripts(contextRoot string) error { + directory := filepath.Join(contextRoot, "custom-install") + if err := os.MkdirAll(directory, 0o755); err != nil { + return err + } + var runner strings.Builder + runner.WriteString("#!/usr/bin/env bash\nset -euo pipefail\n") + for index, configured := range m.Config.Image.CustomInstallScripts { + source, err := m.customInstallScriptHostPath(configured) + if err != nil { + return err + } + name := fmt.Sprintf("%03d-%s", index+1, guestScriptName(filepath.Base(source))) + if err := copyFile(source, filepath.Join(directory, name), 0o755); err != nil { + return err + } + fmt.Fprintf(&runner, "EPAR_CONTAINER_IMAGE_BUILD=true bash /opt/epar/custom-install/%s\n", name) + } + return writeAtomicFile(filepath.Join(directory, "run.sh"), []byte(runner.String()), 0o755) +} + +func (m *Coordinator) prepareDockerSandboxesTrustPolicy(contextRoot string, snapshot hosttrust.Snapshot) error { + hostDirectory := filepath.Join(contextRoot, "host-trust-certificates") + if err := copyHostTrustCertificatesToDir(hostDirectory, snapshot); err != nil { + return err + } + explicitDirectory := filepath.Join(contextRoot, "trusted-ca-certificates") + if err := m.copyTrustedCACertificatesToDir(explicitDirectory); err != nil { + return err + } + metadataDirectory := filepath.Join(contextRoot, "host-trust-metadata") + if err := os.MkdirAll(metadataDirectory, 0o755); err != nil { + return err + } + marker := struct { + SchemaVersion int `json:"schemaVersion"` + Generation string `json:"generation"` + HostOS string `json:"hostOS"` + Mode string `json:"mode"` + Scopes []string `json:"scopes"` + CertificateCount int `json:"certificateCount"` + }{ + SchemaVersion: 1, + Generation: "disabled", + Mode: hosttrust.ModeDisabled, + Scopes: []string{}, + } + if snapshot.Generation != "" { + marker.Generation = snapshot.Generation + marker.HostOS = snapshot.HostOS + marker.Mode = hosttrust.ModeOverlay + marker.Scopes = append([]string(nil), snapshot.Scopes...) + marker.CertificateCount = len(snapshot.Certificates) + } + content, err := json.MarshalIndent(marker, "", " ") + if err != nil { + return err + } + return writeAtomicFile(filepath.Join(metadataDirectory, filepath.Base(hostTrustMarkerGuest)), append(content, '\n'), 0o644) +} + +func copyHostTrustCertificatesToDir(destination string, snapshot hosttrust.Snapshot) error { + if err := os.MkdirAll(destination, 0o755); err != nil { + return err + } + for _, certificate := range snapshot.Certificates { + if err := writeAtomicFile(filepath.Join(destination, certificate.Name), certificate.PEM, 0o644); err != nil { + return err + } + } + return nil +} + +func boundedRedactedLogTail(path string, maximumBytes int64) string { + file, err := os.Open(path) + if err != nil { + return "" + } + defer file.Close() + info, err := file.Stat() + if err != nil { + return "" + } + start := info.Size() - maximumBytes + if start < 0 { + start = 0 + } + if _, err := file.Seek(start, io.SeekStart); err != nil { + return "" + } + content, err := io.ReadAll(io.LimitReader(file, maximumBytes)) + if err != nil { + return "" + } + text := strings.TrimSpace(provider.RedactText(string(content))) + if text == "" { + return "" + } + if start > 0 { + text = "[earlier Buildx output omitted]\n" + text + } + return "\nBuildx error tail:\n" + text +} + +func copyDirectory(source, destination string) error { + return filepath.WalkDir(source, func(path string, entry os.DirEntry, walkErr error) error { + if walkErr != nil { + return walkErr + } + relative, err := filepath.Rel(source, path) + if err != nil { + return err + } + target := filepath.Join(destination, relative) + if entry.IsDir() { + return os.MkdirAll(target, 0o755) + } + info, err := entry.Info() + if err != nil { + return err + } + if !info.Mode().IsRegular() { + return fmt.Errorf("refusing non-regular Docker Sandboxes template input %s", path) + } + return copyFile(path, target, info.Mode().Perm()) + }) +} + +func writeJSONFile(path string, value any) error { + content, err := json.MarshalIndent(value, "", " ") + if err != nil { + return err + } + return writeAtomicFile(path, append(content, '\n'), 0o600) +} + +func readJSONFile(path string, value any) error { + file, err := os.Open(path) + if err != nil { + return err + } + defer file.Close() + decoder := json.NewDecoder(io.LimitReader(file, 16<<20)) + if err := decoder.Decode(value); err != nil { + return err + } + var trailing any + if err := decoder.Decode(&trailing); err != io.EOF { + if err == nil { + return fmt.Errorf("unexpected trailing JSON value") + } + return err + } + return nil +} + +func hashFile(path string) (string, uint64, error) { + file, err := os.Open(path) + if err != nil { + return "", 0, err + } + defer file.Close() + hash := sha256.New() + size, err := io.Copy(hash, file) + if err != nil { + return "", 0, err + } + return "sha256:" + hex.EncodeToString(hash.Sum(nil)), uint64(size), nil +} + +func validSHA256(value string) bool { + if len(value) != len("sha256:")+64 || !strings.HasPrefix(value, "sha256:") { + return false + } + _, err := hex.DecodeString(strings.TrimPrefix(value, "sha256:")) + return err == nil && value == strings.ToLower(value) +} diff --git a/internal/image/docker_sandboxes_scope_test.go b/internal/image/docker_sandboxes_scope_test.go new file mode 100644 index 0000000..ab8573d --- /dev/null +++ b/internal/image/docker_sandboxes_scope_test.go @@ -0,0 +1,31 @@ +package image + +import ( + "path/filepath" + "testing" +) + +func TestDockerSandboxesArtifactWorkspaceIsConfigurationScoped(t *testing.T) { + project := t.TempDir() + first := &Coordinator{ProjectRoot: project, ConfigPath: filepath.Join(project, ".local", "config.yml")} + second := &Coordinator{ProjectRoot: project, ConfigPath: filepath.Join(project, ".local", "config.docker-sandboxes.yml")} + const manifestHash = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + firstPath, err := first.dockerSandboxesArtifactRoot(manifestHash) + if err != nil { + t.Fatal(err) + } + firstAgain, err := first.dockerSandboxesArtifactRoot(manifestHash) + if err != nil { + t.Fatal(err) + } + secondPath, err := second.dockerSandboxesArtifactRoot(manifestHash) + if err != nil { + t.Fatal(err) + } + if firstPath != firstAgain { + t.Fatalf("same configuration workspace is unstable: %q != %q", firstPath, firstAgain) + } + if firstPath == secondPath { + t.Fatalf("different configs share Docker Sandboxes workspace %q", firstPath) + } +} diff --git a/internal/image/docker_sandboxes_test.go b/internal/image/docker_sandboxes_test.go new file mode 100644 index 0000000..90847cc --- /dev/null +++ b/internal/image/docker_sandboxes_test.go @@ -0,0 +1,611 @@ +package image + +import ( + "crypto/sha256" + "encoding/json" + "fmt" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" +) + +func TestDockerSandboxesHelperChecksumsMatchGuestScripts(t *testing.T) { + templateRoot := filepath.Join("..", "..", "templates", "docker-sandboxes") + content, err := os.ReadFile(filepath.Join(templateRoot, "helpers.sha256")) + if err != nil { + t.Fatal(err) + } + for lineNumber, line := range strings.Split(strings.TrimSpace(string(content)), "\n") { + fields := strings.Fields(line) + if len(fields) != 2 || len(fields[0]) != 64 || !strings.HasPrefix(fields[1], "./") { + t.Fatalf("helpers.sha256 line %d is malformed: %q", lineNumber+1, line) + } + guestPath := filepath.Join(templateRoot, "guest", strings.TrimPrefix(fields[1], "./")) + guestContent, err := os.ReadFile(guestPath) + if err != nil { + t.Fatalf("read helper on line %d: %v", lineNumber+1, err) + } + if got := fmt.Sprintf("%x", sha256.Sum256(guestContent)); got != fields[0] { + t.Fatalf("helpers.sha256 line %d digest = %s, want %s for %s", lineNumber+1, fields[0], got, guestPath) + } + } +} + +func TestNormalizeCatthehackerSourceProfilesAndCustomTag(t *testing.T) { + for _, test := range []struct { + input string + want string + }{ + {"", "ghcr.io/catthehacker/ubuntu:full-latest"}, + {"full", "ghcr.io/catthehacker/ubuntu:full-latest"}, + {"act", "ghcr.io/catthehacker/ubuntu:act-latest"}, + {"dotnet", "ghcr.io/catthehacker/ubuntu:dotnet-latest"}, + {"js", "ghcr.io/catthehacker/ubuntu:js-latest"}, + {"go-24.04", "ghcr.io/catthehacker/ubuntu:go-24.04"}, + {"ghcr.io/catthehacker/ubuntu:go-24.04", "ghcr.io/catthehacker/ubuntu:go-24.04"}, + } { + t.Run(test.input, func(t *testing.T) { + got, err := NormalizeCatthehackerSource(test.input) + if err != nil { + t.Fatal(err) + } + if got != test.want { + t.Fatalf("NormalizeCatthehackerSource(%q) = %q, want %q", test.input, got, test.want) + } + }) + } +} + +func TestDockerSandboxesDockerfileUsesVerifiedLocalDownloadsAndInstallsTrustBeforeCustomization(t *testing.T) { + content, err := os.ReadFile(filepath.Join("..", "..", "templates", "docker-sandboxes", "Dockerfile")) + if err != nil { + t.Fatal(err) + } + text := string(content) + if strings.Contains(text, "ADD --") || strings.Contains(text, "ACTIONS_RUNNER_URL") || strings.Contains(text, "TINI_URL") { + t.Fatalf("Docker Sandboxes Dockerfile still delegates remote HTTPS downloads to BuildKit:\n%s", text) + } + for _, want := range []string{ + "COPY --chmod=0755 inputs/tini /usr/local/bin/tini", + "COPY inputs/actions-runner.tar.gz /tmp/actions-runner.tar.gz", + `echo "${TINI_SHA256#sha256:} /usr/local/bin/tini" | sha256sum --check -`, + `echo "${ACTIONS_RUNNER_SHA256#sha256:} /tmp/actions-runner.tar.gz" | sha256sum --check -`, + "RUNNER_TOOL_CACHE=/opt/actions-runner/_work/_tool", + "AGENT_TOOLSDIRECTORY=/opt/actions-runner/_work/_tool", + "DOTNET_INSTALL_DIR=/opt/actions-runner/_work/_tool/dotnet", + } { + if !strings.Contains(text, want) { + t.Fatalf("Docker Sandboxes Dockerfile omitted %q", want) + } + } + trustInstall := strings.Index(text, "/opt/epar/install-trusted-ca-certificates.sh") + customInstall := strings.Index(text, "/opt/epar/custom-install/run.sh") + if trustInstall < 0 || customInstall < 0 || trustInstall >= customInstall { + t.Fatalf("runner trust must be installed before custom scripts:\n%s", text) + } + runnerScript, err := os.ReadFile(filepath.Join("..", "..", "templates", "docker-sandboxes", "guest", "run-runner.sh")) + if err != nil { + t.Fatal(err) + } + for _, want := range []string{ + `tool_cache="${EPAR_RUNNER_TOOL_CACHE:-${runner_dir}/_work/_tool}"`, + `"RUNNER_TOOL_CACHE=${tool_cache}"`, + `"AGENT_TOOLSDIRECTORY=${tool_cache}"`, + `"DOTNET_INSTALL_DIR=${tool_cache}/dotnet"`, + } { + if !strings.Contains(string(runnerScript), want) { + t.Fatalf("Docker Sandboxes runner script omitted %q", want) + } + } +} + +func TestDockerSandboxesRunnerIdentityAndCredentialHygieneContract(t *testing.T) { + templateRoot := filepath.Join("..", "..", "templates", "docker-sandboxes") + readTemplateFile := func(t *testing.T, relativePath string) string { + t.Helper() + content, err := os.ReadFile(filepath.Join(templateRoot, relativePath)) + if err != nil { + t.Fatal(err) + } + return string(content) + } + + t.Run("environment normalization", func(t *testing.T) { + dockerfile := readTemplateFile(t, "Dockerfile") + for _, required := range []string{ + "HOME=/home/agent", + "USER=agent", + "LOGNAME=agent", + "SSH_AUTH_SOCK=", + "SSH_AUTH_SOCK_GATEWAY=", + "SSH_AGENT_PID=", + "XDG_CONFIG_HOME=/home/agent/.config", + "XDG_CACHE_HOME=/home/agent/.cache", + "XDG_DATA_HOME=/home/agent/.local/share", + "XDG_STATE_HOME=/home/agent/.local/state", + "XDG_RUNTIME_DIR=/run/user/1000", + "DOCKER_CONFIG=/home/agent/.docker", + } { + if !strings.Contains(dockerfile, required) { + t.Fatalf("Docker Sandboxes Dockerfile omitted normalized environment %q", required) + } + } + + runner := readTemplateFile(t, filepath.Join("guest", "run-runner.sh")) + for _, required := range []string{ + "unset SSH_AUTH_SOCK SSH_AUTH_SOCK_GATEWAY SSH_AGENT_PID", + "env -i", + `"HOME=${agent_home}"`, + `"USER=agent"`, + `"LOGNAME=agent"`, + `"XDG_CONFIG_HOME=${agent_home}/.config"`, + `"XDG_CACHE_HOME=${agent_home}/.cache"`, + `"XDG_DATA_HOME=${agent_home}/.local/share"`, + `"XDG_STATE_HOME=${agent_home}/.local/state"`, + `"XDG_RUNTIME_DIR=${agent_runtime_dir}"`, + `"DOCKER_CONFIG=${agent_home}/.docker"`, + } { + if !strings.Contains(runner, required) { + t.Fatalf("Docker Sandboxes listener environment omitted %q", required) + } + } + for _, forbidden := range []string{"http_proxy", "https_proxy", "no_proxy", "HTTP_PROXY", "HTTPS_PROXY", "NO_PROXY"} { + if strings.Contains(runner, forbidden) { + t.Fatalf("Docker Sandboxes listener still inherits host forward-proxy variables via %q", forbidden) + } + } + + configure := readTemplateFile(t, filepath.Join("guest", "configure-runner.sh")) + for _, forbidden := range []string{"http_proxy", "https_proxy", "no_proxy", "HTTP_PROXY", "HTTPS_PROXY", "NO_PROXY"} { + if strings.Contains(configure, forbidden) { + t.Fatalf("Docker Sandboxes registration still inherits host forward-proxy variables via %q", forbidden) + } + } + + entrypoint := readTemplateFile(t, filepath.Join("guest", "template-entrypoint.sh")) + for _, required := range []string{ + `-n "${SSH_AUTH_SOCK:-}"`, + `-n "${SSH_AUTH_SOCK_GATEWAY:-}"`, + "-e /run/ssh-agent.sock", + "host SSH-agent forwarding is not permitted", + "unset http_proxy https_proxy no_proxy HTTP_PROXY HTTPS_PROXY NO_PROXY", + `docker info --format '{{.NoProxy}}'`, + "policy-enforced transparent egress", + } { + if !strings.Contains(entrypoint, required) { + t.Fatalf("Docker Sandboxes entrypoint omitted SSH-agent isolation contract %q", required) + } + } + verify := readTemplateFile(t, filepath.Join("guest", "verify-template.sh")) + for _, required := range []string{`[[ -z "${SSH_AUTH_SOCK:-}" ]]`, `[[ -z "${SSH_AUTH_SOCK_GATEWAY:-}" ]]`, `[[ -z "${SSH_AGENT_PID:-}" ]]`, `[[ ! -e /run/ssh-agent.sock && ! -L /run/ssh-agent.sock ]]`} { + if !strings.Contains(verify, required) { + t.Fatalf("Docker Sandboxes template verification omitted SSH-agent isolation contract %q", required) + } + } + }) + + t.Run("private directory ownership", func(t *testing.T) { + prepare := readTemplateFile(t, filepath.Join("guest", "prepare-template.sh")) + for _, required := range []string{ + "install -d -m 0700 -o agent -g agent", + "/home/agent/.docker", + "/home/agent/.config", + "/home/agent/.cache", + "/home/agent/.local/share", + "/home/agent/.local/state", + "/run/user/1000", + } { + if !strings.Contains(prepare, required) { + t.Fatalf("Docker Sandboxes template preparation omitted private-directory contract %q", required) + } + } + + verify := readTemplateFile(t, filepath.Join("guest", "verify-template.sh")) + if !strings.Contains(verify, `stat -c '%U:%G:%a'`) || !strings.Contains(verify, `"agent:agent:700"`) { + t.Fatal("Docker Sandboxes template verification does not enforce restrictive agent directory ownership and mode") + } + entrypoint := readTemplateFile(t, filepath.Join("guest", "template-entrypoint.sh")) + if !strings.Contains(entrypoint, "sudo -n install -d -m 0700 -o agent -g agent /run/user/1000") { + t.Fatal("Docker Sandboxes entrypoint does not recreate the agent runtime directory after a boot-time /run reset") + } + configure := readTemplateFile(t, filepath.Join("guest", "configure-runner.sh")) + if !strings.Contains(configure, "install -d -m 0700 -o agent -g agent") || !strings.Contains(configure, "/run/user/1000") { + t.Fatal("Docker Sandboxes runner configuration does not ensure the agent runtime directory exists") + } + }) + + t.Run("source credential scrubbing", func(t *testing.T) { + prepare := readTemplateFile(t, filepath.Join("guest", "prepare-template.sh")) + for _, required := range []string{ + `passwd_entries="$(getent passwd)"`, + `[[ -z "${passwd_entries}" ]]`, + `done <<<"${passwd_entries}"`, + "for root_home_docker_config in /.docker /.dockercfg", + `rm -rf -- "${credential_home}/.docker"`, + `rm -f -- "${credential_home}/.dockercfg"`, + "failed to scrub source Docker client configuration", + } { + if !strings.Contains(prepare, required) { + t.Fatalf("Docker Sandboxes template preparation omitted credential-scrubbing contract %q", required) + } + } + for _, forbidden := range []string{"cat /root/.docker", "cat /home/runner/.docker", "cat /home/agent/.docker", "cp /root/.docker", "cp /home/runner/.docker"} { + if strings.Contains(prepare, forbidden) { + t.Fatalf("Docker Sandboxes credential hygiene exposes or copies source credential material via %q", forbidden) + } + } + verify := readTemplateFile(t, filepath.Join("guest", "verify-template.sh")) + for _, required := range []string{ + `passwd_entries="$(getent passwd)"`, + `done <<<"${passwd_entries}"`, + `sudo -n test ! -e "${normalized_home}/.dockercfg"`, + `sudo -n test ! -e "${normalized_home}/.docker"`, + } { + if !strings.Contains(verify, required) { + t.Fatalf("Docker Sandboxes template verification omitted foreign-home credential check %q", required) + } + } + }) + + t.Run("foreign runner config rejection", func(t *testing.T) { + for _, relativePath := range []string{ + filepath.Join("guest", "template-entrypoint.sh"), + filepath.Join("guest", "run-runner.sh"), + filepath.Join("guest", "verify-template.sh"), + } { + content := readTemplateFile(t, relativePath) + if !strings.Contains(content, "/home/runner/.docker") { + t.Fatalf("%s does not reject stale foreign Docker client configuration", relativePath) + } + } + }) +} + +func TestDockerSandboxesDockerDaemonUsesPolicyEnforcedTransparentRegistryPath(t *testing.T) { + templateRoot := filepath.Join("..", "..", "templates", "docker-sandboxes") + content, err := os.ReadFile(filepath.Join(templateRoot, "guest", "docker-daemon.json")) + if err != nil { + t.Fatal(err) + } + var configuration struct { + Proxies struct { + HTTPProxy string `json:"http-proxy"` + HTTPSProxy string `json:"https-proxy"` + NoProxy string `json:"no-proxy"` + } `json:"proxies"` + } + if err := json.Unmarshal(content, &configuration); err != nil { + t.Fatalf("parse Docker daemon proxy configuration: %v", err) + } + if configuration.Proxies.HTTPProxy != "http://gateway.docker.internal:3128" || configuration.Proxies.HTTPSProxy != "http://gateway.docker.internal:3128" || configuration.Proxies.NoProxy != "*" { + t.Fatalf("Docker daemon proxy configuration = %#v, want Sandbox forward proxy with daemon-wide transparent routing", configuration.Proxies) + } + prepareContent, err := os.ReadFile(filepath.Join(templateRoot, "guest", "prepare-template.sh")) + if err != nil { + t.Fatal(err) + } + prepare := string(prepareContent) + for _, required := range []string{ + "pinned source image unexpectedly supplies /etc/docker/daemon.json", + "install -m 0644 -o root -g root /opt/epar/docker-daemon.json /etc/docker/daemon.json", + "cmp -s /opt/epar/docker-daemon.json /etc/docker/daemon.json", + "rm -f /etc/sudoers.d/epar-proxy", + } { + if !strings.Contains(prepare, required) { + t.Fatalf("Docker Sandboxes template preparation omitted daemon proxy contract %q", required) + } + } + verifyContent, err := os.ReadFile(filepath.Join(templateRoot, "guest", "verify-template.sh")) + if err != nil { + t.Fatal(err) + } + verify := string(verifyContent) + for _, required := range []string{ + "test ! -L /etc/docker/daemon.json", + `stat -c '%U:%G:%a' /etc/docker/daemon.json`, + `"root:root:644"`, + `.proxies == {`, + `(keys - ["proxies", "registry-mirrors"])`, + `has("registry-mirrors")`, + `docker info --format '{{.NoProxy}}'`, + } { + if !strings.Contains(verify, required) { + t.Fatalf("Docker Sandboxes template verification omitted daemon proxy contract %q", required) + } + } +} + +func TestDockerSandboxesDisabledTrustPolicyIsExplicit(t *testing.T) { + root := t.TempDir() + coordinator := &Coordinator{ProjectRoot: root} + if err := coordinator.prepareDockerSandboxesTrustPolicy(root, hosttrust.Snapshot{}); err != nil { + t.Fatal(err) + } + content, err := os.ReadFile(filepath.Join(root, "host-trust-metadata", "host-trust-generation.json")) + if err != nil { + t.Fatal(err) + } + var marker struct { + SchemaVersion int `json:"schemaVersion"` + Generation string `json:"generation"` + HostOS string `json:"hostOS"` + Mode string `json:"mode"` + Scopes []string `json:"scopes"` + CertificateCount int `json:"certificateCount"` + } + if err := json.Unmarshal(content, &marker); err != nil { + t.Fatal(err) + } + if marker.SchemaVersion != 1 || marker.Generation != "disabled" || marker.HostOS != "" || marker.Mode != hosttrust.ModeDisabled || len(marker.Scopes) != 0 || marker.CertificateCount != 0 { + t.Fatalf("disabled policy marker = %+v", marker) + } +} + +func TestReadVerifiedBuildEvidenceRequiresStableBoundedRegularFile(t *testing.T) { + root := t.TempDir() + path := filepath.Join(root, "evidence.json") + if err := os.WriteFile(path, []byte(`{"ok":true}`), 0o600); err != nil { + t.Fatal(err) + } + content, err := readVerifiedBuildEvidence(path, 64) + if err != nil { + t.Fatal(err) + } + if string(content) != `{"ok":true}` { + t.Fatalf("evidence = %q", content) + } + if _, err := readVerifiedBuildEvidence(path, 4); err == nil { + t.Fatal("oversized evidence was accepted") + } + link := filepath.Join(root, "evidence-link.json") + if err := os.Symlink(path, link); err == nil { + if _, err := readVerifiedBuildEvidence(link, 64); err == nil { + t.Fatal("symlinked evidence was accepted") + } + } +} + +func TestProvenanceValidatorsRequireMaxBuildxAndInTotoSLSAContracts(t *testing.T) { + buildx := []byte(`{"buildType":"https://mobyproject.org/buildkit@v1","materials":[{"uri":"pkg:docker/example"}],"invocation":{"parameters":{"frontend":"gateway.v0"}}}`) + if err := validateBuildxMaxProvenance(buildx); err != nil { + t.Fatal(err) + } + if err := validateBuildxMaxProvenance([]byte(`{"buildType":"buildkit","materials":[],"invocation":{}}`)); err == nil { + t.Fatal("incomplete Buildx provenance was accepted") + } + statement := []byte(`{"_type":"https://in-toto.io/Statement/v1","predicateType":"https://slsa.dev/provenance/v1","subject":[{"name":"software-inventory.txt"}],"predicate":{"buildDefinition":{}}}`) + if err := validateInTotoProvenance(statement); err != nil { + t.Fatal(err) + } + if err := validateInTotoProvenance([]byte(`{"_type":"https://in-toto.io/Statement/v1","predicateType":"unknown","subject":[],"predicate":{}}`)); err == nil { + t.Fatal("invalid in-toto provenance was accepted") + } +} + +func TestValidateInTotoSPDXStreamsAndRejectsMalformedPolicy(t *testing.T) { + path := filepath.Join(t.TempDir(), "sbom.intoto.json") + valid := `{"_type":"https://in-toto.io/Statement/v1","subject":[],"predicateType":"https://spdx.dev/Document","predicate":{"SPDXID":"SPDXRef-DOCUMENT","packages":[{"name":"example"}]}}` + if err := os.WriteFile(path, []byte(valid), 0o600); err != nil { + t.Fatal(err) + } + if err := validateInTotoSPDX(path); err != nil { + t.Fatal(err) + } + invalid := strings.Replace(valid, "https://spdx.dev/Document", "https://example.invalid/Unknown", 1) + if err := os.WriteFile(path, []byte(invalid), 0o600); err != nil { + t.Fatal(err) + } + if err := validateInTotoSPDX(path); err == nil { + t.Fatal("unknown predicate type was accepted") + } +} + +func TestNormalizeCatthehackerSourceRejectsOtherRepositoriesAndInvalidTags(t *testing.T) { + for _, input := range []string{ + "ubuntu:latest", + "ghcr.io/other/ubuntu:full-latest", + "ghcr.io/catthehacker/ubuntu@sha256:" + strings.Repeat("a", 64), + "tag with spaces", + "-leading-dash", + } { + if _, err := NormalizeCatthehackerSource(input); err == nil { + t.Fatalf("NormalizeCatthehackerSource(%q) succeeded", input) + } + } +} + +func TestParseResolvedDockerSourceSelectsExactNativeManifestAndSize(t *testing.T) { + index := dockerManifestDocument{MediaType: "application/vnd.oci.image.index.v1+json"} + amd64 := dockerManifestDescriptor{Digest: "sha256:" + strings.Repeat("a", 64)} + amd64.Platform.OS = "linux" + amd64.Platform.Architecture = "amd64" + arm64 := dockerManifestDescriptor{Digest: "sha256:" + strings.Repeat("b", 64)} + arm64.Platform.OS = "linux" + arm64.Platform.Architecture = "arm64" + index.Manifests = []dockerManifestDescriptor{amd64, arm64} + indexRaw, err := json.Marshal(index) + if err != nil { + t.Fatal(err) + } + manifestRaw := []byte(`{"layers":[{"size":100},{"size":23}]}`) + resolved, err := parseResolvedDockerSource("ghcr.io/catthehacker/ubuntu:full-latest", "linux/arm64", indexRaw, manifestRaw) + if err != nil { + t.Fatal(err) + } + if resolved.PlatformDigest != arm64.Digest || resolved.CompressedLayerBytes != 123 || !strings.HasPrefix(resolved.ImmutableReference, "ghcr.io/catthehacker/ubuntu@sha256:") { + t.Fatalf("resolved source = %+v", resolved) + } + withCommandNewline, err := parseResolvedDockerSource("ghcr.io/catthehacker/ubuntu:full-latest", "linux/arm64", append(append([]byte(nil), indexRaw...), '\n'), append(append([]byte(nil), manifestRaw...), '\n')) + if err != nil { + t.Fatal(err) + } + if withCommandNewline.IndexDigest != resolved.IndexDigest { + t.Fatalf("index digest changed because the CLI appended a newline: %s != %s", withCommandNewline.IndexDigest, resolved.IndexDigest) + } +} + +func TestDockerSandboxesArtifactIdentityChangesWithEveryFreshnessInput(t *testing.T) { + base := Manifest{ + SchemaVersion: ManifestSchemaVersion, + ProviderType: "docker-sandboxes", + ProviderPlatform: "linux/arm64", + SourceType: "docker-image", + SourceImage: "ghcr.io/catthehacker/ubuntu:full-latest", + SourceDigest: "sha256:" + strings.Repeat("a", 64), + SourcePlatformDigest: "sha256:" + strings.Repeat("b", 64), + RunnerVersion: "2.332.0", + TemplateInputs: []FileDigest{{Path: "Dockerfile", SHA256: strings.Repeat("c", 64)}}, + CustomInstallScripts: []FileDigest{{Path: "custom.sh", SHA256: strings.Repeat("d", 64)}}, + } + baseHash, err := ManifestHash(base) + if err != nil { + t.Fatal(err) + } + mutations := []func(*Manifest){ + func(value *Manifest) { value.SourceDigest = "sha256:" + strings.Repeat("e", 64) }, + func(value *Manifest) { value.SourcePlatformDigest = "sha256:" + strings.Repeat("f", 64) }, + func(value *Manifest) { value.RunnerVersion = "2.333.0" }, + func(value *Manifest) { value.TemplateInputs[0].SHA256 = strings.Repeat("1", 64) }, + func(value *Manifest) { value.CustomInstallScripts[0].SHA256 = strings.Repeat("2", 64) }, + func(value *Manifest) { + value.HostTrust = &HostTrustMetadata{Generation: "sha256:" + strings.Repeat("3", 64)} + }, + } + for index, mutate := range mutations { + changed := base + changed.TemplateInputs = append([]FileDigest(nil), base.TemplateInputs...) + changed.CustomInstallScripts = append([]FileDigest(nil), base.CustomInstallScripts...) + mutate(&changed) + hash, err := ManifestHash(changed) + if err != nil { + t.Fatal(err) + } + if hash == baseHash { + t.Fatalf("freshness mutation %d did not change artifact identity", index) + } + } +} + +func TestVerifiedDockerSandboxesBuildArtifactAcceptsOnlyCompleteExactEvidence(t *testing.T) { + root := t.TempDir() + manifestHash := strings.Repeat("a", 64) + source := ResolvedDockerSource{ + Reference: "ghcr.io/catthehacker/ubuntu:full-latest", + ImmutableReference: "ghcr.io/catthehacker/ubuntu@sha256:" + strings.Repeat("b", 64), + IndexDigest: "sha256:" + strings.Repeat("b", 64), + PlatformDigest: "sha256:" + strings.Repeat("c", 64), + Platform: "linux/amd64", + CompressedLayerBytes: 123, + } + fixturePath, templateDigest, _ := writeDockerArchiveFixture(t, false, false) + archivePath := filepath.Join(root, "runner-template.tar") + fixtureContent, err := os.ReadFile(fixturePath) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(archivePath, fixtureContent, 0o600); err != nil { + t.Fatal(err) + } + archiveSHA, archiveBytes, err := hashFile(archivePath) + if err != nil { + t.Fatal(err) + } + metadata := dockerSandboxesTemplateMetadata{ + SchemaVersion: dockerSandboxesMetadataSchema, + Profile: "full-latest", + Platform: source.Platform, + ManifestHash: manifestHash, + Source: source, + Artifacts: make(map[string]artifactEvidence), + } + metadata.Template.Tag = "docker.io/library/epar-template:test-amd64" + metadata.Template.Digest = templateDigest + metadata.Template.CacheID = strings.TrimPrefix(templateDigest, "sha256:")[:12] + metadata.Template.RootDisk = "90GiB" + metadata.Template.Archive = filepath.Base(archivePath) + metadata.Template.ArchiveSHA256 = archiveSHA + metadata.Template.ArchiveBytes = archiveBytes + metadata.Compatibility.TemplateSchemaVersion = 1 + metadata.Compatibility.RunnerExecution = "direct-actions-listener" + metadata.Compatibility.DockerDaemonOwner = "docker-sandboxes-runtime" + metadata.Compatibility.ExpectedDockerDaemonCount = 1 + if err := writeJSONFile(filepath.Join(root, "buildMetadata.json"), dockerSandboxesBuildMetadata{ + ImageDigest: templateDigest, + Provenance: json.RawMessage(`{}`), + BuildRef: strings.Repeat("b", 12), + }); err != nil { + t.Fatal(err) + } + for _, name := range []string{"buildMetadata", "attestationMetadata", "provenance", "sbom", "softwareInventory", "compatibility"} { + path := filepath.Join(root, name+".json") + if name != "buildMetadata" { + if err := os.WriteFile(path, []byte(name), 0o600); err != nil { + t.Fatal(err) + } + } + digest, _, err := hashFile(path) + if err != nil { + t.Fatal(err) + } + metadata.Artifacts[name] = artifactEvidence{Path: filepath.Base(path), SHA256: digest} + } + metadataPath := filepath.Join(root, "template-metadata.json") + if err := writeJSONFile(metadataPath, metadata); err != nil { + t.Fatal(err) + } + _, artifact, _, _, valid, err := verifiedDockerSandboxesBuildArtifact(root, metadataPath, archivePath, manifestHash, source) + if err != nil { + t.Fatal(err) + } + if !valid || artifact.Digest != templateDigest || artifact.Platform != "linux/amd64" || artifact.RootDisk != "90GiB" { + t.Fatalf("verified artifact = %+v, valid=%t", artifact, valid) + } + + if err := os.WriteFile(filepath.Join(root, "sbom.json"), []byte("changed"), 0o600); err != nil { + t.Fatal(err) + } + _, _, _, _, valid, err = verifiedDockerSandboxesBuildArtifact(root, metadataPath, archivePath, manifestHash, source) + if err != nil { + t.Fatal(err) + } + if valid { + t.Fatal("corrupted evidence was accepted for interrupted-build resume") + } +} + +func TestDockerSandboxesBuildUsesDirectArchiveAndInventoryTargets(t *testing.T) { + sourcePath := filepath.Join("docker_sandboxes.go") + content, err := os.ReadFile(sourcePath) + if err != nil { + t.Fatal(err) + } + text := string(content) + for _, required := range []string{ + `"--target", "runner-template", "--output", "type=docker,dest=" + partialArchivePath`, + `"--provenance=false", "--sbom=false"`, + `"--target", "software-inventory-export", "--output", "type=local,dest=" + evidenceExportRoot`, + `"--provenance", "mode=max", "--sbom", "generator=" + platformLock.SBOMGeneratorReference`, + `"-attestation.docker-build.log"`, + } { + if !strings.Contains(text, required) { + t.Fatalf("Docker Sandboxes build path omitted %q", required) + } + } + for _, forbidden := range []string{`"--load"`, `"image", "save"`, `"image", "load"`, `"image", "inspect"`, `"type=image,push=false"`} { + if strings.Contains(text, forbidden) { + t.Fatalf("Docker Sandboxes build path retained forbidden Docker staging operation %q", forbidden) + } + } + dockerfile, err := os.ReadFile(filepath.Join("..", "..", "templates", "docker-sandboxes", "Dockerfile")) + if err != nil { + t.Fatal(err) + } + for _, required := range []string{"AS runner-template", "ARG BUILDKIT_SBOM_SCAN_STAGE=true", "AS software-inventory-export"} { + if !strings.Contains(string(dockerfile), required) { + t.Fatalf("Dockerfile omitted %q", required) + } + } +} diff --git a/internal/image/manifest.go b/internal/image/manifest.go new file mode 100644 index 0000000..25eaee5 --- /dev/null +++ b/internal/image/manifest.go @@ -0,0 +1,154 @@ +package image + +import ( + "crypto/sha256" + "encoding/hex" + "encoding/json" + "os" + "path/filepath" + "strings" +) + +const ( + ManifestSchemaVersion = 2 + ManifestGuestPath = "/opt/epar/image-manifest.json" + ManifestLabel = "org.solutionforest.epar.manifest-sha256" +) + +type HostTrustMetadata struct { + Mode string `json:"mode"` + HostOS string `json:"hostOS"` + Scopes []string `json:"scopes"` + Generation string `json:"generation"` + CertificateCount int `json:"certificateCount"` +} + +type FileDigest struct { + Path string `json:"path"` + SHA256 string `json:"sha256"` +} + +type Manifest struct { + SchemaVersion int `json:"schemaVersion"` + ProviderType string `json:"providerType"` + ProviderPlatform string `json:"providerPlatform,omitempty"` + ProviderRosettaTag string `json:"providerRosettaTag,omitempty"` + SourceType string `json:"sourceType,omitempty"` + SourceImage string `json:"sourceImage"` + SourcePlatform string `json:"sourcePlatform,omitempty"` + SourceDigest string `json:"sourceDigest,omitempty"` + SourcePlatformDigest string `json:"sourcePlatformDigest,omitempty"` + OutputImage string `json:"outputImage"` + RunnerSelector string `json:"runnerSelector"` + RunnerVersion string `json:"runnerVersion"` + RunnerAssetName string `json:"runnerAssetName,omitempty"` + RunnerAssetURL string `json:"runnerAssetUrl,omitempty"` + RunnerAssetDigest string `json:"runnerAssetDigest,omitempty"` + UpstreamCommit string `json:"upstreamCommit,omitempty"` + EPARScripts []FileDigest `json:"eparScripts,omitempty"` + TemplateInputs []FileDigest `json:"templateInputs,omitempty"` + CustomInstallScripts []FileDigest `json:"customInstallScripts,omitempty"` + TrustedCACertificates []FileDigest `json:"trustedCaCertificates,omitempty"` + HostTrust *HostTrustMetadata `json:"hostTrust,omitempty"` +} + +type StoredManifest struct { + Hash string `json:"hash"` + Manifest Manifest `json:"manifest"` +} + +type SourceCacheManifest struct { + SourceImage string `json:"sourceImage"` + SourcePlatform string `json:"sourcePlatform,omitempty"` + SourceDigest string `json:"sourceDigest,omitempty"` +} + +func ManifestHash(manifest Manifest) (string, error) { + content, err := json.Marshal(manifest) + if err != nil { + return "", err + } + sum := sha256.Sum256(content) + return hex.EncodeToString(sum[:]), nil +} + +func StoredManifestContent(manifest Manifest) (string, string, error) { + hash, err := ManifestHash(manifest) + if err != nil { + return "", "", err + } + content, err := json.MarshalIndent(StoredManifest{Hash: hash, Manifest: manifest}, "", " ") + if err != nil { + return "", "", err + } + return string(content) + "\n", hash, nil +} + +func ReadStoredManifest(path string) (StoredManifest, error) { + content, err := os.ReadFile(path) + if err != nil { + return StoredManifest{}, err + } + var stored StoredManifest + if err := json.Unmarshal(content, &stored); err != nil { + return StoredManifest{}, err + } + return stored, nil +} + +func WriteStoredManifest(path string, manifest Manifest) error { + content, _, err := StoredManifestContent(manifest) + if err != nil { + return err + } + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + return err + } + return os.WriteFile(path, []byte(content), 0o644) +} + +func SourceCacheManifestPath(rootfsPath string) string { + return rootfsPath + ".source.json" +} + +func WSLSourceRootfsPath(outputPath string) string { + switch { + case strings.HasSuffix(outputPath, ".tar.gz"): + return strings.TrimSuffix(outputPath, ".tar.gz") + ".source.rootfs.tar" + case strings.HasSuffix(outputPath, ".tgz"): + return strings.TrimSuffix(outputPath, ".tgz") + ".source.rootfs.tar" + case strings.HasSuffix(outputPath, ".tar"): + return strings.TrimSuffix(outputPath, ".tar") + ".source.rootfs.tar" + default: + return outputPath + ".source.rootfs.tar" + } +} + +func WSLImageManifestPath(outputPath string) string { + return outputPath + ".epar-manifest.json" +} + +func SourceCacheMatches(path string, want SourceCacheManifest) bool { + content, err := os.ReadFile(path) + if err != nil { + return false + } + var got SourceCacheManifest + if err := json.Unmarshal(content, &got); err != nil { + return false + } + return got == want +} + +func WriteSourceCacheManifest(path string, manifest SourceCacheManifest) error { + content, err := json.MarshalIndent(manifest, "", " ") + if err != nil { + return err + } + return os.WriteFile(path, append(content, '\n'), 0o644) +} + +func DockerInspectMeansMissing(err error) bool { + text := strings.ToLower(err.Error()) + return strings.Contains(text, "no such image") || strings.Contains(text, "no such object") || strings.Contains(text, "not found") +} diff --git a/internal/image/manifest_paths_test.go b/internal/image/manifest_paths_test.go new file mode 100644 index 0000000..3112ae8 --- /dev/null +++ b/internal/image/manifest_paths_test.go @@ -0,0 +1,20 @@ +package image + +import "testing" + +func TestWSLArtifactPaths(t *testing.T) { + tests := map[string]string{ + "runner.tar": "runner.source.rootfs.tar", + "runner.tar.gz": "runner.source.rootfs.tar", + "runner.tgz": "runner.source.rootfs.tar", + "runner": "runner.source.rootfs.tar", + } + for output, want := range tests { + if got := WSLSourceRootfsPath(output); got != want { + t.Errorf("WSLSourceRootfsPath(%q) = %q, want %q", output, got, want) + } + } + if got, want := WSLImageManifestPath("runner.tar"), "runner.tar.epar-manifest.json"; got != want { + t.Fatalf("WSLImageManifestPath() = %q, want %q", got, want) + } +} diff --git a/internal/image/oci_resolver.go b/internal/image/oci_resolver.go new file mode 100644 index 0000000..2ffa97e --- /dev/null +++ b/internal/image/oci_resolver.go @@ -0,0 +1,42 @@ +package image + +import ( + "context" + "fmt" + "strings" + + "github.com/google/go-containerregistry/pkg/authn" + "github.com/google/go-containerregistry/pkg/name" + "github.com/google/go-containerregistry/pkg/v1/remote" +) + +// resolveTartOCIReference observes only the registry descriptor. Tart remains +// responsible for pulling its VM artifact; EPAR records and clones the exact +// immutable OCI identity selected by the schedule. +func (m *Coordinator) resolveTartOCIReference(ctx context.Context, reference string) (string, error) { + ref, err := name.ParseReference(strings.TrimSpace(reference)) + if err != nil { + return "", fmt.Errorf("parse Tart OCI source %q: %w", reference, err) + } + authenticator, err := authn.DefaultKeychain.Resolve(ref.Context().Registry) + if err != nil { + return "", fmt.Errorf("resolve Tart OCI registry credentials: %w", err) + } + buildTrust, err := m.resolveBuildTrust(ctx) + if err != nil { + return "", err + } + client, err := buildTrustHTTPClient(buildTrust) + if err != nil { + return "", err + } + descriptor, err := remote.Get(ref, remote.WithContext(ctx), remote.WithAuth(authenticator), remote.WithTransport(client.Transport)) + if err != nil { + return "", fmt.Errorf("resolve Tart OCI source %s: %w", reference, err) + } + digest := descriptor.Digest.String() + if !validSHA256(digest) { + return "", fmt.Errorf("Tart OCI source %s returned an invalid immutable digest", reference) + } + return ref.Context().Name() + "@" + digest, nil +} diff --git a/internal/image/operation_progress.go b/internal/image/operation_progress.go new file mode 100644 index 0000000..817ab2d --- /dev/null +++ b/internal/image/operation_progress.go @@ -0,0 +1,70 @@ +package image + +import ( + "fmt" + "os" + "sync" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +var imageOperationHeartbeatInterval = 5 * time.Second + +// runProgressOperation keeps long, otherwise-silent image and storage work +// visible without mixing command output into the manager console. +func (m *Coordinator) runProgressOperation(label string, detail func() string, operation func() error) error { + return runProgressOperation(label, imageOperationHeartbeatInterval, m.infof, detail, operation) +} + +func runProgressOperation(label string, interval time.Duration, logf func(string, ...any), detail func() string, operation func() error) (err error) { + started := time.Now() + logf("%s started\n", label) + + done := make(chan struct{}) + var heartbeat sync.WaitGroup + if interval > 0 { + heartbeat.Add(1) + go func() { + defer heartbeat.Done() + ticker := time.NewTicker(interval) + defer ticker.Stop() + for { + select { + case <-ticker.C: + suffix := "" + if detail != nil { + if current := detail(); current != "" { + suffix = "; " + current + } + } + logf("%s: still working; elapsed %s%s\n", label, time.Since(started).Round(time.Second), suffix) + case <-done: + return + } + } + }() + } + + finished := false + defer func() { + close(done) + heartbeat.Wait() + if finished && err == nil { + logf("%s complete; elapsed %s\n", label, time.Since(started).Round(time.Second)) + } + }() + err = operation() + finished = true + return err +} + +func regularFileSizeDetail(path, description string) func() string { + return func() string { + info, err := os.Lstat(path) + if err != nil || !info.Mode().IsRegular() || info.Size() < 0 { + return "" + } + return fmt.Sprintf("%s %s", description, storage.FormatBytes(uint64(info.Size()))) + } +} diff --git a/internal/image/operation_progress_test.go b/internal/image/operation_progress_test.go new file mode 100644 index 0000000..9ff423d --- /dev/null +++ b/internal/image/operation_progress_test.go @@ -0,0 +1,89 @@ +package image + +import ( + "errors" + "fmt" + "os" + "path/filepath" + "strings" + "sync" + "testing" + "time" +) + +func TestRunProgressOperationReportsHeartbeatAndCompletion(t *testing.T) { + var mu sync.Mutex + var messages []string + heartbeatObserved := make(chan struct{}) + var heartbeatOnce sync.Once + logf := func(format string, args ...any) { + message := fmt.Sprintf(format, args...) + mu.Lock() + messages = append(messages, message) + mu.Unlock() + if strings.Contains(message, "still working") { + heartbeatOnce.Do(func() { close(heartbeatObserved) }) + } + } + + if err := runProgressOperation("Template archive export", 5*time.Millisecond, logf, func() string { + return "archive written 12.00 GiB" + }, func() error { + <-heartbeatObserved + return nil + }); err != nil { + t.Fatal(err) + } + + mu.Lock() + output := strings.Join(messages, "") + mu.Unlock() + for _, wanted := range []string{ + "Template archive export started", + "Template archive export: still working; elapsed", + "archive written 12.00 GiB", + "Template archive export complete; elapsed", + } { + if !strings.Contains(output, wanted) { + t.Fatalf("progress output omitted %q:\n%s", wanted, output) + } + } +} + +func TestRunProgressOperationDoesNotReportFailedOperationAsComplete(t *testing.T) { + var mu sync.Mutex + var output strings.Builder + logf := func(format string, args ...any) { + mu.Lock() + defer mu.Unlock() + fmt.Fprintf(&output, format, args...) + } + expected := errors.New("failed") + err := runProgressOperation("Template import", 0, logf, nil, func() error { + return expected + }) + if !errors.Is(err, expected) { + t.Fatalf("error = %v, want %v", err, expected) + } + mu.Lock() + defer mu.Unlock() + if strings.Contains(output.String(), "Template import complete") { + t.Fatalf("failed operation was reported complete:\n%s", output.String()) + } +} + +func TestImageOperationHeartbeatDefaultIsFiveSeconds(t *testing.T) { + if got, want := imageOperationHeartbeatInterval, 5*time.Second; got != want { + t.Fatalf("image-operation progress heartbeat = %s, want %s", got, want) + } +} + +func TestRegularFileSizeDetailUsesHumanReadableSize(t *testing.T) { + path := filepath.Join(t.TempDir(), "archive.tar") + if err := os.WriteFile(path, make([]byte, 1536), 0o600); err != nil { + t.Fatal(err) + } + if got, want := regularFileSizeDetail(path, "archive written")(), "archive written 1.50 KiB"; got != want { + t.Fatalf("detail = %q, want %q", got, want) + } +} diff --git a/internal/image/runner_release.go b/internal/image/runner_release.go new file mode 100644 index 0000000..a42c9ac --- /dev/null +++ b/internal/image/runner_release.go @@ -0,0 +1,179 @@ +package image + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "os" + "path/filepath" + "runtime" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +const defaultActionsRunnerReleaseAPI = "https://api.github.com/repos/actions/runner/releases" + +type actionsRunnerRelease struct { + TagName string `json:"tag_name"` + Assets []actionsRunnerAsset `json:"assets"` +} + +type actionsRunnerAsset struct { + Name string `json:"name"` + BrowserDownloadURL string `json:"browser_download_url"` + Digest string `json:"digest"` +} + +func normalizedRunnerSelector(value string) string { + value = strings.TrimSpace(strings.TrimPrefix(value, "v")) + if value == "" { + return "latest" + } + return value +} + +func (m *Coordinator) resolveActionsRunner(ctx context.Context, manifest Manifest) (Manifest, error) { + selector := normalizedRunnerSelector(manifest.RunnerSelector) + platform, err := actionsRunnerPlatform(manifest, m.Config.Provider.Type) + if err != nil { + return manifest, err + } + architecture := "" + switch platform { + case "linux/amd64": + architecture = "x64" + case "linux/arm64": + architecture = "arm64" + default: + return manifest, fmt.Errorf("GitHub Actions runner packages are not configured for %s", platform) + } + endpoint := actionsRunnerReleaseAPI() + if selector == "latest" { + endpoint += "/latest" + } else { + endpoint += "/tags/v" + selector + } + buildTrust, err := m.resolveBuildTrust(ctx) + if err != nil { + return manifest, err + } + client, err := buildTrustHTTPClient(buildTrust) + if err != nil { + return manifest, err + } + request, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) + if err != nil { + return manifest, err + } + request.Header.Set("Accept", "application/vnd.github+json") + request.Header.Set("X-GitHub-Api-Version", "2022-11-28") + request.Header.Set("User-Agent", "ephemeral-action-runner") + response, err := client.Do(request) + if err != nil { + return manifest, fmt.Errorf("resolve GitHub Actions runner %s: %w", selector, err) + } + defer response.Body.Close() + if response.StatusCode != http.StatusOK { + return manifest, fmt.Errorf("resolve GitHub Actions runner %s: GitHub returned HTTP %d", selector, response.StatusCode) + } + var release actionsRunnerRelease + if err := json.NewDecoder(response.Body).Decode(&release); err != nil { + return manifest, fmt.Errorf("parse GitHub Actions runner release: %w", err) + } + return applyActionsRunnerRelease(manifest, selector, architecture, release) +} + +func applyActionsRunnerRelease(manifest Manifest, selector, architecture string, release actionsRunnerRelease) (Manifest, error) { + version := strings.TrimPrefix(strings.TrimSpace(release.TagName), "v") + if version == "" { + return manifest, fmt.Errorf("GitHub Actions runner release omitted tag_name") + } + if selector != "latest" && selector != version { + return manifest, fmt.Errorf("GitHub Actions runner release returned version %q for selector %q", version, selector) + } + assetName := fmt.Sprintf("actions-runner-linux-%s-%s.tar.gz", architecture, version) + var selected *actionsRunnerAsset + for index := range release.Assets { + if release.Assets[index].Name == assetName { + selected = &release.Assets[index] + break + } + } + if selected == nil { + return manifest, fmt.Errorf("GitHub Actions runner %s does not provide %s", version, assetName) + } + digest := strings.ToLower(strings.TrimSpace(selected.Digest)) + if !validSHA256(digest) { + return manifest, fmt.Errorf("GitHub Actions runner asset %s omitted a valid SHA-256 digest", assetName) + } + if strings.TrimSpace(selected.BrowserDownloadURL) == "" { + return manifest, fmt.Errorf("GitHub Actions runner asset %s omitted its download URL", assetName) + } + manifest.RunnerSelector = selector + manifest.RunnerVersion = version + manifest.RunnerAssetName = assetName + manifest.RunnerAssetURL = selected.BrowserDownloadURL + manifest.RunnerAssetDigest = digest + return manifest, nil +} + +func actionsRunnerPlatform(manifest Manifest, providerType string) (string, error) { + for _, candidate := range []string{manifest.SourcePlatform, manifest.ProviderPlatform} { + if platform, ok := NormalizedDockerPlatform(candidate, "linux"); ok { + return platform.OS + "/" + platform.Architecture, nil + } + } + if providerType == "tart" { + return "linux/arm64", nil + } + architecture := runtime.GOARCH + if architecture == "386" { + architecture = "amd64" + } + if architecture != "amd64" && architecture != "arm64" { + return "", fmt.Errorf("cannot select a GitHub Actions runner package for host architecture %s", runtime.GOARCH) + } + return "linux/" + architecture, nil +} + +func actionsRunnerReleaseAPI() string { + if value := strings.TrimSpace(os.Getenv("EPAR_TEST_ACTIONS_RUNNER_RELEASE_API")); value != "" { + return strings.TrimRight(value, "/") + } + return defaultActionsRunnerReleaseAPI +} + +func (m *Coordinator) acquireActionsRunner(ctx context.Context, manifest Manifest) (string, error) { + if manifest.RunnerVersion == "" || manifest.RunnerAssetURL == "" || !validSHA256(manifest.RunnerAssetDigest) { + return "", fmt.Errorf("resolved Actions runner identity is incomplete") + } + digest := strings.TrimPrefix(manifest.RunnerAssetDigest, "sha256:") + cachePath := filepath.Join(m.ProjectRoot, ".local", "state", "image", "downloads", "actions-runner", digest, manifest.RunnerAssetName) + buildTrust, err := m.resolveBuildTrust(ctx) + if err != nil { + return "", err + } + client, err := buildTrustHTTPClient(buildTrust) + if err != nil { + return "", err + } + if err := verifiedDownload(ctx, client, manifest.RunnerAssetURL, cachePath, manifest.RunnerAssetDigest, 0o600); err != nil { + return "", fmt.Errorf("acquire GitHub Actions runner %s: %w", manifest.RunnerVersion, err) + } + return cachePath, nil +} + +func (m *Coordinator) installActionsRunnerPackage(ctx context.Context, instance string, manifest Manifest) error { + path, err := m.acquireActionsRunner(ctx, manifest) + if err != nil { + return err + } + const guestPath = "/opt/epar/actions-runner.tar.gz" + if err := provider.CopyFile(ctx, m.Provider, instance, path, guestPath, "0600"); err != nil { + return fmt.Errorf("copy verified Actions runner package into build guest: %w", err) + } + _, err = m.execBuildGuest(ctx, instance, []string{"sudo", "bash", "/opt/epar/install-runner.sh", manifest.RunnerVersion, guestPath, manifest.RunnerAssetDigest}, provider.ExecOptions{}) + return err +} diff --git a/internal/image/runner_release_test.go b/internal/image/runner_release_test.go new file mode 100644 index 0000000..2336f04 --- /dev/null +++ b/internal/image/runner_release_test.go @@ -0,0 +1,81 @@ +package image + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +func TestApplyActionsRunnerReleaseSelectsExactArchitectureAndDigest(t *testing.T) { + digest := "sha256:" + strings.Repeat("a", 64) + release := actionsRunnerRelease{ + TagName: "v2.332.0", + Assets: []actionsRunnerAsset{ + {Name: "actions-runner-linux-x64-2.332.0.tar.gz", BrowserDownloadURL: "https://example.invalid/x64", Digest: digest}, + {Name: "actions-runner-linux-arm64-2.332.0.tar.gz", BrowserDownloadURL: "https://example.invalid/arm64", Digest: "sha256:" + strings.Repeat("b", 64)}, + }, + } + manifest, err := applyActionsRunnerRelease(Manifest{}, "latest", "x64", release) + if err != nil { + t.Fatal(err) + } + if manifest.RunnerSelector != "latest" || manifest.RunnerVersion != "2.332.0" { + t.Fatalf("runner identity = selector %q version %q", manifest.RunnerSelector, manifest.RunnerVersion) + } + if manifest.RunnerAssetName != "actions-runner-linux-x64-2.332.0.tar.gz" || manifest.RunnerAssetURL != "https://example.invalid/x64" || manifest.RunnerAssetDigest != digest { + t.Fatalf("selected asset = %+v", manifest) + } +} + +func TestApplyActionsRunnerReleaseRejectsWrongVersionAndMissingIntegrity(t *testing.T) { + release := actionsRunnerRelease{ + TagName: "v2.332.0", + Assets: []actionsRunnerAsset{{ + Name: "actions-runner-linux-x64-2.332.0.tar.gz", + BrowserDownloadURL: "https://example.invalid/x64", + }}, + } + if _, err := applyActionsRunnerRelease(Manifest{}, "2.331.0", "x64", release); err == nil || !strings.Contains(err.Error(), "returned version") { + t.Fatalf("wrong-version error = %v", err) + } + if _, err := applyActionsRunnerRelease(Manifest{}, "latest", "x64", release); err == nil || !strings.Contains(err.Error(), "valid SHA-256") { + t.Fatalf("missing-integrity error = %v", err) + } +} + +func TestActionsRunnerPlatformUsesConfiguredSourceArchitecture(t *testing.T) { + for _, test := range []struct { + platform string + want string + }{ + {"linux/amd64", "linux/amd64"}, + {"linux/arm64", "linux/arm64"}, + } { + got, err := actionsRunnerPlatform(Manifest{SourcePlatform: test.platform}, "docker-container") + if err != nil { + t.Fatal(err) + } + if got != test.want { + t.Fatalf("actionsRunnerPlatform(%q) = %q, want %q", test.platform, got, test.want) + } + } +} + +func TestRunnerInstallScriptRequiresVerifiedLocalPackage(t *testing.T) { + content, err := os.ReadFile(filepath.Join("..", "..", "scripts", "guest", "ubuntu", "install-runner.sh")) + if err != nil { + t.Fatal(err) + } + text := string(content) + for _, forbidden := range []string{"releases/latest", "api.github.com", "curl ", "wget "} { + if strings.Contains(text, forbidden) { + t.Fatalf("install-runner.sh still performs guest-side remote resolution: found %q", forbidden) + } + } + for _, required := range []string{"", "sha256sum --check", "sudo -u runner -H ./bin/Runner.Listener --version"} { + if !strings.Contains(text, required) { + t.Fatalf("install-runner.sh omitted %q", required) + } + } +} diff --git a/internal/image/storage_catalog.go b/internal/image/storage_catalog.go new file mode 100644 index 0000000..9166f76 --- /dev/null +++ b/internal/image/storage_catalog.go @@ -0,0 +1,1404 @@ +package image + +import ( + "context" + "crypto/sha256" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "regexp" + "runtime" + "sort" + "strconv" + "strings" + "sync" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + sandboxcapacity "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/capacity" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +var buildxUsageSizePattern = regexp.MustCompile(`^([0-9]+(?:\.[0-9]+)?)([kMGTPE]?i?B)$`) + +const ( + catalogDockerImageKind = "docker-image" + catalogSandboxTemplateKind = "sandbox-template" + catalogWSLArtifactKind = "provider-image" + catalogTartImageKind = "tart-image" + catalogTemplateStagingKind = "template-staging-directory" + catalogDockerOutputTagClaimKind = "docker-output-tag-claim" + dockerOutputTagClaimLifetime = 2 * time.Minute + dockerOutputTagClaimRefresh = 30 * time.Second + dockerOutputTagClaimLockTimeout = 15 * time.Second +) + +func (m *Coordinator) effectiveConfigPath() string { + if strings.TrimSpace(m.ConfigPath) != "" { + return m.ConfigPath + } + return filepath.Join(m.ProjectRoot, ".local", "config.yml") +} + +func (m *Coordinator) hostCatalog() (*storagecatalog.Store, error) { + return storagecatalog.Open("") +} + +func (m *Coordinator) catalogInstallationID(now time.Time) (string, error) { + store, err := m.hostCatalog() + if err != nil { + return "", err + } + var installationID string + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + record, err := storagecatalog.RegisterConfig(value, m.ProjectRoot, m.effectiveConfigPath(), now) + if err == nil { + installationID = record.InstallationID + err = m.applyCatalogConfigSettings(value, record.ID) + } + return err + }) + if err != nil { + return "", err + } + if installationID == "" { + return "", errors.New("registered EPAR installation identity disappeared") + } + return installationID, nil +} + +func (m *Coordinator) applyCatalogConfigSettings(value *storagecatalog.Catalog, configID string) error { + configured := strings.TrimSpace(m.Config.Storage.BuildCacheLimit) + if configured == "" { + configured = "20GiB" + } + limit, err := config.ParseByteSize(configured) + if err != nil { + return err + } + for index := range value.Configs { + if value.Configs[index].ID == configID { + value.Configs[index].BuildCacheLimitBytes = uint64(limit) + return nil + } + } + return fmt.Errorf("registered catalog configuration %s disappeared", configID) +} + +func (m *Coordinator) dockerBackendID(ctx context.Context) (string, error) { + id, err := m.runHostOutput(ctx, "docker", "info", "--format", "{{.ID}}") + if err != nil { + return "", err + } + id = strings.TrimSpace(id) + if id == "" { + return "", errors.New("Docker Engine returned an empty daemon identity") + } + return "docker:" + id, nil +} + +func (m *Coordinator) acquireDockerBackendLock(ctx context.Context) (string, func(), error) { + backendID, err := m.dockerBackendID(ctx) + if err != nil { + return "", nil, err + } + store, err := m.hostCatalog() + if err != nil { + return "", nil, err + } + lock, err := store.AcquireBackendLock(ctx, backendID) + if err != nil { + return "", nil, err + } + return backendID, func() { + if closeErr := lock.Close(); closeErr != nil { + m.warnf("EPAR Docker backend lock release warning: %v\n", closeErr) + } + }, nil +} + +// claimDockerOutputTag records a short-lived, exact intent to publish a Docker +// image tag. The catalog claim closes the window between checking active +// configuration references and mutating Docker's globally shared tag. A +// different configuration may share the tag only when it requests the exact +// same immutable manifest. +func (m *Coordinator) claimDockerOutputTag(ctx context.Context, tag, manifestHash string) (context.Context, func() error, error) { + tag = normalizedDockerTag(tag) + if tag == "" || strings.TrimSpace(manifestHash) == "" { + return nil, nil, errors.New("Docker output tag and manifest hash are required") + } + backendID, releaseBackend, err := m.acquireDockerBackendLock(ctx) + if err != nil { + return nil, nil, err + } + claimErr := m.claimDockerOutputTagLocked(backendID, tag, manifestHash, time.Now().UTC()) + releaseBackend() + if claimErr != nil { + return nil, nil, claimErr + } + claimContext, cancelClaim := context.WithCancelCause(ctx) + stopRefresh := make(chan struct{}) + refreshDone := make(chan struct{}) + var refreshOnce sync.Once + var refreshErr error + go func() { + defer close(refreshDone) + ticker := time.NewTicker(dockerOutputTagClaimRefresh) + defer ticker.Stop() + for { + select { + case <-stopRefresh: + return + case now := <-ticker.C: + refreshLockContext, cancelRefreshLock := context.WithTimeout(claimContext, dockerOutputTagClaimLockTimeout) + refreshBackendID, releaseRefreshLock, refreshLockErr := m.acquireDockerBackendLock(refreshLockContext) + cancelRefreshLock() + if refreshLockErr != nil { + refreshErr = fmt.Errorf("refresh Docker output-tag claim lock: %w", refreshLockErr) + cancelClaim(refreshErr) + return + } + if refreshBackendID != backendID { + releaseRefreshLock() + refreshErr = fmt.Errorf("Docker backend changed while publishing output tag %s", tag) + cancelClaim(refreshErr) + return + } + refreshErr = m.claimDockerOutputTagLocked(backendID, tag, manifestHash, now.UTC()) + releaseRefreshLock() + if refreshErr != nil { + refreshErr = fmt.Errorf("refresh Docker output-tag claim: %w", refreshErr) + cancelClaim(refreshErr) + return + } + } + } + }() + return claimContext, func() error { + refreshOnce.Do(func() { close(stopRefresh) }) + <-refreshDone + cancelClaim(nil) + releaseContext, cancelRelease := context.WithTimeout(context.Background(), dockerOutputTagClaimLockTimeout) + defer cancelRelease() + backendID, releaseBackend, err := m.acquireDockerBackendLock(releaseContext) + if err != nil { + return errors.Join(refreshErr, err) + } + defer releaseBackend() + return errors.Join(refreshErr, m.releaseDockerOutputTagClaim(backendID, tag, time.Now().UTC())) + }, nil +} + +func (m *Coordinator) claimDockerOutputTagLocked(backendID, tag, manifestHash string, now time.Time) error { + tag = normalizedDockerTag(tag) + if tag == "" || strings.TrimSpace(manifestHash) == "" { + return errors.New("Docker output tag and manifest hash are required") + } + store, err := m.hostCatalog() + if err != nil { + return err + } + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + configRecord, err := storagecatalog.RegisterConfig(value, m.ProjectRoot, m.effectiveConfigPath(), now) + if err != nil { + return err + } + if err := m.applyCatalogConfigSettings(value, configRecord.ID); err != nil { + return err + } + for _, resource := range value.Resources { + if resource.BackendID != backendID || resource.Kind != catalogDockerImageKind || normalizedDockerTag(resource.Locator) != tag { + continue + } + for _, reference := range resource.References { + if reference.ConfigID == configRecord.ID || reference.Role != "provider-artifact" { + continue + } + referencedManifest := reference.ManifestHash + if referencedManifest == "" { + referencedManifest = resource.ManifestHash + } + if referencedManifest != "" && referencedManifest != manifestHash { + return dockerOutputTagConflict(value, reference.ConfigID, tag, referencedManifest, manifestHash) + } + } + } + claimIdentity := "tag:" + tag + claimKey := storagecatalog.ResourceKey(backendID, catalogDockerOutputTagClaimKind, claimIdentity) + var existingReferences []storagecatalog.Reference + for resourceIndex := range value.Resources { + resource := &value.Resources[resourceIndex] + if resource.Key != claimKey { + continue + } + activeReferences := resource.References[:0] + for _, reference := range resource.References { + if !reference.UpdatedAt.Add(dockerOutputTagClaimLifetime).After(now) { + continue + } + activeReferences = append(activeReferences, reference) + if reference.ConfigID == configRecord.ID || reference.ManifestHash == manifestHash { + continue + } + return dockerOutputTagConflict(value, reference.ConfigID, tag, reference.ManifestHash, manifestHash) + } + resource.References = activeReferences + existingReferences = append(existingReferences, activeReferences...) + } + claim := storagecatalog.Resource{ + Key: claimKey, BackendID: backendID, Kind: catalogDockerOutputTagClaimKind, + Provider: "docker-container", Role: "runtime-image-tag-claim", Locator: tag, Identity: claimIdentity, + Custody: storagecatalog.CustodyGenerated, ManifestHash: manifestHash, State: storagecatalog.StateCurrent, + References: existingReferences, CreatedAt: now, LastSeenAt: now, + } + if err := storagecatalog.UpsertResource(value, claim); err != nil { + return err + } + storagecatalog.ReplaceConfigRoleReferences(value, configRecord.ID, "docker-output-tag-claim", map[string]storagecatalog.Reference{ + claimKey: {ManifestHash: manifestHash}, + }, now) + return nil + }) + return err +} + +func (m *Coordinator) releaseDockerOutputTagClaim(backendID, tag string, now time.Time) error { + if strings.TrimSpace(backendID) == "" || normalizedDockerTag(tag) == "" { + return errors.New("Docker backend and output tag are required to release an output-tag claim") + } + store, err := m.hostCatalog() + if err != nil { + return err + } + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + configRecord, err := storagecatalog.RegisterConfig(value, m.ProjectRoot, m.effectiveConfigPath(), now) + if err != nil { + return err + } + storagecatalog.ReplaceConfigRoleReferences(value, configRecord.ID, "docker-output-tag-claim", nil, now) + return nil + }) + return err +} + +func normalizedDockerTag(tag string) string { + tag = strings.TrimSpace(tag) + if tag == "" || strings.Contains(tag, "@") { + return tag + } + if strings.LastIndex(tag, ":") <= strings.LastIndex(tag, "/") { + tag += ":latest" + } + tagSeparator := strings.LastIndex(tag, ":") + name, suffix := tag[:tagSeparator], tag[tagSeparator:] + parts := strings.Split(name, "/") + hasRegistry := len(parts) > 1 && (strings.Contains(parts[0], ".") || strings.Contains(parts[0], ":") || parts[0] == "localhost") + if !hasRegistry { + if len(parts) == 1 { + name = "docker.io/library/" + name + } else { + name = "docker.io/" + name + } + } else { + if parts[0] == "index.docker.io" || parts[0] == "registry-1.docker.io" { + parts[0] = "docker.io" + } + if parts[0] == "docker.io" && len(parts) == 2 { + parts = []string{"docker.io", "library", parts[1]} + } + name = strings.Join(parts, "/") + } + return strings.ToLower(name) + suffix +} + +func dockerOutputTagConflict(value *storagecatalog.Catalog, configID, tag, existingManifest, requestedManifest string) error { + path := configID + for _, configRecord := range value.Configs { + if configRecord.ID == configID { + path = configRecord.DisplayPath + if path == "" { + path = configRecord.Path + } + break + } + } + return fmt.Errorf("Docker output image tag %q is actively claimed by configuration %s with manifest %s; requested manifest %s differs. Configure a unique image.outputImage or use matching immutable image inputs", tag, path, existingManifest, requestedManifest) +} + +func backendPathID(kind, path string) (string, error) { + absolute, err := filepath.Abs(path) + if err != nil { + return "", err + } + canonical := filepath.Clean(absolute) + if runtime.GOOS == "windows" { + canonical = strings.ToLower(canonical) + } + sum := sha256.Sum256([]byte(canonical)) + return fmt.Sprintf("%s:%x", kind, sum[:12]), nil +} + +func sandboxBackendID() (string, error) { + root, err := sandboxcapacity.DockerSandboxesStorageRoot() + if err != nil { + return "", err + } + return backendPathID("sandbox", root) +} + +func (m *Coordinator) withSandboxBackendLock(ctx context.Context, operation func() error) error { + backendID, err := sandboxBackendID() + if err != nil { + return err + } + store, err := m.hostCatalog() + if err != nil { + return err + } + lock, err := store.AcquireBackendLock(ctx, backendID) + if err != nil { + return err + } + defer func() { + if closeErr := lock.Close(); closeErr != nil { + m.warnf("EPAR Docker Sandboxes backend lock release warning: %v\n", closeErr) + } + }() + return operation() +} + +func tartBackendID() (string, error) { + root := strings.TrimSpace(os.Getenv("TART_HOME")) + if root == "" { + home, err := os.UserHomeDir() + if err != nil { + return "", err + } + root = filepath.Join(home, ".tart") + } + return backendPathID("tart", root) +} + +func (m *Coordinator) withTartBackendLock(ctx context.Context, operation func() error) error { + backendID, err := tartBackendID() + if err != nil { + return err + } + store, err := m.hostCatalog() + if err != nil { + return err + } + lock, err := store.AcquireBackendLock(ctx, backendID) + if err != nil { + return err + } + defer func() { + if closeErr := lock.Close(); closeErr != nil { + m.warnf("EPAR Tart backend lock release warning: %v\n", closeErr) + } + }() + return operation() +} + +func (m *Coordinator) recordCurrentArtifact(ctx context.Context, manifestHash string) error { + if m.DryRun { + return nil + } + now := time.Now().UTC() + var resource storagecatalog.Resource + switch m.Config.Provider.Type { + case "docker-container": + reference := strings.TrimSpace(m.Config.Image.OutputImage) + identity, err := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", reference) + if err != nil { + return fmt.Errorf("read current Docker Container image identity: %w", err) + } + backendID, err := m.dockerBackendID(ctx) + if err != nil { + return err + } + resource = storagecatalog.Resource{ + BackendID: backendID, Kind: catalogDockerImageKind, Provider: "docker-container", Role: "runtime-image", + Locator: reference, Identity: strings.TrimSpace(identity), Custody: storagecatalog.CustodyGenerated, + ManifestHash: manifestHash, IntroducedTags: []string{reference}, State: storagecatalog.StateCurrent, + CreatedAt: now, LastSeenAt: now, + } + case "wsl": + path := filepath.Clean(configPath(m.ProjectRoot, m.Config.Image.OutputImage)) + target, err := storage.SnapshotFilesystemTarget(path) + if err != nil { + return fmt.Errorf("read current WSL artifact identity: %w", err) + } + resource = storagecatalog.Resource{ + BackendID: "filesystem:" + filepath.VolumeName(target.Locator), Kind: catalogWSLArtifactKind, Provider: "wsl", Role: "runtime-rootfs", + Locator: target.Locator, Identity: target.Identity, Fingerprint: target.Fingerprint, Custody: storagecatalog.CustodyGenerated, + ManifestHash: manifestHash, State: storagecatalog.StateCurrent, CreatedAt: now, LastSeenAt: now, + } + case "tart": + output := strings.TrimSpace(m.Config.Image.OutputImage) + items, err := m.Lifecycle.Inventory(ctx) + if err != nil { + return fmt.Errorf("read current Tart image identity: %w", err) + } + var exact provider.Instance + for _, item := range items { + if item.Instance.Name == output { + exact = item.Instance + break + } + } + if exact.ProviderID == "" { + return fmt.Errorf("current Tart image %q has no immutable provider identity", output) + } + backendID, err := tartBackendID() + if err != nil { + return err + } + resource = storagecatalog.Resource{ + BackendID: backendID, Kind: catalogTartImageKind, Provider: "tart", Role: "runtime-image", + Locator: output, Identity: exact.ProviderID, Custody: storagecatalog.CustodyGenerated, + ManifestHash: manifestHash, State: storagecatalog.StateCurrent, CreatedAt: now, LastSeenAt: now, + } + default: + return nil + } + if err := m.registerCurrentCatalogResource(ctx, resource, manifestHash, now); err != nil { + return err + } + return m.releaseCatalogRole("build-source", now) +} + +// recordTartStagingImage records an exact temporary Tart image while the +// caller holds the Tart backend lock. Its per-config staging reference protects +// rollback evidence until startup reconciliation has restored or confirmed the +// configured output image. +func (m *Coordinator) recordTartStagingImage(ctx context.Context, name, role string) error { + items, err := m.Provider.List(ctx) + if err != nil { + return err + } + exact, found := findTartImage(items, name) + if !found || exact.ProviderID == "" { + return fmt.Errorf("Tart image %q has no exact immutable identity", name) + } + backendID, err := tartBackendID() + if err != nil { + return err + } + store, err := m.hostCatalog() + if err != nil { + return err + } + now := time.Now().UTC() + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + configRecord, err := storagecatalog.RegisterConfig(value, m.ProjectRoot, m.effectiveConfigPath(), now) + if err != nil { + return err + } + if err := m.applyCatalogConfigSettings(value, configRecord.ID); err != nil { + return err + } + resource := storagecatalog.Resource{ + BackendID: backendID, InstallationIDs: []string{configRecord.InstallationID}, Kind: catalogTartImageKind, + Provider: "tart", Role: role, Locator: name, Identity: exact.ProviderID, + Custody: storagecatalog.CustodyGenerated, State: storagecatalog.StateStaging, + CreatedAt: now, LastSeenAt: now, + References: []storagecatalog.Reference{{ + ConfigID: configRecord.ID, Role: "tart-staging", UpdatedAt: now, + }}, + } + return storagecatalog.UpsertResource(value, resource) + }) + return err +} + +func configPath(projectRoot, path string) string { + if filepath.IsAbs(path) { + return path + } + return filepath.Join(projectRoot, path) +} + +func (m *Coordinator) recordCurrentSandboxArtifact(ctx context.Context, artifact provider.TemplateArtifact, manifestHash string, now time.Time) error { + backendID, err := sandboxBackendID() + if err != nil { + return err + } + resource := storagecatalog.Resource{ + BackendID: backendID, Kind: catalogSandboxTemplateKind, Provider: "docker-sandboxes", Role: "runtime-template", + Locator: artifact.Reference, Identity: artifact.CacheID, Fingerprint: artifact.Digest, + Custody: storagecatalog.CustodyGenerated, ManifestHash: manifestHash, State: storagecatalog.StateCurrent, + CreatedAt: now, LastSeenAt: now, + } + if err := m.registerCurrentCatalogResource(ctx, resource, manifestHash, now); err != nil { + return err + } + return m.releaseCatalogRole("build-source", now) +} + +func (m *Coordinator) recordSandboxWorkspace(ctx context.Context, workspacePath, manifestHash string, state storagecatalog.State, now time.Time) error { + if state != storagecatalog.StateStaging && state != storagecatalog.StateSuperseded { + return fmt.Errorf("unsupported Docker Sandboxes workspace state %q", state) + } + stagingDirectory, err := storage.SnapshotFilesystemTarget(workspacePath) + if err != nil { + return fmt.Errorf("read Docker Sandboxes staging directory identity: %w", err) + } + resource := storagecatalog.Resource{ + BackendID: "filesystem:" + filepath.VolumeName(stagingDirectory.Locator), Kind: catalogTemplateStagingKind, Provider: "docker-sandboxes", Role: "template-archive-workspace", + Locator: stagingDirectory.Locator, Identity: stagingDirectory.Identity, Fingerprint: stagingDirectory.Fingerprint, Custody: storagecatalog.CustodyGenerated, + ManifestHash: manifestHash, State: state, CreatedAt: now, LastSeenAt: now, + } + if state == storagecatalog.StateSuperseded { + supersededAt := now.UTC() + resource.SupersededAt = &supersededAt + } + store, err := m.hostCatalog() + if err != nil { + return err + } + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + configRecord, err := storagecatalog.RegisterConfig(value, m.ProjectRoot, m.effectiveConfigPath(), now) + if err != nil { + return err + } + if err := m.applyCatalogConfigSettings(value, configRecord.ID); err != nil { + return err + } + resource.InstallationIDs = unionStrings(resource.InstallationIDs, []string{configRecord.InstallationID}) + resource.Key = storagecatalog.ResourceKey(resource.BackendID, resource.Kind, resource.Identity) + for index := range value.Resources { + candidate := &value.Resources[index] + filtered := candidate.References[:0] + for _, reference := range candidate.References { + if reference.ConfigID == configRecord.ID && reference.Role == "template-staging" { + continue + } + filtered = append(filtered, reference) + } + candidate.References = filtered + if candidate.Key != resource.Key && len(candidate.References) == 0 && candidate.State == storagecatalog.StateStaging { + supersededAt := now.UTC() + candidate.State = storagecatalog.StateSuperseded + candidate.SupersededAt = &supersededAt + } + if candidate.Key == resource.Key { + resource.References = append(resource.References, candidate.References...) + } + } + if state == storagecatalog.StateStaging { + resource.References = append(resource.References, storagecatalog.Reference{ + ConfigID: configRecord.ID, ManifestHash: manifestHash, Role: "template-staging", UpdatedAt: now.UTC(), + }) + resource.SupersededAt = nil + } else if len(resource.References) != 0 { + resource.State = storagecatalog.StateStaging + resource.SupersededAt = nil + } + if err := storagecatalog.UpsertResource(value, resource); err != nil { + return err + } + return nil + }) + return err +} + +func (m *Coordinator) recordDockerSourceAcquisition(ctx context.Context, reference, previousID, currentID string, now time.Time) error { + return m.recordDockerRoleAcquisition(ctx, "build-source", reference, previousID, currentID, now) +} + +func (m *Coordinator) recordDockerRoleAcquisition(ctx context.Context, role, reference, previousID, currentID string, now time.Time) error { + if m.DryRun || currentID == "" { + return nil + } + backendID, err := m.dockerBackendID(ctx) + if err != nil { + return err + } + tag := reference + if strings.LastIndex(tag, ":") <= strings.LastIndex(tag, "/") { + tag += ":latest" + } + resource := storagecatalog.Resource{ + BackendID: backendID, Kind: catalogDockerImageKind, Role: role, Locator: tag, Identity: currentID, + Custody: storagecatalog.CustodyAcquired, State: storagecatalog.StateStaging, + CreatedAt: now, LastSeenAt: now, + } + if previousID == "" { + resource.IntroducedTags = []string{tag} + } + store, err := m.hostCatalog() + if err != nil { + return err + } + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + configRecord, err := storagecatalog.RegisterConfig(value, m.ProjectRoot, m.effectiveConfigPath(), now) + if err != nil { + return err + } + if err := m.applyCatalogConfigSettings(value, configRecord.ID); err != nil { + return err + } + resource.Key = storagecatalog.ResourceKey(resource.BackendID, resource.Kind, resource.Identity) + if currentID == previousID { + found := false + for _, existing := range value.Resources { + if existing.Key == resource.Key && existing.Custody == storagecatalog.CustodyAcquired { + resource = existing + resource.Role = role + resource.LastSeenAt = now + found = true + break + } + } + if !found { + completeDockerAcquisitionJournal(value, dockerAcquisitionJournalID(configRecord.ID, backendID, role, tag), now) + return nil + } + } + resource.InstallationIDs = unionStrings(resource.InstallationIDs, []string{configRecord.InstallationID}) + if err := storagecatalog.UpsertResource(value, resource); err != nil { + return err + } + storagecatalog.ReplaceConfigRoleReferences(value, configRecord.ID, role, map[string]storagecatalog.Reference{ + resource.Key: {}, + }, now) + completeDockerAcquisitionJournal(value, dockerAcquisitionJournalID(configRecord.ID, backendID, role, tag), now) + return nil + }) + return err +} + +func (m *Coordinator) beginDockerRoleAcquisition(backendID, role, reference, previousID string, now time.Time) error { + tag := reference + if strings.LastIndex(tag, ":") <= strings.LastIndex(tag, "/") { + tag += ":latest" + } + store, err := m.hostCatalog() + if err != nil { + return err + } + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + configRecord, registerErr := storagecatalog.RegisterConfig(value, m.ProjectRoot, m.effectiveConfigPath(), now) + if registerErr != nil { + return registerErr + } + if settingsErr := m.applyCatalogConfigSettings(value, configRecord.ID); settingsErr != nil { + return settingsErr + } + id := dockerAcquisitionJournalID(configRecord.ID, backendID, role, tag) + for index := range value.Journals { + if value.Journals[index].ID == id { + value.Journals[index].Phase = "acquiring" + value.Journals[index].PreviousIdentity = previousID + value.Journals[index].UpdatedAt = now + value.Journals[index].Error = "" + return nil + } + } + value.Journals = append(value.Journals, storagecatalog.Journal{ + ID: id, Operation: "docker-image-acquisition", BackendID: backendID, ConfigID: configRecord.ID, + Role: role, Locator: tag, PreviousIdentity: previousID, Phase: "acquiring", StartedAt: now, UpdatedAt: now, + }) + return nil + }) + return err +} + +func dockerAcquisitionJournalID(configID, backendID, role, locator string) string { + sum := sha256.Sum256([]byte(configID + "\x00" + backendID + "\x00" + role + "\x00" + locator)) + return fmt.Sprintf("acquire-%x", sum[:12]) +} + +func completeDockerAcquisitionJournal(value *storagecatalog.Catalog, id string, now time.Time) { + for index := range value.Journals { + if value.Journals[index].ID == id { + value.Journals[index].Phase = "complete" + value.Journals[index].UpdatedAt = now + value.Journals[index].Error = "" + return + } + } +} + +func (m *Coordinator) releaseCatalogRole(role string, now time.Time) error { + store, err := m.hostCatalog() + if err != nil { + return err + } + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + configRecord, err := storagecatalog.RegisterConfig(value, m.ProjectRoot, m.effectiveConfigPath(), now) + if err != nil { + return err + } + if err := m.applyCatalogConfigSettings(value, configRecord.ID); err != nil { + return err + } + storagecatalog.ReplaceConfigRoleReferences(value, configRecord.ID, role, nil, now) + return nil + }) + return err +} + +func (m *Coordinator) registerCurrentCatalogResource(ctx context.Context, resource storagecatalog.Resource, manifestHash string, now time.Time) error { + store, err := m.hostCatalog() + if err != nil { + return err + } + backendLock, err := store.AcquireBackendLock(ctx, resource.BackendID) + if err != nil { + return err + } + defer backendLock.Close() + _, err = store.WithLock(now, func(value *storagecatalog.Catalog) error { + configRecord, err := storagecatalog.RegisterConfig(value, m.ProjectRoot, m.effectiveConfigPath(), now) + if err != nil { + return err + } + if err := m.applyCatalogConfigSettings(value, configRecord.ID); err != nil { + return err + } + resource.Key = storagecatalog.ResourceKey(resource.BackendID, resource.Kind, resource.Identity) + var existingReferences []storagecatalog.Reference + for _, existing := range value.Resources { + if existing.Key == resource.Key { + existingReferences = append(existingReferences, existing.References...) + if resource.CreatedAt.IsZero() { + resource.CreatedAt = existing.CreatedAt + } + resource.IntroducedTags = unionStrings(existing.IntroducedTags, resource.IntroducedTags) + resource.InstallationIDs = unionStrings(existing.InstallationIDs, resource.InstallationIDs) + break + } + } + resource.InstallationIDs = unionStrings(resource.InstallationIDs, []string{configRecord.InstallationID}) + resource.References = existingReferences + if err := storagecatalog.UpsertResource(value, resource); err != nil { + return err + } + storagecatalog.ReplaceConfigRoleReferences(value, configRecord.ID, "provider-artifact", map[string]storagecatalog.Reference{ + resource.Key: {ManifestHash: manifestHash}, + }, now) + return nil + }) + return err +} + +func unionStrings(left, right []string) []string { + set := make(map[string]struct{}, len(left)+len(right)) + for _, value := range append(append([]string(nil), left...), right...) { + if strings.TrimSpace(value) != "" { + set[value] = struct{}{} + } + } + out := make([]string, 0, len(set)) + for value := range set { + out = append(out, value) + } + sort.Strings(out) + return out +} + +func (m *Coordinator) cleanupSupersededCatalog(ctx context.Context) error { + if m.DryRun || strings.EqualFold(m.Config.Storage.AutomaticHousekeeping, "disabled") { + return nil + } + if err := m.reconcileInterruptedDockerAcquisitions(ctx); err != nil { + m.warnf("EPAR interrupted Docker acquisition reconciliation deferred: %v\n", err) + } + if err := m.reconcileInterruptedTartArtifacts(ctx); err != nil { + return fmt.Errorf("reconcile interrupted Tart artifact activation: %w", err) + } + store, err := m.hostCatalog() + if err != nil { + return err + } + if err := m.enforceDedicatedBuildxCache(ctx); err != nil { + m.warnf("EPAR BuildKit cache housekeeping deferred: %v\n", err) + } + now := time.Now().UTC() + var reconcileWarnings []string + value, err := store.WithLock(now, func(value *storagecatalog.Catalog) error { + reconcileWarnings = storagecatalog.Compact(value, now, func(resource storagecatalog.Resource) (bool, error) { + return m.catalogResourceExists(ctx, resource) + }) + return nil + }) + if err != nil { + return err + } + for _, warning := range reconcileWarnings { + m.warnf("EPAR storage catalog reconciliation: %s\n", warning) + } + if m.Config.Storage.KeepPrevious > 0 { + m.infof("automatic artifact retirement is deferred because storage.keepPrevious=%d; use storage prune to preview the retention policy\n", m.Config.Storage.KeepPrevious) + return nil + } + for _, resource := range value.Resources { + if len(resource.References) != 0 || (resource.State != storagecatalog.StateSuperseded && resource.State != storagecatalog.StateCleanupPending) { + continue + } + backendLock, lockErr := store.AcquireBackendLock(ctx, resource.BackendID) + if lockErr != nil { + m.warnf("EPAR storage cleanup backend lock deferred for %s %s: %v\n", resource.Kind, resource.Identity, lockErr) + continue + } + startedAt := time.Now().UTC() + removeCandidate := resource + shouldRemove := false + if _, journalErr := store.WithLock(startedAt, func(current *storagecatalog.Catalog) error { + for _, candidate := range current.Resources { + if candidate.Key != resource.Key { + continue + } + if len(candidate.References) != 0 || (candidate.State != storagecatalog.StateSuperseded && candidate.State != storagecatalog.StateCleanupPending) { + return nil + } + removeCandidate = candidate + shouldRemove = true + upsertCleanupJournal(current, candidate, "remove-started", "", startedAt) + return nil + } + return nil + }); journalErr != nil { + _ = backendLock.Close() + return journalErr + } + if !shouldRemove { + if closeErr := backendLock.Close(); closeErr != nil { + m.warnf("EPAR storage cleanup backend lock release warning: %v\n", closeErr) + } + continue + } + cleanupLabel := fmt.Sprintf("EPAR superseded %s cleanup for %s", removeCandidate.Kind, removeCandidate.Identity) + removeErr := m.runProgressOperation(cleanupLabel, nil, func() error { + return m.removeCatalogResource(ctx, removeCandidate) + }) + _, updateErr := store.WithLock(time.Now().UTC(), func(current *storagecatalog.Catalog) error { + for index := range current.Resources { + if current.Resources[index].Key != removeCandidate.Key { + continue + } + if len(current.Resources[index].References) != 0 { + return nil + } + if removeErr == nil { + current.Resources = append(current.Resources[:index], current.Resources[index+1:]...) + upsertCleanupJournal(current, removeCandidate, "complete", "", time.Now().UTC()) + } else { + current.Resources[index].State = storagecatalog.StateCleanupPending + current.Resources[index].CleanupError = removeErr.Error() + upsertCleanupJournal(current, removeCandidate, "cleanup-pending", removeErr.Error(), time.Now().UTC()) + } + return nil + } + return nil + }) + closeErr := backendLock.Close() + if updateErr != nil { + return updateErr + } + if closeErr != nil { + m.warnf("EPAR storage cleanup backend lock release warning: %v\n", closeErr) + } + if removeErr != nil { + m.warnf("EPAR storage cleanup deferred for %s %s: %v\n", removeCandidate.Kind, removeCandidate.Identity, removeErr) + } + } + return nil +} + +func (m *Coordinator) reconcileInterruptedTartArtifacts(ctx context.Context) error { + if m.Config.Provider.Type != "tart" { + return nil + } + return m.withTartBackendLock(ctx, func() error { + store, err := m.hostCatalog() + if err != nil { + return err + } + now := time.Now().UTC() + value, err := store.Load(now) + if err != nil { + return err + } + configID, err := storagecatalog.ConfigID(m.ProjectRoot, m.effectiveConfigPath()) + if err != nil { + return err + } + var staging []storagecatalog.Resource + for _, resource := range value.Resources { + if resource.Kind != catalogTartImageKind || resource.State != storagecatalog.StateStaging { + continue + } + for _, reference := range resource.References { + if reference.ConfigID == configID && reference.Role == "tart-staging" { + staging = append(staging, resource) + break + } + } + } + if len(staging) == 0 { + return nil + } + instances, err := m.Provider.List(ctx) + if err != nil { + return err + } + outputName := strings.TrimSpace(m.Config.Image.OutputImage) + if _, outputExists := findTartImage(instances, outputName); !outputExists { + var rollback *storagecatalog.Resource + for index := range staging { + resource := &staging[index] + if resource.Role != "activation-rollback" { + continue + } + instance, exists := findTartImage(instances, resource.Locator) + if exists && instance.ProviderID == resource.Identity { + rollback = resource + break + } + } + if rollback == nil { + return fmt.Errorf("configured Tart output %q is missing and no exact rollback image is available", outputName) + } + if err := m.Provider.Clone(ctx, rollback.Locator, outputName); err != nil { + return fmt.Errorf("restore interrupted Tart activation from %q: %w", rollback.Locator, err) + } + if err := m.verifyTartImageIdentity(ctx, outputName); err != nil { + return err + } + if m.environment != nil { + m.warnf("restored Tart output image %s after an interrupted activation; the desired artifact will be reconciled next\n", outputName) + } + } + _, err = store.WithLock(time.Now().UTC(), func(current *storagecatalog.Catalog) error { + when := time.Now().UTC() + for index := range current.Resources { + resource := ¤t.Resources[index] + if resource.Kind != catalogTartImageKind || resource.State != storagecatalog.StateStaging { + continue + } + filtered := resource.References[:0] + removed := false + for _, reference := range resource.References { + if reference.ConfigID == configID && reference.Role == "tart-staging" { + removed = true + continue + } + filtered = append(filtered, reference) + } + resource.References = filtered + if removed && len(resource.References) == 0 { + resource.State = storagecatalog.StateSuperseded + resource.SupersededAt = &when + } + } + return nil + }) + return err + }) +} + +func (m *Coordinator) StorageCleanupPending() (bool, error) { + store, err := m.hostCatalog() + if err != nil { + return false, err + } + value, err := store.Load(time.Now().UTC()) + if err != nil { + return false, err + } + for _, resource := range value.Resources { + if resource.State == storagecatalog.StateCleanupPending { + return true, nil + } + } + for _, journal := range value.Journals { + if journal.Phase == "cleanup-pending" { + return true, nil + } + } + return false, nil +} + +func (m *Coordinator) reconcileInterruptedDockerAcquisitions(ctx context.Context) error { + store, err := m.hostCatalog() + if err != nil { + return err + } + value, err := store.Load(time.Now().UTC()) + if err != nil { + return err + } + for _, journal := range value.Journals { + if journal.Operation != "docker-image-acquisition" || journal.Phase == "complete" { + continue + } + lock, lockErr := store.AcquireBackendLock(ctx, journal.BackendID) + if lockErr != nil { + return lockErr + } + currentID, inspectErr := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", journal.Locator) + if inspectErr != nil && !dockerInspectMeansMissing(inspectErr) { + _ = lock.Close() + return inspectErr + } + currentID = strings.TrimSpace(currentID) + now := time.Now().UTC() + _, updateErr := store.WithLock(now, func(current *storagecatalog.Catalog) error { + if currentID != "" && currentID != journal.PreviousIdentity { + installationIDs := []string{} + for _, configRecord := range current.Configs { + if configRecord.ID == journal.ConfigID { + installationIDs = append(installationIDs, configRecord.InstallationID) + break + } + } + resource := storagecatalog.Resource{ + BackendID: journal.BackendID, InstallationIDs: installationIDs, Kind: catalogDockerImageKind, + Role: journal.Role, Locator: journal.Locator, Identity: currentID, Custody: storagecatalog.CustodyAcquired, + State: storagecatalog.StateSuperseded, CreatedAt: now, LastSeenAt: now, + } + if journal.PreviousIdentity == "" { + resource.IntroducedTags = []string{journal.Locator} + } + when := now.UTC() + resource.SupersededAt = &when + if upsertErr := storagecatalog.UpsertResource(current, resource); upsertErr != nil { + return upsertErr + } + } + completeDockerAcquisitionJournal(current, journal.ID, now) + return nil + }) + closeErr := lock.Close() + if updateErr != nil { + return updateErr + } + if closeErr != nil { + return closeErr + } + } + return nil +} + +func upsertCleanupJournal(value *storagecatalog.Catalog, resource storagecatalog.Resource, phase, message string, now time.Time) { + id := "cleanup-" + resource.Key + for index := range value.Journals { + if value.Journals[index].ID == id { + value.Journals[index].Phase = phase + value.Journals[index].UpdatedAt = now + value.Journals[index].Error = message + return + } + } + value.Journals = append(value.Journals, storagecatalog.Journal{ + ID: id, Operation: "remove-exact-resource", ResourceKey: resource.Key, Phase: phase, + StartedAt: now, UpdatedAt: now, Error: message, + }) +} + +func (m *Coordinator) catalogResourceExists(ctx context.Context, resource storagecatalog.Resource) (bool, error) { + switch resource.Kind { + case catalogDockerImageKind: + identity, err := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", resource.Identity) + if err != nil { + if dockerInspectMeansMissing(err) { + return false, nil + } + return false, err + } + return strings.TrimSpace(identity) == resource.Identity, nil + case catalogSandboxTemplateKind: + observer, ok := m.Lifecycle.(provider.TemplateArtifactObserver) + if !ok { + return true, nil + } + return observer.ObserveTemplate(ctx, provider.TemplateArtifact{ + Reference: resource.Locator, + CacheID: resource.Identity, + Digest: resource.Fingerprint, + }) + case catalogWSLArtifactKind: + target, err := storage.SnapshotFilesystemTarget(resource.Locator) + if errors.Is(err, os.ErrNotExist) { + return false, nil + } + if err != nil { + return false, err + } + return target.Identity == resource.Identity && target.Fingerprint == resource.Fingerprint, nil + case catalogTemplateStagingKind: + target, err := storage.SnapshotFilesystemTarget(resource.Locator) + if errors.Is(err, os.ErrNotExist) { + return false, nil + } + if err != nil { + return false, err + } + return target.Identity == resource.Identity && target.Fingerprint == resource.Fingerprint, nil + case catalogDockerOutputTagClaimKind: + // Claims are catalog-only intent records. They have no Docker object to + // inspect and are removed exactly when their reference is released. + return true, nil + case catalogTartImageKind: + if m.Config.Provider.Type != "tart" { + return true, nil + } + items, err := m.Lifecycle.Inventory(ctx) + if err != nil { + return false, err + } + for _, item := range items { + if item.Instance.Name == resource.Locator && item.Instance.ProviderID == resource.Identity { + return true, nil + } + } + return false, nil + default: + return true, nil + } +} + +func (m *Coordinator) enforceDedicatedBuildxCache(ctx context.Context) error { + metadata, err := LoadBuildxMetadataForConfig(m.ProjectRoot, m.effectiveConfigPath()) + if errors.Is(err, os.ErrNotExist) { + return nil + } + if err != nil { + return err + } + limitBytes, err := m.effectiveBuildCacheLimit() + if err != nil { + return err + } + if _, err := m.runHostOutput(ctx, "docker", "buildx", "inspect", metadata.Builder); err != nil { + return nil + } + usageOutput, err := m.runHostOutput(ctx, "docker", "buildx", "du", "--builder", metadata.Builder, "--format", "json") + if err != nil { + return err + } + usageBytes, err := parseBuildxUsageBytes([]byte(usageOutput)) + if err != nil { + return err + } + if usageBytes <= limitBytes { + return nil + } + return m.runHostQuiet(ctx, "docker", "buildx", "prune", "--builder", metadata.Builder, "--force", "--max-used-space", strconv.FormatUint(limitBytes, 10)+"B") +} + +func (m *Coordinator) effectiveBuildCacheLimit() (uint64, error) { + configured := strings.TrimSpace(m.Config.Storage.BuildCacheLimit) + if configured == "" { + configured = "20GiB" + } + current, err := config.ParseByteSize(configured) + if err != nil { + return 0, err + } + return uint64(current), nil +} + +func parseBuildxUsageBytes(content []byte) (uint64, error) { + type record struct { + ID string `json:"ID"` + Size json.RawMessage `json:"Size"` + Total json.RawMessage `json:"Total"` + } + trimmed := strings.TrimSpace(string(content)) + if trimmed == "" { + return 0, nil + } + var records []record + if strings.HasPrefix(trimmed, "[") { + if err := json.Unmarshal([]byte(trimmed), &records); err != nil { + return 0, err + } + } else { + for _, line := range strings.Split(trimmed, "\n") { + var value record + if err := json.Unmarshal([]byte(strings.TrimSpace(line)), &value); err != nil { + return 0, err + } + records = append(records, value) + } + } + parse := func(raw json.RawMessage) (uint64, bool) { + var numeric uint64 + if err := json.Unmarshal(raw, &numeric); err == nil { + return numeric, true + } + var text string + if err := json.Unmarshal(raw, &text); err == nil { + if parsed, parseErr := config.ParseByteSize(text); parseErr == nil && parsed >= 0 { + return uint64(parsed), true + } + if match := buildxUsageSizePattern.FindStringSubmatch(text); match != nil { + if parsed, ok := parseBuildxBytes(match[1], match[2]); ok && parsed >= 0 { + return uint64(parsed), true + } + } + } + return 0, false + } + var sum uint64 + for _, value := range records { + if total, ok := parse(value.Total); ok { + return total, nil + } + if value.ID == "" { + continue + } + size, ok := parse(value.Size) + if !ok || sum > ^uint64(0)-size { + return 0, errors.New("Buildx disk-usage output is incomplete or overflows") + } + sum += size + } + return sum, nil +} + +func (m *Coordinator) removeCatalogResource(ctx context.Context, resource storagecatalog.Resource) error { + switch resource.Kind { + case catalogDockerImageKind: + containers, err := m.runHostOutput(ctx, "docker", "ps", "-a", "--filter", "ancestor="+resource.Identity, "--format", "{{.ID}}") + if err != nil { + return err + } + if strings.TrimSpace(containers) != "" { + return errors.New("a Docker container still references the image") + } + tagsJSON, err := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{json .RepoTags}}", resource.Identity) + if err != nil { + if dockerInspectMeansMissing(err) { + return nil + } + return err + } + var tags []string + if err := json.Unmarshal([]byte(strings.TrimSpace(tagsJSON)), &tags); err != nil { + return err + } + introduced := make(map[string]bool, len(resource.IntroducedTags)) + for _, tag := range resource.IntroducedTags { + introduced[tag] = true + } + for _, tag := range resource.IntroducedTags { + tagID, inspectErr := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", tag) + if inspectErr == nil && strings.TrimSpace(tagID) == resource.Identity { + if err := m.runHostQuiet(ctx, "docker", "image", "rm", tag); err != nil { + return err + } + } + } + remainingTagsJSON, inspectErr := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{json .RepoTags}}", resource.Identity) + if inspectErr == nil { + var remaining []string + if err := json.Unmarshal([]byte(strings.TrimSpace(remainingTagsJSON)), &remaining); err != nil { + return err + } + for _, tag := range remaining { + if !introduced[tag] { + return nil + } + } + if len(remaining) != 0 { + return fmt.Errorf("EPAR-introduced Docker image tags remain after exact removal: %s", strings.Join(remaining, ", ")) + } + if err := m.runHostQuiet(ctx, "docker", "image", "rm", resource.Identity); err != nil { + return err + } + } + return nil + case catalogSandboxTemplateKind: + if resource.Locator == "docker.io/docker/sandbox-templates:shell-docker" || resource.Locator == "docker/sandbox-templates:shell-docker" { + return errors.New("the Docker Sandboxes shell-docker base template is protected") + } + cleaner, ok := m.Lifecycle.(provider.TemplateArtifactCleaner) + if !ok { + return errors.New("provider does not expose exact template cleanup") + } + return cleaner.RemoveTemplate(ctx, provider.TemplateArtifact{Reference: resource.Locator, CacheID: resource.Identity, Digest: resource.Fingerprint}) + case catalogWSLArtifactKind: + absolute, err := filepath.Abs(resource.Locator) + if err != nil { + return err + } + root, err := filepath.Abs(m.ProjectRoot) + if err != nil { + return err + } + relative, err := filepath.Rel(root, absolute) + if err != nil || relative == "." || relative == ".." || filepath.IsAbs(relative) || strings.HasPrefix(relative, ".."+string(filepath.Separator)) { + return errors.New("filesystem artifact belongs to another project and is deferred to its own controller") + } + target, err := storage.SnapshotFilesystemTarget(absolute) + if errors.Is(err, os.ErrNotExist) { + return nil + } + if err != nil { + return err + } + if target.Identity != resource.Identity || target.Fingerprint != resource.Fingerprint || target.Kind != storage.TargetFile { + return errors.New("filesystem artifact identity changed") + } + return os.Remove(absolute) + case catalogTemplateStagingKind: + executor, err := storage.NewFilesystemExecutor(filepath.Join(m.ProjectRoot, "work", "template-builds", "docker-sandboxes")) + if err != nil { + return err + } + target := storage.Target{Kind: storage.TargetDirectory, Locator: resource.Locator, Identity: resource.Identity, Fingerprint: resource.Fingerprint, Match: storage.MatchExact} + observation, err := executor.ObserveExact(ctx, target) + if err != nil { + return err + } + if !observation.Exists { + return nil + } + if observation.Target != target { + return errors.New("template staging directory exact identity changed") + } + return executor.RemoveExact(ctx, storage.Removal{Target: target}) + case catalogTartImageKind: + if m.Config.Provider.Type != "tart" { + return errors.New("Tart image cleanup is deferred to a Tart controller on macOS") + } + items, err := m.Lifecycle.Inventory(ctx) + if err != nil { + return err + } + var exact *provider.Instance + for index := range items { + instance := items[index].Instance + if instance.Source == resource.Locator && instance.Name != resource.Locator { + return fmt.Errorf("Tart instance %q still references image %q", instance.Name, resource.Locator) + } + if instance.Name == resource.Locator { + if instance.ProviderID != resource.Identity { + return errors.New("Tart image identity changed") + } + copy := instance + exact = © + } + } + if exact == nil { + return nil + } + if strings.EqualFold(exact.State, "running") { + return errors.New("Tart image is running") + } + return m.Lifecycle.Delete(ctx, *exact) + case catalogDockerOutputTagClaimKind: + return nil + default: + return fmt.Errorf("automatic cleanup is not implemented for catalog resource kind %q", resource.Kind) + } +} diff --git a/internal/image/storage_catalog_test.go b/internal/image/storage_catalog_test.go new file mode 100644 index 0000000..f428af7 --- /dev/null +++ b/internal/image/storage_catalog_test.go @@ -0,0 +1,211 @@ +package image + +import ( + "context" + "os" + "path/filepath" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +func TestBeginDockerAcquisitionPersistsPreexistingIdentityBeforePull(t *testing.T) { + stateRoot := filepath.Join(t.TempDir(), "host-state") + t.Setenv("EPAR_STATE_HOME", stateRoot) + projectRoot := t.TempDir() + configPath := filepath.Join(projectRoot, "config.yml") + if err := os.WriteFile(configPath, []byte("provider:\n type: docker-container\n"), 0o600); err != nil { + t.Fatal(err) + } + coordinator := Coordinator{Config: config.Default(), ProjectRoot: projectRoot, ConfigPath: configPath} + now := time.Date(2026, 7, 29, 1, 2, 3, 0, time.UTC) + if err := coordinator.beginDockerRoleAcquisition("docker:daemon", "build-source", "example/source:latest", "sha256:before", now); err != nil { + t.Fatal(err) + } + store, err := storagecatalog.Open("") + if err != nil { + t.Fatal(err) + } + value, err := store.Load(now) + if err != nil { + t.Fatal(err) + } + if len(value.Journals) != 1 { + t.Fatalf("journals = %#v, want one acquisition journal", value.Journals) + } + journal := value.Journals[0] + if journal.Operation != "docker-image-acquisition" || journal.Phase != "acquiring" || journal.BackendID != "docker:daemon" || journal.Locator != "example/source:latest" || journal.PreviousIdentity != "sha256:before" { + t.Fatalf("acquisition journal omitted pre-pull evidence: %#v", journal) + } + if len(value.Configs) != 1 || value.Configs[0].InstallationID == "" { + t.Fatalf("acquisition journal did not register its exact EPAR installation: %#v", value.Configs) + } +} + +func TestNonTartControllerPreservesTartCatalogEvidence(t *testing.T) { + coordinator := Coordinator{Config: config.Default()} + coordinator.Config.Provider.Type = "docker-container" + exists, err := coordinator.catalogResourceExists(context.Background(), storagecatalog.Resource{ + Kind: catalogTartImageKind, + Locator: "epar-tart-image", + Identity: "tart-mac:001122334455", + }) + if err != nil { + t.Fatal(err) + } + if !exists { + t.Fatal("non-Tart controller discarded Tart catalog evidence") + } +} + +func TestCurrentReferenceUpdateWaitsForBackendLock(t *testing.T) { + t.Setenv("EPAR_STATE_HOME", filepath.Join(t.TempDir(), "host-state")) + projectRoot := t.TempDir() + configPath := filepath.Join(projectRoot, "config.yml") + if err := os.WriteFile(configPath, []byte("provider:\n type: wsl\n"), 0o600); err != nil { + t.Fatal(err) + } + coordinator := Coordinator{Config: config.Default(), ProjectRoot: projectRoot, ConfigPath: configPath} + store, err := storagecatalog.Open("") + if err != nil { + t.Fatal(err) + } + backendID := "filesystem:test" + backendLock, err := store.AcquireBackendLock(context.Background(), backendID) + if err != nil { + t.Fatal(err) + } + result := make(chan error, 1) + go func() { + result <- coordinator.registerCurrentCatalogResource(context.Background(), storagecatalog.Resource{ + BackendID: backendID, + Kind: catalogWSLArtifactKind, + Provider: "wsl", + Role: "runtime-rootfs", + Locator: filepath.Join(projectRoot, "runner.tar"), + Identity: "filesystem-id", + Custody: storagecatalog.CustodyGenerated, + State: storagecatalog.StateCurrent, + }, "manifest", time.Now().UTC()) + }() + select { + case err := <-result: + t.Fatalf("reference update bypassed the held backend lock: %v", err) + case <-time.After(100 * time.Millisecond): + } + if err := backendLock.Close(); err != nil { + t.Fatal(err) + } + select { + case err := <-result: + if err != nil { + t.Fatal(err) + } + case <-time.After(2 * time.Second): + t.Fatal("reference update did not proceed after backend lock release") + } +} + +func TestSandboxWorkspaceIsCatalogedBeforeBuildAndSupersededExactly(t *testing.T) { + t.Setenv("EPAR_STATE_HOME", filepath.Join(t.TempDir(), "host-state")) + projectRoot := t.TempDir() + configPath := filepath.Join(projectRoot, "config.yml") + if err := os.WriteFile(configPath, []byte("provider:\n type: docker-sandboxes\n"), 0o600); err != nil { + t.Fatal(err) + } + first := filepath.Join(projectRoot, "work", "template-builds", "docker-sandboxes", "first") + second := filepath.Join(projectRoot, "work", "template-builds", "docker-sandboxes", "second") + for _, path := range []string{first, second} { + if err := os.MkdirAll(path, 0o700); err != nil { + t.Fatal(err) + } + } + coordinator := Coordinator{Config: config.Default(), ProjectRoot: projectRoot, ConfigPath: configPath} + now := time.Date(2026, 7, 30, 1, 2, 3, 0, time.UTC) + if err := coordinator.recordSandboxWorkspace(context.Background(), first, "manifest-first", storagecatalog.StateStaging, now); err != nil { + t.Fatal(err) + } + if err := coordinator.recordSandboxWorkspace(context.Background(), second, "manifest-second", storagecatalog.StateStaging, now.Add(time.Minute)); err != nil { + t.Fatal(err) + } + store, err := storagecatalog.Open("") + if err != nil { + t.Fatal(err) + } + value, err := store.Load(now.Add(time.Minute)) + if err != nil { + t.Fatal(err) + } + firstTarget, err := storage.SnapshotFilesystemTarget(first) + if err != nil { + t.Fatal(err) + } + secondTarget, err := storage.SnapshotFilesystemTarget(second) + if err != nil { + t.Fatal(err) + } + firstResource := findCatalogResourceByLocator(t, value.Resources, firstTarget.Locator) + secondResource := findCatalogResourceByLocator(t, value.Resources, secondTarget.Locator) + if firstResource.State != storagecatalog.StateSuperseded || len(firstResource.References) != 0 || firstResource.SupersededAt == nil { + t.Fatalf("replaced staging workspace = %#v", firstResource) + } + if secondResource.State != storagecatalog.StateStaging || len(secondResource.References) != 1 || secondResource.References[0].Role != "template-staging" { + t.Fatalf("current staging workspace = %#v", secondResource) + } + if err := coordinator.recordSandboxWorkspace(context.Background(), second, "manifest-second", storagecatalog.StateSuperseded, now.Add(2*time.Minute)); err != nil { + t.Fatal(err) + } + value, err = store.Load(now.Add(2 * time.Minute)) + if err != nil { + t.Fatal(err) + } + secondResource = findCatalogResourceByLocator(t, value.Resources, secondTarget.Locator) + if secondResource.State != storagecatalog.StateSuperseded || len(secondResource.References) != 0 || secondResource.SupersededAt == nil { + t.Fatalf("completed staging workspace = %#v", secondResource) + } +} + +func findCatalogResourceByLocator(t *testing.T, resources []storagecatalog.Resource, locator string) storagecatalog.Resource { + t.Helper() + for _, resource := range resources { + if filepath.Clean(resource.Locator) == filepath.Clean(locator) { + return resource + } + } + t.Fatalf("catalog resource %q not found in %#v", locator, resources) + return storagecatalog.Resource{} +} + +func TestTartCurrentStateRequiresExactCatalogManifestReference(t *testing.T) { + value := storagecatalog.Catalog{Resources: []storagecatalog.Resource{{ + Kind: catalogTartImageKind, + Locator: "epar-tart-image", + Identity: "tart-mac:001122334455", + References: []storagecatalog.Reference{{ + ConfigID: "config-one", + Role: "provider-artifact", + ManifestHash: "manifest-one", + }}, + }}} + if !tartCatalogReferenceMatches(value, "config-one", "epar-tart-image", "tart-mac:001122334455", "manifest-one") { + t.Fatal("exact Tart catalog reference was not recognized") + } + for _, mismatch := range []struct { + configID string + locator string + identity string + manifestHash string + }{ + {configID: "config-two", locator: "epar-tart-image", identity: "tart-mac:001122334455", manifestHash: "manifest-one"}, + {configID: "config-one", locator: "other-image", identity: "tart-mac:001122334455", manifestHash: "manifest-one"}, + {configID: "config-one", locator: "epar-tart-image", identity: "tart-mac:aabbccddeeff", manifestHash: "manifest-one"}, + {configID: "config-one", locator: "epar-tart-image", identity: "tart-mac:001122334455", manifestHash: "manifest-two"}, + } { + if tartCatalogReferenceMatches(value, mismatch.configID, mismatch.locator, mismatch.identity, mismatch.manifestHash) { + t.Fatalf("mismatched Tart catalog reference was accepted: %+v", mismatch) + } + } +} diff --git a/internal/image/storage_plan.go b/internal/image/storage_plan.go new file mode 100644 index 0000000..6511d7a --- /dev/null +++ b/internal/image/storage_plan.go @@ -0,0 +1,193 @@ +package image + +import ( + "context" + "errors" + "fmt" + "math" + "os" + "path/filepath" + "strconv" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +const ( + ExpandedSizeFallbackMultiplier uint64 = 5 + CustomizationAllowanceBytes = 5 * storage.GiB + SandboxWritableHeadroomBytes = 20 * storage.GiB + SandboxRootRoundingBytes = 10 * storage.GiB + SandboxMinimumRootBytes = 20 * storage.GiB +) + +type EstimateConfidence string + +const ( + EstimateExact EstimateConfidence = "exact" + EstimateDerived EstimateConfidence = "derived" + EstimateFallback EstimateConfidence = "conservative-fallback" +) + +type SourceSizeEstimate struct { + CompressedBytes uint64 `json:"compressedBytes"` + ExpandedBytes uint64 `json:"expandedBytes"` + Confidence EstimateConfidence `json:"confidence"` +} + +type ArtifactStoragePlan struct { + Provider string `json:"provider"` + CompressedDownloadBytes uint64 `json:"compressedDownloadBytes"` + ExpandedSourceBytes uint64 `json:"expandedSourceBytes"` + CustomizationBytes uint64 `json:"customizationBytes"` + EstimatedIncrementalPeak uint64 `json:"estimatedIncrementalPeak"` + Confidence EstimateConfidence `json:"confidence"` + LogicalRootMaximumBytes uint64 `json:"logicalRootMaximumBytes,omitempty"` + LogicalDockerMaximumBytes uint64 `json:"logicalDockerMaximumBytes,omitempty"` + LogicalLimitsSparse bool `json:"logicalLimitsSparse,omitempty"` + Notes []string `json:"notes,omitempty"` +} + +func EstimateSourceSize(compressedBytes, exactExpandedBytes uint64) (SourceSizeEstimate, error) { + if compressedBytes == 0 && exactExpandedBytes == 0 { + return SourceSizeEstimate{}, errors.New("source size cannot be estimated without compressed or expanded bytes") + } + if exactExpandedBytes > 0 { + return SourceSizeEstimate{CompressedBytes: compressedBytes, ExpandedBytes: exactExpandedBytes, Confidence: EstimateExact}, nil + } + if compressedBytes > math.MaxUint64/ExpandedSizeFallbackMultiplier { + return SourceSizeEstimate{}, errors.New("expanded source-size estimate overflows uint64") + } + return SourceSizeEstimate{CompressedBytes: compressedBytes, ExpandedBytes: compressedBytes * ExpandedSizeFallbackMultiplier, Confidence: EstimateFallback}, nil +} + +func AutomaticDockerSandboxesRootBytes(expandedSourceBytes uint64) (uint64, error) { + if expandedSourceBytes > math.MaxUint64-CustomizationAllowanceBytes { + return 0, errors.New("Docker Sandboxes customization allowance overflows uint64") + } + required := expandedSourceBytes + CustomizationAllowanceBytes + if required > math.MaxUint64-SandboxWritableHeadroomBytes { + return 0, errors.New("Docker Sandboxes writable headroom overflows uint64") + } + required += SandboxWritableHeadroomBytes + if required < SandboxMinimumRootBytes { + required = SandboxMinimumRootBytes + } + remainder := required % SandboxRootRoundingBytes + if remainder == 0 { + return required, nil + } + increment := SandboxRootRoundingBytes - remainder + if required > math.MaxUint64-increment { + return 0, errors.New("Docker Sandboxes root-disk rounding overflows uint64") + } + return required + increment, nil +} + +func PlanArtifactStorage(providerType string, source SourceSizeEstimate, cached bool, dockerDiskBytes uint64) (ArtifactStoragePlan, error) { + plan := ArtifactStoragePlan{ + Provider: providerType, + CompressedDownloadBytes: source.CompressedBytes, + ExpandedSourceBytes: source.ExpandedBytes, + CustomizationBytes: CustomizationAllowanceBytes, + Confidence: source.Confidence, + } + if cached { + plan.EstimatedIncrementalPeak = 0 + plan.Notes = append(plan.Notes, "A verified current artifact can be reused.") + return plan, nil + } + add := func(value uint64) error { + if plan.EstimatedIncrementalPeak > math.MaxUint64-value { + return errors.New("artifact storage estimate overflows uint64") + } + plan.EstimatedIncrementalPeak += value + return nil + } + switch providerType { + case "docker-container": + if err := add(source.CompressedBytes); err != nil { + return ArtifactStoragePlan{}, err + } + if err := add(source.ExpandedBytes); err != nil { + return ArtifactStoragePlan{}, err + } + if err := add(CustomizationAllowanceBytes); err != nil { + return ArtifactStoragePlan{}, err + } + plan.Notes = append(plan.Notes, "Physical estimate covers Docker Engine source/output growth and customization.") + case "docker-sandboxes": + rootBytes, err := AutomaticDockerSandboxesRootBytes(source.ExpandedBytes) + if err != nil { + return ArtifactStoragePlan{}, err + } + for _, value := range []uint64{source.CompressedBytes, source.ExpandedBytes, source.ExpandedBytes, CustomizationAllowanceBytes} { + if err := add(value); err != nil { + return ArtifactStoragePlan{}, err + } + } + plan.LogicalRootMaximumBytes = rootBytes + plan.LogicalDockerMaximumBytes = dockerDiskBytes + plan.LogicalLimitsSparse = true + plan.Notes = append(plan.Notes, "Physical estimate covers dedicated BuildKit state, one directly exported archive, Sandbox template-cache import, and customization; no Docker Engine output image is created.", "The root and inner-Docker sizes are independent sparse logical limits and are not added to immediate host growth.") + case "wsl": + for _, value := range []uint64{source.CompressedBytes, source.ExpandedBytes, source.ExpandedBytes, CustomizationAllowanceBytes} { + if err := add(value); err != nil { + return ArtifactStoragePlan{}, err + } + } + plan.Notes = append(plan.Notes, "Physical estimate covers Docker Engine build data, rootfs export, temporary build distribution, and customization.") + default: + return ArtifactStoragePlan{}, fmt.Errorf("provider %q does not use the shared Docker-image storage plan", providerType) + } + return plan, nil +} + +func (m *Coordinator) configuredArtifactStoragePlan(ctx context.Context, cached bool) (ArtifactStoragePlan, error) { + if m.Config.Provider.Type == "tart" { + return ArtifactStoragePlan{ + Provider: m.Config.Provider.Type, + CustomizationBytes: CustomizationAllowanceBytes, + EstimatedIncrementalPeak: CustomizationAllowanceBytes, + Confidence: EstimateDerived, + Notes: []string{"Tart does not use the shared Docker-image plan; only the customization allowance is admitted here."}, + }, nil + } + if m.Config.Provider.Type == "wsl" && m.Config.Image.SourceType == config.ImageSourceRootFSTar { + sourcePath := config.ProjectPath(m.ProjectRoot, m.Config.Image.SourceImage) + info, err := os.Stat(filepath.Clean(sourcePath)) + if err != nil { + return ArtifactStoragePlan{}, fmt.Errorf("measure WSL rootfs source %s: %w", sourcePath, err) + } + if info.Size() < 0 { + return ArtifactStoragePlan{}, fmt.Errorf("measure WSL rootfs source %s: negative size", sourcePath) + } + estimate, err := EstimateSourceSize(uint64(info.Size()), uint64(info.Size())) + if err != nil { + return ArtifactStoragePlan{}, err + } + return PlanArtifactStorage(m.Config.Provider.Type, estimate, cached, 0) + } + source, err := m.resolveDockerSandboxesSource(ctx) + if err != nil { + return ArtifactStoragePlan{}, err + } + var exactExpanded uint64 + if output, inspectErr := m.runHostOutput(ctx, "docker", "image", "inspect", "--format", "{{.Size}}", source.Reference); inspectErr == nil { + exactExpanded, _ = strconv.ParseUint(strings.TrimSpace(output), 10, 64) + } + estimate, err := EstimateSourceSize(source.CompressedLayerBytes, exactExpanded) + if err != nil { + return ArtifactStoragePlan{}, err + } + dockerDiskBytes := uint64(0) + if m.Config.Provider.Type == "docker-sandboxes" { + parsed, err := config.ParseByteSize(m.Config.DockerSandboxes.DockerDisk) + if err != nil { + return ArtifactStoragePlan{}, err + } + dockerDiskBytes = uint64(parsed) + } + return PlanArtifactStorage(m.Config.Provider.Type, estimate, cached, dockerDiskBytes) +} diff --git a/internal/image/storage_plan_test.go b/internal/image/storage_plan_test.go new file mode 100644 index 0000000..b3bbd8e --- /dev/null +++ b/internal/image/storage_plan_test.go @@ -0,0 +1,76 @@ +package image + +import ( + "math" + "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +func TestEstimateSourceSizeUsesExactOrFiveTimesCompressed(t *testing.T) { + exact, err := EstimateSourceSize(2*storage.GiB, 7*storage.GiB) + if err != nil { + t.Fatal(err) + } + if exact.ExpandedBytes != 7*storage.GiB || exact.Confidence != EstimateExact { + t.Fatalf("exact estimate = %+v", exact) + } + fallback, err := EstimateSourceSize(2*storage.GiB, 0) + if err != nil { + t.Fatal(err) + } + if fallback.ExpandedBytes != 10*storage.GiB || fallback.Confidence != EstimateFallback { + t.Fatalf("fallback estimate = %+v", fallback) + } + if _, err := EstimateSourceSize(math.MaxUint64, 0); err == nil { + t.Fatal("EstimateSourceSize accepted overflowing fallback") + } +} + +func TestAutomaticDockerSandboxesRootTracksSourceSize(t *testing.T) { + act, err := AutomaticDockerSandboxesRootBytes(5 * storage.GiB) + if err != nil { + t.Fatal(err) + } + full, err := AutomaticDockerSandboxesRootBytes(75 * storage.GiB) + if err != nil { + t.Fatal(err) + } + if act != 30*storage.GiB { + t.Fatalf("act root = %d, want 30GiB", act) + } + if full != 100*storage.GiB { + t.Fatalf("full root = %d, want 100GiB", full) + } + if full <= act { + t.Fatalf("full root %d must exceed act root %d", full, act) + } +} + +func TestDockerSandboxesPlanDoesNotAddSparseLogicalLimitsToPhysicalPeak(t *testing.T) { + source := SourceSizeEstimate{CompressedBytes: 16 * storage.GiB, ExpandedBytes: 75 * storage.GiB, Confidence: EstimateExact} + plan, err := PlanArtifactStorage("docker-sandboxes", source, false, 50*storage.GiB) + if err != nil { + t.Fatal(err) + } + wantPhysical := 16*storage.GiB + 75*storage.GiB + 75*storage.GiB + CustomizationAllowanceBytes + if plan.EstimatedIncrementalPeak != wantPhysical { + t.Fatalf("physical peak = %d, want %d", plan.EstimatedIncrementalPeak, wantPhysical) + } + if plan.LogicalRootMaximumBytes != 100*storage.GiB || plan.LogicalDockerMaximumBytes != 50*storage.GiB || !plan.LogicalLimitsSparse { + t.Fatalf("logical limits = %+v", plan) + } +} + +func TestVerifiedCachedArtifactHasZeroIncrementalPeak(t *testing.T) { + source := SourceSizeEstimate{CompressedBytes: storage.GiB, ExpandedBytes: 5 * storage.GiB, Confidence: EstimateFallback} + for _, providerType := range []string{"docker-container", "docker-sandboxes", "wsl"} { + plan, err := PlanArtifactStorage(providerType, source, true, 50*storage.GiB) + if err != nil { + t.Fatal(err) + } + if plan.EstimatedIncrementalPeak != 0 { + t.Fatalf("%s cached peak = %d, want zero", providerType, plan.EstimatedIncrementalPeak) + } + } +} diff --git a/internal/image/tart_activation_test.go b/internal/image/tart_activation_test.go new file mode 100644 index 0000000..6de325e --- /dev/null +++ b/internal/image/tart_activation_test.go @@ -0,0 +1,154 @@ +package image + +import ( + "context" + "errors" + "fmt" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +type fakeTartActivationProvider struct { + instances map[string]provider.Instance + nextID int + failSource string + failTarget string +} + +func (p *fakeTartActivationProvider) Clone(_ context.Context, source, name string) error { + if source == p.failSource && name == p.failTarget { + return errors.New("injected clone failure") + } + if _, exists := p.instances[source]; !exists { + return fmt.Errorf("source %q does not exist", source) + } + p.nextID++ + p.instances[name] = provider.Instance{Name: name, Source: source, ProviderID: fmt.Sprintf("tart-mac:%012d", p.nextID), State: "stopped"} + return nil +} + +func (p *fakeTartActivationProvider) Start(context.Context, string, provider.StartOptions) (*provider.RunningProcess, error) { + return nil, errors.New("unexpected start") +} + +func (p *fakeTartActivationProvider) Exec(context.Context, string, []string, provider.ExecOptions) (provider.ExecResult, error) { + return provider.ExecResult{}, errors.New("unexpected exec") +} + +func (p *fakeTartActivationProvider) IP(context.Context, string, int) (string, error) { + return "", errors.New("unexpected IP") +} + +func (p *fakeTartActivationProvider) Stop(_ context.Context, name string) error { + instance, exists := p.instances[name] + if !exists { + return fmt.Errorf("instance %q does not exist", name) + } + instance.State = "stopped" + p.instances[name] = instance + return nil +} + +func (p *fakeTartActivationProvider) Delete(_ context.Context, name string) error { + if _, exists := p.instances[name]; !exists { + return fmt.Errorf("instance %q does not exist", name) + } + delete(p.instances, name) + return nil +} + +func (p *fakeTartActivationProvider) List(context.Context) ([]provider.Instance, error) { + result := make([]provider.Instance, 0, len(p.instances)) + for _, instance := range p.instances { + result = append(result, instance) + } + return result, nil +} + +func TestTartActivationRetainsRollbackUntilReplacementReadback(t *testing.T) { + t.Setenv("EPAR_STATE_HOME", filepath.Join(t.TempDir(), "state")) + t.Setenv("TART_HOME", filepath.Join(t.TempDir(), "tart")) + previous := provider.Instance{Name: "runner-image", ProviderID: "tart-mac:000000000001", State: "stopped"} + fake := &fakeTartActivationProvider{instances: map[string]provider.Instance{ + previous.Name: previous, + "runner-build-hash": {Name: "runner-build-hash", ProviderID: "tart-mac:000000000002", State: "stopped"}, + }, nextID: 2} + coordinator := Coordinator{Config: config.Default(), Provider: fake, ProjectRoot: t.TempDir()} + if err := coordinator.activateTartImage(context.Background(), previous, true, "runner-build-hash", previous.Name); err != nil { + t.Fatal(err) + } + if current, exists := fake.instances[previous.Name]; !exists || current.ProviderID == previous.ProviderID { + t.Fatalf("replacement was not activated: %#v", fake.instances) + } + if _, exists := fake.instances["runner-build-hash"]; exists { + t.Fatal("verified Tart build candidate was not retired after activation") + } + if _, exists := fake.instances[tartBackupName(previous.Name, previous.ProviderID)]; exists { + t.Fatal("Tart rollback image was retired only after readback but still remains") + } +} + +func TestTartActivationFailureRestoresPreviousGeneration(t *testing.T) { + t.Setenv("EPAR_STATE_HOME", filepath.Join(t.TempDir(), "state")) + t.Setenv("TART_HOME", filepath.Join(t.TempDir(), "tart")) + previous := provider.Instance{Name: "runner-image", ProviderID: "tart-mac:000000000001", State: "stopped"} + fake := &fakeTartActivationProvider{instances: map[string]provider.Instance{ + previous.Name: previous, + "runner-build-hash": {Name: "runner-build-hash", ProviderID: "tart-mac:000000000002", State: "stopped"}, + }, nextID: 2, failSource: "runner-build-hash", failTarget: previous.Name} + coordinator := Coordinator{Config: config.Default(), Provider: fake, ProjectRoot: t.TempDir()} + err := coordinator.activateTartImage(context.Background(), previous, true, "runner-build-hash", previous.Name) + if err == nil || !strings.Contains(err.Error(), "previous image was restored") { + t.Fatalf("activation error = %v", err) + } + if current, exists := fake.instances[previous.Name]; !exists || current.ProviderID == "" { + t.Fatalf("previous generation was not restored: %#v", fake.instances) + } +} + +func TestTartStartupReconciliationRestoresCatalogedRollbackBeforeCleanup(t *testing.T) { + t.Setenv("EPAR_STATE_HOME", filepath.Join(t.TempDir(), "state")) + t.Setenv("TART_HOME", filepath.Join(t.TempDir(), "tart")) + projectRoot := t.TempDir() + configPath := filepath.Join(projectRoot, "config.yml") + if err := os.WriteFile(configPath, []byte("provider:\n type: tart\n"), 0o600); err != nil { + t.Fatal(err) + } + backupName := "runner-image-epar-previous-000000000001" + fake := &fakeTartActivationProvider{instances: map[string]provider.Instance{ + backupName: {Name: backupName, ProviderID: "tart-mac:000000000002", State: "stopped"}, + }, nextID: 2} + cfg := config.Default() + cfg.Provider.Type = "tart" + cfg.Image.OutputImage = "runner-image" + coordinator := Coordinator{Config: cfg, Provider: fake, ProjectRoot: projectRoot, ConfigPath: configPath} + if err := coordinator.recordTartStagingImage(context.Background(), backupName, "activation-rollback"); err != nil { + t.Fatal(err) + } + if err := coordinator.reconcileInterruptedTartArtifacts(context.Background()); err != nil { + t.Fatal(err) + } + if current, exists := fake.instances[cfg.Image.OutputImage]; !exists || current.ProviderID == "" { + t.Fatalf("startup did not restore the configured Tart output: %#v", fake.instances) + } + store, err := storagecatalog.Open("") + if err != nil { + t.Fatal(err) + } + value, err := store.Load(time.Now().UTC()) + if err != nil { + t.Fatal(err) + } + for _, resource := range value.Resources { + if resource.Locator == backupName && (resource.State != storagecatalog.StateSuperseded || len(resource.References) != 0) { + t.Fatalf("restored rollback staging reference was not released: %#v", resource) + } + } +} diff --git a/internal/image/terminology_test.go b/internal/image/terminology_test.go new file mode 100644 index 0000000..4d04ff3 --- /dev/null +++ b/internal/image/terminology_test.go @@ -0,0 +1,45 @@ +package image + +import ( + "os" + "path/filepath" + "regexp" + "strings" + "testing" +) + +func TestLegacyDockerSandboxesDevelopmentLabelsDoNotReturn(t *testing.T) { + projectRoot := filepath.Clean(filepath.Join("..", "..")) + pattern := regexp.MustCompile(`(?i)\bcandidate[\s_-]*[ab]\b`) + allowedExtensions := map[string]bool{ + ".go": true, ".md": true, ".json": true, ".yml": true, ".yaml": true, ".ps1": true, ".sh": true, + } + err := filepath.WalkDir(projectRoot, func(path string, entry os.DirEntry, walkErr error) error { + if walkErr != nil { + return walkErr + } + if entry.IsDir() { + switch entry.Name() { + case ".git", ".local", "work": + if path != projectRoot { + return filepath.SkipDir + } + } + return nil + } + if !allowedExtensions[strings.ToLower(filepath.Ext(path))] { + return nil + } + content, err := os.ReadFile(path) + if err != nil { + return err + } + if pattern.Match(content) { + t.Errorf("legacy Docker Sandboxes development label found in %s", path) + } + return nil + }) + if err != nil { + t.Fatal(err) + } +} diff --git a/internal/pool/trusted_ca.go b/internal/image/trusted_ca.go similarity index 86% rename from internal/pool/trusted_ca.go rename to internal/image/trusted_ca.go index dde68f6..578ef36 100644 --- a/internal/pool/trusted_ca.go +++ b/internal/image/trusted_ca.go @@ -1,4 +1,4 @@ -package pool +package image import ( "bytes" @@ -19,13 +19,13 @@ import ( const trustedCAGuestDir = "/usr/local/share/ca-certificates/epar" -type trustedCACertificate struct { +type TrustedCACertificate struct { DestinationName string PEM []byte } -func (m *Manager) trustedCACertificates() ([]trustedCACertificate, error) { - byName := make(map[string]trustedCACertificate) +func (m *Coordinator) trustedCACertificates() ([]TrustedCACertificate, error) { + byName := make(map[string]TrustedCACertificate) for _, configuredPath := range m.Config.Image.TrustedCACertificatePaths { path := config.ProjectPath(m.ProjectRoot, strings.TrimSpace(configuredPath)) info, err := os.Stat(path) @@ -49,7 +49,7 @@ func (m *Manager) trustedCACertificates() ([]trustedCACertificate, error) { } sum := sha256.Sum256(certificate.Raw) name := "epar-" + hex.EncodeToString(sum[:12]) + ".crt" - byName[name] = trustedCACertificate{ + byName[name] = TrustedCACertificate{ DestinationName: name, PEM: pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: certificate.Raw}), } @@ -60,13 +60,18 @@ func (m *Manager) trustedCACertificates() ([]trustedCACertificate, error) { names = append(names, name) } sort.Strings(names) - out := make([]trustedCACertificate, 0, len(names)) + out := make([]TrustedCACertificate, 0, len(names)) for _, name := range names { out = append(out, byName[name]) } return out, nil } +// TrustedCACertificates returns validated CA material for provider-specific guest installation. +func (m *Coordinator) TrustedCACertificates() ([]TrustedCACertificate, error) { + return m.trustedCACertificates() +} + func parseTrustedCACertificateFile(content []byte) ([]*x509.Certificate, error) { trimmed := bytes.TrimSpace(content) if !bytes.HasPrefix(trimmed, []byte("-----BEGIN")) { @@ -105,7 +110,7 @@ func parseTrustedCACertificateFile(content []byte) ([]*x509.Certificate, error) return nil, fmt.Errorf("no PEM certificates found") } -func (m *Manager) copyTrustedCACertificatesToDir(destination string) error { +func (m *Coordinator) copyTrustedCACertificatesToDir(destination string) error { certificates, err := m.trustedCACertificates() if err != nil { return err @@ -121,7 +126,7 @@ func (m *Manager) copyTrustedCACertificatesToDir(destination string) error { return nil } -func (m *Manager) installTrustedCACertificates(ctx context.Context, vmName string) error { +func (m *Coordinator) installTrustedCACertificates(ctx context.Context, vmName string) error { certificates, err := m.trustedCACertificates() if err != nil { return err diff --git a/internal/image/update_policy.go b/internal/image/update_policy.go new file mode 100644 index 0000000..9abefc4 --- /dev/null +++ b/internal/image/update_policy.go @@ -0,0 +1,534 @@ +package image + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +const updatePolicyStateSchemaVersion = 1 + +// UpdatePolicyState is the per-configuration durable record for remote image +// and Actions runner freshness. Local desired inputs are represented by +// LocalInputHash and are never allowed to become stale merely because a remote +// check is not due. +type UpdatePolicyState struct { + SchemaVersion int `json:"schemaVersion"` + LocalInputHash string `json:"localInputHash"` + LastAttemptAt time.Time `json:"lastAttemptAt,omitempty"` + LastSuccessfulCheckAt time.Time `json:"lastSuccessfulCheckAt,omitempty"` + NextEligibleAt time.Time `json:"nextEligibleAt,omitempty"` + NextRetryAt time.Time `json:"nextRetryAt,omitempty"` + ConsecutiveFailures int `json:"consecutiveFailures,omitempty"` + TimeZone string `json:"timeZone,omitempty"` + PolicyFrequency string `json:"policyFrequency,omitempty"` + PolicyTime string `json:"policyTime,omitempty"` + LastResolvedManifest *Manifest `json:"lastResolvedManifest,omitempty"` + LastResolvedSource *ResolvedDockerSource `json:"lastResolvedSource,omitempty"` + PendingManifest *Manifest `json:"pendingManifest,omitempty"` + PendingSource *ResolvedDockerSource `json:"pendingSource,omitempty"` + DeferredReason string `json:"deferredReason,omitempty"` + LastError string `json:"lastError,omitempty"` +} + +// UpdatePolicyStatus is safe to expose through status and console output. +type UpdatePolicyStatus struct { + Frequency string + UpdateTime string + LastSuccessfulCheckAt time.Time + NextEligibleAt time.Time + NextRetryAt time.Time + Pending bool + PendingIdentity string + DeferredReason string + LastError string +} + +type RemoteUpdateCheck struct { + Due bool + Changed bool + CurrentManifest *Manifest + PendingManifest *Manifest + NextEligibleAt time.Time + NextRetryAt time.Time +} + +func UpdatePolicyStatePath(projectRoot, configPath string) (string, error) { + if strings.TrimSpace(configPath) == "" { + configPath = filepath.Join(projectRoot, ".local", "config.yml") + } + configID, err := storagecatalog.ConfigID(projectRoot, configPath) + if err != nil { + return "", err + } + return filepath.Join(projectRoot, ".local", "state", "image", configID, "update-policy.json"), nil +} + +func ReadUpdatePolicyState(projectRoot, configPath string) (UpdatePolicyState, error) { + path, err := UpdatePolicyStatePath(projectRoot, configPath) + if err != nil { + return UpdatePolicyState{}, err + } + content, err := os.ReadFile(path) + if err != nil { + return UpdatePolicyState{}, err + } + var state UpdatePolicyState + if err := json.Unmarshal(content, &state); err != nil { + return UpdatePolicyState{}, fmt.Errorf("parse image update state: %w", err) + } + if state.SchemaVersion != updatePolicyStateSchemaVersion { + return UpdatePolicyState{}, fmt.Errorf("unsupported image update state schema %d", state.SchemaVersion) + } + return state, nil +} + +func writeUpdatePolicyState(projectRoot, configPath string, state UpdatePolicyState) error { + path, err := UpdatePolicyStatePath(projectRoot, configPath) + if err != nil { + return err + } + state.SchemaVersion = updatePolicyStateSchemaVersion + content, err := json.MarshalIndent(state, "", " ") + if err != nil { + return err + } + return writeAtomicFile(path, append(content, '\n'), 0o600) +} + +func (m *Coordinator) readUpdatePolicyState() (UpdatePolicyState, error) { + state, err := ReadUpdatePolicyState(m.ProjectRoot, m.effectiveConfigPath()) + if errors.Is(err, os.ErrNotExist) { + return UpdatePolicyState{SchemaVersion: updatePolicyStateSchemaVersion}, nil + } + return state, err +} + +func (m *Coordinator) writeUpdatePolicyState(state UpdatePolicyState) error { + return writeUpdatePolicyState(m.ProjectRoot, m.effectiveConfigPath(), state) +} + +func bootstrapUpdatePolicyState(state *UpdatePolicyState, image config.ImageConfig, localManifest, resolvedManifest Manifest, source *ResolvedDockerSource, activatedAt time.Time, location *time.Location) (bool, error) { + if state.LocalInputHash != "" || state.LastResolvedManifest != nil || activatedAt.IsZero() { + return false, nil + } + localHash, err := ManifestHash(localManifest) + if err != nil { + return false, err + } + projected := resolvedManifest + projected.SourceImage = localManifest.SourceImage + projected.SourcePlatform = localManifest.SourcePlatform + projected.SourceDigest = localManifest.SourceDigest + projected.SourcePlatformDigest = localManifest.SourcePlatformDigest + projected.RunnerSelector = localManifest.RunnerSelector + projected.RunnerVersion = localManifest.RunnerVersion + projected.RunnerAssetName = localManifest.RunnerAssetName + projected.RunnerAssetURL = localManifest.RunnerAssetURL + projected.RunnerAssetDigest = localManifest.RunnerAssetDigest + projectedHash, err := ManifestHash(projected) + if err != nil { + return false, err + } + if projectedHash != localHash { + return false, nil + } + if location == nil { + location = time.Local + } + state.LocalInputHash = localHash + state.LastResolvedManifest = &resolvedManifest + if source != nil { + copy := *source + state.LastResolvedSource = © + } + if err := scheduleNextSuccess(state, image, activatedAt.In(location)); err != nil { + return false, err + } + return true, nil +} + +func (m *Coordinator) UpdatePolicyStatus() (UpdatePolicyStatus, error) { + state, err := m.readUpdatePolicyState() + if err != nil { + return UpdatePolicyStatus{}, err + } + recalculateScheduleForTimeZone(&state, m.Config.Image, time.Local) + pendingIdentity := "" + if state.PendingManifest != nil { + hash, hashErr := ManifestHash(*state.PendingManifest) + if hashErr != nil { + return UpdatePolicyStatus{}, hashErr + } + pendingIdentity = shortUpdateIdentity(*state.PendingManifest, hash) + } + return UpdatePolicyStatus{ + Frequency: m.Config.Image.UpdateFrequency, + UpdateTime: m.Config.Image.UpdateTime, + LastSuccessfulCheckAt: state.LastSuccessfulCheckAt, + NextEligibleAt: state.NextEligibleAt, + NextRetryAt: state.NextRetryAt, + Pending: state.PendingManifest != nil, + PendingIdentity: pendingIdentity, + DeferredReason: state.DeferredReason, + LastError: state.LastError, + }, nil +} + +// CheckRemoteUpdate performs only the cheap immutable remote observation. It +// never builds, imports, activates, or retires an artifact. +func (m *Coordinator) CheckRemoteUpdate(ctx context.Context, now time.Time) (RemoteUpdateCheck, error) { + state, err := m.readUpdatePolicyState() + if err != nil { + return RemoteUpdateCheck{}, err + } + if recalculateScheduleForTimeZone(&state, m.Config.Image, now.Location()) { + if err := m.writeUpdatePolicyState(state); err != nil { + return RemoteUpdateCheck{}, err + } + } + if !updateCheckDue(state, m.Config.Image, now) { + return RemoteUpdateCheck{ + CurrentManifest: state.LastResolvedManifest, + PendingManifest: state.PendingManifest, + NextEligibleAt: state.NextEligibleAt, + NextRetryAt: state.NextRetryAt, + }, nil + } + var ( + localManifest Manifest + manifest Manifest + source *ResolvedDockerSource + ) + if m.Config.Provider.Type == "docker-sandboxes" { + localManifest, err = m.dockerSandboxesLocalManifest(ctx) + if err == nil { + var resolved ResolvedDockerSource + manifest, resolved, err = m.dockerSandboxesDesiredManifest(ctx) + source = &resolved + } + } else { + localManifest, err = m.desiredLocalImageManifest(ctx) + if err == nil { + manifest, err = m.resolveRemoteImageManifest(ctx, localManifest) + } + } + if err != nil { + scheduleUpdateFailure(&state, now, err) + _ = m.writeUpdatePolicyState(state) + return RemoteUpdateCheck{Due: true, NextRetryAt: state.NextRetryAt}, err + } + localHash, err := ManifestHash(localManifest) + if err != nil { + return RemoteUpdateCheck{}, err + } + if state.LocalInputHash != localHash { + return RemoteUpdateCheck{}, fmt.Errorf("local image inputs changed while the pool was running; restart EPAR to apply the new configuration safely") + } + resolvedHash, err := ManifestHash(manifest) + if err != nil { + return RemoteUpdateCheck{}, err + } + currentHash := "" + if state.LastResolvedManifest != nil { + currentHash, err = ManifestHash(*state.LastResolvedManifest) + if err != nil { + return RemoteUpdateCheck{}, err + } + } + if currentHash == resolvedHash { + if err := scheduleNextSuccess(&state, m.Config.Image, now); err != nil { + return RemoteUpdateCheck{}, err + } + state.PendingManifest = nil + state.PendingSource = nil + if err := m.writeUpdatePolicyState(state); err != nil { + return RemoteUpdateCheck{}, err + } + return RemoteUpdateCheck{ + Due: true, + CurrentManifest: state.LastResolvedManifest, + NextEligibleAt: state.NextEligibleAt, + }, nil + } + state.LastAttemptAt = now.UTC() + state.PendingManifest = &manifest + state.PendingSource = source + state.DeferredReason = "waiting for the common pool maintenance drain" + state.LastError = "" + if err := m.writeUpdatePolicyState(state); err != nil { + return RemoteUpdateCheck{}, err + } + return RemoteUpdateCheck{ + Due: true, + Changed: true, + CurrentManifest: state.LastResolvedManifest, + PendingManifest: &manifest, + }, nil +} + +// ApplyPendingUpdate performs the build and activation selected by +// CheckRemoteUpdate after the common pool lifecycle has drained all runners. +func (m *Coordinator) ApplyPendingUpdate(ctx context.Context, now time.Time) error { + state, err := m.readUpdatePolicyState() + if err != nil { + return err + } + if state.PendingManifest == nil { + return nil + } + manifest := *state.PendingManifest + if m.Config.Provider.Type == "docker-sandboxes" { + if state.PendingSource == nil { + return fmt.Errorf("pending Docker Sandboxes update is missing its immutable source observation") + } + err = m.ensureDockerSandboxesTemplateResolved(ctx, false, manifest, *state.PendingSource) + } else { + err = m.buildResolvedImage(ctx, manifest) + if err == nil { + hash, hashErr := ManifestHash(manifest) + if hashErr != nil { + err = hashErr + } else if recordErr := m.recordCurrentArtifact(ctx, hash); recordErr != nil { + err = recordErr + } + } + } + if err != nil { + scheduleUpdateFailure(&state, now, err) + state.DeferredReason = "scheduled build failed; the previous verified generation remains active" + _ = m.writeUpdatePolicyState(state) + return err + } + state.LastResolvedManifest = &manifest + state.LastResolvedSource = state.PendingSource + state.PendingManifest = nil + state.PendingSource = nil + if err := scheduleNextSuccess(&state, m.Config.Image, now); err != nil { + return err + } + if err := m.writeUpdatePolicyState(state); err != nil { + return err + } + return m.cleanupSupersededCatalog(ctx) +} + +func (m *Coordinator) DeferPendingUpdate(reason string) error { + state, err := m.readUpdatePolicyState() + if err != nil { + return err + } + if state.PendingManifest == nil { + return nil + } + state.DeferredReason = reason + return m.writeUpdatePolicyState(state) +} + +func NextImageUpdateAt(last time.Time, frequency, wallClock string, location *time.Location) (time.Time, error) { + if frequency == config.ImageUpdateFrequencyManual { + return time.Time{}, nil + } + if location == nil { + location = time.Local + } + clock, err := time.Parse("15:04", wallClock) + if err != nil { + return time.Time{}, fmt.Errorf("parse image update time: %w", err) + } + localLast := last.In(location) + year, month, day := localLast.Date() + switch frequency { + case config.ImageUpdateFrequencyDaily: + day++ + case config.ImageUpdateFrequencyWeekly: + day += 7 + case config.ImageUpdateFrequencyBiweekly: + day += 14 + case config.ImageUpdateFrequencyMonthly: + year, month, day = addCalendarMonthClamped(year, month, day) + default: + return time.Time{}, fmt.Errorf("unsupported image update frequency %q", frequency) + } + next := time.Date(year, month, day, clock.Hour(), clock.Minute(), 0, 0, location) + return normalizeLocalWallClock(next, clock.Hour(), clock.Minute(), location), nil +} + +func addCalendarMonthClamped(year int, month time.Month, day int) (int, time.Month, int) { + targetMonth := month + 1 + targetYear := year + if targetMonth > time.December { + targetMonth = time.January + targetYear++ + } + lastDay := time.Date(targetYear, targetMonth+1, 0, 12, 0, 0, 0, time.UTC).Day() + if day > lastDay { + day = lastDay + } + return targetYear, targetMonth, day +} + +// normalizeLocalWallClock handles DST gaps by selecting the first valid local +// instant after the requested wall-clock time. Repeated times use Go's earlier +// occurrence, which is deterministic and still runs only once because state is +// persisted immediately after the check. +func normalizeLocalWallClock(candidate time.Time, hour, minute int, location *time.Location) time.Time { + local := candidate.In(location) + if local.Hour() == hour && local.Minute() == minute { + return candidate + } + for i := 0; i < 180; i++ { + candidate = candidate.Add(time.Minute) + local = candidate.In(location) + if local.Hour() > hour || (local.Hour() == hour && local.Minute() >= minute) { + return candidate + } + } + return candidate +} + +func updateCheckDue(state UpdatePolicyState, image config.ImageConfig, now time.Time) bool { + if image.UpdateFrequency == config.ImageUpdateFrequencyManual { + return false + } + if state.LastResolvedManifest != nil && !manifestHasMutableRemoteInputs(*state.LastResolvedManifest) { + return false + } + if !state.NextRetryAt.IsZero() && now.Before(state.NextRetryAt) { + return false + } + if state.LastSuccessfulCheckAt.IsZero() || state.NextEligibleAt.IsZero() { + return true + } + return !now.Before(state.NextEligibleAt) +} + +func pendingUpdateReady(state UpdatePolicyState, now time.Time) bool { + if state.PendingManifest == nil { + return false + } + return state.NextRetryAt.IsZero() || !now.Before(state.NextRetryAt) +} + +func scheduleNextSuccess(state *UpdatePolicyState, image config.ImageConfig, now time.Time) error { + location := now.Location() + if location == nil || location == time.UTC { + location = time.Local + } + next, err := NextImageUpdateAt(now, image.UpdateFrequency, image.UpdateTime, location) + if err != nil { + return err + } + if state.LastResolvedManifest != nil && !manifestHasMutableRemoteInputs(*state.LastResolvedManifest) { + next = time.Time{} + } + state.LastAttemptAt = now.UTC() + state.LastSuccessfulCheckAt = now.UTC() + state.NextEligibleAt = next.UTC() + state.NextRetryAt = time.Time{} + state.ConsecutiveFailures = 0 + state.TimeZone = location.String() + state.PolicyFrequency = image.UpdateFrequency + state.PolicyTime = image.UpdateTime + state.LastError = "" + state.DeferredReason = "" + return nil +} + +func manifestHasMutableRemoteInputs(manifest Manifest) bool { + if normalizedRunnerSelector(manifest.RunnerSelector) == "latest" { + return true + } + if manifest.SourceType == config.ImageSourceDockerImage && !strings.Contains(strings.ToLower(manifest.SourceImage), "@sha256:") { + return true + } + return false +} + +func recalculateScheduleForTimeZone(state *UpdatePolicyState, image config.ImageConfig, location *time.Location) bool { + if location == nil { + location = time.Local + } + policyChanged := state.PolicyFrequency != image.UpdateFrequency || state.PolicyTime != image.UpdateTime + if image.UpdateFrequency == config.ImageUpdateFrequencyManual { + if !policyChanged && state.NextEligibleAt.IsZero() && state.TimeZone == location.String() { + return false + } + state.PolicyFrequency = image.UpdateFrequency + state.PolicyTime = image.UpdateTime + state.TimeZone = location.String() + state.NextEligibleAt = time.Time{} + return true + } + if state.LastSuccessfulCheckAt.IsZero() { + if !policyChanged && state.TimeZone == location.String() { + return false + } + state.PolicyFrequency = image.UpdateFrequency + state.PolicyTime = image.UpdateTime + state.TimeZone = location.String() + return true + } + if !policyChanged && state.TimeZone == location.String() { + return false + } + next, err := NextImageUpdateAt(state.LastSuccessfulCheckAt, image.UpdateFrequency, image.UpdateTime, location) + if err != nil { + return false + } + state.TimeZone = location.String() + state.PolicyFrequency = image.UpdateFrequency + state.PolicyTime = image.UpdateTime + state.NextEligibleAt = next.UTC() + return true +} + +func shortUpdateIdentity(manifest Manifest, hash string) string { + parts := []string{"manifest=" + shortDigest(hash)} + if manifest.SourcePlatformDigest != "" { + parts = append(parts, "source="+shortDigest(manifest.SourcePlatformDigest)) + } else if manifest.SourceDigest != "" { + parts = append(parts, "source="+shortDigest(manifest.SourceDigest)) + } + if manifest.RunnerVersion != "" { + parts = append(parts, "runner="+manifest.RunnerVersion) + } + return strings.Join(parts, " ") +} + +func shortDigest(value string) string { + value = strings.TrimSpace(strings.TrimPrefix(value, "sha256:")) + if len(value) > 16 { + value = value[:16] + } + return value +} + +func scheduleUpdateFailure(state *UpdatePolicyState, now time.Time, failure error) { + state.LastAttemptAt = now.UTC() + state.ConsecutiveFailures++ + delay := time.Hour + for i := 1; i < state.ConsecutiveFailures && delay < 24*time.Hour; i++ { + delay *= 2 + } + if delay > 24*time.Hour { + delay = 24 * time.Hour + } + state.NextRetryAt = now.Add(delay).UTC() + state.LastError = failure.Error() +} + +func formatUpdateTime(value time.Time) string { + if value.IsZero() { + return "manual" + } + return value.In(time.Local).Format("2006-01-02 15:04 MST") +} diff --git a/internal/image/update_policy_test.go b/internal/image/update_policy_test.go new file mode 100644 index 0000000..1def271 --- /dev/null +++ b/internal/image/update_policy_test.go @@ -0,0 +1,282 @@ +package image + +import ( + "errors" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" +) + +func TestNextImageUpdateAtIntervals(t *testing.T) { + location := time.FixedZone("test", 8*60*60) + last := time.Date(2026, time.July, 30, 13, 45, 0, 0, location) + tests := []struct { + frequency string + want time.Time + }{ + {config.ImageUpdateFrequencyDaily, time.Date(2026, time.July, 31, 7, 0, 0, 0, location)}, + {config.ImageUpdateFrequencyWeekly, time.Date(2026, time.August, 6, 7, 0, 0, 0, location)}, + {config.ImageUpdateFrequencyBiweekly, time.Date(2026, time.August, 13, 7, 0, 0, 0, location)}, + {config.ImageUpdateFrequencyMonthly, time.Date(2026, time.August, 30, 7, 0, 0, 0, location)}, + } + for _, test := range tests { + got, err := NextImageUpdateAt(last, test.frequency, "07:00", location) + if err != nil { + t.Fatalf("%s: %v", test.frequency, err) + } + if !got.Equal(test.want) { + t.Fatalf("%s next = %s, want %s", test.frequency, got, test.want) + } + } +} + +func TestNextImageUpdateAtClampsCalendarMonth(t *testing.T) { + location := time.UTC + tests := []struct { + last time.Time + want time.Time + }{ + { + time.Date(2025, time.January, 31, 12, 0, 0, 0, location), + time.Date(2025, time.February, 28, 7, 0, 0, 0, location), + }, + { + time.Date(2024, time.January, 31, 12, 0, 0, 0, location), + time.Date(2024, time.February, 29, 7, 0, 0, 0, location), + }, + { + time.Date(2026, time.December, 31, 12, 0, 0, 0, location), + time.Date(2027, time.January, 31, 7, 0, 0, 0, location), + }, + } + for _, test := range tests { + got, err := NextImageUpdateAt(test.last, config.ImageUpdateFrequencyMonthly, "07:00", location) + if err != nil { + t.Fatal(err) + } + if !got.Equal(test.want) { + t.Fatalf("next = %s, want %s", got, test.want) + } + } +} + +func TestNextImageUpdateAtManual(t *testing.T) { + got, err := NextImageUpdateAt(time.Now(), config.ImageUpdateFrequencyManual, "07:00", time.Local) + if err != nil { + t.Fatal(err) + } + if !got.IsZero() { + t.Fatalf("manual next = %s, want zero", got) + } +} + +func TestUpdateFailureBackoffIsBounded(t *testing.T) { + now := time.Date(2026, time.July, 30, 7, 0, 0, 0, time.UTC) + state := UpdatePolicyState{} + for attempt := 1; attempt <= 8; attempt++ { + scheduleUpdateFailure(&state, now, errors.New("registry unavailable")) + delay := state.NextRetryAt.Sub(now) + if delay > 24*time.Hour { + t.Fatalf("attempt %d delay = %s, want at most 24h", attempt, delay) + } + } + if state.NextRetryAt.Sub(now) != 24*time.Hour { + t.Fatalf("final delay = %s, want 24h", state.NextRetryAt.Sub(now)) + } +} + +func TestUpdateCheckDueHonorsManualAndRetry(t *testing.T) { + now := time.Date(2026, time.July, 30, 7, 0, 0, 0, time.UTC) + image := config.Default().Image + state := UpdatePolicyState{} + if !updateCheckDue(state, image, now) { + t.Fatal("new automatic policy should be due") + } + image.UpdateFrequency = config.ImageUpdateFrequencyManual + if updateCheckDue(state, image, now) { + t.Fatal("manual policy should not be due") + } + image.UpdateFrequency = config.ImageUpdateFrequencyWeekly + state.NextRetryAt = now.Add(time.Hour) + if updateCheckDue(state, image, now) { + t.Fatal("retry backoff should defer check") + } +} + +func TestUpdateCheckDueSkipsImmutableRemoteInputs(t *testing.T) { + now := time.Date(2026, time.July, 30, 7, 0, 0, 0, time.UTC) + image := config.Default().Image + state := UpdatePolicyState{ + LastSuccessfulCheckAt: now.Add(-8 * 24 * time.Hour), + NextEligibleAt: now.Add(-24 * time.Hour), + LastResolvedManifest: &Manifest{ + SourceType: config.ImageSourceDockerImage, + SourceImage: "ghcr.io/example/runner@sha256:" + strings.Repeat("a", 64), + RunnerSelector: "2.332.0", + }, + } + if updateCheckDue(state, image, now) { + t.Fatal("immutable source and pinned runner should not schedule a remote check") + } + state.LastResolvedManifest.RunnerSelector = "latest" + if !updateCheckDue(state, image, now) { + t.Fatal("runnerVersion latest should remain scheduled") + } +} + +func TestNextImageUpdateAtHandlesDSTGapAndRepeat(t *testing.T) { + location, err := time.LoadLocation("America/New_York") + if err != nil { + t.Skipf("timezone database unavailable: %v", err) + } + gap, err := NextImageUpdateAt(time.Date(2026, time.March, 7, 7, 0, 0, 0, location), config.ImageUpdateFrequencyDaily, "02:30", location) + if err != nil { + t.Fatal(err) + } + gapLocal := gap.In(location) + if gapLocal.Year() != 2026 || gapLocal.Month() != time.March || gapLocal.Day() != 8 || gapLocal.Hour() != 3 || gapLocal.Minute() != 0 { + t.Fatalf("DST gap next = %s, want first valid local instant at 03:00", gapLocal) + } + repeat, err := NextImageUpdateAt(time.Date(2026, time.October, 31, 7, 0, 0, 0, location), config.ImageUpdateFrequencyDaily, "01:30", location) + if err != nil { + t.Fatal(err) + } + repeatLocal := repeat.In(location) + if repeatLocal.Year() != 2026 || repeatLocal.Month() != time.November || repeatLocal.Day() != 1 || repeatLocal.Hour() != 1 || repeatLocal.Minute() != 30 { + t.Fatalf("DST repeat next = %s, want local 01:30", repeatLocal) + } +} + +func TestRecalculateScheduleForTimeZone(t *testing.T) { + singapore := time.FixedZone("Asia/Singapore", 8*60*60) + tokyo := time.FixedZone("Asia/Tokyo", 9*60*60) + success := time.Date(2026, time.July, 30, 7, 0, 0, 0, singapore) + state := UpdatePolicyState{ + LastSuccessfulCheckAt: success.UTC(), + NextEligibleAt: success.AddDate(0, 0, 7).UTC(), + TimeZone: singapore.String(), + PolicyFrequency: config.ImageUpdateFrequencyWeekly, + PolicyTime: "07:00", + } + image := config.Default().Image + if !recalculateScheduleForTimeZone(&state, image, tokyo) { + t.Fatal("timezone change did not recalculate the next check") + } + want := time.Date(2026, time.August, 6, 7, 0, 0, 0, tokyo) + if !state.NextEligibleAt.Equal(want) { + t.Fatalf("next after timezone change = %s, want %s", state.NextEligibleAt, want) + } +} + +func TestRecalculateScheduleAfterPolicyChange(t *testing.T) { + location := time.FixedZone("local", 8*60*60) + success := time.Date(2026, time.July, 30, 7, 0, 0, 0, location) + state := UpdatePolicyState{ + LastSuccessfulCheckAt: success.UTC(), + NextEligibleAt: success.AddDate(0, 0, 7).UTC(), + TimeZone: location.String(), + PolicyFrequency: config.ImageUpdateFrequencyWeekly, + PolicyTime: "07:00", + } + image := config.Default().Image + image.UpdateFrequency = config.ImageUpdateFrequencyDaily + image.UpdateTime = "06:30" + if !recalculateScheduleForTimeZone(&state, image, location) { + t.Fatal("policy change did not recalculate the next check") + } + want := time.Date(2026, time.July, 31, 6, 30, 0, 0, location) + if !state.NextEligibleAt.Equal(want) { + t.Fatalf("next after policy change = %s, want %s", state.NextEligibleAt, want) + } + image.UpdateFrequency = config.ImageUpdateFrequencyManual + if !recalculateScheduleForTimeZone(&state, image, location) || !state.NextEligibleAt.IsZero() { + t.Fatalf("manual policy did not clear next automatic check: %+v", state) + } +} + +func TestBootstrapUpdatePolicyStateFromMatchingVerifiedArtifact(t *testing.T) { + location := time.FixedZone("local", 8*60*60) + activated := time.Date(2026, time.July, 1, 7, 0, 0, 0, location) + local := Manifest{ + SchemaVersion: ManifestSchemaVersion, + ProviderType: "docker-sandboxes", + SourceType: config.ImageSourceDockerImage, + SourceImage: "ghcr.io/catthehacker/ubuntu:act-latest", + RunnerSelector: "latest", + EPARScripts: []FileDigest{{Path: "runtime.sh", SHA256: "local"}}, + } + resolved := local + resolved.SourcePlatform = "linux/amd64" + resolved.SourceDigest = "ghcr.io/catthehacker/ubuntu@sha256:" + strings.Repeat("a", 64) + resolved.SourcePlatformDigest = "sha256:" + strings.Repeat("b", 64) + resolved.RunnerVersion = "2.332.0" + resolved.RunnerAssetName = "actions-runner-linux-x64-2.332.0.tar.gz" + resolved.RunnerAssetURL = "https://example.invalid/runner.tar.gz" + resolved.RunnerAssetDigest = "sha256:" + strings.Repeat("c", 64) + source := ResolvedDockerSource{Reference: local.SourceImage, Platform: "linux/amd64", PlatformDigest: resolved.SourcePlatformDigest} + state := UpdatePolicyState{SchemaVersion: updatePolicyStateSchemaVersion} + image := config.ImageConfig{UpdateFrequency: config.ImageUpdateFrequencyWeekly, UpdateTime: "07:00"} + + bootstrapped, err := bootstrapUpdatePolicyState(&state, image, local, resolved, &source, activated, location) + if err != nil { + t.Fatal(err) + } + if !bootstrapped || state.LastResolvedManifest == nil || state.LastResolvedSource == nil { + t.Fatalf("bootstrap result = %t, state = %+v", bootstrapped, state) + } + if got := state.NextEligibleAt.In(location); !got.Equal(activated.AddDate(0, 0, 7)) { + t.Fatalf("next eligible = %v, want %v", got, activated.AddDate(0, 0, 7)) + } + + changed := local + changed.EPARScripts = []FileDigest{{Path: "runtime.sh", SHA256: "changed"}} + rejected := UpdatePolicyState{SchemaVersion: updatePolicyStateSchemaVersion} + bootstrapped, err = bootstrapUpdatePolicyState(&rejected, image, changed, resolved, &source, activated, location) + if err != nil { + t.Fatal(err) + } + if bootstrapped || rejected.LastResolvedManifest != nil { + t.Fatalf("changed local inputs bootstrapped stale artifact: %+v", rejected) + } +} + +func TestUpdatePolicyStateUsesPerConfigPathAndAtomicJSON(t *testing.T) { + root := t.TempDir() + configA := filepath.Join(root, ".local", "a.yml") + configB := filepath.Join(root, ".local", "b.yml") + if err := os.MkdirAll(filepath.Dir(configA), 0o700); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(configA, []byte("provider:\n type: docker-container\n"), 0o600); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(configB, []byte("provider:\n type: wsl\n"), 0o600); err != nil { + t.Fatal(err) + } + pathA, err := UpdatePolicyStatePath(root, configA) + if err != nil { + t.Fatal(err) + } + pathB, err := UpdatePolicyStatePath(root, configB) + if err != nil { + t.Fatal(err) + } + if pathA == pathB { + t.Fatalf("different configs share update-policy path %q", pathA) + } + state := UpdatePolicyState{LocalInputHash: "local", LastError: "registry unavailable"} + if err := writeUpdatePolicyState(root, configA, state); err != nil { + t.Fatal(err) + } + got, err := ReadUpdatePolicyState(root, configA) + if err != nil { + t.Fatal(err) + } + if got.SchemaVersion != updatePolicyStateSchemaVersion || got.LocalInputHash != "local" || got.LastError != "registry unavailable" { + t.Fatalf("read state = %+v", got) + } +} diff --git a/internal/image/verified_download.go b/internal/image/verified_download.go new file mode 100644 index 0000000..a5f7769 --- /dev/null +++ b/internal/image/verified_download.go @@ -0,0 +1,142 @@ +package image + +import ( + "context" + "crypto/sha256" + "crypto/tls" + "crypto/x509" + "encoding/hex" + "fmt" + "io" + "net/http" + "os" + "path/filepath" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" +) + +func verifiedDownload(ctx context.Context, client *http.Client, sourceURL, destination, expectedSHA256 string, mode os.FileMode) error { + expectedSHA256 = strings.ToLower(strings.TrimPrefix(strings.TrimSpace(expectedSHA256), "sha256:")) + if len(expectedSHA256) != sha256.Size*2 { + return fmt.Errorf("invalid locked SHA-256 for %s", sourceURL) + } + if ok, err := fileMatchesSHA256(destination, expectedSHA256); err != nil { + return err + } else if ok { + return nil + } + if err := os.MkdirAll(filepath.Dir(destination), 0o700); err != nil { + return err + } + partial := destination + ".partial" + var offset int64 + if info, err := os.Lstat(partial); err == nil { + if !info.Mode().IsRegular() { + return fmt.Errorf("partial download %s is not a regular file", partial) + } + offset = info.Size() + } else if !os.IsNotExist(err) { + return err + } + request, err := http.NewRequestWithContext(ctx, http.MethodGet, sourceURL, nil) + if err != nil { + return err + } + if offset > 0 { + request.Header.Set("Range", fmt.Sprintf("bytes=%d-", offset)) + } + response, err := client.Do(request) + if err != nil { + return fmt.Errorf("download %s: %w", sourceURL, err) + } + defer response.Body.Close() + flags := os.O_CREATE | os.O_WRONLY + if response.StatusCode == http.StatusPartialContent && offset > 0 { + flags |= os.O_APPEND + } else { + offset = 0 + flags |= os.O_TRUNC + } + if response.StatusCode != http.StatusOK && response.StatusCode != http.StatusPartialContent { + return fmt.Errorf("download %s: HTTP %s", sourceURL, response.Status) + } + file, err := os.OpenFile(partial, flags, 0o600) + if err != nil { + return err + } + _, copyErr := io.Copy(file, response.Body) + syncErr := file.Sync() + closeErr := file.Close() + if copyErr != nil { + return fmt.Errorf("download %s after %d bytes: %w", sourceURL, offset, copyErr) + } + if syncErr != nil { + return syncErr + } + if closeErr != nil { + return closeErr + } + ok, err := fileMatchesSHA256(partial, expectedSHA256) + if err != nil { + return err + } + if !ok { + return fmt.Errorf("download %s failed locked SHA-256 verification; partial content retained at %s", sourceURL, partial) + } + if err := os.Chmod(partial, mode); err != nil { + return err + } + if runtimeRenameReplace(partial, destination); err != nil { + return err + } + return nil +} + +func fileMatchesSHA256(path, expected string) (bool, error) { + info, err := os.Lstat(path) + if os.IsNotExist(err) { + return false, nil + } + if err != nil { + return false, err + } + if !info.Mode().IsRegular() { + return false, fmt.Errorf("download target %s is not a regular file", path) + } + file, err := os.Open(path) + if err != nil { + return false, err + } + defer file.Close() + hash := sha256.New() + if _, err := io.Copy(hash, file); err != nil { + return false, err + } + return hex.EncodeToString(hash.Sum(nil)) == expected, nil +} + +func buildTrustHTTPClient(snapshot hosttrust.Snapshot) (*http.Client, error) { + roots, err := x509.SystemCertPool() + if err != nil || roots == nil { + roots = x509.NewCertPool() + } + for _, certificate := range snapshot.Certificates { + if !roots.AppendCertsFromPEM(certificate.PEM) { + return nil, fmt.Errorf("append operational build CA %s", certificate.Name) + } + } + transport := http.DefaultTransport.(*http.Transport).Clone() + transport.TLSClientConfig = &tls.Config{MinVersion: tls.VersionTLS12, RootCAs: roots} + return &http.Client{Transport: transport}, nil +} + +func runtimeRenameReplace(source, destination string) error { + if err := os.Rename(source, destination); err == nil { + return nil + } + if err := os.Remove(destination); err != nil && !os.IsNotExist(err) { + return err + } + return os.Rename(source, destination) +} diff --git a/internal/image/verified_download_test.go b/internal/image/verified_download_test.go new file mode 100644 index 0000000..de1a116 --- /dev/null +++ b/internal/image/verified_download_test.go @@ -0,0 +1,79 @@ +package image + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "fmt" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strconv" + "strings" + "testing" +) + +func TestVerifiedDownloadResumesAndPublishesOnlyLockedContent(t *testing.T) { + content := []byte(strings.Repeat("verified-actions-runner-content\n", 128)) + sum := sha256.Sum256(content) + var requestedRange string + server := httptest.NewServer(http.HandlerFunc(func(response http.ResponseWriter, request *http.Request) { + requestedRange = request.Header.Get("Range") + start := 0 + if requestedRange != "" { + if _, err := fmt.Sscanf(requestedRange, "bytes=%d-", &start); err != nil { + t.Errorf("invalid Range header %q: %v", requestedRange, err) + response.WriteHeader(http.StatusBadRequest) + return + } + response.Header().Set("Content-Range", fmt.Sprintf("bytes %d-%d/%d", start, len(content)-1, len(content))) + response.WriteHeader(http.StatusPartialContent) + } + _, _ = response.Write(content[start:]) + })) + defer server.Close() + + destination := filepath.Join(t.TempDir(), "inputs", "actions-runner.tar.gz") + if err := os.MkdirAll(filepath.Dir(destination), 0o700); err != nil { + t.Fatal(err) + } + partialBytes := 197 + if err := os.WriteFile(destination+".partial", content[:partialBytes], 0o600); err != nil { + t.Fatal(err) + } + if err := verifiedDownload(context.Background(), server.Client(), server.URL, destination, hex.EncodeToString(sum[:]), 0o600); err != nil { + t.Fatal(err) + } + if requestedRange != "bytes="+strconv.Itoa(partialBytes)+"-" { + t.Fatalf("Range = %q, want resume from %d", requestedRange, partialBytes) + } + got, err := os.ReadFile(destination) + if err != nil { + t.Fatal(err) + } + if string(got) != string(content) { + t.Fatal("verified download content does not match") + } + if _, err := os.Stat(destination + ".partial"); !os.IsNotExist(err) { + t.Fatalf("partial download still exists after publication: %v", err) + } +} + +func TestVerifiedDownloadRetainsFailedPartialAndDoesNotPublish(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(response http.ResponseWriter, _ *http.Request) { + _, _ = response.Write([]byte("wrong content")) + })) + defer server.Close() + destination := filepath.Join(t.TempDir(), "tini") + err := verifiedDownload(context.Background(), server.Client(), server.URL, destination, strings.Repeat("0", 64), 0o700) + if err == nil || !strings.Contains(err.Error(), "failed locked SHA-256 verification") { + t.Fatalf("checksum failure = %v", err) + } + if _, err := os.Stat(destination); !os.IsNotExist(err) { + t.Fatalf("unverified destination was published: %v", err) + } + if _, err := os.Stat(destination + ".partial"); err != nil { + t.Fatalf("failed partial was not retained for diagnosis/resume: %v", err) + } +} diff --git a/internal/image/wsl_artifact.go b/internal/image/wsl_artifact.go new file mode 100644 index 0000000..84abf61 --- /dev/null +++ b/internal/image/wsl_artifact.go @@ -0,0 +1,180 @@ +package image + +import ( + "errors" + "fmt" + "os" +) + +const wslPreviousArtifactSuffix = ".epar-previous" + +func activateWSLArtifact(candidatePath, outputPath string) (err error) { + candidateSidecar := wslImageManifestSidecarPath(candidatePath) + outputSidecar := wslImageManifestSidecarPath(outputPath) + previousPath := outputPath + wslPreviousArtifactSuffix + previousSidecar := outputSidecar + wslPreviousArtifactSuffix + + if err := requireRegularFile(candidatePath); err != nil { + return fmt.Errorf("candidate archive: %w", err) + } + if err := requireRegularFile(candidateSidecar); err != nil { + return fmt.Errorf("candidate manifest: %w", err) + } + if err := recoverWSLArtifactSwap(outputPath); err != nil { + return err + } + + hadOutput, err := regularFileExists(outputPath) + if err != nil { + return err + } + hadSidecar, err := regularFileExists(outputSidecar) + if err != nil { + return err + } + if hadOutput != hadSidecar { + return fmt.Errorf("existing WSL artifact is incomplete: archive=%t manifest=%t", hadOutput, hadSidecar) + } + + rollbackNeeded := false + rollback := func() { + _ = removeRegularFileIfPresent(outputPath) + _ = removeRegularFileIfPresent(outputSidecar) + if hadOutput { + _ = os.Rename(previousPath, outputPath) + } + if hadSidecar { + _ = os.Rename(previousSidecar, outputSidecar) + } + } + defer func() { + if err != nil && rollbackNeeded { + rollback() + } + }() + + if hadOutput { + if err = os.Rename(outputPath, previousPath); err != nil { + return err + } + if err = os.Rename(outputSidecar, previousSidecar); err != nil { + _ = os.Rename(previousPath, outputPath) + return err + } + } + rollbackNeeded = true + if err = os.Rename(candidatePath, outputPath); err != nil { + return err + } + if err = os.Rename(candidateSidecar, outputSidecar); err != nil { + return err + } + if err = requireRegularFile(outputPath); err != nil { + return err + } + if _, err = readStoredImageManifest(outputSidecar); err != nil { + return err + } + rollbackNeeded = false + if err = removeRegularFileIfPresent(previousPath); err != nil { + return err + } + if err = removeRegularFileIfPresent(previousSidecar); err != nil { + return err + } + return nil +} + +func recoverWSLArtifactSwap(outputPath string) error { + outputSidecar := wslImageManifestSidecarPath(outputPath) + previousPath := outputPath + wslPreviousArtifactSuffix + previousSidecar := outputSidecar + wslPreviousArtifactSuffix + + hasPrevious, err := regularFileExists(previousPath) + if err != nil { + return err + } + hasPreviousSidecar, err := regularFileExists(previousSidecar) + if err != nil { + return err + } + if !hasPrevious && !hasPreviousSidecar { + return nil + } + hasOutput, err := regularFileExists(outputPath) + if err != nil { + return err + } + hasOutputSidecar, err := regularFileExists(outputSidecar) + if err != nil { + return err + } + if hasOutput && hasOutputSidecar { + if _, readErr := readStoredImageManifest(outputSidecar); readErr == nil { + if err := removeRegularFileIfPresent(previousPath); err != nil { + return err + } + return removeRegularFileIfPresent(previousSidecar) + } + } + if hasPrevious && !hasPreviousSidecar { + if !hasOutput && hasOutputSidecar { + return os.Rename(previousPath, outputPath) + } + return fmt.Errorf("incomplete WSL activation recovery evidence: previous archive without its manifest") + } + if !hasPrevious && hasPreviousSidecar { + return fmt.Errorf("incomplete WSL activation recovery evidence: previous manifest without its archive") + } + + if err := removeRegularFileIfPresent(outputPath); err != nil { + return err + } + if err := removeRegularFileIfPresent(outputSidecar); err != nil { + return err + } + if err := os.Rename(previousPath, outputPath); err != nil { + return err + } + if err := os.Rename(previousSidecar, outputSidecar); err != nil { + _ = os.Rename(outputPath, previousPath) + return err + } + return nil +} + +func requireRegularFile(path string) error { + info, err := os.Lstat(path) + if err != nil { + return err + } + if !info.Mode().IsRegular() { + return fmt.Errorf("%s is not a regular file", path) + } + return nil +} + +func regularFileExists(path string) (bool, error) { + info, err := os.Lstat(path) + if errors.Is(err, os.ErrNotExist) { + return false, nil + } + if err != nil { + return false, err + } + if !info.Mode().IsRegular() { + return false, fmt.Errorf("%s is not a regular file", path) + } + return true, nil +} + +func removeRegularFileIfPresent(path string) error { + exists, err := regularFileExists(path) + if err != nil { + return err + } + if !exists { + return nil + } + return os.Remove(path) +} diff --git a/internal/image/wsl_artifact_test.go b/internal/image/wsl_artifact_test.go new file mode 100644 index 0000000..5024177 --- /dev/null +++ b/internal/image/wsl_artifact_test.go @@ -0,0 +1,97 @@ +package image + +import ( + "os" + "path/filepath" + "testing" +) + +func TestActivateWSLArtifactReplacesVerifiedPair(t *testing.T) { + dir := t.TempDir() + output := filepath.Join(dir, "runner.tar") + candidate := output + ".epar-candidate-test" + writeWSLArtifactTestPair(t, output, "old", ImageManifest{SchemaVersion: 1, OutputImage: "old"}) + writeWSLArtifactTestPair(t, candidate, "new", ImageManifest{SchemaVersion: 1, OutputImage: "new"}) + + if err := activateWSLArtifact(candidate, output); err != nil { + t.Fatal(err) + } + content, err := os.ReadFile(output) + if err != nil { + t.Fatal(err) + } + if string(content) != "new" { + t.Fatalf("active archive = %q, want new", content) + } + stored, err := readStoredImageManifest(wslImageManifestSidecarPath(output)) + if err != nil { + t.Fatal(err) + } + if stored.Manifest.OutputImage != "new" { + t.Fatalf("active manifest output = %q, want new", stored.Manifest.OutputImage) + } + if _, err := os.Stat(output + wslPreviousArtifactSuffix); !os.IsNotExist(err) { + t.Fatalf("previous archive still exists: %v", err) + } +} + +func TestRecoverWSLArtifactSwapRestoresPreviousPair(t *testing.T) { + dir := t.TempDir() + output := filepath.Join(dir, "runner.tar") + previous := output + wslPreviousArtifactSuffix + previousSidecar := wslImageManifestSidecarPath(output) + wslPreviousArtifactSuffix + writeWSLArtifactTestPair(t, previous, "old", ImageManifest{SchemaVersion: 1, OutputImage: "old"}) + if err := os.Rename(wslImageManifestSidecarPath(previous), previousSidecar); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(output, []byte("partial-new"), 0o600); err != nil { + t.Fatal(err) + } + + if err := recoverWSLArtifactSwap(output); err != nil { + t.Fatal(err) + } + content, err := os.ReadFile(output) + if err != nil { + t.Fatal(err) + } + if string(content) != "old" { + t.Fatalf("recovered archive = %q, want old", content) + } + stored, err := readStoredImageManifest(wslImageManifestSidecarPath(output)) + if err != nil { + t.Fatal(err) + } + if stored.Manifest.OutputImage != "old" { + t.Fatalf("recovered manifest output = %q, want old", stored.Manifest.OutputImage) + } +} + +func TestActivateWSLArtifactRejectsSymlinkCandidate(t *testing.T) { + dir := t.TempDir() + target := filepath.Join(dir, "target.tar") + if err := os.WriteFile(target, []byte("candidate"), 0o600); err != nil { + t.Fatal(err) + } + candidate := filepath.Join(dir, "candidate.tar") + if err := os.Symlink(target, candidate); err != nil { + t.Skipf("symlinks unavailable: %v", err) + } + if err := writeStoredImageManifest(wslImageManifestSidecarPath(candidate), ImageManifest{SchemaVersion: 1}); err != nil { + t.Fatal(err) + } + + if err := activateWSLArtifact(candidate, filepath.Join(dir, "output.tar")); err == nil { + t.Fatal("activateWSLArtifact accepted a symlink candidate") + } +} + +func writeWSLArtifactTestPair(t *testing.T, output, content string, manifest ImageManifest) { + t.Helper() + if err := os.WriteFile(output, []byte(content), 0o600); err != nil { + t.Fatal(err) + } + if err := writeStoredImageManifest(wslImageManifestSidecarPath(output), manifest); err != nil { + t.Fatal(err) + } +} diff --git a/internal/invocation/invocation.go b/internal/invocation/invocation.go new file mode 100644 index 0000000..7587e06 --- /dev/null +++ b/internal/invocation/invocation.go @@ -0,0 +1,58 @@ +package invocation + +import ( + "os" + "path/filepath" + "strings" +) + +// Environment identifies the user-facing entry point selected by an EPAR +// wrapper. Values are deliberately closed so an inherited environment variable +// cannot inject arbitrary text into a suggested command. +const Environment = "EPAR_INVOCATION" + +// Command returns a command line that uses the same entry point as the current +// process. +func Command(args ...string) string { + parts := append([]string{commandPrefix(os.Getenv(Environment), os.Args[0], executablePath())}, args...) + return strings.Join(parts, " ") +} + +func executablePath() string { + path, err := os.Executable() + if err != nil { + return "" + } + return path +} + +func commandPrefix(marker, arg0, executable string) string { + switch marker { + case "start": + return "./start" + case "run-with-docker": + return "scripts/run-with-docker.sh" + case "run-with-docker-powershell": + return `scripts\run-with-docker.ps1` + } + if isGoRunExecutable(executable) { + return "go run ./cmd/ephemeral-action-runner" + } + if strings.TrimSpace(arg0) == "" { + return "ephemeral-action-runner" + } + if strings.ContainsAny(arg0, " \t") { + return `"` + arg0 + `"` + } + return arg0 +} + +func isGoRunExecutable(path string) bool { + normalized := strings.ToLower(filepath.ToSlash(strings.ReplaceAll(path, `\`, "/"))) + for _, segment := range strings.Split(normalized, "/") { + if strings.HasPrefix(segment, "go-build") { + return true + } + } + return false +} diff --git a/internal/invocation/invocation_test.go b/internal/invocation/invocation_test.go new file mode 100644 index 0000000..3b6d6a2 --- /dev/null +++ b/internal/invocation/invocation_test.go @@ -0,0 +1,35 @@ +package invocation + +import "testing" + +func TestCommandPrefix(t *testing.T) { + tests := []struct { + name string + marker string + arg0 string + executable string + want string + }{ + {name: "start wrapper", marker: "start", arg0: "ignored", want: "./start"}, + {name: "docker shell wrapper", marker: "run-with-docker", arg0: "ignored", want: "scripts/run-with-docker.sh"}, + {name: "docker PowerShell wrapper", marker: "run-with-docker-powershell", arg0: "ignored", want: `scripts\run-with-docker.ps1`}, + {name: "go run Unix", arg0: "/tmp/go-build123/b001/exe/ephemeral-action-runner", executable: "/tmp/go-build123/b001/exe/ephemeral-action-runner", want: "go run ./cmd/ephemeral-action-runner"}, + {name: "go run Windows", arg0: `C:\Temp\go-build123\b001\exe\ephemeral-action-runner.exe`, executable: `C:\Temp\go-build123\b001\exe\ephemeral-action-runner.exe`, want: "go run ./cmd/ephemeral-action-runner"}, + {name: "direct binary", arg0: `.\bin\ephemeral-action-runner.exe`, executable: `C:\repo\bin\ephemeral-action-runner.exe`, want: `.\bin\ephemeral-action-runner.exe`}, + {name: "direct binary with spaces", arg0: `C:\EPAR Tools\ephemeral-action-runner.exe`, executable: `C:\EPAR Tools\ephemeral-action-runner.exe`, want: `"C:\EPAR Tools\ephemeral-action-runner.exe"`}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + if got := commandPrefix(test.marker, test.arg0, test.executable); got != test.want { + t.Fatalf("commandPrefix() = %q, want %q", got, test.want) + } + }) + } +} + +func TestUnknownMarkerCannotReplaceCommand(t *testing.T) { + got := commandPrefix("arbitrary command", "ephemeral-action-runner", `C:\bin\ephemeral-action-runner.exe`) + if got != "ephemeral-action-runner" { + t.Fatalf("commandPrefix() = %q, want direct binary", got) + } +} diff --git a/internal/logging/paths.go b/internal/logging/paths.go index c2f1607..8530c93 100644 --- a/internal/logging/paths.go +++ b/internal/logging/paths.go @@ -90,7 +90,7 @@ func validBuildComponent(component string) bool { func validInstanceComponent(component string) bool { switch component { - case "guest", "docker-dind", "wsl", "tart": + case "guest", "docker-container", "docker-sandboxes", "wsl", "tart": return true default: return false diff --git a/internal/logging/paths_test.go b/internal/logging/paths_test.go index 4221055..f538fd5 100644 --- a/internal/logging/paths_test.go +++ b/internal/logging/paths_test.go @@ -13,12 +13,13 @@ func TestPathHelpersAndRecognition(t *testing.T) { path string category Category }{ - {mustPath(InstancePath(root, "runner-1", "docker-dind")), CategoryInstances}, + {mustPath(InstancePath(root, "runner-1", "docker-container")), CategoryInstances}, + {mustPath(InstancePath(root, "runner-1", "docker-sandboxes")), CategoryInstances}, {mustPath(InstancePath(root, "runner-1", "guest")), CategoryInstances}, {mustPath(BuildPath(root, "ubuntu-24.04", "docker-build")), CategoryBuilds}, {mustPath(BuildPath(root, "ubuntu-24.04", "guest")), CategoryBuilds}, {ErrorPath(root, timestamp), CategoryErrors}, - {mustPath(BenchmarkPath(root, timestamp, "docker-dind")), CategoryBenchmarks}, + {mustPath(BenchmarkPath(root, timestamp, "docker-container")), CategoryBenchmarks}, } for _, test := range tests { recognized, ok := recognizePath(root, test.path) @@ -56,11 +57,11 @@ func TestLegacyFlatRecognitionIsConstrained(t *testing.T) { category Category }{ {"epar-pool-20260715-010203-007.guest.log", CategoryInstances}, - {"epar-pool-20260715-010203-007.docker-dind.log", CategoryInstances}, + {"epar-pool-20260715-010203-007.docker-container.log", CategoryInstances}, {"ubuntu.docker-build.log", CategoryBuilds}, {"ubuntu.source.log", CategoryBuilds}, {"epar-20260715-010203-error.log", CategoryErrors}, - {"20260715T010203.000000123Z-docker-dind.jsonl", CategoryBenchmarks}, + {"20260715T010203.000000123Z-docker-container.jsonl", CategoryBenchmarks}, {"epar-2026-07-15T01-02-03.004.log.gz", CategoryManager}, } for _, test := range tests { diff --git a/internal/logging/recognition.go b/internal/logging/recognition.go index d1b5cab..3ebdcfb 100644 --- a/internal/logging/recognition.go +++ b/internal/logging/recognition.go @@ -9,10 +9,10 @@ import ( var ( lumberjackSuffixPattern = regexp.MustCompile(`-\d{4}-\d{2}-\d{2}[Tt]\d{2}-\d{2}-\d{2}\.\d{3}(\.log|\.jsonl)$`) errorNamePattern = regexp.MustCompile(`^epar-\d{8}-\d{6}-error\.log$`) - instanceNamePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]*\.(guest|docker-dind|wsl|tart)\.log$`) - legacyInstancePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]*-\d{8}-\d{6}-\d{3}\.(guest|docker-dind|wsl|tart)\.log$`) - buildNamePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]*\.(docker-build|wsl-build|build|source|refresh|wsl-refresh|guest)\.log$`) - legacyBuildNamePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]*\.(docker-build|wsl-build|build|source|refresh|wsl-refresh)\.log$`) + instanceNamePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]*\.(guest|docker-container|docker-sandboxes|wsl|tart)\.log$`) + legacyInstancePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]*-\d{8}-\d{6}-\d{3}\.(guest|docker-container|docker-sandboxes|wsl|tart)\.log$`) + buildNamePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]*\.(docker-build|docker-pull|wsl-build|build|source|refresh|wsl-refresh|guest)\.log$`) + legacyBuildNamePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]*\.(docker-build|docker-pull|wsl-build|build|source|refresh|wsl-refresh)\.log$`) benchmarkNamePattern = regexp.MustCompile(`^\d{8}[Tt]\d{6}\.\d{9}[Zz]-[A-Za-z0-9][A-Za-z0-9_-]*\.jsonl$`) ) diff --git a/internal/logging/runtime_test.go b/internal/logging/runtime_test.go index c6d22f6..cbe8b32 100644 --- a/internal/logging/runtime_test.go +++ b/internal/logging/runtime_test.go @@ -72,7 +72,7 @@ func TestDefaultManagerTextFormatIsHumanReadable(t *testing.T) { if err != nil { t.Fatal(err) } - runtime.Manager().Info("runner ready", "provider", "docker-dind", "attempt", 2) + runtime.Manager().Info("runner ready", "provider", "docker-container", "attempt", 2) if err := runtime.Close(); err != nil { t.Fatal(err) } @@ -222,11 +222,11 @@ func TestTranscriptCoalescesRawWritesAndFlushesContextualPartialLines(t *testing if err != nil { t.Fatalf("NewRuntime: %v", err) } - path, err := InstancePath(root, "runner-1", "docker-dind") + path, err := InstancePath(root, "runner-1", "docker-container") if err != nil { t.Fatalf("InstancePath: %v", err) } - transcript, err := runtime.OpenTranscript(TranscriptMetadata{SessionID: "session-1", Category: CategoryInstances, Instance: "runner-1", Component: "provider", Provider: "docker-dind"}, path) + transcript, err := runtime.OpenTranscript(TranscriptMetadata{SessionID: "session-1", Category: CategoryInstances, Instance: "runner-1", Component: "provider", Provider: "docker-container"}, path) if err != nil { t.Fatalf("OpenTranscript: %v", err) } diff --git a/internal/pool/architecture_test.go b/internal/pool/architecture_test.go new file mode 100644 index 0000000..3cf3c16 --- /dev/null +++ b/internal/pool/architecture_test.go @@ -0,0 +1,71 @@ +package pool + +import ( + "go/parser" + "go/token" + "os" + "path/filepath" + "runtime" + "strconv" + "strings" + "testing" +) + +func TestPoolDoesNotImportProviderImplementations(t *testing.T) { + _, filename, _, ok := runtime.Caller(0) + if !ok { + t.Fatal("could not locate pool package") + } + root := filepath.Dir(filename) + err := filepath.WalkDir(root, func(path string, entry os.DirEntry, walkErr error) error { + if walkErr != nil { + return walkErr + } + if entry.IsDir() || filepath.Ext(path) != ".go" || strings.HasSuffix(path, "_test.go") { + return nil + } + parsed, err := parser.ParseFile(token.NewFileSet(), path, nil, parser.ImportsOnly) + if err != nil { + return err + } + for _, imported := range parsed.Imports { + value, err := strconv.Unquote(imported.Path.Value) + if err != nil { + return err + } + if strings.HasPrefix(value, "github.com/solutionforest/ephemeral-action-runner/internal/provider/") { + t.Errorf("%s imports provider implementation %q; pool may depend only on provider contracts", path, value) + } + } + return nil + }) + if err != nil { + t.Fatal(err) + } +} + +func TestImageImplementationsStayOutsidePool(t *testing.T) { + _, filename, _, ok := runtime.Caller(0) + if !ok { + t.Fatal("could not locate pool package") + } + poolRoot := filepath.Dir(filename) + internalRoot := filepath.Dir(poolRoot) + for _, name := range []string{"image.go", "artifact_lifecycle.go", "image_acquisition.go", "trusted_ca.go", "docker_pull.go", "image_manifest.go"} { + path := filepath.Join(poolRoot, name) + if _, err := os.Stat(path); !os.IsNotExist(err) { + t.Errorf("image implementation must not live in pool: %s", path) + } + } + for _, name := range []string{"build.go", "artifact_lifecycle.go", "acquisition.go", "trusted_ca.go", "docker_pull.go", "manifest.go"} { + path := filepath.Join(internalRoot, "image", name) + info, err := os.Stat(path) + if err != nil { + t.Errorf("required image implementation %s: %v", path, err) + continue + } + if info.IsDir() { + t.Errorf("image implementation is not a file: %s", path) + } + } +} diff --git a/internal/pool/controller_lock.go b/internal/pool/controller_lock.go index c0c259a..820aada 100644 --- a/internal/pool/controller_lock.go +++ b/internal/pool/controller_lock.go @@ -3,45 +3,159 @@ package pool import ( "crypto/sha256" "encoding/hex" + "encoding/json" + "errors" "fmt" "io" "os" "path/filepath" - "runtime" "strings" + "time" - "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + "github.com/solutionforest/ephemeral-action-runner/internal/filelock" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" ) -// AcquirePoolControllerLock excludes another mutating controller for the same -// canonical configuration, provider, and pool prefix on this host. +const poolControllerLockDirectory = "pool-controller-locks" + +type poolControllerLockOwner struct { + ConfigPath string `json:"configPath"` + Provider string `json:"provider"` + NamePrefix string `json:"namePrefix"` + PID int `json:"pid"` + StartedAt time.Time `json:"startedAt"` +} + +type poolControllerLocks struct { + config *filelock.Lock + prefix *filelock.Lock +} + +func (locks *poolControllerLocks) Close() error { + if locks == nil { + return nil + } + var result error + if locks.prefix != nil { + result = errors.Join(result, locks.prefix.Close()) + } + if locks.config != nil { + result = errors.Join(result, locks.config.Close()) + } + return result +} + +// AcquirePoolControllerLock excludes another mutating controller that uses +// either this canonical configuration path or this normalized pool prefix. +// Config locking prevents a second controller from bypassing an active one by +// editing provider or prefix fields in place; prefix locking protects global +// provider and GitHub runner identities across configurations and projects. func (m *Manager) AcquirePoolControllerLock() (io.Closer, error) { providerType := strings.TrimSpace(strings.ToLower(m.Config.Provider.Type)) namePrefix := strings.TrimSpace(strings.ToLower(m.Config.Pool.NamePrefix)) if providerType == "" || namePrefix == "" { return nil, fmt.Errorf("acquire pool controller lock: provider.type and pool.namePrefix are required") } + canonicalConfig, err := m.canonicalPoolControllerConfigPath() + if err != nil { + return nil, err + } + root, err := storagecatalog.DefaultRoot() + if err != nil { + return nil, fmt.Errorf("resolve EPAR state root for pool controller lock: %w", err) + } + lockRoot := filepath.Join(root, poolControllerLockDirectory) + if err := os.MkdirAll(lockRoot, 0o700); err != nil { + return nil, fmt.Errorf("create pool controller lock directory: %w", err) + } + configPath := poolControllerLockPath(lockRoot, "config", canonicalConfig) + prefixPath := poolControllerLockPath(lockRoot, "prefix", namePrefix) + configLock, err := filelock.Acquire(configPath) + if err != nil { + return nil, poolControllerLockError("configuration", canonicalConfig, configPath, err) + } + prefixLock, err := filelock.Acquire(prefixPath) + if err != nil { + _ = configLock.Close() + return nil, poolControllerLockError("pool.namePrefix", namePrefix, prefixPath, err) + } + locks := &poolControllerLocks{config: configLock, prefix: prefixLock} + owner := poolControllerLockOwner{ + ConfigPath: canonicalConfig, + Provider: providerType, + NamePrefix: namePrefix, + PID: os.Getpid(), + StartedAt: time.Now().UTC(), + } + if err := writePoolControllerLockOwner(configLock, owner); err != nil { + _ = locks.Close() + return nil, err + } + if err := writePoolControllerLockOwner(prefixLock, owner); err != nil { + _ = locks.Close() + return nil, err + } + if m.LifecycleStateEnabled && m.LifecycleState == nil { + lifecycleState, err := OpenLifecycleState(m.ProjectRoot, m.ConfigPath) + if err != nil { + _ = locks.Close() + return nil, fmt.Errorf("open lifecycle state after acquiring pool controller locks: %w", err) + } + m.LifecycleState = lifecycleState + } + return locks, nil +} + +func (m *Manager) canonicalPoolControllerConfigPath() (string, error) { configPath := strings.TrimSpace(m.ConfigPath) if configPath == "" { configPath = filepath.Join(m.ProjectRoot, ".local", "config.yml") } - canonicalConfig, err := filepath.Abs(configPath) + canonicalConfig, err := storagecatalog.CanonicalPath(configPath) + if err != nil { + return "", fmt.Errorf("acquire pool controller lock: resolve config path: %w", err) + } + return canonicalConfig, nil +} + +func poolControllerLockPath(root, kind, identity string) string { + sum := sha256.Sum256([]byte(identity)) + return filepath.Join(root, kind+"-"+hex.EncodeToString(sum[:])+".lock") +} + +func writePoolControllerLockOwner(lock *filelock.Lock, owner poolControllerLockOwner) error { + content, err := json.Marshal(owner) if err != nil { - return nil, fmt.Errorf("acquire pool controller lock: resolve config path: %w", err) + return fmt.Errorf("encode pool controller lock owner: %w", err) } - if resolved, resolveErr := filepath.EvalSymlinks(canonicalConfig); resolveErr == nil { - canonicalConfig = resolved + if err := lock.ReplaceContent(append(content, '\n')); err != nil { + return fmt.Errorf("write pool controller lock owner metadata: %w", err) } - canonicalConfig = filepath.Clean(canonicalConfig) - if runtime.GOOS == "windows" { - canonicalConfig = strings.ToLower(canonicalConfig) + return nil +} + +func poolControllerLockError(kind, identity, path string, lockErr error) error { + if !errors.Is(lockErr, filelock.ErrLocked) { + return fmt.Errorf("acquire pool controller %s lock for %q: %w", kind, identity, lockErr) } - identity := canonicalConfig + "\x00" + providerType + "\x00" + namePrefix - sum := sha256.Sum256([]byte(identity)) - syntheticPath := filepath.Join(os.TempDir(), "ephemeral-action-runner", "pool-controller", hex.EncodeToString(sum[:])+".identity") - lock, err := hosttrust.AcquireConfigLock(syntheticPath) + message := fmt.Sprintf("pool controller %s lock is already held for %q", kind, identity) + if owner, err := readPoolControllerLockOwner(path); err == nil { + message += fmt.Sprintf(" (owner config=%q provider=%q prefix=%q pid=%d startedAt=%s)", owner.ConfigPath, owner.Provider, owner.NamePrefix, owner.PID, owner.StartedAt.UTC().Format(time.RFC3339)) + } + return errors.New(message) +} + +func readPoolControllerLockOwner(path string) (poolControllerLockOwner, error) { + content, err := os.ReadFile(path) if err != nil { - return nil, fmt.Errorf("acquire pool controller lock for provider %q prefix %q: %w", providerType, namePrefix, err) + return poolControllerLockOwner{}, err + } + var owner poolControllerLockOwner + if err := json.Unmarshal(content, &owner); err != nil { + return poolControllerLockOwner{}, err + } + if owner.ConfigPath == "" || owner.Provider == "" || owner.NamePrefix == "" || owner.PID <= 0 || owner.StartedAt.IsZero() { + return poolControllerLockOwner{}, errors.New("invalid pool controller lock owner metadata") } - return lock, nil + return owner, nil } diff --git a/internal/pool/controller_lock_test.go b/internal/pool/controller_lock_test.go index 93a8fa8..818776e 100644 --- a/internal/pool/controller_lock_test.go +++ b/internal/pool/controller_lock_test.go @@ -2,15 +2,23 @@ package pool import ( "context" + "encoding/json" + "errors" + "io" + "os" + "path/filepath" "strings" + "sync" "testing" + "time" "github.com/solutionforest/ephemeral-action-runner/internal/config" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" ) -func TestPoolControllerLockConflictsForSameProviderAndPrefix(t *testing.T) { - t.Setenv("LOCALAPPDATA", t.TempDir()) - manager := Manager{ConfigPath: "config.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-dind"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-same"}}} +func TestPoolControllerLockConflictsForSameConfig(t *testing.T) { + setPoolControllerLockStateHome(t) + manager := testPoolControllerManager(t, t.TempDir(), "config.yml", "docker-container", "epar-lock-same") first, err := manager.AcquirePoolControllerLock() if err != nil { t.Fatal(err) @@ -18,13 +26,229 @@ func TestPoolControllerLockConflictsForSameProviderAndPrefix(t *testing.T) { defer first.Close() if second, err := manager.AcquirePoolControllerLock(); err == nil { _ = second.Close() - t.Fatal("second controller acquired the same provider/prefix lock") + t.Fatal("second controller acquired the same configuration lock") + } +} + +func TestPoolControllerLockConflictsWhenSameConfigChangesProviderOrPrefix(t *testing.T) { + setPoolControllerLockStateHome(t) + projectRoot := t.TempDir() + firstManager := testPoolControllerManager(t, projectRoot, "config.yml", "docker-container", "epar-lock-before") + secondManager := testPoolControllerManager(t, projectRoot, "config.yml", "docker-sandboxes", "epar-lock-after") + first, err := firstManager.AcquirePoolControllerLock() + if err != nil { + t.Fatal(err) + } + defer first.Close() + if second, err := secondManager.AcquirePoolControllerLock(); err == nil { + _ = second.Close() + t.Fatal("in-place configuration mutation bypassed the configuration lock") + } +} + +func TestPoolControllerLockConflictsForSamePrefixAcrossConfigProviderAndProject(t *testing.T) { + setPoolControllerLockStateHome(t) + firstManager := testPoolControllerManager(t, t.TempDir(), "first.yml", "docker-container", "epar-lock-shared-prefix") + secondManager := testPoolControllerManager(t, t.TempDir(), "second.yml", "docker-sandboxes", "epar-lock-shared-prefix") + first, err := firstManager.AcquirePoolControllerLock() + if err != nil { + t.Fatal(err) + } + defer first.Close() + if second, err := secondManager.AcquirePoolControllerLock(); err == nil { + _ = second.Close() + t.Fatal("second controller acquired a shared pool prefix across projects/providers") + } else if !strings.Contains(err.Error(), "pool.namePrefix") || !strings.Contains(err.Error(), "owner config=") { + t.Fatalf("prefix conflict error = %v, want owner diagnostic", err) + } +} + +func TestPoolControllerLockAllowsDistinctConfigAndPrefix(t *testing.T) { + setPoolControllerLockStateHome(t) + projectRoot := t.TempDir() + firstManager := testPoolControllerManager(t, projectRoot, "first.yml", "docker-container", "epar-lock-first") + secondManager := testPoolControllerManager(t, projectRoot, "second.yml", "docker-sandboxes", "epar-lock-second") + first, err := firstManager.AcquirePoolControllerLock() + if err != nil { + t.Fatal(err) + } + defer first.Close() + second, err := secondManager.AcquirePoolControllerLock() + if err != nil { + t.Fatalf("independent pool identity was blocked: %v", err) + } + defer second.Close() +} + +func TestPoolControllerLockReleaseAllowsReacquire(t *testing.T) { + setPoolControllerLockStateHome(t) + manager := testPoolControllerManager(t, t.TempDir(), "config.yml", "docker-container", "epar-lock-release") + first, err := manager.AcquirePoolControllerLock() + if err != nil { + t.Fatal(err) + } + if err := first.Close(); err != nil { + t.Fatal(err) + } + second, err := manager.AcquirePoolControllerLock() + if err != nil { + t.Fatalf("reacquire after release: %v", err) + } + defer second.Close() +} + +func TestPoolControllerLockOverwritesStaleMetadataAfterBothLocksAreAcquired(t *testing.T) { + stateHome := setPoolControllerLockStateHome(t) + manager := testPoolControllerManager(t, t.TempDir(), "config.yml", "docker-container", "epar-lock-stale") + canonicalConfig, err := manager.canonicalPoolControllerConfigPath() + if err != nil { + t.Fatal(err) + } + lockRoot := filepath.Join(stateHome, poolControllerLockDirectory) + if err := os.MkdirAll(lockRoot, 0o700); err != nil { + t.Fatal(err) + } + stale := poolControllerLockOwner{ConfigPath: "stale.yml", Provider: "tart", NamePrefix: "stale-prefix", PID: 1, StartedAt: time.Unix(1, 0).UTC()} + for _, path := range []string{ + poolControllerLockPath(lockRoot, "config", canonicalConfig), + poolControllerLockPath(lockRoot, "prefix", "epar-lock-stale"), + } { + if err := os.WriteFile(path, mustJSON(t, stale), 0o600); err != nil { + t.Fatal(err) + } + } + lock, err := manager.AcquirePoolControllerLock() + if err != nil { + t.Fatal(err) + } + defer lock.Close() + owner, err := readPoolControllerLockOwner(poolControllerLockPath(lockRoot, "prefix", "epar-lock-stale")) + if err != nil { + t.Fatal(err) + } + if owner.ConfigPath != canonicalConfig || owner.NamePrefix != "epar-lock-stale" || owner.PID != os.Getpid() { + t.Fatalf("stale metadata was not replaced after lock acquisition: %#v", owner) + } +} + +func TestPoolControllerLockRaceAllowsOneOwner(t *testing.T) { + setPoolControllerLockStateHome(t) + manager := testPoolControllerManager(t, t.TempDir(), "config.yml", "docker-container", "epar-lock-race") + const contenders = 8 + locks := make(chan io.Closer, contenders) + errs := make(chan error, contenders) + var start sync.WaitGroup + start.Add(1) + for range contenders { + go func() { + start.Wait() + lock, err := manager.AcquirePoolControllerLock() + if err != nil { + errs <- err + return + } + locks <- lock + }() + } + start.Done() + var winners []io.Closer + for range contenders { + select { + case lock := <-locks: + winners = append(winners, lock) + case <-errs: + } + } + for _, lock := range winners { + defer lock.Close() + } + if len(winners) != 1 { + t.Fatalf("concurrent lock winners = %d, want 1", len(winners)) + } +} + +func TestPoolControllerLockMigratesLifecycleStateOnlyAfterExclusiveOwnership(t *testing.T) { + setPoolControllerLockStateHome(t) + root := t.TempDir() + project := filepath.Join(root, "project") + alias := filepath.Join(root, "project-alias") + if err := os.MkdirAll(filepath.Join(project, ".local"), 0o700); err != nil { + t.Fatal(err) + } + if err := os.Symlink(project, alias); err != nil { + t.Skipf("symlinks unavailable: %v", err) + } + configPath := filepath.Join(alias, ".local", "config.yml") + if err := os.WriteFile(filepath.Join(project, ".local", "config.yml"), []byte("provider: {}\n"), 0o600); err != nil { + t.Fatal(err) + } + legacyConfig, err := filepath.Abs(configPath) + if err != nil { + t.Fatal(err) + } + legacyDirectory := lifecycleStateDirectory(project, filepath.Clean(legacyConfig)) + canonicalConfig, err := storagecatalog.CanonicalPath(configPath) + if err != nil { + t.Fatal(err) + } + canonicalDirectory := lifecycleStateDirectory(project, canonicalConfig) + if legacyDirectory == canonicalDirectory { + t.Fatal("test requires distinct legacy and canonical lifecycle namespaces") + } + if err := os.MkdirAll(legacyDirectory, 0o700); err != nil { + t.Fatal(err) + } + marker := filepath.Join(legacyDirectory, "migration-marker") + if err := os.WriteFile(marker, []byte("preserve"), 0o600); err != nil { + t.Fatal(err) + } + + blocker := testPoolControllerManager(t, project, configPath, "docker-container", "epar-lock-migration") + held, err := blocker.AcquirePoolControllerLock() + if err != nil { + t.Fatal(err) + } + candidate := testPoolControllerManager(t, project, configPath, "docker-container", "epar-lock-migration") + candidate.LifecycleStateEnabled = true + if lock, err := candidate.AcquirePoolControllerLock(); err == nil { + _ = lock.Close() + t.Fatal("candidate acquired controller locks while blocker was active") + } + if candidate.LifecycleState != nil { + t.Fatal("lifecycle state opened before controller locks were acquired") + } + if _, err := os.Stat(marker); err != nil { + t.Fatalf("failed lock attempt moved legacy lifecycle state: %v", err) + } + if _, err := os.Stat(canonicalDirectory); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("failed lock attempt created canonical lifecycle state: %v", err) + } + if err := held.Close(); err != nil { + t.Fatal(err) + } + + lock, err := candidate.AcquirePoolControllerLock() + if err != nil { + t.Fatal(err) + } + defer lock.Close() + if candidate.LifecycleState == nil { + t.Fatal("lifecycle state was not opened after acquiring controller locks") + } + if filepath.Dir(candidate.LifecycleState.Path()) != canonicalDirectory { + t.Fatalf("lifecycle state path = %q, want canonical directory %q", candidate.LifecycleState.Path(), canonicalDirectory) + } + if _, err := os.Stat(filepath.Join(canonicalDirectory, "migration-marker")); err != nil { + t.Fatalf("legacy lifecycle content was not migrated under the controller locks: %v", err) + } + if _, err := os.Stat(legacyDirectory); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("legacy lifecycle namespace still exists after locked migration: %v", err) } } func TestVerifyRequiresPoolControllerLock(t *testing.T) { - t.Setenv("LOCALAPPDATA", t.TempDir()) - manager := Manager{ProjectRoot: t.TempDir(), ConfigPath: "verify.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-dind", SourceImage: "image"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-verify"}}, Provider: &fakeProvider{}} + setPoolControllerLockStateHome(t) + manager := Manager{ProjectRoot: t.TempDir(), ConfigPath: "verify.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container", SourceImage: "image"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-verify"}}, Provider: &fakeProvider{}} held, err := manager.AcquirePoolControllerLock() if err != nil { t.Fatal(err) @@ -37,8 +261,8 @@ func TestVerifyRequiresPoolControllerLock(t *testing.T) { } func TestCleanupRequiresPoolControllerLock(t *testing.T) { - t.Setenv("LOCALAPPDATA", t.TempDir()) - manager := Manager{ProjectRoot: t.TempDir(), ConfigPath: "cleanup.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-dind"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-cleanup"}}, Provider: &fakeProvider{}} + setPoolControllerLockStateHome(t) + manager := Manager{ProjectRoot: t.TempDir(), ConfigPath: "cleanup.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-cleanup"}}, Provider: &fakeProvider{}} held, err := manager.AcquirePoolControllerLock() if err != nil { t.Fatal(err) @@ -51,8 +275,8 @@ func TestCleanupRequiresPoolControllerLock(t *testing.T) { } func TestProvisionPoolRequiresPoolControllerLock(t *testing.T) { - t.Setenv("LOCALAPPDATA", t.TempDir()) - manager := Manager{ProjectRoot: t.TempDir(), ConfigPath: "provision.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-dind", SourceImage: "image"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-provision"}}, Provider: &fakeProvider{}} + setPoolControllerLockStateHome(t) + manager := Manager{ProjectRoot: t.TempDir(), ConfigPath: "provision.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container", SourceImage: "image"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-provision"}}, Provider: &fakeProvider{}} held, err := manager.AcquirePoolControllerLock() if err != nil { t.Fatal(err) @@ -64,34 +288,34 @@ func TestProvisionPoolRequiresPoolControllerLock(t *testing.T) { } } -func TestPoolControllerLockIsIndependentAcrossPoolIdentity(t *testing.T) { - t.Setenv("LOCALAPPDATA", t.TempDir()) - firstManager := Manager{ConfigPath: "first.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-dind"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-first"}}} - secondManager := Manager{ConfigPath: "second.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-dind"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-second"}}} - first, err := firstManager.AcquirePoolControllerLock() - if err != nil { - t.Fatal(err) +func setPoolControllerLockStateHome(t *testing.T) string { + t.Helper() + root := t.TempDir() + t.Setenv("EPAR_STATE_HOME", root) + return root +} + +func testPoolControllerManager(t *testing.T, projectRoot, configName, providerType, prefix string) *Manager { + t.Helper() + manager := &Manager{ + ProjectRoot: projectRoot, + ConfigPath: configName, + Config: config.Config{ + Provider: config.ProviderConfig{Type: providerType}, + Pool: config.PoolConfig{NamePrefix: prefix}, + }, } - defer first.Close() - second, err := secondManager.AcquirePoolControllerLock() - if err != nil { - t.Fatalf("independent pool identity was blocked: %v", err) + if !filepath.IsAbs(configName) { + manager.ConfigPath = filepath.Join(projectRoot, configName) } - defer second.Close() + return manager } -func TestPoolControllerLockIncludesCanonicalConfigIdentity(t *testing.T) { - t.Setenv("LOCALAPPDATA", t.TempDir()) - firstManager := Manager{ConfigPath: "first.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-dind"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-shared-prefix"}}} - secondManager := Manager{ConfigPath: "second.yml", Config: config.Config{Provider: config.ProviderConfig{Type: "docker-dind"}, Pool: config.PoolConfig{NamePrefix: "epar-lock-shared-prefix"}}} - first, err := firstManager.AcquirePoolControllerLock() +func mustJSON(t *testing.T, value any) []byte { + t.Helper() + content, err := json.Marshal(value) if err != nil { t.Fatal(err) } - defer first.Close() - second, err := secondManager.AcquirePoolControllerLock() - if err != nil { - t.Fatalf("distinct canonical config identity was blocked: %v", err) - } - defer second.Close() + return append(content, '\n') } diff --git a/internal/pool/docker_pull.go b/internal/pool/docker_pull.go deleted file mode 100644 index dbf327e..0000000 --- a/internal/pool/docker_pull.go +++ /dev/null @@ -1,358 +0,0 @@ -package pool - -import ( - "context" - "encoding/base64" - "encoding/json" - "fmt" - "io" - "os" - "runtime" - "strings" - "time" - - "github.com/google/go-containerregistry/pkg/authn" - "github.com/google/go-containerregistry/pkg/name" - gcrv1 "github.com/google/go-containerregistry/pkg/v1" - "github.com/google/go-containerregistry/pkg/v1/remote" - "github.com/moby/moby/api/types/jsonstream" - "github.com/moby/moby/api/types/registry" - "github.com/moby/moby/client" - ocispec "github.com/opencontainers/image-spec/specs-go/v1" - "golang.org/x/term" -) - -const dockerPullProgressInterval = 250 * time.Millisecond - -type dockerSourcePullOptions struct { - Image string - Platform string - LogPath string - AnnounceRemoteSize bool -} - -type dockerLayerProgress struct { - current int64 - total int64 - completed bool -} - -// pullDockerSourceCommand is kept as a small seam for existing command-based -// image preparation tests. Production always uses Manager.pullDockerSource. -var pullDockerSourceCommand = (*Manager).pullDockerSource - -func (m *Manager) pullDockerSource(ctx context.Context, opts dockerSourcePullOptions) error { - cli, err := client.New(client.FromEnv) - if err != nil { - return m.pullDockerSourceWithCLI(ctx, opts, fmt.Errorf("initialize Docker Engine client: %w", err)) - } - if _, err := cli.Ping(ctx, client.PingOptions{}); err != nil { - return m.pullDockerSourceWithCLI(ctx, opts, fmt.Errorf("connect to Docker Engine: %w", err)) - } - - platform, err := m.resolveDockerPullPlatform(ctx, cli, opts.Platform) - if err != nil { - return m.pullDockerSourceWithCLI(ctx, opts, err) - } - if opts.AnnounceRemoteSize { - if size, err := remoteCompressedLayerSize(opts.Image, platform); err != nil { - m.writeDockerPullNotice(opts.LogPath, "warning: could not determine remote compressed layer size: "+sanitizeTimingError(err)) - } else { - m.writeDockerPullNotice(opts.LogPath, fmt.Sprintf("Remote compressed layers: %s; actual transfer may be lower when Docker reuses layers.", formatDockerPullBytes(size))) - } - } - - registryAuth, err := dockerRegistryAuth(opts.Image) - if err != nil { - m.writeDockerPullNotice(opts.LogPath, "warning: could not load Docker registry credentials; continuing without explicit credentials: "+sanitizeTimingError(err)) - } - response, err := cli.ImagePull(ctx, opts.Image, client.ImagePullOptions{ - RegistryAuth: registryAuth, - Platforms: []ocispec.Platform{platform}, - }) - if err != nil { - return fmt.Errorf("Docker Engine pull %s: %w", opts.Image, err) - } - if err := m.renderDockerPullProgress(ctx, response, opts.LogPath); err != nil { - return fmt.Errorf("Docker Engine pull %s: %w", opts.Image, err) - } - m.writeDockerPullNotice(opts.LogPath, "Docker source pull complete: "+opts.Image) - return nil -} - -func (m *Manager) pullDockerSourceWithCLI(ctx context.Context, opts dockerSourcePullOptions, apiErr error) error { - m.warnf("warning: %v; falling back to docker pull CLI\n", apiErr) - m.writeDockerPullNotice(opts.LogPath, "warning: "+sanitizeTimingError(apiErr)+"; falling back to docker pull CLI") - args := []string{"pull"} - if opts.Platform != "" { - args = append(args, "--platform", opts.Platform) - } - args = append(args, opts.Image) - return m.runHostLogged(ctx, opts.LogPath, "docker", args...) -} - -func (m *Manager) resolveDockerPullPlatform(ctx context.Context, cli *client.Client, configured string) (ocispec.Platform, error) { - if platform, ok := normalizedDockerPlatform(configured, ""); ok { - return platform, nil - } - if platform, ok := normalizedDockerPlatform(m.Config.Provider.Platform, ""); ok { - return platform, nil - } - info, err := cli.Info(ctx, client.InfoOptions{}) - if err != nil { - return ocispec.Platform{}, fmt.Errorf("inspect Docker Engine platform: %w", err) - } - if platform, ok := normalizedDockerPlatform(info.Info.OSType+"/"+info.Info.Architecture, ""); ok { - return platform, nil - } - if platform, ok := normalizedDockerPlatform(runtime.GOOS+"/"+runtime.GOARCH, ""); ok { - return platform, nil - } - return ocispec.Platform{}, fmt.Errorf("Docker Engine did not report a usable platform") -} - -func normalizedDockerPlatform(value, fallbackOS string) (ocispec.Platform, bool) { - parts := strings.Split(strings.Trim(strings.ToLower(value), "/"), "/") - if len(parts) == 0 || len(parts) > 3 || parts[0] == "" { - return ocispec.Platform{}, false - } - platform := ocispec.Platform{OS: fallbackOS} - if len(parts) == 1 { - platform.Architecture = normalizeDockerArchitecture(parts[0]) - } else { - platform.OS = parts[0] - platform.Architecture = normalizeDockerArchitecture(parts[1]) - if len(parts) == 3 { - platform.Variant = parts[2] - } - } - if platform.OS == "" { - platform.OS = "linux" - } - if platform.Architecture == "" { - return ocispec.Platform{}, false - } - return platform, true -} - -func normalizeDockerArchitecture(architecture string) string { - switch architecture { - case "x86_64", "x64": - return "amd64" - case "aarch64": - return "arm64" - default: - return architecture - } -} - -func remoteCompressedLayerSize(image string, platform ocispec.Platform) (int64, error) { - ref, authenticator, err := dockerImageReferenceAndAuth(image) - if err != nil { - return 0, err - } - remoteImage, err := remote.Image(ref, remote.WithAuth(authenticator), remote.WithPlatform(gcrv1.Platform{ - OS: platform.OS, - Architecture: platform.Architecture, - Variant: platform.Variant, - })) - if err != nil { - return 0, err - } - layers, err := remoteImage.Layers() - if err != nil { - return 0, err - } - var total int64 - for _, layer := range layers { - size, err := layer.Size() - if err != nil { - return 0, err - } - total += size - } - return total, nil -} - -func dockerRegistryAuth(image string) (string, error) { - ref, authenticator, err := dockerImageReferenceAndAuth(image) - if err != nil { - return "", err - } - credentials, err := authenticator.Authorization() - if err != nil { - return "", err - } - content, err := json.Marshal(registry.AuthConfig{ - Username: credentials.Username, - Password: credentials.Password, - Auth: credentials.Auth, - ServerAddress: ref.Context().RegistryStr(), - IdentityToken: credentials.IdentityToken, - RegistryToken: credentials.RegistryToken, - }) - if err != nil { - return "", err - } - return base64.RawURLEncoding.EncodeToString(content), nil -} - -func dockerImageReferenceAndAuth(image string) (name.Reference, authn.Authenticator, error) { - ref, err := name.ParseReference(image) - if err != nil { - return nil, nil, err - } - authenticator, err := authn.DefaultKeychain.Resolve(ref.Context().Registry) - if err != nil { - return nil, nil, err - } - return ref, authenticator, nil -} - -func (m *Manager) renderDockerPullProgress(ctx context.Context, response client.ImagePullResponse, logPath string) error { - transcript, err := m.transcript(logPath, "", "docker-pull") - if err != nil { - return err - } - interactive := term.IsTerminal(int(os.Stdout.Fd())) && containsString(m.Config.Logging.TranscriptSinks, "console") && m.Config.Logging.TranscriptConsoleFormat == "text" - eventWriter := transcript.Stdout - if interactive { - eventWriter = transcript.File - } - layers := map[string]dockerLayerProgress{} - lastRender := time.Time{} - rendered := false - for message, streamErr := range response.JSONMessages(ctx) { - if streamErr != nil { - return streamErr - } - writeDockerPullEvent(eventWriter, message.ID, message.Status, message.Progress, message.Stream, message.Error) - if message.Error != nil { - return message.Error - } - if message.ID != "" { - layer := layers[message.ID] - if message.Progress != nil { - layer.current = message.Progress.Current - if message.Progress.Total > 0 { - layer.total = message.Progress.Total - } - } - if dockerPullLayerComplete(message.Status) { - layer.completed = true - if layer.total > 0 { - layer.current = layer.total - } - } - layers[message.ID] = layer - } - if time.Since(lastRender) >= dockerPullProgressInterval { - if interactive { - writeDockerPullSummary(os.Stdout, true, layers) - } else { - writeDockerPullSummary(transcript.Stdout, false, layers) - } - lastRender = time.Now() - rendered = true - } - } - if rendered { - if interactive { - writeDockerPullSummary(os.Stdout, true, layers) - } else { - writeDockerPullSummary(transcript.Stdout, false, layers) - } - if interactive { - fmt.Fprintln(os.Stdout) - } - } - return nil -} - -func (m *Manager) writeDockerPullNotice(logPath, message string) { - transcript, err := m.transcript(logPath, "", "docker-pull") - if err != nil { - m.logger().Warn("docker pull transcript unavailable", "operation", "docker-pull", "logPath", logPath, "error", err) - return - } - _, _ = fmt.Fprintf(transcript.Stdout, "%s\n", message) -} - -func containsString(values []string, wanted string) bool { - for _, value := range values { - if value == wanted { - return true - } - } - return false -} - -func writeDockerPullEvent(logFile io.Writer, id, status string, progress *jsonstream.Progress, stream string, pullErr error) { - if logFile == nil { - return - } - parts := make([]string, 0, 4) - if id != "" { - parts = append(parts, id) - } - if status != "" { - parts = append(parts, status) - } - if progress != nil { - parts = append(parts, fmt.Sprintf("progress=%d/%d", progress.Current, progress.Total)) - } - if stream != "" { - parts = append(parts, strings.TrimSpace(stream)) - } - if pullErr != nil { - parts = append(parts, "error="+sanitizeTimingError(pullErr)) - } - fmt.Fprintf(logFile, "%s %s\n", time.Now().UTC().Format(time.RFC3339Nano), strings.Join(parts, " ")) -} - -func dockerPullLayerComplete(status string) bool { - status = strings.ToLower(strings.TrimSpace(status)) - return status == "pull complete" || status == "already exists" || status == "exists" -} - -func writeDockerPullSummary(w io.Writer, interactive bool, layers map[string]dockerLayerProgress) { - var complete, known int - var currentBytes, totalBytes int64 - for _, layer := range layers { - if layer.completed { - complete++ - } - if layer.total > 0 { - known++ - totalBytes += layer.total - currentBytes += min(layer.current, layer.total) - } - } - line := fmt.Sprintf("Docker source pull: %d/%d layers complete; %s/%s", complete, len(layers), formatDockerPullBytes(currentBytes), formatDockerPullBytes(totalBytes)) - if totalBytes > 0 { - line += fmt.Sprintf(" (%.0f%%)", float64(currentBytes)*100/float64(totalBytes)) - } - if known < len(layers) { - line += fmt.Sprintf("; %d layer(s) size pending", len(layers)-known) - } - if interactive { - fmt.Fprintf(w, "\r\033[2K%s", line) - return - } - fmt.Fprintf(w, "%s %s\n", time.Now().UTC().Format(time.RFC3339Nano), line) -} - -func formatDockerPullBytes(value int64) string { - const unit = 1024 - if value < unit { - return fmt.Sprintf("%d B", value) - } - units := []string{"KiB", "MiB", "GiB", "TiB"} - size := float64(value) - index := -1 - for size >= unit && index+1 < len(units) { - size /= unit - index++ - } - return fmt.Sprintf("%.1f %s", size, units[index]) -} diff --git a/internal/pool/host_trust.go b/internal/pool/host_trust.go index 295c46e..c8b925d 100644 --- a/internal/pool/host_trust.go +++ b/internal/pool/host_trust.go @@ -1,8 +1,11 @@ package pool import ( + "archive/tar" + "bytes" "context" "encoding/json" + "errors" "fmt" "io" "os" @@ -13,6 +16,7 @@ import ( "time" "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + artifactimage "github.com/solutionforest/ephemeral-action-runner/internal/image" "github.com/solutionforest/ephemeral-action-runner/internal/provider" ) @@ -20,22 +24,24 @@ const ( hostTrustGuestDir = "/usr/local/share/ca-certificates/epar-host" hostTrustMarkerGuest = "/opt/epar/host-trust-generation.json" hostTrustLeaseGuest = "/run/epar/host-trust-lease.json" - hostTrustLeaseLifetime = 20 * time.Second + hostTrustLeaseLifetime = 90 * time.Second + hostTrustHandoffLease = 2 * time.Minute hostTrustMaximumAge = 30 * time.Second hostTrustNativePoll = 15 * time.Second ) -var hostTrustRefreshInterval = 5 * time.Second +// HostTrustMarkerGuest is the stable guest path for verified host-trust metadata. +const HostTrustMarkerGuest = hostTrustMarkerGuest + +// HostTrustLeaseLifetime is the default shared guest lease duration. +const HostTrustLeaseLifetime = hostTrustLeaseLifetime + +var hostTrustRefreshInterval = 30 * time.Second +var hostTrustWriteTimeout = 10 * time.Second var hostTrustControllerInContainer = linuxControllerInContainer var hostTrustControllerOS = runtime.GOOS -type hostTrustImageMetadata struct { - Mode string `json:"mode"` - HostOS string `json:"hostOS"` - Scopes []string `json:"scopes"` - Generation string `json:"generation"` - CertificateCount int `json:"certificateCount"` -} +type hostTrustImageMetadata = artifactimage.HostTrustMetadata type hostTrustMarker struct { SchemaVersion int `json:"schemaVersion"` @@ -55,6 +61,12 @@ type hostTrustLease struct { ExpiresAt string `json:"expiresAt"` } +// HostTrustMarker and HostTrustLease are shared payloads for specialized +// providers that install the same verified host-trust contract by another +// transport. +type HostTrustMarker = hostTrustMarker +type HostTrustLease = hostTrustLease + func (m *Manager) hostTrustEnabled() bool { return hosttrust.Enabled(m.Config.Image.HostTrustMode) } @@ -125,6 +137,52 @@ func (m *Manager) resolveHostTrust(ctx context.Context) (hosttrust.Snapshot, err return validateHostTrustSnapshot(snapshot, time.Now().UTC()) } +func (m *Manager) resolveBuildTrust(ctx context.Context) (hosttrust.Snapshot, error) { + if m.buildTrustResolver != nil { + snapshot, err := m.buildTrustResolver(ctx) + if err != nil { + return hosttrust.Snapshot{}, err + } + return validateHostTrustSnapshot(snapshot, time.Now().UTC()) + } + if m.hostTrustResolver != nil { + snapshot, err := m.hostTrustResolver(ctx) + if err != nil { + return hosttrust.Snapshot{}, err + } + return validateHostTrustSnapshot(snapshot, time.Now().UTC()) + } + scopes := buildTrustScopes(m.Config.Image.HostTrustMode, m.Config.Image.HostTrustScopes) + feedPath := strings.TrimSpace(os.Getenv("EPAR_BUILD_TRUST_FEED")) + controllerHostOS := strings.TrimSpace(os.Getenv("EPAR_CONTROLLER_HOST_OS")) + if feedPath == "" && hostTrustControllerOS == "linux" && hostTrustControllerInContainer() { + return hosttrust.Snapshot{}, fmt.Errorf("operational BuildKit trust requires EPAR_BUILD_TRUST_FEED when the EPAR controller runs in a container; use an official no-Go wrapper") + } + snapshot, err := hosttrust.Resolve(ctx, hosttrust.Options{ + Mode: hosttrust.ModeOverlay, + Scopes: scopes, + FeedPath: feedPath, + ControllerHostOS: controllerHostOS, + }) + if err != nil { + return hosttrust.Snapshot{}, fmt.Errorf("resolve operational BuildKit trust: %w", err) + } + return validateHostTrustSnapshot(snapshot, time.Now().UTC()) +} + +func buildTrustScopes(runnerMode string, runnerScopes []string) []string { + scopes := []string{hosttrust.ScopeSystem} + if hosttrust.Enabled(runnerMode) { + for _, scope := range runnerScopes { + if strings.EqualFold(strings.TrimSpace(scope), hosttrust.ScopeUser) { + scopes = append(scopes, hosttrust.ScopeUser) + break + } + } + } + return scopes +} + func linuxControllerInContainer() bool { return linuxContainerEvidence( func(path string) bool { _, err := os.Stat(path); return err == nil }, @@ -184,6 +242,11 @@ func validateHostTrustSnapshot(snapshot hosttrust.Snapshot, now time.Time) (host return snapshot, nil } +// ValidateHostTrustSnapshot verifies the shared host-trust freshness contract. +func ValidateHostTrustSnapshot(snapshot hosttrust.Snapshot, now time.Time) (hosttrust.Snapshot, error) { + return validateHostTrustSnapshot(snapshot, now) +} + func hostTrustMetadata(snapshot hosttrust.Snapshot) *hostTrustImageMetadata { if snapshot.Generation == "" { return nil @@ -243,16 +306,23 @@ func validateHostTrustMarkerAgainstSnapshot(marker hostTrustMarker, snapshot hos } func hostTrustLeaseJSON(snapshot hosttrust.Snapshot, now time.Time) ([]byte, error) { + return hostTrustLeaseJSONWithLifetime(snapshot, now, hostTrustLeaseLifetime) +} + +func hostTrustLeaseJSONWithLifetime(snapshot hosttrust.Snapshot, now time.Time, lifetime time.Duration) ([]byte, error) { if _, err := validateHostTrustSnapshot(snapshot, now); err != nil { return nil, err } + if lifetime <= 0 { + return nil, fmt.Errorf("host trust lease lifetime must be positive") + } return json.MarshalIndent(hostTrustLease{ SchemaVersion: 1, Generation: snapshot.Generation, HostOS: snapshot.HostOS, Mode: hosttrust.ModeOverlay, Scopes: append([]string(nil), snapshot.Scopes...), - ExpiresAt: now.Add(hostTrustLeaseLifetime).UTC().Format(time.RFC3339Nano), + ExpiresAt: now.Add(lifetime).UTC().Format(time.RFC3339Nano), }, "", " ") } @@ -268,6 +338,53 @@ func copyHostTrustCertificatesToDir(destination string, snapshot hosttrust.Snaps return nil } +func hostTrustCertificateArchive(snapshot hosttrust.Snapshot) (string, error) { + var buffer bytes.Buffer + writer := tar.NewWriter(&buffer) + for _, certificate := range snapshot.Certificates { + header := &tar.Header{ + Name: certificate.Name, + Mode: 0644, + Size: int64(len(certificate.PEM)), + } + if err := writer.WriteHeader(header); err != nil { + return "", err + } + if _, err := writer.Write(certificate.PEM); err != nil { + return "", err + } + } + if err := writer.Close(); err != nil { + return "", err + } + return buffer.String(), nil +} + +func (m *Manager) installHostTrustRuntime(ctx context.Context, instanceName string, snapshot hosttrust.Snapshot) error { + if !m.hostTrustEnabled() { + return nil + } + if _, err := validateHostTrustSnapshot(snapshot, time.Now().UTC()); err != nil { + return err + } + archive, err := hostTrustCertificateArchive(snapshot) + if err != nil { + return fmt.Errorf("archive host trust certificates: %w", err) + } + script := fmt.Sprintf("sudo install -d -m 0755 %s && sudo find %s -maxdepth 1 -type f -name 'epar-*.crt' -delete && sudo tar -x -f - --no-same-owner --no-same-permissions -C %s && sudo update-ca-certificates", shellQuote(hostTrustGuestDir), shellQuote(hostTrustGuestDir), shellQuote(hostTrustGuestDir)) + if _, err := m.execGuest(ctx, instanceName, provider.ShellCommand(script), provider.ExecOptions{Stdin: archive}); err != nil { + return fmt.Errorf("install host trust certificates in runtime: %w", err) + } + content, err := hostTrustMarkerJSON(snapshot) + if err != nil { + return err + } + if err := m.copyTextGuest(ctx, instanceName, hostTrustMarkerGuest, "0644", string(content)+"\n", true); err != nil { + return fmt.Errorf("install host trust generation marker: %w", err) + } + return nil +} + func (m *Manager) writeHostTrustBuildInputs(buildContext string, snapshot hosttrust.Snapshot) error { if err := copyHostTrustCertificatesToDir(filepath.Join(buildContext, "host-trust-certificates"), snapshot); err != nil { return err @@ -287,24 +404,39 @@ func (m *Manager) writeHostTrustBuildInputs(buildContext string, snapshot hosttr } func (m *Manager) issueHostTrustLease(ctx context.Context, instanceName string, snapshot hosttrust.Snapshot) error { + return m.issueHostTrustLeaseWithLifetime(ctx, instanceName, snapshot, hostTrustLeaseLifetime) +} + +func (m *Manager) issueHostTrustLeaseWithLifetime(ctx context.Context, instanceName string, snapshot hosttrust.Snapshot, lifetime time.Duration) error { if !m.hostTrustEnabled() { return nil } - now := time.Now().UTC() - content, err := hostTrustLeaseJSON(snapshot, now) + content, err := hostTrustLeaseJSONWithLifetime(snapshot, time.Now().UTC(), lifetime) if err != nil { return err } - if _, err := m.execGuest(ctx, instanceName, provider.ShellCommand("if command -v sudo >/dev/null 2>&1; then sudo install -d -m 0755 /run/epar; else install -d -m 0755 /run/epar; fi"), provider.ExecOptions{}); err != nil { + writeCtx, cancel := context.WithTimeout(ctx, hostTrustWriteTimeout) + defer cancel() + staging := hostTrustLeaseGuest + ".tmp" + script := fmt.Sprintf("cat > /tmp/epar-host-trust-lease && if command -v sudo >/dev/null 2>&1; then sudo install -d -m 0755 /run/epar && sudo install -m 0644 /tmp/epar-host-trust-lease %s && sudo mv -f %s %s; else install -d -m 0755 /run/epar && install -m 0644 /tmp/epar-host-trust-lease %s && mv -f %s %s; fi && rm -f /tmp/epar-host-trust-lease", shellQuote(staging), shellQuote(staging), shellQuote(hostTrustLeaseGuest), shellQuote(staging), shellQuote(staging), shellQuote(hostTrustLeaseGuest)) + if _, err := m.execGuest(writeCtx, instanceName, provider.ShellCommand(script), provider.ExecOptions{Stdin: string(content) + "\n"}); err != nil { + if errors.Is(err, context.DeadlineExceeded) { + return fmt.Errorf("host trust lease write exceeded %s: %w", hostTrustWriteTimeout, err) + } return err } - return provider.CopyTextAtomic(ctx, m.Provider, instanceName, hostTrustLeaseGuest, "0644", string(content)+"\n") + return nil } -func (m *Manager) reconcileHostTrustRunners(ctx context.Context, active map[string]ProvisionedInstance, current hosttrust.Snapshot) int { +func (m *Manager) reconcileHostTrustRunners(ctx context.Context, active map[string]ProvisionedInstance, current hosttrust.Snapshot, busyHandoff map[string]bool) int { if m.GitHub == nil { return 0 } + for name := range busyHandoff { + if _, found := active[name]; !found { + delete(busyHandoff, name) + } + } retired := 0 for name, instance := range active { if instance.HostTrustGeneration != current.Generation { @@ -317,19 +449,40 @@ func (m *Manager) reconcileHostTrustRunners(ctx context.Context, active map[stri } runner, found, err := m.GitHub.RunnerByName(ctx, name) if err != nil { - m.warnf("[%s] host trust reconciliation warning; lease not refreshed: %v\n", name, err) + if isTransientGitHubLivenessError(err) { + m.warnf("[%s] GitHub API is temporarily unavailable during host trust refresh; the existing lease will expire closed and EPAR will retry: %v\n", name, err) + } else { + m.warnf("[%s] host trust reconciliation warning; lease not refreshed: %v\n", name, err) + } continue } if instance.HostTrustGeneration == current.Generation { - if !found || runner.Busy { + if !found { + delete(busyHandoff, name) + continue + } + if runner.Busy { + if busyHandoff[name] { + continue + } + // GitHub can mark a runner busy before the job-start hook executes. + // Issue one bounded handoff lease for that transition, but never + // renew it while the job remains busy. + if err := m.issueHostTrustLeaseWithLifetime(ctx, name, current, hostTrustHandoffLease); err != nil { + m.warnf("[%s] host trust job handoff lease warning: %v\n", name, err) + continue + } + busyHandoff[name] = true continue } + delete(busyHandoff, name) if err := m.issueHostTrustLease(ctx, name, current); err != nil { m.warnf("[%s] host trust lease refresh warning: %v\n", name, err) } continue } if found && runner.Busy { + delete(busyHandoff, name) m.infof("[%s] draining busy runner on old host trust generation %s\n", name, instance.HostTrustGeneration) instance.Phase = LifecycleDraining active[name] = instance @@ -341,6 +494,7 @@ func (m *Manager) reconcileHostTrustRunners(ctx context.Context, active map[stri continue } delete(active, name) + delete(busyHandoff, name) retired++ } return retired diff --git a/internal/pool/host_trust_test.go b/internal/pool/host_trust_test.go index 809b6da..b703ba4 100644 --- a/internal/pool/host_trust_test.go +++ b/internal/pool/host_trust_test.go @@ -1,11 +1,19 @@ package pool import ( + "archive/tar" + "bytes" "context" + "crypto/sha256" "encoding/json" + "errors" + "fmt" "io" + "net/http" + "net/http/httptest" "os" "path/filepath" + "slices" "strings" "sync/atomic" "testing" @@ -14,9 +22,10 @@ import ( "github.com/solutionforest/ephemeral-action-runner/internal/config" gh "github.com/solutionforest/ephemeral-action-runner/internal/github" "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" ) -func TestDockerDindBuildContextKeepsHostAndExplicitTrustSeparate(t *testing.T) { +func TestDockerContainerBuildContextKeepsHostAndExplicitTrustSeparate(t *testing.T) { root := t.TempDir() for _, dir := range []string{ filepath.Join(root, "scripts", "guest", "ubuntu"), @@ -40,7 +49,7 @@ func TestDockerDindBuildContextKeepsHostAndExplicitTrustSeparate(t *testing.T) { ProjectRoot: root, } buildContext := t.TempDir() - if err := manager.prepareDockerDindBuildContextWithHostTrust(buildContext, t.TempDir(), `{"hash":"test"}`+"\n", snapshot); err != nil { + if err := manager.prepareDockerContainerBuildContextWithHostTrust(buildContext, t.TempDir(), `{"hash":"test"}`+"\n", snapshot); err != nil { t.Fatal(err) } assertSingleCertificateFile(t, filepath.Join(buildContext, "trusted-ca-certificates")) @@ -106,6 +115,46 @@ func TestHostTrustLeaseMatchesMarkerAndExpires(t *testing.T) { } } +func TestHostTrustCertificateArchiveContainsOnlyExactCertificates(t *testing.T) { + snapshot := hosttrust.Snapshot{ + Generation: "generation-one", + HostOS: "windows", + Scopes: []string{"system", "user"}, + Certificates: []hosttrust.Certificate{ + {Name: "epar-system-a.crt", PEM: []byte("system")}, + {Name: "epar-user-b.crt", PEM: []byte("user")}, + }, + CollectedAt: time.Now(), + } + archive, err := hostTrustCertificateArchive(snapshot) + if err != nil { + t.Fatal(err) + } + reader := tar.NewReader(bytes.NewBufferString(archive)) + var names []string + for { + header, err := reader.Next() + if errors.Is(err, io.EOF) { + break + } + if err != nil { + t.Fatal(err) + } + names = append(names, header.Name) + if header.Mode != 0644 { + t.Fatalf("certificate mode = %o, want 0644", header.Mode) + } + } + if len(names) != len(snapshot.Certificates) { + t.Fatalf("archive entries = %d, want %d", len(names), len(snapshot.Certificates)) + } + for index, certificate := range snapshot.Certificates { + if names[index] != certificate.Name { + t.Fatalf("archive entry %d = %q, want %q", index, names[index], certificate.Name) + } + } +} + func TestValidateHostTrustMarkerAgainstSnapshotRejectsCloningRace(t *testing.T) { snapshot := hosttrust.Snapshot{ Generation: "g2", HostOS: "windows", Scopes: []string{"system", "user"}, @@ -239,7 +288,7 @@ func TestHostTrustReconciliationRevokesAndRetiresIdleOldGeneration(t *testing.T) CollectedAt: time.Now().UTC(), } active := map[string]ProvisionedInstance{"runner-1": {Name: "runner-1", RunnerID: 42, HostTrustGeneration: "g1"}} - manager.reconcileHostTrustRunners(context.Background(), active, current) + manager.reconcileHostTrustRunners(context.Background(), active, current, make(map[string]bool)) if len(active) != 0 { t.Fatalf("active runners = %#v, want old idle runner retired", active) } @@ -273,7 +322,7 @@ func TestHostTrustReconciliationRevokesButDoesNotRetireBusyOldGeneration(t *test CollectedAt: time.Now().UTC(), } active := map[string]ProvisionedInstance{"runner-1": {Name: "runner-1", RunnerID: 42, HostTrustGeneration: "g1"}} - manager.reconcileHostTrustRunners(context.Background(), active, current) + manager.reconcileHostTrustRunners(context.Background(), active, current, make(map[string]bool)) if len(active) != 1 { t.Fatal("busy old-generation runner was retired before its job completed") } @@ -291,7 +340,154 @@ func TestHostTrustReconciliationRevokesButDoesNotRetireBusyOldGeneration(t *test } } +func TestHostTrustReconciliationPreservesRunnerDuringGitHub503(t *testing.T) { + fake := &fakeProvider{} + github := &fakeGitHub{runnerErr: &gh.HTTPError{StatusCode: http.StatusServiceUnavailable}} + manager := Manager{ + Config: config.Config{Image: config.ImageConfig{HostTrustMode: config.HostTrustModeOverlay, HostTrustScopes: []string{"system"}}}, + Provider: fake, + GitHub: github, + } + current := hosttrust.Snapshot{ + Generation: "g1", + HostOS: "linux", + Scopes: []string{"system"}, + Certificates: []hosttrust.Certificate{{Name: "root.crt", PEM: []byte("pem")}}, + CollectedAt: time.Now().UTC(), + } + active := map[string]ProvisionedInstance{"runner-1": {Name: "runner-1", RunnerID: 42, HostTrustGeneration: "g1"}} + if retired := manager.reconcileHostTrustRunners(context.Background(), active, current, make(map[string]bool)); retired != 0 { + t.Fatalf("retired runners = %d, want 0 during GitHub 503", retired) + } + if len(active) != 1 { + t.Fatal("GitHub 503 removed the active runner") + } + if got := atomic.LoadInt32(&fake.execCalls); got != 0 { + t.Fatalf("guest lease commands = %d, want 0 when GitHub status is unknown", got) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 0 { + t.Fatalf("provider delete calls = %d, want 0 during GitHub 503", got) + } +} + +func TestHostTrustReconciliationIssuesOneBoundedBusyHandoffLease(t *testing.T) { + provider := &fakeProvider{} + github := &fakeGitHub{runner: gh.Runner{Name: "runner-1", ID: 42, Status: "online", Busy: true}, found: true} + manager := Manager{ + Config: config.Config{Image: config.ImageConfig{HostTrustMode: config.HostTrustModeOverlay, HostTrustScopes: []string{"system"}}}, + Provider: provider, + GitHub: github, + } + current := hosttrust.Snapshot{ + Generation: "g1", + HostOS: "linux", + Scopes: []string{"system"}, + Certificates: []hosttrust.Certificate{{Name: "root.crt", PEM: []byte("pem")}}, + CollectedAt: time.Now().UTC(), + } + active := map[string]ProvisionedInstance{"runner-1": {Name: "runner-1", RunnerID: 42, HostTrustGeneration: "g1"}} + busyHandoff := make(map[string]bool) + + manager.reconcileHostTrustRunners(context.Background(), active, current, busyHandoff) + manager.reconcileHostTrustRunners(context.Background(), active, current, busyHandoff) + + leases := hostTrustLeaseInputs(provider) + if len(leases) != 1 { + t.Fatalf("busy handoff lease writes = %d, want 1", len(leases)) + } + var lease hostTrustLease + if err := json.Unmarshal([]byte(leases[0]), &lease); err != nil { + t.Fatal(err) + } + expires, err := time.Parse(time.RFC3339Nano, lease.ExpiresAt) + if err != nil { + t.Fatal(err) + } + if remaining := time.Until(expires); remaining < hostTrustHandoffLease-5*time.Second || remaining > hostTrustHandoffLease+time.Second { + t.Fatalf("busy handoff lease remaining lifetime = %s, want about %s", remaining, hostTrustHandoffLease) + } + + github.runner.Busy = false + manager.reconcileHostTrustRunners(context.Background(), active, current, busyHandoff) + github.runner.Busy = true + manager.reconcileHostTrustRunners(context.Background(), active, current, busyHandoff) + if leases = hostTrustLeaseInputs(provider); len(leases) != 3 { + t.Fatalf("lease writes after idle and second busy transition = %d, want 3", len(leases)) + } +} + +func TestHostTrustLeaseUsesOneBoundedAtomicGuestCommand(t *testing.T) { + fake := &fakeProvider{} + manager := Manager{ + Config: config.Config{Image: config.ImageConfig{HostTrustMode: config.HostTrustModeOverlay, HostTrustScopes: []string{"system"}}}, + Provider: fake, + } + snapshot := hosttrust.Snapshot{ + Generation: "g1", + HostOS: "linux", + Scopes: []string{"system"}, + Certificates: []hosttrust.Certificate{{Name: "root.crt", PEM: []byte("pem")}}, + CollectedAt: time.Now().UTC(), + } + if err := manager.issueHostTrustLease(context.Background(), "runner-1", snapshot); err != nil { + t.Fatal(err) + } + if got := atomic.LoadInt32(&fake.execCalls); got != 1 { + t.Fatalf("guest commands = %d, want one atomic lease write", got) + } + fake.mu.Lock() + defer fake.mu.Unlock() + if len(fake.commands) != 1 || !strings.Contains(fake.commands[0], "install -d -m 0755 /run/epar") || !strings.Contains(fake.commands[0], "mv -f") { + t.Fatalf("lease command = %q, want directory creation and atomic rename in one command", strings.Join(fake.commands, "\n")) + } + if len(fake.execOptions) != 1 || !strings.Contains(fake.execOptions[0].Stdin, `"expiresAt"`) { + t.Fatal("lease payload was not supplied to the atomic guest command") + } +} + +func TestHostTrustLeaseWriteTimeoutIsReportedWithoutBlocking(t *testing.T) { + oldTimeout := hostTrustWriteTimeout + hostTrustWriteTimeout = 5 * time.Millisecond + t.Cleanup(func() { hostTrustWriteTimeout = oldTimeout }) + fake := &fakeProvider{execFunc: func(ctx context.Context, _ string, _ []string, _ provider.ExecOptions) (provider.ExecResult, error) { + <-ctx.Done() + return provider.ExecResult{}, ctx.Err() + }} + manager := Manager{ + Config: config.Config{Image: config.ImageConfig{HostTrustMode: config.HostTrustModeOverlay, HostTrustScopes: []string{"system"}}}, + Provider: fake, + } + snapshot := hosttrust.Snapshot{ + Generation: "g1", + HostOS: "linux", + Scopes: []string{"system"}, + Certificates: []hosttrust.Certificate{{Name: "root.crt", PEM: []byte("pem")}}, + CollectedAt: time.Now().UTC(), + } + started := time.Now() + err := manager.issueHostTrustLease(context.Background(), "runner-1", snapshot) + if err == nil || !strings.Contains(err.Error(), "host trust lease write exceeded") { + t.Fatalf("issueHostTrustLease() error = %v, want bounded-timeout detail", err) + } + if elapsed := time.Since(started); elapsed > time.Second { + t.Fatalf("lease timeout took %s, want less than one second", elapsed) + } +} + +func hostTrustLeaseInputs(provider *fakeProvider) []string { + provider.mu.Lock() + defer provider.mu.Unlock() + var leases []string + for _, options := range provider.execOptions { + if strings.Contains(options.Stdin, `"expiresAt"`) { + leases = append(leases, options.Stdin) + } + } + return leases +} + func TestHostTrustImageBuildRetriesChangedGenerationBeforePublishing(t *testing.T) { + t.Setenv("EPAR_STATE_HOME", filepath.Join(t.TempDir(), "host-state")) root := t.TempDir() for _, dir := range []string{ filepath.Join(root, "scripts", "guest", "ubuntu"), @@ -307,6 +503,10 @@ func TestHostTrustImageBuildRetriesChangedGenerationBeforePublishing(t *testing. writeTestCACertificate(t, secondPath, "Host Root G2") g1 := hostTrustSnapshotFromFile(t, firstPath, "windows", []string{"system", "user"}) g2 := hostTrustSnapshotFromFile(t, secondPath, "windows", []string{"system", "user"}) + var buildTrustBundleContent strings.Builder + for _, certificate := range g1.Certificates { + buildTrustBundleContent.Write(certificate.PEM) + } sequence := []hosttrust.Snapshot{g1, g2, g2, g2} manager := Manager{ Config: config.Config{ @@ -319,7 +519,7 @@ func TestHostTrustImageBuildRetriesChangedGenerationBeforePublishing(t *testing. HostTrustMode: config.HostTrustModeOverlay, HostTrustScopes: []string{"system", "user"}, }, - Provider: config.ProviderConfig{Type: "docker-dind"}, + Provider: config.ProviderConfig{Type: "docker-container"}, Runner: config.RunnerConfig{Ephemeral: true}, Logging: config.LoggingConfig{Directory: "work/logs"}, }, @@ -335,6 +535,11 @@ func TestHostTrustImageBuildRetriesChangedGenerationBeforePublishing(t *testing. value.CollectedAt = time.Now().UTC() return value, nil } + manager.buildTrustResolver = func(context.Context) (hosttrust.Snapshot, error) { + value := g1 + value.CollectedAt = time.Now().UTC() + return value, nil + } oldLogged := runHostLoggedCommand oldOutput := runHostOutputCommand oldQuiet := runHostQuietCommand @@ -350,12 +555,21 @@ func TestHostTrustImageBuildRetriesChangedGenerationBeforePublishing(t *testing. builds := 0 tagged := false runHostLoggedCommand = func(_ context.Context, _ string, _, _ io.Writer, name string, args ...string) error { - if name == "docker" && len(args) > 0 && args[0] == "build" { + if name == "docker" && len(args) > 1 && args[0] == "buildx" && args[1] == "build" { builds++ } return nil } - runHostOutputCommand = func(context.Context, string, ...string) (string, error) { + runHostOutputCommand = func(_ context.Context, _ string, args ...string) (string, error) { + if len(args) > 1 && args[0] == "buildx" && args[1] == "inspect" { + return "", errors.New("builder not found") + } + if len(args) > 3 && args[0] == "exec" && strings.Contains(args[3], "/certs/") { + return buildTrustBundleContent.String(), nil + } + if len(args) > 1 && args[0] == "exec" { + return "# epar-build-trust-generation=" + g1.Generation + "\n[registry.\"docker.io\"]\n", nil + } return `["source@sha256:1234"]`, nil } runHostQuietCommand = func(context.Context, string, ...string) error { return nil } @@ -366,7 +580,25 @@ func TestHostTrustImageBuildRetriesChangedGenerationBeforePublishing(t *testing. } return nil } - if err := manager.buildDockerDindImage(context.Background(), ImageBuildOptions{Replace: true}, t.TempDir()); err != nil { + runnerPackage := []byte("test-actions-runner-package") + runnerServer := httptest.NewServer(http.HandlerFunc(func(response http.ResponseWriter, _ *http.Request) { + _, _ = response.Write(runnerPackage) + })) + defer runnerServer.Close() + runnerDigest := sha256.Sum256(runnerPackage) + manifest := ImageManifest{ + SchemaVersion: imageManifestSchemaVersion, + ProviderType: "docker-container", + SourceType: config.ImageSourceDockerImage, + SourceImage: "source:latest", + OutputImage: "output:latest", + RunnerSelector: "latest", + RunnerVersion: "2.332.0", + RunnerAssetName: "actions-runner-linux-x64-2.332.0.tar.gz", + RunnerAssetURL: runnerServer.URL, + RunnerAssetDigest: fmt.Sprintf("sha256:%x", runnerDigest), + } + if err := manager.buildDockerContainerImage(context.Background(), ImageBuildOptions{Replace: true, Manifest: &manifest}, t.TempDir()); err != nil { t.Fatal(err) } if builds != 2 { @@ -377,6 +609,15 @@ func TestHostTrustImageBuildRetriesChangedGenerationBeforePublishing(t *testing. } } +func TestBuildTrustScopesAreIndependentFromDisabledRunnerOverlay(t *testing.T) { + if got := buildTrustScopes(config.HostTrustModeDisabled, []string{hosttrust.ScopeSystem, hosttrust.ScopeUser}); !slices.Equal(got, []string{hosttrust.ScopeSystem}) { + t.Fatalf("disabled runner build scopes = %v, want system only", got) + } + if got := buildTrustScopes(config.HostTrustModeOverlay, []string{hosttrust.ScopeUser}); !slices.Equal(got, []string{hosttrust.ScopeSystem, hosttrust.ScopeUser}) { + t.Fatalf("user-overlay build scopes = %v, want mandatory system plus opted-in user", got) + } +} + func hostTrustSnapshotFromFile(t *testing.T, path, hostOS string, scopes []string) hosttrust.Snapshot { t.Helper() content, err := os.ReadFile(path) diff --git a/internal/pool/image_acquisition_test.go b/internal/pool/image_acquisition_test.go new file mode 100644 index 0000000..e44ea53 --- /dev/null +++ b/internal/pool/image_acquisition_test.go @@ -0,0 +1,265 @@ +package pool + +import ( + "bytes" + "context" + "encoding/json" + "io" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/image" + "github.com/solutionforest/ephemeral-action-runner/internal/logging" +) + +func TestBuildxProgressHeartbeatDefaultIsFiveSeconds(t *testing.T) { + if got, want := buildxProgressHeartbeatInterval, 5*time.Second; got != want { + t.Fatalf("Buildx progress heartbeat = %s, want %s", got, want) + } +} + +func TestBuildxProgressUsesManagerLoggerAndPreservesRawTranscript(t *testing.T) { + root := t.TempDir() + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: root, + ManagerSinks: logging.SinkConsole, + TranscriptSinks: logging.SinkFile, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + + cfg := config.Default() + cfg.Provider.Type = "docker-sandboxes" + cfg.Logging.Directory = root + cfg.Logging.ManagerSinks = []string{"console"} + cfg.Logging.TranscriptSinks = []string{"file"} + manager := Manager{Config: cfg, Logging: runtime} + logPath := filepath.Join(root, "builds", "docker-sandboxes-test.docker-build.log") + previousTerminal := dockerPullProgressTerminal + dockerPullProgressTerminal = func() bool { return false } + t.Cleanup(func() { dockerPullProgressTerminal = previousTerminal }) + previousLogged := runHostLoggedCommand + runHostLoggedCommand = func(_ context.Context, _ string, stdout, stderr io.Writer, name string, args ...string) error { + if name != "docker" || len(args) < 2 || args[0] != "buildx" || args[1] != "build" { + t.Fatalf("unexpected command: %s %v", name, args) + } + digest := "sha256:" + strings.Repeat("d", 64) + _, _ = io.WriteString(stderr, "#12 "+digest+" 1.00GB / 2.00GB 10.0s\n") + _, _ = io.WriteString(stdout, "#12 "+digest+" 2.00GB / 2.00GB 20.0s done\n") + return nil + } + t.Cleanup(func() { runHostLoggedCommand = previousLogged }) + + if err := manager.runHostBuildxLogged(context.Background(), logPath, "docker", "buildx", "build"); err != nil { + t.Fatal(err) + } + if err := manager.releaseTranscript(logPath); err != nil { + t.Fatal(err) + } + + consoleText := console.String() + if !strings.Contains(consoleText, "Docker Sandboxes template build: 953.7 MiB/1.9 GiB (50%); 0/1 layer downloads complete; BuildKit step #12") { + t.Fatalf("manager console did not receive Buildx progress: %q", consoleText) + } + if !strings.Contains(consoleText, "Docker Sandboxes template Buildx phase complete; finalizing evidence and importing the template next") { + t.Fatalf("manager console did not receive Buildx completion: %q", consoleText) + } + raw, err := os.ReadFile(logPath) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(string(raw), "1.00GB / 2.00GB") || !strings.Contains(string(raw), "2.00GB / 2.00GB") { + t.Fatalf("raw Buildx transcript was not preserved: %q", raw) + } +} + +func TestBuildxProgressHeartbeatShowsLongSilentStepIsAlive(t *testing.T) { + root := t.TempDir() + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: root, + ManagerSinks: logging.SinkConsole, + TranscriptSinks: logging.SinkFile, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + + cfg := config.Default() + cfg.Provider.Type = "docker-sandboxes" + cfg.Logging.Directory = root + cfg.Logging.ManagerSinks = []string{"console"} + cfg.Logging.TranscriptSinks = []string{"file"} + manager := Manager{Config: cfg, Logging: runtime} + previousTerminal := dockerPullProgressTerminal + dockerPullProgressTerminal = func() bool { return false } + t.Cleanup(func() { dockerPullProgressTerminal = previousTerminal }) + previousInterval := buildxProgressHeartbeatInterval + buildxProgressHeartbeatInterval = 5 * time.Millisecond + t.Cleanup(func() { buildxProgressHeartbeatInterval = previousInterval }) + previousLogged := runHostLoggedCommand + runHostLoggedCommand = func(_ context.Context, _ string, _, _ io.Writer, _ string, _ ...string) error { + time.Sleep(20 * time.Millisecond) + return nil + } + t.Cleanup(func() { runHostLoggedCommand = previousLogged }) + + if err := manager.runHostBuildxLogged(context.Background(), filepath.Join(root, "builds", "docker-sandboxes-heartbeat.docker-build.log"), "docker", "buildx", "build"); err != nil { + t.Fatal(err) + } + if !strings.Contains(console.String(), "Docker Sandboxes template build: elapsed") { + t.Fatalf("long silent Buildx step had no heartbeat: %q", console.String()) + } +} + +func TestDockerPullProgressUsesManagerLoggerAndPreservesSourceTranscript(t *testing.T) { + root := t.TempDir() + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: root, + ManagerSinks: logging.SinkConsole, + TranscriptSinks: logging.SinkFile, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + + cfg := config.Default() + cfg.Logging.Directory = root + manager := Manager{Config: cfg, Logging: runtime} + logPath := filepath.Join(root, "builds", "source.docker-pull.log") + previousTerminal := dockerPullProgressTerminal + dockerPullProgressTerminal = func() bool { return false } + t.Cleanup(func() { dockerPullProgressTerminal = previousTerminal }) + manager.writeDockerPullProgress(logPath, map[string]image.DockerPullProgress{ + "layer-a": {Current: 512, Total: 1024, Completed: false}, + "layer-b": {Completed: true}, + }) + transcriptWriter, err := manager.transcript(logPath, "", "docker-pull") + if err != nil { + t.Fatal(err) + } + writeDockerPullEvent(transcriptWriter.Stdout, image.DockerPullEvent{ID: "layer-a", Status: "Downloading"}) + manager.writeDockerPullNotice(logPath, "Docker source pull complete: example.invalid/source:latest") + if err := manager.releaseTranscript(logPath); err != nil { + t.Fatal(err) + } + + consoleText := console.String() + if !strings.Contains(consoleText, "Docker source pull: 1/2 layers complete; 512 B/1.0 KiB (50%); 1 layer(s) size pending") { + t.Fatalf("manager console did not receive pull progress: %q", consoleText) + } + if !strings.Contains(consoleText, "Docker source pull complete: example.invalid/source:latest") { + t.Fatalf("manager console did not receive pull completion notice: %q", consoleText) + } + transcript, err := os.ReadFile(logPath) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(string(transcript), "Docker source pull complete: example.invalid/source:latest") { + t.Fatalf("source transcript did not retain pull completion notice: %q", transcript) + } + if !strings.Contains(string(transcript), "layer-a Downloading") { + t.Fatalf("source transcript did not retain raw pull event: %q", transcript) + } +} + +func TestDockerPullProgressHonorsManagerJSONConsoleFormat(t *testing.T) { + root := t.TempDir() + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: root, + ManagerSinks: logging.SinkConsole, + ManagerConsoleFormat: logging.FormatJSON, + TranscriptSinks: logging.SinkFile, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + + cfg := config.Default() + cfg.Logging.Directory = root + cfg.Logging.ManagerConsoleFormat = "json" + cfg.Provider.Type = "docker-container" + manager := Manager{Config: cfg, Logging: runtime} + previousTerminal := dockerPullProgressTerminal + dockerPullProgressTerminal = func() bool { return true } + t.Cleanup(func() { dockerPullProgressTerminal = previousTerminal }) + logPath := filepath.Join(root, "builds", "source.docker-pull.log") + manager.writeDockerPullProgress(logPath, map[string]image.DockerPullProgress{"layer-a": {Current: 1, Total: 2}}) + + var record map[string]any + if err := json.Unmarshal(console.Bytes(), &record); err != nil { + t.Fatalf("decode manager JSON console: %v: %q", err, console.String()) + } + if record["msg"] != "Docker source pull: 0/1 layers complete; 1 B/2 B (50%)" || record["provider"] != "docker-container" || record["operation"] != "docker-pull" || record["logPath"] != logPath { + t.Fatalf("manager JSON console missing pull context: %#v", record) + } +} + +func TestDockerPullProgressUsesSingleLineTerminalDisplayForTextManagerConsole(t *testing.T) { + root := t.TempDir() + var managerConsole, terminalConsole bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: root, + ManagerSinks: logging.SinkConsole, + TranscriptSinks: logging.SinkFile, + Stdout: &managerConsole, + Stderr: &managerConsole, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + + previousTerminal, previousConsole := dockerPullProgressTerminal, dockerPullProgressConsole + dockerPullProgressTerminal = func() bool { return true } + dockerPullProgressConsole = &terminalConsole + t.Cleanup(func() { + dockerPullProgressTerminal = previousTerminal + dockerPullProgressConsole = previousConsole + }) + + manager := Manager{Config: config.Default(), Logging: runtime} + manager.writeDockerPullProgress("source.docker-pull.log", map[string]image.DockerPullProgress{"layer-a": {Current: 1, Total: 2}}) + + if got, want := terminalConsole.String(), "\r\033[2KDocker source pull: 0/1 layers complete; 1 B/2 B (50%)"; got != want { + t.Fatalf("terminal pull progress = %q, want %q", got, want) + } + if got := managerConsole.String(); got != "" { + t.Fatalf("interactive pull progress was duplicated through manager logger: %q", got) + } +} + +func TestBuildxDockerArchiveDestinationRecognizesDirectExporter(t *testing.T) { + path := filepath.Join(t.TempDir(), "runner-template.tar.partial") + for _, args := range [][]string{ + {"buildx", "build", "--output", "type=docker,dest=" + path}, + {"buildx", "build", "--output=type=docker,dest=" + path}, + } { + if got := buildxDockerArchiveDestination(args); got != path { + t.Fatalf("buildxDockerArchiveDestination(%v) = %q, want %q", args, got, path) + } + } + if got := buildxDockerArchiveDestination([]string{"buildx", "build", "--load"}); got != "" { + t.Fatalf("non-archive output returned %q", got) + } +} diff --git a/internal/pool/image_manifest.go b/internal/pool/image_manifest.go deleted file mode 100644 index d37dc3f..0000000 --- a/internal/pool/image_manifest.go +++ /dev/null @@ -1,503 +0,0 @@ -package pool - -import ( - "context" - "crypto/sha256" - "encoding/hex" - "encoding/json" - "errors" - "fmt" - "os" - "path/filepath" - "sort" - "strings" - - "github.com/solutionforest/ephemeral-action-runner/internal/config" - "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" - "github.com/solutionforest/ephemeral-action-runner/internal/provider" -) - -const ( - imageManifestSchemaVersion = 1 - imageManifestGuestPath = "/opt/epar/image-manifest.json" - imageManifestLabel = "org.solutionforest.epar.manifest-sha256" -) - -type ImageManifest struct { - SchemaVersion int `json:"schemaVersion"` - ProviderType string `json:"providerType"` - ProviderPlatform string `json:"providerPlatform,omitempty"` - ProviderRosettaTag string `json:"providerRosettaTag,omitempty"` - SourceType string `json:"sourceType,omitempty"` - SourceImage string `json:"sourceImage"` - SourcePlatform string `json:"sourcePlatform,omitempty"` - SourceDigest string `json:"sourceDigest,omitempty"` - OutputImage string `json:"outputImage"` - RunnerVersion string `json:"runnerVersion"` - UpstreamCommit string `json:"upstreamCommit,omitempty"` - EPARScripts []fileDigest `json:"eparScripts,omitempty"` - CustomInstallScripts []fileDigest `json:"customInstallScripts,omitempty"` - TrustedCACertificates []fileDigest `json:"trustedCaCertificates,omitempty"` - HostTrust *hostTrustImageMetadata `json:"hostTrust,omitempty"` -} - -type fileDigest struct { - Path string `json:"path"` - SHA256 string `json:"sha256"` -} - -type storedImageManifest struct { - Hash string `json:"hash"` - Manifest ImageManifest `json:"manifest"` -} - -type sourceCacheManifest struct { - SourceImage string `json:"sourceImage"` - SourcePlatform string `json:"sourcePlatform,omitempty"` - SourceDigest string `json:"sourceDigest,omitempty"` -} - -func (m *Manager) EnsureImage(ctx context.Context) error { - manifest, err := m.desiredImageManifest(ctx) - if err != nil { - return err - } - hash, err := imageManifestHash(manifest) - if err != nil { - return err - } - if m.DryRun { - m.infof("[dry-run] would ensure image %s has manifest %s\n", m.Config.Image.OutputImage, hash) - return m.BuildImage(ctx, ImageBuildOptions{Replace: true, Manifest: &manifest}) - } - state, err := m.currentImageState(ctx, hash) - if err != nil { - return err - } - switch state { - case imageStateCurrent: - m.infof("image is current: %s\n", m.Config.Image.OutputImage) - return nil - case imageStateMissing: - m.infof("image is missing; building %s\n", m.Config.Image.OutputImage) - case imageStateOutdated: - m.infof("image is outdated or not aligned with config; rebuilding %s\n", m.Config.Image.OutputImage) - } - return m.BuildImage(ctx, ImageBuildOptions{Replace: true, Manifest: &manifest}) -} - -type imageState int - -const ( - imageStateMissing imageState = iota - imageStateOutdated - imageStateCurrent -) - -func (m *Manager) currentImageState(ctx context.Context, wantHash string) (imageState, error) { - switch m.Config.Provider.Type { - case "docker-dind": - got, exists, err := m.currentDockerDindManifestHash(ctx) - if err != nil { - return imageStateMissing, err - } - if !exists { - return imageStateMissing, nil - } - if got != wantHash { - return imageStateOutdated, nil - } - return imageStateCurrent, nil - case "wsl": - got, exists, err := m.currentWSLManifestHash() - if err != nil { - return imageStateMissing, err - } - if !exists { - return imageStateMissing, nil - } - if got != wantHash { - return imageStateOutdated, nil - } - return imageStateCurrent, nil - case "tart": - return m.currentTartImageState(ctx) - default: - return imageStateMissing, fmt.Errorf("unsupported provider.type %q", m.Config.Provider.Type) - } -} - -func (m *Manager) currentDockerDindManifestHash(ctx context.Context) (string, bool, error) { - output := strings.TrimSpace(m.Config.Image.OutputImage) - if output == "" { - return "", false, fmt.Errorf("image.outputImage is required") - } - out, err := runHostOutputCommand(ctx, "docker", "image", "inspect", "--format", "{{json .Config.Labels}}", output) - if err != nil { - if dockerInspectMeansMissing(err) { - return "", false, nil - } - return "", false, err - } - labels := map[string]string{} - if err := json.Unmarshal([]byte(strings.TrimSpace(out)), &labels); err != nil { - return "", true, fmt.Errorf("parse Docker image labels for %s: %w", output, err) - } - return labels[imageManifestLabel], true, nil -} - -func (m *Manager) currentWSLManifestHash() (string, bool, error) { - outputPath := config.ProjectPath(m.ProjectRoot, m.Config.Image.OutputImage) - if _, err := os.Stat(outputPath); err != nil { - if errors.Is(err, os.ErrNotExist) { - return "", false, nil - } - return "", false, err - } - stored, err := readStoredImageManifest(wslImageManifestSidecarPath(outputPath)) - if err != nil { - if errors.Is(err, os.ErrNotExist) { - return "", true, nil - } - return "", true, err - } - return stored.Hash, true, nil -} - -func (m *Manager) currentTartImageState(ctx context.Context) (imageState, error) { - instances, err := m.Provider.List(ctx) - if err != nil { - return imageStateMissing, err - } - for _, instance := range instances { - if instance.Name == m.Config.Image.OutputImage || instance.Source == m.Config.Image.OutputImage { - return imageStateCurrent, nil - } - } - return imageStateMissing, nil -} - -func (m *Manager) desiredImageManifest(ctx context.Context) (ImageManifest, error) { - snapshot, err := m.resolveHostTrust(ctx) - if err != nil { - return ImageManifest{}, err - } - return m.desiredImageManifestWithHostTrust(ctx, snapshot) -} - -func (m *Manager) desiredImageManifestWithHostTrust(ctx context.Context, snapshot hosttrust.Snapshot) (ImageManifest, error) { - sourceType := m.Config.Image.SourceType - if sourceType == "" { - sourceType = config.ImageSourceRootFSTar - if m.Config.Provider.Type == "docker-dind" { - sourceType = config.ImageSourceDockerImage - } - } - manifest := ImageManifest{ - SchemaVersion: imageManifestSchemaVersion, - ProviderType: m.Config.Provider.Type, - ProviderPlatform: m.Config.Provider.Platform, - ProviderRosettaTag: m.Config.Provider.RosettaTag, - SourceType: sourceType, - SourceImage: m.Config.Image.SourceImage, - SourcePlatform: m.Config.Image.SourcePlatform, - OutputImage: m.Config.Image.OutputImage, - RunnerVersion: m.Config.Image.RunnerVersion, - HostTrust: hostTrustMetadata(snapshot), - } - switch sourceType { - case config.ImageSourceDockerImage: - if m.Config.Provider.Type == "docker-dind" || m.Config.Provider.Type == "wsl" { - digest, err := m.refreshDockerSourceDigest(ctx) - if err != nil { - return manifest, err - } - manifest.SourceDigest = digest - } - case config.ImageSourceRootFSTar: - if m.Config.Provider.Type == "wsl" { - digest, err := m.fileSHA256(config.ProjectPath(m.ProjectRoot, m.Config.Image.SourceImage)) - if err != nil { - return manifest, err - } - manifest.SourceDigest = digest - } - case "": - default: - return manifest, fmt.Errorf("unsupported image.sourceType %q", sourceType) - } - scripts, err := m.eparScriptDigests() - if err != nil { - return manifest, err - } - manifest.EPARScripts = scripts - customScripts, err := m.customInstallScriptDigests() - if err != nil { - return manifest, err - } - manifest.CustomInstallScripts = customScripts - trustedCACertificates, err := m.trustedCACertificateDigests() - if err != nil { - return manifest, err - } - manifest.TrustedCACertificates = trustedCACertificates - if m.runnerImagesCopyMode() != runnerImagesCopyNone { - commit, err := m.runnerImagesCommit() - if err != nil { - return manifest, err - } - manifest.UpstreamCommit = commit - } - return manifest, nil -} - -func (m *Manager) refreshDockerSourceDigest(ctx context.Context) (string, error) { - var digest string - err := m.timeStartupStage("source_image_pull", func() error { - var err error - digest, err = m.refreshDockerSourceDigestUntimed(ctx) - return err - }) - return digest, err -} - -func (m *Manager) refreshDockerSourceDigestUntimed(ctx context.Context) (string, error) { - if m.DryRun { - return "dry-run", nil - } - image := strings.TrimSpace(m.Config.Image.SourceImage) - if image == "" { - return "", fmt.Errorf("image.sourceImage is required when image.sourceType=docker-image") - } - platform := strings.TrimSpace(m.Config.Image.SourcePlatform) - logPath := m.buildLogPath(imageLogStem(m.Config.Image.OutputImage) + ".source.log") - defer m.releaseTranscript(logPath) - m.infof("refreshing Docker source image %s\n", image) - if err := pullDockerSourceCommand(m, ctx, dockerSourcePullOptions{ - Image: image, - Platform: platform, - LogPath: logPath, - AnnounceRemoteSize: true, - }); err != nil { - return "", fmt.Errorf("refresh Docker source image %s: %w", image, err) - } - digestsJSON, err := runHostOutputCommand(ctx, "docker", "image", "inspect", "--format", "{{json .RepoDigests}}", image) - if err != nil { - return "", err - } - var digests []string - if err := json.Unmarshal([]byte(strings.TrimSpace(digestsJSON)), &digests); err != nil { - return "", fmt.Errorf("parse Docker source RepoDigests for %s: %w", image, err) - } - sort.Strings(digests) - if len(digests) > 0 { - digest := digests[0] - m.writeDockerPullNotice(logPath, "Docker source image digest: "+digest) - return digest, nil - } - imageID, err := runHostOutputCommand(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", image) - if err != nil { - return "", err - } - digest := strings.TrimSpace(imageID) - m.writeDockerPullNotice(logPath, "Docker source image ID: "+digest) - return digest, nil -} - -func (m *Manager) eparScriptDigests() ([]fileDigest, error) { - var roots []string - switch m.Config.Provider.Type { - case "docker-dind": - roots = []string{ - filepath.Join(m.ProjectRoot, "scripts", "guest", "ubuntu"), - filepath.Join(m.ProjectRoot, "scripts", "container", "ubuntu"), - } - case "wsl", "tart": - roots = []string{filepath.Join(m.ProjectRoot, "scripts", "guest", "ubuntu")} - default: - return nil, nil - } - var out []fileDigest - for _, root := range roots { - digests, err := m.fileDigestsUnder(root) - if err != nil { - return nil, err - } - out = append(out, digests...) - } - sortFileDigests(out) - return out, nil -} - -func (m *Manager) customInstallScriptDigests() ([]fileDigest, error) { - var out []fileDigest - for _, script := range m.Config.Image.CustomInstallScripts { - path, err := m.customInstallScriptHostPath(script) - if err != nil { - return nil, err - } - digest, err := m.fileDigest(path) - if err != nil { - return nil, err - } - out = append(out, digest) - } - sortFileDigests(out) - return out, nil -} - -func (m *Manager) trustedCACertificateDigests() ([]fileDigest, error) { - if _, err := m.trustedCACertificates(); err != nil { - return nil, err - } - var out []fileDigest - for _, configuredPath := range m.Config.Image.TrustedCACertificatePaths { - path := config.ProjectPath(m.ProjectRoot, strings.TrimSpace(configuredPath)) - digest, err := m.fileDigest(path) - if err != nil { - return nil, err - } - out = append(out, digest) - } - sortFileDigests(out) - return out, nil -} - -func (m *Manager) fileDigestsUnder(root string) ([]fileDigest, error) { - var out []fileDigest - if err := filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error { - if err != nil { - return err - } - if d.IsDir() || !strings.HasSuffix(d.Name(), ".sh") { - return nil - } - digest, err := m.fileDigest(path) - if err != nil { - return err - } - out = append(out, digest) - return nil - }); err != nil { - return nil, err - } - sortFileDigests(out) - return out, nil -} - -func (m *Manager) fileDigest(path string) (fileDigest, error) { - sha, err := m.fileSHA256(path) - if err != nil { - return fileDigest{}, err - } - rel, err := filepath.Rel(m.ProjectRoot, path) - if err != nil || strings.HasPrefix(rel, ".."+string(filepath.Separator)) || rel == ".." || filepath.IsAbs(rel) { - rel = path - } - return fileDigest{Path: filepath.ToSlash(filepath.Clean(rel)), SHA256: sha}, nil -} - -func (m *Manager) fileSHA256(path string) (string, error) { - if m.DryRun { - return "dry-run", nil - } - content, err := os.ReadFile(path) - if err != nil { - return "", err - } - sum := sha256.Sum256(content) - return hex.EncodeToString(sum[:]), nil -} - -func sortFileDigests(values []fileDigest) { - sort.Slice(values, func(i, j int) bool { - return values[i].Path < values[j].Path - }) -} - -func imageManifestHash(manifest ImageManifest) (string, error) { - content, err := json.Marshal(manifest) - if err != nil { - return "", err - } - sum := sha256.Sum256(content) - return hex.EncodeToString(sum[:]), nil -} - -func storedImageManifestContent(manifest ImageManifest) (string, string, error) { - hash, err := imageManifestHash(manifest) - if err != nil { - return "", "", err - } - content, err := json.MarshalIndent(storedImageManifest{Hash: hash, Manifest: manifest}, "", " ") - if err != nil { - return "", "", err - } - return string(content) + "\n", hash, nil -} - -func readStoredImageManifest(path string) (storedImageManifest, error) { - content, err := os.ReadFile(path) - if err != nil { - return storedImageManifest{}, err - } - var stored storedImageManifest - if err := json.Unmarshal(content, &stored); err != nil { - return storedImageManifest{}, err - } - return stored, nil -} - -func writeStoredImageManifest(path string, manifest ImageManifest) error { - content, _, err := storedImageManifestContent(manifest) - if err != nil { - return err - } - if err := os.MkdirAll(filepath.Dir(path), 0755); err != nil { - return err - } - return os.WriteFile(path, []byte(content), 0644) -} - -func (m *Manager) installImageManifest(ctx context.Context, vmName string, manifest ImageManifest) error { - content, _, err := storedImageManifestContent(manifest) - if err != nil { - return err - } - return provider.CopyText(ctx, m.Provider, vmName, imageManifestGuestPath, "0644", content) -} - -func wslImageManifestSidecarPath(outputPath string) string { - return outputPath + ".epar-manifest.json" -} - -func sourceCacheManifestPath(rootfsPath string) string { - return rootfsPath + ".source.json" -} - -func sourceCacheMatches(path string, want sourceCacheManifest) bool { - content, err := os.ReadFile(path) - if err != nil { - return false - } - var got sourceCacheManifest - if err := json.Unmarshal(content, &got); err != nil { - return false - } - return got == want -} - -func writeSourceCacheManifest(path string, manifest sourceCacheManifest) error { - content, err := json.MarshalIndent(manifest, "", " ") - if err != nil { - return err - } - return os.WriteFile(path, append(content, '\n'), 0644) -} - -func dockerInspectMeansMissing(err error) bool { - text := strings.ToLower(err.Error()) - return strings.Contains(text, "no such image") || - strings.Contains(text, "no such object") || - strings.Contains(text, "not found") -} diff --git a/internal/pool/image_manifest_test.go b/internal/pool/image_manifest_test.go index ddd03a6..c3841b0 100644 --- a/internal/pool/image_manifest_test.go +++ b/internal/pool/image_manifest_test.go @@ -41,7 +41,7 @@ func TestImageManifestHashChangesWithImageInputs(t *testing.T) { } hash := func() string { t.Helper() - manifest, err := manager.desiredImageManifest(context.Background()) + manifest, err := manager.desiredLocalImageManifest(context.Background()) if err != nil { t.Fatal(err) } @@ -87,14 +87,14 @@ func TestImageManifestHashChangesWithImageInputs(t *testing.T) { } } -func TestDockerDindImageStateUsesManifestLabel(t *testing.T) { +func TestDockerContainerImageStateUsesManifestLabel(t *testing.T) { oldOutput := runHostOutputCommand t.Cleanup(func() { runHostOutputCommand = oldOutput }) manager := Manager{Config: config.Config{ Image: config.ImageConfig{OutputImage: "epar-test"}, - Provider: config.ProviderConfig{Type: "docker-dind"}, + Provider: config.ProviderConfig{Type: "docker-container"}, }} runHostOutputCommand = func(context.Context, string, ...string) (string, error) { diff --git a/internal/pool/image_service.go b/internal/pool/image_service.go new file mode 100644 index 0000000..d175039 --- /dev/null +++ b/internal/pool/image_service.go @@ -0,0 +1,427 @@ +package pool + +import ( + "context" + "fmt" + "io" + "os" + "os/exec" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" + artifactimage "github.com/solutionforest/ephemeral-action-runner/internal/image" + "github.com/solutionforest/ephemeral-action-runner/internal/logging" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + "golang.org/x/term" +) + +type ImageBuildOptions = artifactimage.ImageBuildOptions +type ImageManifest = artifactimage.Manifest +type imageState = artifactimage.ImageState +type runnerImagesCopyMode = artifactimage.RunnerImagesCopyMode +type dockerSourcePullOptions = artifactimage.DockerSourcePullOptions +type sourceCacheManifest = artifactimage.SourceCacheManifest +type TrustedCACertificate = artifactimage.TrustedCACertificate + +const ( + imageStateMissing = artifactimage.ImageStateMissing + imageStateCurrent = artifactimage.ImageStateCurrent + imageStateOutdated = artifactimage.ImageStateOutdated + imageManifestSchemaVersion = artifactimage.ManifestSchemaVersion + imageManifestLabel = artifactimage.ManifestLabel + runnerImagesCopyNone = artifactimage.RunnerImagesCopyNone + runnerImagesCopySubset = artifactimage.RunnerImagesCopySubset + trustedCAGuestDir = "/usr/local/share/ca-certificates/epar" +) + +var pullDockerSourceCommand = (*Manager).pullDockerSource + +var dockerPullProgressTerminal = func() bool { + return term.IsTerminal(int(os.Stdout.Fd())) +} + +var dockerPullProgressConsole io.Writer = os.Stdout + +var ( + runHostCommand = runHost + runHostLoggedCommand = runHostLogged + runHostOutputCommand = runHostOutput + runHostOutputToCommand = runHostOutputTo + runHostQuietCommand = runHostQuiet +) + +func (m *Manager) imageCoordinator() *artifactimage.Coordinator { + coordinator := artifactimage.NewCoordinator(m.Config, m.Provider, m.Lifecycle, m.ProjectRoot, m.DryRun, imageEnvironment{manager: m}) + coordinator.ConfigPath = m.ConfigPath + coordinator.Clock = m.currentTime + return coordinator +} + +func (m *Manager) UpdateUpstream(ctx context.Context) error { + return m.imageCoordinator().UpdateUpstream(ctx) +} + +func (m *Manager) BuildImage(ctx context.Context, options ImageBuildOptions) error { + return m.imageCoordinator().BuildImage(ctx, options) +} + +func (m *Manager) EnsureImage(ctx context.Context) error { + m.imageEnsureMu.Lock() + defer m.imageEnsureMu.Unlock() + if m.imageEnsured { + return nil + } + if err := m.imageCoordinator().EnsureImage(ctx); err != nil { + return err + } + if err := m.pinDockerRuntimeImage(ctx); err != nil { + return err + } + m.imageEnsured = true + return nil +} + +func (m *Manager) UpdateImage(ctx context.Context) error { + m.imageEnsureMu.Lock() + defer m.imageEnsureMu.Unlock() + if err := m.imageCoordinator().UpdateImage(ctx); err != nil { + return err + } + if err := m.pinDockerRuntimeImage(ctx); err != nil { + return err + } + m.imageEnsured = true + return nil +} + +func (m *Manager) CheckRemoteImageUpdate(ctx context.Context, now time.Time) (artifactimage.RemoteUpdateCheck, error) { + m.imageEnsureMu.Lock() + defer m.imageEnsureMu.Unlock() + return m.imageCoordinator().CheckRemoteUpdate(ctx, now) +} + +func (m *Manager) ApplyPendingImageUpdate(ctx context.Context, now time.Time) error { + m.imageEnsureMu.Lock() + defer m.imageEnsureMu.Unlock() + if err := m.imageCoordinator().ApplyPendingUpdate(ctx, now); err != nil { + return err + } + if err := m.pinDockerRuntimeImage(ctx); err != nil { + return err + } + m.imageEnsured = true + return nil +} + +func (m *Manager) ImageUpdatePolicyStatus() (artifactimage.UpdatePolicyStatus, error) { + return m.imageCoordinator().UpdatePolicyStatus() +} + +func (m *Manager) DeferPendingImageUpdate(reason string) error { + m.imageEnsureMu.Lock() + defer m.imageEnsureMu.Unlock() + return m.imageCoordinator().DeferPendingUpdate(reason) +} + +func (m *Manager) pinDockerRuntimeImage(ctx context.Context) error { + if m.Config.Provider.Type != "docker-container" || m.DryRun { + return nil + } + identity, err := runHostOutputCommand(ctx, "docker", "image", "inspect", "--format", "{{.Id}}", m.Config.Image.OutputImage) + if err != nil { + return fmt.Errorf("resolve immutable Docker Container runtime image: %w", err) + } + identity = strings.TrimSpace(identity) + if identity == "" { + return fmt.Errorf("Docker Container runtime image returned an empty immutable identity") + } + // Keep user configuration as the desired mutable name while the live + // manager and all replacements consume the exact verified image ID. + m.Config.Provider.SourceImage = identity + return nil +} + +func (m *Manager) RefreshScripts(ctx context.Context) error { + return m.imageCoordinator().RefreshScripts(ctx) +} + +func (m *Manager) pullDockerSource(ctx context.Context, options dockerSourcePullOptions) error { + return m.imageCoordinator().PullDockerSource(ctx, options) +} + +func (m *Manager) writeDockerPullNotice(logPath, message string) { + m.imageCoordinator().WriteDockerPullNotice(logPath, message) +} + +func (m *Manager) writeDockerPullProgress(logPath string, layers map[string]artifactimage.DockerPullProgress) { + m.imageCoordinator().WriteDockerPullProgress(logPath, layers) +} + +func writeDockerPullEvent(writer io.Writer, event artifactimage.DockerPullEvent) { + artifactimage.WriteDockerPullEvent(writer, event) +} + +func (m *Manager) desiredImageManifest(ctx context.Context) (ImageManifest, error) { + return m.imageCoordinator().DesiredImageManifest(ctx) +} + +func (m *Manager) desiredLocalImageManifest(ctx context.Context) (ImageManifest, error) { + return m.imageCoordinator().DesiredLocalImageManifest(ctx) +} + +func (m *Manager) currentImageState(ctx context.Context, wantedHash string) (imageState, error) { + return m.imageCoordinator().CurrentImageState(ctx, wantedHash) +} + +func imageManifestHash(manifest ImageManifest) (string, error) { + return artifactimage.ImageManifestHash(manifest) +} + +func writeStoredImageManifest(path string, manifest ImageManifest) error { + return artifactimage.WriteStoredManifest(path, manifest) +} + +func readStoredImageManifest(path string) (artifactimage.StoredManifest, error) { + return artifactimage.ReadStoredManifest(path) +} + +func wslImageManifestSidecarPath(outputPath string) string { + return artifactimage.WSLImageManifestPath(outputPath) +} + +func sourceCacheManifestPath(rootfsPath string) string { + return artifactimage.SourceCacheManifestPath(rootfsPath) +} + +func writeSourceCacheManifest(path string, manifest sourceCacheManifest) error { + return artifactimage.WriteSourceCacheManifest(path, manifest) +} + +func wslDockerSourceRootfsPath(outputPath string) string { + return artifactimage.WSLSourceRootfsPath(outputPath) +} + +func sourceImageEnvContent(environment []string) string { + return artifactimage.SourceImageEnvContent(environment) +} + +func (m *Manager) prepareDockerContainerBuildContext(buildContext, upstreamDirectory, manifestContent string) error { + return m.imageCoordinator().PrepareDockerContainerBuildContext(buildContext, upstreamDirectory, manifestContent) +} + +func (m *Manager) prepareDockerContainerBuildContextWithHostTrust(buildContext, upstreamDirectory, manifestContent string, snapshot hosttrust.Snapshot) error { + return m.imageCoordinator().PrepareDockerContainerBuildContextWithHostTrust(buildContext, upstreamDirectory, manifestContent, snapshot) +} + +func (m *Manager) buildDockerContainerImage(ctx context.Context, options ImageBuildOptions, upstreamDirectory string) error { + return m.imageCoordinator().BuildDockerContainerImage(ctx, options, upstreamDirectory) +} + +func (m *Manager) trustedCACertificates() ([]TrustedCACertificate, error) { + return m.imageCoordinator().TrustedCACertificates() +} + +func (m *Manager) TrustedCACertificates() ([]TrustedCACertificate, error) { + return m.trustedCACertificates() +} + +func (m *Manager) prepareWSLDockerSourceRootfs(ctx context.Context, outputPath, buildLogPath string, manifest ImageManifest) (string, string, error) { + return m.imageCoordinator().PrepareWSLDockerSourceRootfs(ctx, outputPath, buildLogPath, manifest) +} + +func (m *Manager) runnerImageBuildScripts() []string { + return m.imageCoordinator().RunnerImageBuildScripts() +} + +func (m *Manager) runnerImagesCopyMode() runnerImagesCopyMode { + return m.imageCoordinator().RunnerImagesCopyMode() +} + +func (m *Manager) prepareWSLDockerSourceGuest(ctx context.Context, instance string) error { + return m.imageCoordinator().PrepareWSLDockerSourceGuest(ctx, instance) +} + +func (m *Manager) installCustomInstallScripts(ctx context.Context, instance string) error { + return m.imageCoordinator().InstallCustomInstallScripts(ctx, instance) +} + +func (m *Manager) customInstallScriptHostPath(script string) (string, error) { + return m.imageCoordinator().CustomInstallScriptHostPath(script) +} + +func (m *Manager) enableWSLSystemd(ctx context.Context, instance string) error { + return m.imageCoordinator().EnableWSLSystemd(ctx, instance) +} + +func (m *Manager) startOptions(logPath, instance string) (provider.StartOptions, error) { + transcript, err := m.transcript(logPath, instance, transcriptComponent(logPath)) + if err != nil { + return provider.StartOptions{}, err + } + return provider.StartOptions{ + Network: m.Config.Provider.Network, + RosettaTag: m.Config.Provider.RosettaTag, + LogPath: logPath, + Stdout: transcript.Stdout, + Stderr: transcript.Stderr, + }, nil +} + +type imageEnvironment struct { + manager *Manager +} + +func (environment imageEnvironment) PreflightStorage(operation string, peakBytes uint64) error { + return environment.manager.preflightStorage(operation, peakBytes) +} + +func (environment imageEnvironment) BuildLogPath(name string) string { + return environment.manager.buildLogPath(name) +} + +func (environment imageEnvironment) ReleaseTranscript(path string) error { + return environment.manager.releaseTranscript(path) +} + +func (environment imageEnvironment) Infof(format string, args ...any) { + environment.manager.infof(format, args...) +} + +func (environment imageEnvironment) Warnf(format string, args ...any) { + environment.manager.warnf(format, args...) +} + +func (environment imageEnvironment) RunHostLogged(ctx context.Context, logPath, name string, args ...string) error { + return environment.manager.runHostLogged(ctx, logPath, name, args...) +} + +func (environment imageEnvironment) RunHostBuildxLogged(ctx context.Context, logPath, name string, args ...string) error { + return environment.manager.runHostBuildxLogged(ctx, logPath, name, args...) +} + +func (environment imageEnvironment) RunHost(ctx context.Context, name string, args ...string) error { + return runHostCommand(ctx, name, args...) +} + +func (environment imageEnvironment) RunHostOutput(ctx context.Context, name string, args ...string) (string, error) { + return runHostOutputCommand(ctx, name, args...) +} + +func (environment imageEnvironment) RunHostOutputTo(ctx context.Context, output io.Writer, name string, args ...string) error { + return runHostOutputToCommand(ctx, output, name, args...) +} + +func (environment imageEnvironment) RunHostQuiet(ctx context.Context, name string, args ...string) error { + return runHostQuietCommand(ctx, name, args...) +} + +func (environment imageEnvironment) TimeStartupStage(stage string, fn func() error) error { + return environment.manager.timeStartupStage(stage, fn) +} + +func (environment imageEnvironment) HostTrustEnabled() bool { + return environment.manager.hostTrustEnabled() +} + +func (environment imageEnvironment) ResolveHostTrust(ctx context.Context) (hosttrust.Snapshot, error) { + return environment.manager.resolveHostTrust(ctx) +} + +func (environment imageEnvironment) ResolveBuildTrust(ctx context.Context) (hosttrust.Snapshot, error) { + return environment.manager.resolveBuildTrust(ctx) +} + +func (environment imageEnvironment) WriteHostTrustBuildInputs(buildContext string, snapshot hosttrust.Snapshot) error { + return environment.manager.writeHostTrustBuildInputs(buildContext, snapshot) +} + +func (environment imageEnvironment) ValidateRuntime(ctx context.Context, instance string) error { + return environment.manager.validateRuntime(ctx, instance) +} + +func (environment imageEnvironment) ExecGuest(ctx context.Context, instance string, command []string, options provider.ExecOptions) (provider.ExecResult, error) { + return environment.manager.execGuest(ctx, instance, command, options) +} + +func (environment imageEnvironment) Transcript(path, instance, component string) (*logging.Transcript, error) { + return environment.manager.transcript(path, instance, component) +} + +func (environment imageEnvironment) PullDockerSource(ctx context.Context, options artifactimage.DockerSourcePullOptions) error { + return pullDockerSourceCommand(environment.manager, ctx, options) +} + +func (environment imageEnvironment) LogInfo(message string, args ...any) { + environment.manager.logger().Info(message, args...) +} + +func (environment imageEnvironment) LogWarn(message string, args ...any) { + environment.manager.logger().Warn(message, args...) +} + +func (environment imageEnvironment) ProgressTerminal() bool { + return dockerPullProgressTerminal() +} + +func (environment imageEnvironment) ProgressConsole() io.Writer { + return dockerPullProgressConsole +} + +func (environment imageEnvironment) RunnerName(prefix string, sequence int, now time.Time) string { + return RunnerName(prefix, sequence, now) +} + +func (environment imageEnvironment) TranscriptComponent(path string) string { + return transcriptComponent(path) +} + +func guestText(content []byte) string { + return strings.ReplaceAll(string(content), "\r\n", "\n") +} + +func shellQuote(value string) string { + return "'" + strings.ReplaceAll(value, "'", "'\"'\"'") + "'" +} + +func runHost(ctx context.Context, name string, args ...string) error { + return exec.CommandContext(ctx, name, args...).Run() +} + +func runHostOutput(ctx context.Context, name string, args ...string) (string, error) { + command := exec.CommandContext(ctx, name, args...) + output, err := command.CombinedOutput() + if err != nil { + return "", fmt.Errorf("%s %s failed: %w: %s", name, strings.Join(args, " "), err, strings.TrimSpace(string(output))) + } + return string(output), nil +} + +func runHostOutputTo(ctx context.Context, output io.Writer, name string, args ...string) error { + command := exec.CommandContext(ctx, name, args...) + var stderr strings.Builder + command.Stdout = output + command.Stderr = &stderr + if err := command.Run(); err != nil { + return fmt.Errorf("%s %s failed: %w: %s", name, strings.Join(args, " "), err, strings.TrimSpace(stderr.String())) + } + return nil +} + +func runHostQuiet(ctx context.Context, name string, args ...string) error { + return exec.CommandContext(ctx, name, args...).Run() +} + +func runHostLogged(ctx context.Context, _ string, stdout, stderr io.Writer, name string, args ...string) error { + command := exec.CommandContext(ctx, name, args...) + command.Stdout = stdout + command.Stderr = stderr + if err := command.Run(); err != nil { + return fmt.Errorf("%s %s failed: %w", name, strings.Join(args, " "), err) + } + return nil +} + +func copyFile(source, destination string, mode os.FileMode) error { + return artifactimage.CopyFile(source, destination, mode) +} diff --git a/internal/pool/image_test.go b/internal/pool/image_test.go index b5f057f..22ad183 100644 --- a/internal/pool/image_test.go +++ b/internal/pool/image_test.go @@ -48,14 +48,14 @@ func TestNeedsRunnerImagesSubsetOnlyForBuiltInScripts(t *testing.T) { } } -func TestDockerDindBaseImageDoesNotRequireRunnerImages(t *testing.T) { - manager := Manager{Config: config.Config{Provider: config.ProviderConfig{Type: "docker-dind"}}} +func TestDockerContainerBaseImageDoesNotRequireRunnerImages(t *testing.T) { + manager := Manager{Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}}} if got := manager.runnerImagesCopyMode(); got != runnerImagesCopyNone { t.Fatalf("runnerImagesCopyMode() = %v, want none", got) } } -func TestDockerDindDockerfileRunsBuildStepsAsRoot(t *testing.T) { +func TestDockerContainerDockerfileRunsBuildStepsAsRoot(t *testing.T) { root := t.TempDir() for _, dir := range []string{ filepath.Join(root, "scripts", "guest", "ubuntu"), @@ -74,7 +74,7 @@ func TestDockerDindDockerfileRunsBuildStepsAsRoot(t *testing.T) { }, ProjectRoot: root, } - if err := manager.prepareDockerDindBuildContext(buildCtx, t.TempDir(), `{"hash":"test"}`+"\n"); err != nil { + if err := manager.prepareDockerContainerBuildContext(buildCtx, t.TempDir(), `{"hash":"test"}`+"\n"); err != nil { t.Fatal(err) } content, err := os.ReadFile(filepath.Join(buildCtx, "Dockerfile")) @@ -95,7 +95,7 @@ func TestDockerDindDockerfileRunsBuildStepsAsRoot(t *testing.T) { } } -func TestDockerDindBuildUsesLegacyBuilderCompatibleArgs(t *testing.T) { +func TestDockerContainerBuildUsesDedicatedBuildxBuilder(t *testing.T) { root := t.TempDir() for _, dir := range []string{ filepath.Join(root, "scripts", "guest", "ubuntu"), @@ -110,26 +110,26 @@ func TestDockerDindBuildUsesLegacyBuilderCompatibleArgs(t *testing.T) { Config: config.Config{ Image: config.ImageConfig{ SourceImage: "ghcr.io/catthehacker/ubuntu:full-latest", - OutputImage: "epar-docker-dind-catthehacker-ubuntu", + OutputImage: "epar-docker-container-catthehacker-ubuntu", RunnerVersion: "latest", }, Logging: config.LoggingConfig{Directory: "logs"}, - Provider: config.ProviderConfig{Type: "docker-dind", Platform: "linux/amd64"}, + Provider: config.ProviderConfig{Type: "docker-container", Platform: "linux/amd64"}, }, ProjectRoot: root, DryRun: true, } manifest := ImageManifest{ SchemaVersion: imageManifestSchemaVersion, - ProviderType: "docker-dind", + ProviderType: "docker-container", SourceType: config.ImageSourceDockerImage, SourceImage: "ghcr.io/catthehacker/ubuntu:full-latest", - OutputImage: "epar-docker-dind-catthehacker-ubuntu", + OutputImage: "epar-docker-container-catthehacker-ubuntu", RunnerVersion: "latest", } out, err := capturePoolStdout(t, func() error { - return manager.buildDockerDindImage(context.Background(), ImageBuildOptions{Replace: true, Manifest: &manifest}, filepath.Join(root, "third_party", "runner-images")) + return manager.buildDockerContainerImage(context.Background(), ImageBuildOptions{Replace: true, Manifest: &manifest}, filepath.Join(root, "third_party", "runner-images")) }) if err != nil { t.Fatal(err) @@ -137,8 +137,8 @@ func TestDockerDindBuildUsesLegacyBuilderCompatibleArgs(t *testing.T) { if strings.Contains(out, "--progress") { t.Fatalf("docker build command should not require BuildKit progress support:\n%s", out) } - if !strings.Contains(out, "docker build -t epar-docker-dind-catthehacker-ubuntu --platform linux/amd64") { - t.Fatalf("docker build command missing expected base args:\n%s", out) + if !strings.Contains(out, "docker buildx build --builder epar-") || !strings.Contains(out, " --load -t epar-docker-container-catthehacker-ubuntu --platform linux/amd64") { + t.Fatalf("dedicated Buildx command missing expected base args:\n%s", out) } } diff --git a/internal/pool/lifecycle_cleanup.go b/internal/pool/lifecycle_cleanup.go new file mode 100644 index 0000000..4e11d42 --- /dev/null +++ b/internal/pool/lifecycle_cleanup.go @@ -0,0 +1,282 @@ +package pool + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "time" + + poolstate "github.com/solutionforest/ephemeral-action-runner/internal/pool/state" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +func (m *Manager) cleanupOwnedLifecycle(ctx context.Context) error { + records, err := m.LifecycleState.List(ctx) + if err != nil { + return fmt.Errorf("read provider-neutral lifecycle state: %w", err) + } + inventory, err := m.inventoryProvider(ctx) + if err != nil { + return fmt.Errorf("read provider inventory for exact cleanup: %w", err) + } + byName := inventoryByName(inventory) + ownedNames := make(map[string]struct{}, len(records)) + for _, record := range records { + if record.ProviderType == m.Config.Provider.Type && record.Phase != poolstate.PhaseTombstoned { + ownedNames[record.Name] = struct{}{} + } + } + for _, item := range inventory { + if !HasPrefix(item.Instance.Name, m.Config.Pool.NamePrefix) { + continue + } + if _, found := ownedNames[item.Instance.Name]; found { + continue + } + if err := m.reportUnknownInventory(ctx, item); err != nil { + return err + } + } + + var firstErr error + for _, record := range records { + if record.ProviderType != m.Config.Provider.Type || record.Phase == poolstate.PhaseTombstoned { + continue + } + if err := m.cleanupLifecycleRecord(ctx, record, byName[record.Name]); err != nil { + firstErr = errors.Join(firstErr, fmt.Errorf("cleanup %s: %w", record.Name, err)) + } + } + return firstErr +} + +func inventoryByName(items []provider.InventoryItem) map[string][]provider.InventoryItem { + result := make(map[string][]provider.InventoryItem) + for _, item := range items { + result[item.Instance.Name] = append(result[item.Instance.Name], item) + } + return result +} + +func (m *Manager) reportUnknownInventory(ctx context.Context, item provider.InventoryItem) error { + providerID := item.Instance.ProviderID + if providerID == "" { + providerID = "unidentified:" + item.Instance.Name + } + payload, _ := json.Marshal(map[string]string{"state": item.State, "source": item.Source}) + _, err := m.LifecycleState.ReportUnknown(ctx, poolstate.Discovery{ + ProviderType: m.Config.Provider.Type, + ProviderID: providerID, + ExactName: item.Instance.Name, + Receipt: poolstate.Receipt{Version: "v1", Payload: payload}, + }) + if err != nil { + return fmt.Errorf("quarantine unowned provider instance %q: %w", item.Instance.Name, err) + } + m.warnf("cleanup: quarantined unowned instance %s id=%s; prefix-only or unidentified resources are report-only\n", item.Instance.Name, providerID) + return nil +} + +func (m *Manager) cleanupLifecycleRecord(ctx context.Context, initial poolstate.Record, sameName []provider.InventoryItem) error { + return m.cleanupLifecycleRecordWithRemoteAbsence(ctx, initial, sameName, false) +} + +func (m *Manager) cleanupLifecycleRecordWithRemoteAbsence(ctx context.Context, initial poolstate.Record, sameName []provider.InventoryItem, remoteKnownAbsent bool) error { + for { + record, err := m.LifecycleState.Read(ctx, initial.Name) + if err != nil { + return err + } + if lease, protected := activeLifecycleLease(record.Leases, m.currentTime()); protected { + return fmt.Errorf("active %s lease held by %s protects the instance until %s; refusing cleanup", lease.Purpose, lease.Holder, lease.ExpiresAt.Format(time.RFC3339)) + } + switch record.Phase { + case poolstate.PhaseTombstoned: + return nil + case poolstate.PhaseReserved, poolstate.PhaseCreating: + if len(sameName) != 0 { + for _, item := range sameName { + if reportErr := m.reportUnknownInventory(ctx, item); reportErr != nil { + return reportErr + } + } + m.quarantineLifecycle(ctx, record.Name, fmt.Errorf("create was interrupted before an immutable provider identity was recorded")) + return fmt.Errorf("unidentified same-name instance is quarantined and was not deleted") + } + if m.GitHub != nil { + if _, found, err := m.GitHub.RunnerByName(ctx, record.GitHub.ExactName); err != nil { + return err + } else if found { + m.quarantineLifecycle(ctx, record.Name, fmt.Errorf("create was interrupted and a same-name GitHub runner exists without a recorded immutable id")) + return fmt.Errorf("unidentified same-name GitHub runner is quarantined and was not deleted") + } + } + _, err = m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionAbandonCreate}) + return err + case poolstate.PhaseCleanupPending: + if _, err := m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionResumeCleanup}); err != nil { + return err + } + case poolstate.PhaseQuarantined: + if record.ProviderID == "" { + if len(sameName) != 0 { + for _, item := range sameName { + if reportErr := m.reportUnknownInventory(ctx, item); reportErr != nil { + return reportErr + } + } + return fmt.Errorf("unidentified same-name instance is quarantined and was not deleted") + } + if m.GitHub == nil { + return fmt.Errorf("record has no immutable provider identity and GitHub absence cannot be verified") + } + if _, found, err := m.GitHub.RunnerByName(ctx, record.GitHub.ExactName); err != nil { + return err + } else if found { + return fmt.Errorf("unidentified same-name GitHub runner is quarantined and was not deleted") + } + _, err = m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionAbandonCreate}) + return err + } + if _, err := m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionFenceIntent}); err != nil { + return err + } + case poolstate.PhaseCreated, poolstate.PhaseValidating, poolstate.PhaseStandby, poolstate.PhaseRegistering, poolstate.PhaseReady, poolstate.PhaseBusy, poolstate.PhaseDraining: + if record.ProviderID == "" { + return fmt.Errorf("record has no immutable provider identity and remains report-only") + } + if _, err := m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionFenceIntent}); err != nil { + return err + } + case poolstate.PhaseFencing: + if _, err := m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionFenced}); err != nil { + return err + } + case poolstate.PhaseFenced: + if _, err := m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionVerifyRemoteIntent}); err != nil { + return err + } + case poolstate.PhaseRemoteReconciling: + if !remoteKnownAbsent { + if err := m.removeExactGitHubRunner(ctx, record); err != nil { + m.markCleanupPending(ctx, record.Name) + return err + } + } + if _, err := m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionRemoteAbsent}); err != nil { + return err + } + case poolstate.PhaseRemoteAbsent: + if _, err := m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionRemoveLocalIntent}); err != nil { + return err + } + case poolstate.PhaseLocalRemoving: + if err := m.removeExactProviderInstance(ctx, record, sameName); err != nil { + m.markCleanupPending(ctx, record.Name) + return err + } + if _, err := m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionLocalAbsent}); err != nil { + return err + } + case poolstate.PhaseLocalAbsent: + if _, err := m.LifecycleState.Transition(ctx, record.Name, poolstate.Transition{Action: poolstate.ActionTombstone}); err != nil { + return err + } + paths := ProvisionedInstance{Name: record.Name, LogPath: m.instanceLogPath(record.Name, "."+m.Config.Provider.Type+".log"), GuestLogPath: m.instanceLogPath(record.Name, ".guest.log")} + if releaseErr := m.releaseInstanceTranscripts(paths); releaseErr != nil { + m.logger().Warn("instance transcript close failed after cleanup", "provider", m.Config.Provider.Type, "instance", record.Name, "operation", "cleanup", "error", releaseErr) + } + default: + return fmt.Errorf("unsupported cleanup phase %s", record.Phase) + } + } +} + +func activeLifecycleLease(leases []poolstate.Lease, now time.Time) (poolstate.Lease, bool) { + for _, lease := range leases { + if lease.ExpiresAt.After(now) { + return lease, true + } + } + return poolstate.Lease{}, false +} + +func (m *Manager) removeExactGitHubRunner(ctx context.Context, record poolstate.Record) error { + if m.GitHub == nil { + if record.GitHub.RunnerID != 0 { + return fmt.Errorf("cannot verify GitHub runner id=%d without a GitHub client", record.GitHub.RunnerID) + } + return nil + } + runner, found, err := m.GitHub.RunnerByName(ctx, record.GitHub.ExactName) + if err != nil { + return err + } + if !found { + return nil + } + if record.GitHub.RunnerID == 0 || runner.ID != record.GitHub.RunnerID { + return fmt.Errorf("same-name GitHub runner id=%d does not match recorded id=%d; refusing deletion", runner.ID, record.GitHub.RunnerID) + } + deleteCtx, cancel := context.WithTimeout(ctx, 60*time.Second) + defer cancel() + if err := m.GitHub.DeleteRunnerIfExists(deleteCtx, runner.ID); err != nil { + return err + } + after, found, err := m.GitHub.RunnerByName(ctx, record.GitHub.ExactName) + if err != nil { + return err + } + if found { + return fmt.Errorf("GitHub runner remains after exact deletion: name=%s id=%d", after.Name, after.ID) + } + m.infof("cleanup: deleted exact GitHub runner %s id=%d\n", runner.Name, runner.ID) + return nil +} + +func (m *Manager) removeExactProviderInstance(ctx context.Context, record poolstate.Record, sameName []provider.InventoryItem) error { + var exact *provider.Instance + for i := range sameName { + item := sameName[i] + if item.Instance.ProviderID == record.ProviderID { + copy := item.Instance + exact = © + continue + } + if item.Instance.Name == record.Name { + return fmt.Errorf("same-name provider instance id=%s does not match recorded id=%s; refusing deletion", item.Instance.ProviderID, record.ProviderID) + } + } + if exact == nil { + return nil + } + exact.ReceiptVersion = record.Receipt.Version + exact.Receipt = append([]byte(nil), record.Receipt.Payload...) + stopCtx, stopCancel := context.WithTimeout(ctx, 60*time.Second) + _ = m.stopProviderInstance(stopCtx, *exact) + stopCancel() + deleteCtx, deleteCancel := context.WithTimeout(ctx, 60*time.Second) + err := m.deleteProviderInstance(deleteCtx, *exact) + deleteCancel() + if err != nil { + return err + } + remaining, err := m.inventoryProvider(ctx) + if err != nil { + return err + } + for _, item := range remaining { + if item.Instance.ProviderID == record.ProviderID { + return fmt.Errorf("provider instance remains after exact deletion: name=%s id=%s", item.Instance.Name, item.Instance.ProviderID) + } + } + m.infof("cleanup: deleted exact owned instance %s id=%s\n", record.Name, record.ProviderID) + return nil +} + +func (m *Manager) markCleanupPending(ctx context.Context, name string) { + if _, err := m.LifecycleState.Transition(ctx, name, poolstate.Transition{Action: poolstate.ActionCleanupPending}); err != nil && !errors.Is(err, poolstate.ErrInvalidTransition) { + m.warnf("[%s] failed to persist cleanup-pending phase: %v\n", name, err) + } +} diff --git a/internal/pool/lifecycle_cleanup_test.go b/internal/pool/lifecycle_cleanup_test.go new file mode 100644 index 0000000..20aa9d4 --- /dev/null +++ b/internal/pool/lifecycle_cleanup_test.go @@ -0,0 +1,464 @@ +package pool + +import ( + "bytes" + "context" + "strings" + "sync/atomic" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + gh "github.com/solutionforest/ephemeral-action-runner/internal/github" + "github.com/solutionforest/ephemeral-action-runner/internal/logging" + poolstate "github.com/solutionforest/ephemeral-action-runner/internal/pool/state" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +func TestLifecycleCleanupRefusesRecreatedSameNameProviderInstance(t *testing.T) { + store, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + const name = "epar-test-recreated" + if _, err := store.Reserve(context.Background(), poolstate.CreateSpec{Name: name, ProviderType: "docker-container", GitHub: poolstate.GitHubIdentity{ExactName: name}}); err != nil { + t.Fatal(err) + } + for _, transition := range []poolstate.Transition{ + {Action: poolstate.ActionCreateIntent}, + {Action: poolstate.ActionCreated, ProviderID: "docker:old-id", Receipt: poolstate.Receipt{Version: "v1", Payload: []byte(`{"providerId":"docker:old-id"}`)}}, + {Action: poolstate.ActionValidateIntent}, + {Action: poolstate.ActionValidated}, + } { + if _, err := store.Transition(context.Background(), name, transition); err != nil { + t.Fatal(err) + } + } + fake := &fakeProvider{instances: []provider.Instance{{Name: name, ProviderID: "docker:new-id", State: "running"}}} + manager := Manager{ + Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}, Pool: config.PoolConfig{NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}}, + Provider: fake, + Lifecycle: provider.AdaptLegacy(fake), + LifecycleState: store, + ProjectRoot: t.TempDir(), + } + err = manager.cleanupOwnedLifecycle(context.Background()) + if err == nil || !strings.Contains(err.Error(), "does not match recorded") { + t.Fatalf("cleanup error = %v, want immutable identity mismatch", err) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 0 { + t.Fatalf("delete calls = %d, want 0", got) + } +} + +func TestLifecycleCleanupRefusesActiveLeaseBeforeSideEffects(t *testing.T) { + store, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + const name = "epar-test-busy" + if _, err := store.Reserve(context.Background(), poolstate.CreateSpec{Name: name, ProviderType: "docker-container", GitHub: poolstate.GitHubIdentity{ExactName: name}}); err != nil { + t.Fatal(err) + } + for _, transition := range []poolstate.Transition{ + {Action: poolstate.ActionCreateIntent}, + {Action: poolstate.ActionCreated, ProviderID: "docker:busy-id", Receipt: poolstate.Receipt{Version: "v1", Payload: []byte(`{"providerId":"docker:busy-id"}`)}}, + {Action: poolstate.ActionValidateIntent}, + {Action: poolstate.ActionValidated}, + } { + if _, err := store.Transition(context.Background(), name, transition); err != nil { + t.Fatal(err) + } + } + if _, err := store.AcquireLease(context.Background(), name, poolstate.Lease{Purpose: "job", Holder: "controller-test", ExpiresAt: time.Now().Add(time.Hour)}); err != nil { + t.Fatal(err) + } + fake := &fakeProvider{instances: []provider.Instance{{Name: name, ProviderID: "docker:busy-id", State: "running"}}} + manager := Manager{ + Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}, Pool: config.PoolConfig{NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}}, + Provider: fake, + Lifecycle: provider.AdaptLegacy(fake), + LifecycleState: store, + ProjectRoot: t.TempDir(), + } + err = manager.cleanupOwnedLifecycle(context.Background()) + if err == nil || !strings.Contains(err.Error(), "active job lease") { + t.Fatalf("cleanup error = %v, want active lease protection", err) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 0 { + t.Fatalf("delete calls = %d, want 0", got) + } + record, err := store.Read(context.Background(), name) + if err != nil { + t.Fatal(err) + } + if record.Phase != poolstate.PhaseStandby { + t.Fatalf("phase = %s, want %s", record.Phase, poolstate.PhaseStandby) + } +} + +func TestLifecycleCleanupTombstonesIdentitylessQuarantineAfterExactAbsence(t *testing.T) { + store, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + const name = "epar-test-identityless" + if _, err := store.Reserve(context.Background(), poolstate.CreateSpec{Name: name, ProviderType: "docker-container", GitHub: poolstate.GitHubIdentity{ExactName: name}}); err != nil { + t.Fatal(err) + } + if _, err := store.Transition(context.Background(), name, poolstate.Transition{Action: poolstate.ActionQuarantine, Reason: "post-create identity was lost"}); err != nil { + t.Fatal(err) + } + fake := &fakeProvider{} + manager := Manager{ + Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}, Pool: config.PoolConfig{NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}}, + Provider: fake, + Lifecycle: provider.AdaptLegacy(fake), + LifecycleState: store, + GitHub: &fakeGitHub{}, + ProjectRoot: t.TempDir(), + } + if err := manager.cleanupOwnedLifecycle(context.Background()); err != nil { + t.Fatal(err) + } + record, err := store.Read(context.Background(), name) + if err != nil { + t.Fatal(err) + } + if record.Phase != poolstate.PhaseTombstoned || record.Cleanup.RemoteAbsentAt == nil || record.Cleanup.LocalAbsentAt == nil { + t.Fatalf("cleanup record = %#v, want exact absence tombstone", record) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 0 { + t.Fatalf("delete calls = %d, want no name-only deletion", got) + } +} + +func TestLifecycleCleanupKeepsIdentitylessQuarantineWhenSameNameExists(t *testing.T) { + store, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + const name = "epar-test-identityless-present" + if _, err := store.Reserve(context.Background(), poolstate.CreateSpec{Name: name, ProviderType: "docker-container", GitHub: poolstate.GitHubIdentity{ExactName: name}}); err != nil { + t.Fatal(err) + } + if _, err := store.Transition(context.Background(), name, poolstate.Transition{Action: poolstate.ActionQuarantine, Reason: "post-create identity was lost"}); err != nil { + t.Fatal(err) + } + fake := &fakeProvider{instances: []provider.Instance{{Name: name, ProviderID: "docker:unknown-same-name", State: "running"}}} + manager := Manager{ + Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}, Pool: config.PoolConfig{NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}}, + Provider: fake, + Lifecycle: provider.AdaptLegacy(fake), + LifecycleState: store, + GitHub: &fakeGitHub{}, + ProjectRoot: t.TempDir(), + } + err = manager.cleanupOwnedLifecycle(context.Background()) + if err == nil || !strings.Contains(err.Error(), "same-name instance is quarantined") { + t.Fatalf("cleanup error = %v, want same-name refusal", err) + } + record, readErr := store.Read(context.Background(), name) + if readErr != nil { + t.Fatal(readErr) + } + if record.Phase != poolstate.PhaseQuarantined { + t.Fatalf("phase = %s, want %s", record.Phase, poolstate.PhaseQuarantined) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 0 { + t.Fatalf("delete calls = %d, want no name-only deletion", got) + } +} + +func TestIdentitylessQuarantineAlwaysVerifiesGitHubAbsence(t *testing.T) { + store, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + const name = "epar-test-identityless-remote" + if _, err := store.Reserve(context.Background(), poolstate.CreateSpec{Name: name, ProviderType: "docker-container", GitHub: poolstate.GitHubIdentity{ExactName: name}}); err != nil { + t.Fatal(err) + } + record, err := store.Transition(context.Background(), name, poolstate.Transition{Action: poolstate.ActionQuarantine, Reason: "post-create identity was lost"}) + if err != nil { + t.Fatal(err) + } + fake := &fakeProvider{} + manager := Manager{ + Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}, Pool: config.PoolConfig{NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}}, + Provider: fake, + Lifecycle: provider.AdaptLegacy(fake), + LifecycleState: store, + GitHub: &fakeGitHub{runner: gh.Runner{Name: name, ID: 9123}, found: true}, + ProjectRoot: t.TempDir(), + } + err = manager.cleanupLifecycleRecordWithRemoteAbsence(context.Background(), record, nil, true) + if err == nil || !strings.Contains(err.Error(), "same-name GitHub runner") { + t.Fatalf("cleanup error = %v, want GitHub absence refusal", err) + } + record, readErr := store.Read(context.Background(), name) + if readErr != nil { + t.Fatal(readErr) + } + if record.Phase != poolstate.PhaseQuarantined { + t.Fatalf("phase = %s, want %s", record.Phase, poolstate.PhaseQuarantined) + } +} + +func TestCleanupRecoversInterruptedProvisionLeaseAfterExclusiveLock(t *testing.T) { + store, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + const name = "epar-test-interrupted" + if _, err := store.Reserve(context.Background(), poolstate.CreateSpec{Name: name, ProviderType: "docker-container", GitHub: poolstate.GitHubIdentity{ExactName: name}}); err != nil { + t.Fatal(err) + } + for _, transition := range []poolstate.Transition{ + {Action: poolstate.ActionCreateIntent}, + {Action: poolstate.ActionCreated, ProviderID: "docker:interrupted-id", Receipt: poolstate.Receipt{Version: "v1", Payload: []byte(`{"providerId":"docker:interrupted-id"}`)}}, + {Action: poolstate.ActionValidateIntent}, + } { + if _, err := store.Transition(context.Background(), name, transition); err != nil { + t.Fatal(err) + } + } + if _, err := store.AcquireLease(context.Background(), name, poolstate.Lease{Purpose: "provision", Holder: "controller", ExpiresAt: time.Now().Add(time.Hour)}); err != nil { + t.Fatal(err) + } + fake := &fakeProvider{instances: []provider.Instance{{Name: name, ProviderID: "docker:interrupted-id", State: "running"}}} + projectRoot := t.TempDir() + manager := Manager{ + Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}, Pool: config.PoolConfig{NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}}, + ConfigPath: "interrupted.yml", + Provider: fake, + Lifecycle: provider.AdaptLegacy(fake), + LifecycleState: store, + ProjectRoot: projectRoot, + } + + if err := manager.Cleanup(context.Background()); err != nil { + t.Fatal(err) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 1 { + t.Fatalf("delete calls = %d, want 1", got) + } + record, err := store.Read(context.Background(), name) + if err != nil { + t.Fatal(err) + } + if record.Phase != poolstate.PhaseTombstoned { + t.Fatalf("phase = %s, want %s", record.Phase, poolstate.PhaseTombstoned) + } + if len(record.Leases) != 0 { + t.Fatalf("leases = %+v, want none", record.Leases) + } +} + +func TestInterruptedProvisionRecoveryPreservesJobLease(t *testing.T) { + manager, store, name := readyLifecycleManager(t) + if _, err := store.Transition(context.Background(), name, poolstate.Transition{Action: poolstate.ActionJobStarted}); err != nil { + t.Fatal(err) + } + if _, err := store.AcquireLease(context.Background(), name, poolstate.Lease{Purpose: "provision", Holder: "controller", ExpiresAt: time.Now().Add(time.Hour)}); err != nil { + t.Fatal(err) + } + if _, err := store.AcquireLease(context.Background(), name, poolstate.Lease{Purpose: "job", Holder: "github-42", ExpiresAt: time.Now().Add(time.Hour)}); err != nil { + t.Fatal(err) + } + + if err := manager.recoverInterruptedProvisionLeases(context.Background()); err != nil { + t.Fatal(err) + } + + record, err := store.Read(context.Background(), name) + if err != nil { + t.Fatal(err) + } + if len(record.Leases) != 1 || record.Leases[0].Purpose != "job" || record.Leases[0].Holder != "github-42" { + t.Fatalf("leases = %+v, want exact job lease only", record.Leases) + } +} + +func TestRemoteAbsenceReleasesExactJobLease(t *testing.T) { + manager, store, name := readyLifecycleManager(t) + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: t.TempDir(), + ManagerSinks: logging.SinkConsole, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + manager.Logging = runtime + if _, err := store.Transition(context.Background(), name, poolstate.Transition{Action: poolstate.ActionJobStarted}); err != nil { + t.Fatal(err) + } + holder := "github-42" + if _, err := store.AcquireLease(context.Background(), name, poolstate.Lease{Purpose: "job", Holder: holder, ExpiresAt: time.Now().Add(time.Hour)}); err != nil { + t.Fatal(err) + } + + if err := manager.recordLifecycleRemoteAbsence(context.Background(), name); err != nil { + t.Fatal(err) + } + + record, err := store.Read(context.Background(), name) + if err != nil { + t.Fatal(err) + } + if record.Phase != poolstate.PhaseDraining { + t.Fatalf("phase = %s, want %s", record.Phase, poolstate.PhaseDraining) + } + if lease, active := activeLifecycleLease(record.Leases, time.Now()); active { + t.Fatalf("job lease remained active after exact remote absence: %+v", lease) + } + if message := console.String(); !strings.Contains(message, "["+name+"] Job finished and GitHub released the ephemeral runner; GitHub Actions has the success or failure result.") { + t.Fatalf("remote-absence lifecycle output omitted job completion: %q", message) + } +} + +func TestReconciliationLogsJobFinishBeforeExactCleanup(t *testing.T) { + manager, store, name := readyLifecycleManager(t) + if _, err := store.Transition(context.Background(), name, poolstate.Transition{Action: poolstate.ActionJobStarted}); err != nil { + t.Fatal(err) + } + if _, err := store.AcquireLease(context.Background(), name, poolstate.Lease{Purpose: "job", Holder: "github-42", ExpiresAt: time.Now().Add(time.Hour)}); err != nil { + t.Fatal(err) + } + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: t.TempDir(), + ManagerSinks: logging.SinkConsole, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + fake := &fakeProvider{instances: []provider.Instance{{Name: name, ProviderID: "docker:ready-id", State: "running"}}} + manager.Logging = runtime + manager.Provider = fake + manager.Lifecycle = provider.AdaptLegacy(fake) + manager.GitHub = &fakeGitHub{} + + active, err := manager.reconcilePhysicalPool(context.Background(), nil, true) + if err != nil { + t.Fatal(err) + } + if len(active) != 0 { + t.Fatalf("active instances = %+v, want none after completed ephemeral runner cleanup", active) + } + output := console.String() + finishedAt := strings.Index(output, "["+name+"] Job finished and GitHub released the ephemeral runner") + cleanupAt := strings.Index(output, "cleanup: deleted exact owned instance "+name) + if finishedAt < 0 || cleanupAt < 0 || finishedAt >= cleanupAt { + t.Fatalf("job completion was not logged before exact cleanup: %q", output) + } +} + +func TestLifecycleJobObservationLogsStartAndFinishOnce(t *testing.T) { + manager, store, name := readyLifecycleManager(t) + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: t.TempDir(), + ManagerSinks: logging.SinkConsole, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + manager.Logging = runtime + + runner := gh.Runner{Name: name, ID: 42, Status: "online", Busy: true} + if err := manager.recordLifecycleJobObservation(context.Background(), runner); err != nil { + t.Fatal(err) + } + if err := manager.recordLifecycleJobObservation(context.Background(), runner); err != nil { + t.Fatal(err) + } + runner.Busy = false + if err := manager.recordLifecycleJobObservation(context.Background(), runner); err != nil { + t.Fatal(err) + } + + record, err := store.Read(context.Background(), name) + if err != nil { + t.Fatal(err) + } + if record.Phase != poolstate.PhaseDraining { + t.Fatalf("phase = %s, want %s", record.Phase, poolstate.PhaseDraining) + } + output := console.String() + if count := strings.Count(output, "["+name+"] Job started; GitHub assigned work to this runner."); count != 1 { + t.Fatalf("job-start log count = %d, want 1: %q", count, output) + } + if count := strings.Count(output, "["+name+"] Job finished; GitHub Actions has the success or failure result."); count != 1 { + t.Fatalf("job-finish log count = %d, want 1: %q", count, output) + } +} + +func TestReconciliationPreservesStoppedBusyRunnerProtectedByJobLease(t *testing.T) { + manager, store, name := readyLifecycleManager(t) + if _, err := store.Transition(context.Background(), name, poolstate.Transition{Action: poolstate.ActionJobStarted}); err != nil { + t.Fatal(err) + } + if _, err := store.AcquireLease(context.Background(), name, poolstate.Lease{Purpose: "job", Holder: "github-42", ExpiresAt: time.Now().Add(time.Hour)}); err != nil { + t.Fatal(err) + } + fake := &fakeProvider{instances: []provider.Instance{{Name: name, ProviderID: "docker:ready-id", State: "stopped"}}} + github := &fakeGitHub{listRunners: []gh.Runner{{Name: name, ID: 42, Status: "offline", Busy: true}}} + manager.Provider = fake + manager.Lifecycle = provider.AdaptLegacy(fake) + manager.GitHub = github + + active, err := manager.reconcilePhysicalPool(context.Background(), nil, true) + if err != nil { + t.Fatal(err) + } + if active[name].Phase != LifecycleCleanupPending { + t.Fatalf("phase = %s, want %s", active[name].Phase, LifecycleCleanupPending) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 0 { + t.Fatalf("provider delete calls = %d, want 0 while job lease is active", got) + } + if got := atomic.LoadInt32(&github.deleteCalls); got != 0 { + t.Fatalf("GitHub delete calls = %d, want 0 while exact lifecycle cleanup is pending", got) + } +} + +func readyLifecycleManager(t *testing.T) (*Manager, *poolstate.Store, string) { + t.Helper() + store, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + const name = "epar-test-ready" + if _, err := store.Reserve(context.Background(), poolstate.CreateSpec{Name: name, ProviderType: "docker-container", GitHub: poolstate.GitHubIdentity{ExactName: name}}); err != nil { + t.Fatal(err) + } + for _, transition := range []poolstate.Transition{ + {Action: poolstate.ActionCreateIntent}, + {Action: poolstate.ActionCreated, ProviderID: "docker:ready-id", Receipt: poolstate.Receipt{Version: "v1", Payload: []byte(`{"providerId":"docker:ready-id"}`)}}, + {Action: poolstate.ActionValidateIntent}, + {Action: poolstate.ActionValidated}, + {Action: poolstate.ActionRegisterIntent}, + {Action: poolstate.ActionRegistered, RunnerID: 42}, + } { + if _, err := store.Transition(context.Background(), name, transition); err != nil { + t.Fatal(err) + } + } + manager := &Manager{ + Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}, Pool: config.PoolConfig{NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}}, + LifecycleState: store, + ProjectRoot: t.TempDir(), + } + return manager, store, name +} diff --git a/internal/pool/lifecycle_state.go b/internal/pool/lifecycle_state.go new file mode 100644 index 0000000..931a866 --- /dev/null +++ b/internal/pool/lifecycle_state.go @@ -0,0 +1,346 @@ +package pool + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "strconv" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/filelock" + gh "github.com/solutionforest/ephemeral-action-runner/internal/github" + poolstate "github.com/solutionforest/ephemeral-action-runner/internal/pool/state" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +// OpenLifecycleState opens the provider-neutral state namespace for one exact +// configuration file. The namespace hash prevents two configurations in the +// same checkout from claiming each other's instances. Production callers must +// hold the canonical configuration and normalized prefix controller locks so a +// legacy namespace cannot be renamed beneath another active controller. +func OpenLifecycleState(projectRoot, configPath string) (*poolstate.Store, error) { + legacyConfig, err := filepath.Abs(configPath) + if err != nil { + return nil, fmt.Errorf("resolve legacy lifecycle config path: %w", err) + } + legacyConfig = filepath.Clean(legacyConfig) + canonicalConfig, err := storagecatalog.CanonicalPath(configPath) + if err != nil { + return nil, fmt.Errorf("resolve lifecycle config path: %w", err) + } + legacyDirectory := lifecycleStateDirectory(projectRoot, legacyConfig) + canonicalDirectory := lifecycleStateDirectory(projectRoot, canonicalConfig) + if err := os.MkdirAll(filepath.Dir(canonicalDirectory), 0o700); err != nil { + return nil, fmt.Errorf("create lifecycle state root: %w", err) + } + migrationLock, err := acquireLifecycleMigrationLock(canonicalDirectory + ".migration.lock") + if err != nil { + return nil, err + } + defer migrationLock.Close() + if legacyDirectory != canonicalDirectory { + if err := migrateLegacyLifecycleState(legacyDirectory, canonicalDirectory); err != nil { + return nil, err + } + } + return poolstate.Open(canonicalDirectory) +} + +func acquireLifecycleMigrationLock(path string) (*filelock.Lock, error) { + deadline := time.Now().Add(15 * time.Second) + for { + lock, err := filelock.Acquire(path) + if err == nil { + return lock, nil + } + if !errors.Is(err, filelock.ErrLocked) { + return nil, fmt.Errorf("acquire lifecycle migration lock %s: %w", path, err) + } + if !time.Now().Before(deadline) { + return nil, fmt.Errorf("timed out waiting for lifecycle migration lock %s", path) + } + time.Sleep(50 * time.Millisecond) + } +} + +func lifecycleStateDirectory(projectRoot, canonicalConfig string) string { + sum := sha256.Sum256([]byte(canonicalConfig)) + namespace := hex.EncodeToString(sum[:8]) + return filepath.Join(projectRoot, ".local", "state", "pools", namespace) +} + +func migrateLegacyLifecycleState(legacyDirectory, canonicalDirectory string) error { + legacyInfo, legacyErr := os.Lstat(legacyDirectory) + if errors.Is(legacyErr, os.ErrNotExist) { + return nil + } + if legacyErr != nil { + return fmt.Errorf("inspect legacy lifecycle state %s: %w", legacyDirectory, legacyErr) + } + if !legacyInfo.IsDir() || legacyInfo.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("legacy lifecycle state is not a real directory: %s", legacyDirectory) + } + if _, canonicalErr := os.Lstat(canonicalDirectory); canonicalErr == nil { + return fmt.Errorf("both legacy and canonical lifecycle state exist; refusing to choose between %s and %s", legacyDirectory, canonicalDirectory) + } else if !errors.Is(canonicalErr, os.ErrNotExist) { + return fmt.Errorf("inspect canonical lifecycle state %s: %w", canonicalDirectory, canonicalErr) + } + if err := os.Rename(legacyDirectory, canonicalDirectory); err != nil { + _, legacyRetryErr := os.Lstat(legacyDirectory) + _, canonicalRetryErr := os.Lstat(canonicalDirectory) + if errors.Is(legacyRetryErr, os.ErrNotExist) && canonicalRetryErr == nil { + return nil + } + return fmt.Errorf("migrate legacy lifecycle state %s to %s: %w", legacyDirectory, canonicalDirectory, err) + } + return nil +} + +func (m *Manager) reserveLifecycle(ctx context.Context, name string) error { + if m.LifecycleState == nil { + return nil + } + _, err := m.LifecycleState.Reserve(ctx, poolstate.CreateSpec{ + Name: name, + ProviderType: m.Config.Provider.Type, + GitHub: poolstate.GitHubIdentity{ExactName: name}, + }) + if err != nil { + return fmt.Errorf("reserve provider-neutral lifecycle record: %w", err) + } + _, err = m.LifecycleState.Transition(ctx, name, poolstate.Transition{Action: poolstate.ActionCreateIntent}) + if err != nil { + return fmt.Errorf("record provider create intent: %w", err) + } + return nil +} + +func (m *Manager) acquireLifecycleLease(ctx context.Context, name, purpose, holder string, lifetime time.Duration) error { + if m.LifecycleState == nil { + return nil + } + _, err := m.LifecycleState.AcquireLease(ctx, name, poolstate.Lease{Purpose: purpose, Holder: holder, ExpiresAt: m.currentTime().Add(lifetime)}) + return err +} + +func (m *Manager) releaseLifecycleLease(ctx context.Context, name, purpose, holder string) { + if m.LifecycleState == nil { + return + } + if _, err := m.LifecycleState.ReleaseLease(ctx, name, purpose, holder); err != nil && !errors.Is(err, poolstate.ErrNotFound) { + m.warnf("[%s] release %s lifecycle lease failed: %v\n", name, purpose, err) + } +} + +// recoverInterruptedProvisionLeases removes only leases owned by a previous +// common controller after the caller has acquired the exclusive pool lock. +// Job and provider-specific leases remain authoritative cleanup barriers. +func (m *Manager) recoverInterruptedProvisionLeases(ctx context.Context) error { + if m.LifecycleState == nil { + return nil + } + records, err := m.LifecycleState.List(ctx) + if err != nil { + return fmt.Errorf("read lifecycle state for interrupted provisioning recovery: %w", err) + } + for _, record := range records { + if record.ProviderType != m.Config.Provider.Type || record.Phase == poolstate.PhaseTombstoned { + continue + } + for _, lease := range record.Leases { + if lease.Purpose != "provision" || lease.Holder != "controller" { + continue + } + if _, err := m.LifecycleState.ReleaseLease(ctx, record.Name, lease.Purpose, lease.Holder); err != nil { + return fmt.Errorf("release interrupted provisioning lease for %s: %w", record.Name, err) + } + m.warnf("[%s] recovered interrupted provisioning lease after acquiring the exclusive pool lock\n", record.Name) + break + } + } + return nil +} + +func (m *Manager) recordLifecycleJobObservation(ctx context.Context, runner gh.Runner) error { + if m.LifecycleState == nil { + return nil + } + record, err := m.LifecycleState.Read(ctx, runner.Name) + if err != nil { + return err + } + holder := "github-" + strconv.FormatInt(runner.ID, 10) + if runner.Busy { + started := false + if record.Phase == poolstate.PhaseReady { + if _, err := m.LifecycleState.Transition(ctx, runner.Name, poolstate.Transition{Action: poolstate.ActionJobStarted}); err != nil { + return err + } + started = true + } + if err := m.acquireLifecycleLease(ctx, runner.Name, "job", holder, 10*time.Minute); err != nil { + return err + } + if started { + m.infof("[%s] Job started; GitHub assigned work to this runner.\n", runner.Name) + } + return nil + } + if record.Phase == poolstate.PhaseBusy { + if _, err := m.LifecycleState.Transition(ctx, runner.Name, poolstate.Transition{Action: poolstate.ActionJobFinished}); err != nil { + return err + } + m.releaseLifecycleLease(ctx, runner.Name, "job", holder) + m.infof("[%s] Job finished; GitHub Actions has the success or failure result.\n", runner.Name) + return nil + } + for _, lease := range record.Leases { + if lease.Purpose == "job" && lease.Holder == holder { + m.releaseLifecycleLease(ctx, runner.Name, "job", holder) + break + } + } + return nil +} + +func (m *Manager) recordLifecycleRemoteAbsence(ctx context.Context, name string) error { + if m.LifecycleState == nil { + return nil + } + record, err := m.LifecycleState.Read(ctx, name) + if err != nil { + return err + } + jobFinished := record.Phase == poolstate.PhaseBusy + if record.Phase == poolstate.PhaseBusy { + if _, err := m.LifecycleState.Transition(ctx, name, poolstate.Transition{Action: poolstate.ActionJobFinished}); err != nil { + return err + } + } + if record.GitHub.RunnerID != 0 { + m.releaseLifecycleLease(ctx, name, "job", "github-"+strconv.FormatInt(record.GitHub.RunnerID, 10)) + } + if jobFinished { + m.infof("[%s] Job finished and GitHub released the ephemeral runner; GitHub Actions has the success or failure result.\n", name) + } else if record.Phase == poolstate.PhaseReady { + m.infof("[%s] GitHub runner registration disappeared before EPAR observed a job start; starting exact cleanup.\n", name) + } + return nil +} + +func (m *Manager) recordLifecycleCreated(ctx context.Context, instance provider.Instance) error { + if m.LifecycleState == nil { + return nil + } + version := instance.ReceiptVersion + payload := append([]byte(nil), instance.Receipt...) + if version == "" || len(payload) == 0 { + version = "v1" + var err error + payload, err = json.Marshal(map[string]string{"exactName": instance.Name, "providerId": instance.ProviderID, "source": instance.Source}) + if err != nil { + return fmt.Errorf("encode provider lifecycle receipt: %w", err) + } + } + _, err := m.LifecycleState.Transition(ctx, instance.Name, poolstate.Transition{ + Action: poolstate.ActionCreated, + ProviderID: instance.ProviderID, + Receipt: poolstate.Receipt{Version: version, Payload: payload}, + }) + if err != nil { + return fmt.Errorf("record provider instance creation: %w", err) + } + return nil +} + +func (m *Manager) recordLifecycleValidationIntent(ctx context.Context, name string) error { + if m.LifecycleState == nil { + return nil + } + _, err := m.LifecycleState.Transition(ctx, name, poolstate.Transition{Action: poolstate.ActionValidateIntent}) + return err +} + +func (m *Manager) recordLifecycleValidated(ctx context.Context, name string) error { + if m.LifecycleState == nil { + return nil + } + _, err := m.LifecycleState.Transition(ctx, name, poolstate.Transition{Action: poolstate.ActionValidated}) + return err +} + +func (m *Manager) recordLifecycleRegistrationIntent(ctx context.Context, name string) error { + if m.LifecycleState == nil { + return nil + } + _, err := m.LifecycleState.Transition(ctx, name, poolstate.Transition{Action: poolstate.ActionRegisterIntent}) + return err +} + +func (m *Manager) recordLifecycleRegistered(ctx context.Context, name string, runnerID int64) error { + if m.LifecycleState == nil { + return nil + } + _, err := m.LifecycleState.Transition(ctx, name, poolstate.Transition{Action: poolstate.ActionRegistered, RunnerID: runnerID}) + return err +} + +func (m *Manager) quarantineLifecycle(ctx context.Context, name string, cause error) { + if m.LifecycleState == nil || cause == nil { + return + } + if _, err := m.LifecycleState.Transition(ctx, name, poolstate.Transition{Action: poolstate.ActionQuarantine, Reason: cause.Error()}); err != nil && !errors.Is(err, poolstate.ErrInvalidTransition) { + m.warnf("[%s] provider-neutral lifecycle quarantine failed: %v\n", name, err) + } +} + +func (m *Manager) lifecycleOwns(ctx context.Context, name, providerID string) (bool, error) { + if m.LifecycleState == nil { + return true, nil + } + record, err := m.LifecycleState.Read(ctx, name) + if errors.Is(err, poolstate.ErrNotFound) { + return false, nil + } + if err != nil { + return false, err + } + return record.ProviderType == m.Config.Provider.Type && record.ProviderID != "" && record.ProviderID == providerID && record.Phase != poolstate.PhaseTombstoned, nil +} + +func (m *Manager) lifecycleOwnsRunner(ctx context.Context, name string, runnerID int64) (bool, error) { + if m.LifecycleState == nil { + return true, nil + } + record, err := m.LifecycleState.Read(ctx, name) + if errors.Is(err, poolstate.ErrNotFound) { + return false, nil + } + if err != nil { + return false, err + } + return record.ProviderType == m.Config.Provider.Type && record.GitHub.RunnerID != 0 && record.GitHub.RunnerID == runnerID && record.Phase != poolstate.PhaseTombstoned, nil +} + +func (m *Manager) reportUnknownLifecycle(ctx context.Context, name, providerID, source, observedState string) error { + if m.LifecycleState == nil { + return nil + } + payload, err := json.Marshal(map[string]string{"source": source, "state": observedState}) + if err != nil { + return err + } + _, err = m.LifecycleState.ReportUnknown(ctx, poolstate.Discovery{ + ProviderType: m.Config.Provider.Type, + ProviderID: providerID, + ExactName: name, + Receipt: poolstate.Receipt{Version: "v1", Payload: payload}, + }) + return err +} diff --git a/internal/pool/lifecycle_state_identity_test.go b/internal/pool/lifecycle_state_identity_test.go new file mode 100644 index 0000000..479c44e --- /dev/null +++ b/internal/pool/lifecycle_state_identity_test.go @@ -0,0 +1,78 @@ +package pool + +import ( + "crypto/sha256" + "encoding/hex" + "errors" + "os" + "path/filepath" + "testing" +) + +func TestLifecycleStateUsesCanonicalConfigurationIdentity(t *testing.T) { + project := t.TempDir() + realConfig := filepath.Join(project, ".local", "config.yml") + linkConfig := filepath.Join(project, ".local", "config-link.yml") + if err := os.MkdirAll(filepath.Dir(realConfig), 0o700); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(realConfig, []byte("provider: {}\n"), 0o600); err != nil { + t.Fatal(err) + } + if err := os.Symlink(realConfig, linkConfig); err != nil { + t.Skipf("symlinks unavailable: %v", err) + } + realStore, err := OpenLifecycleState(project, realConfig) + if err != nil { + t.Fatal(err) + } + linkStore, err := OpenLifecycleState(project, linkConfig) + if err != nil { + t.Fatal(err) + } + if realStore.Path() != linkStore.Path() { + t.Fatalf("one configuration received split lifecycle state through a symlink: %q != %q", realStore.Path(), linkStore.Path()) + } +} + +func TestLifecycleStateMigratesAncestorSymlinkNamespace(t *testing.T) { + root := t.TempDir() + project := filepath.Join(root, "project") + alias := filepath.Join(root, "project-alias") + if err := os.MkdirAll(filepath.Join(project, ".local"), 0o700); err != nil { + t.Fatal(err) + } + if err := os.Symlink(project, alias); err != nil { + t.Skipf("symlinks unavailable: %v", err) + } + configPath := filepath.Join(alias, ".local", "config.yml") + if err := os.WriteFile(filepath.Join(project, ".local", "config.yml"), []byte("provider: {}\n"), 0o600); err != nil { + t.Fatal(err) + } + legacyAbsolute, err := filepath.Abs(configPath) + if err != nil { + t.Fatal(err) + } + legacySum := sha256.Sum256([]byte(filepath.Clean(legacyAbsolute))) + legacyDirectory := filepath.Join(project, ".local", "state", "pools", hex.EncodeToString(legacySum[:8])) + if err := os.MkdirAll(legacyDirectory, 0o700); err != nil { + t.Fatal(err) + } + marker := filepath.Join(legacyDirectory, "migration-marker") + if err := os.WriteFile(marker, []byte("preserve"), 0o600); err != nil { + t.Fatal(err) + } + store, err := OpenLifecycleState(project, configPath) + if err != nil { + t.Fatal(err) + } + if store.Path() == filepath.Join(legacyDirectory, "state-v1.json") { + t.Fatal("ancestor-symlink lifecycle state remained in the legacy namespace") + } + if _, err := os.Stat(filepath.Join(filepath.Dir(store.Path()), "migration-marker")); err != nil { + t.Fatalf("legacy lifecycle state content was not migrated: %v", err) + } + if _, err := os.Stat(legacyDirectory); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("legacy lifecycle namespace still exists after migration: %v", err) + } +} diff --git a/internal/pool/logging.go b/internal/pool/logging.go index f898f85..4e0672a 100644 --- a/internal/pool/logging.go +++ b/internal/pool/logging.go @@ -9,11 +9,15 @@ import ( "os" "path/filepath" "strings" + "sync" "time" + artifactimage "github.com/solutionforest/ephemeral-action-runner/internal/image" "github.com/solutionforest/ephemeral-action-runner/internal/logging" ) +var buildxProgressHeartbeatInterval = 5 * time.Second + func (m *Manager) logger() *slog.Logger { if m != nil && m.Logging != nil { return m.Logging.Logger() @@ -34,6 +38,25 @@ func (m *Manager) warnf(format string, args ...any) { m.logger().Warn(fmt.Sprintf(strings.TrimSuffix(format, "\n"), args...)) } +func (m *Manager) logPoolRunning(label string) { + m.infof("%s is running. Press Ctrl-C once to stop, then wait for cleanup to finish before closing this window.\n", label) +} + +func (m *Manager) logReplacementReady(label, name string) { + m.infof("Replacement runner %s is online; %s is ready for the next job. Press Ctrl-C once to stop, then wait for cleanup to finish before closing this window.\n", name, label) +} + +func (m *Manager) cleanupPoolWithStatus(resources string, cleanup func() error) error { + m.infof("Stopping EPAR pool. Cleaning up %s. Please wait; do not press Ctrl-C again or close this window.\n", resources) + err := cleanup() + if err != nil { + m.warnf("Cleanup did not fully complete. EPAR retained its cleanup state and will reconcile it on the next run: %v\n", err) + return err + } + m.infof("Cleanup complete. EPAR can now exit safely.\n") + return nil +} + func (m *Manager) Close() error { if m == nil || m.Logging == nil { return nil @@ -128,6 +151,100 @@ func (m *Manager) runHostLogged(ctx context.Context, logPath, name string, args return runHostLoggedCommand(ctx, logPath, transcript.Stdout, transcript.Stderr, name, args...) } +func (m *Manager) runHostBuildxLogged(ctx context.Context, logPath, name string, args ...string) error { + transcript, err := m.transcript(logPath, "", transcriptComponent(logPath)) + if err != nil { + return err + } + prefix := "Docker image build" + if m.Config.Provider.Type == "docker-sandboxes" { + prefix = "Docker Sandboxes template build" + } + attributes := []any{"provider", m.Config.Provider.Type, "operation", "buildx-build", "logPath", logPath} + rawTranscriptOnConsole := stringSliceContains(m.Config.Logging.TranscriptSinks, "console") + interactive := dockerPullProgressTerminal() && stringSliceContains(m.Config.Logging.ManagerSinks, "console") && m.Config.Logging.ManagerConsoleFormat == "text" && !rawTranscriptOnConsole + archiveExportPath := buildxDockerArchiveDestination(args) + var reportMu sync.Mutex + report := func(snapshot artifactimage.BuildxProgressSnapshot) { + if rawTranscriptOnConsole { + return + } + reportMu.Lock() + defer reportMu.Unlock() + line := artifactimage.FormatBuildxProgress(prefix, snapshot) + if archiveExportPath != "" { + if info, err := os.Lstat(archiveExportPath); err == nil && info.Mode().IsRegular() { + line += "; archive written " + artifactimage.FormatDockerPullBytes(info.Size()) + } + } + if interactive { + _, _ = fmt.Fprintf(dockerPullProgressConsole, "\r\033[2K%s", line) + return + } + m.logger().Info(line, attributes...) + } + monitor := artifactimage.NewBuildxProgressMonitor(report) + stdoutProgress := monitor.NewStream() + stderrProgress := monitor.NewStream() + heartbeatDone := make(chan struct{}) + var heartbeat sync.WaitGroup + if !rawTranscriptOnConsole && buildxProgressHeartbeatInterval > 0 { + heartbeat.Add(1) + go func() { + defer heartbeat.Done() + ticker := time.NewTicker(buildxProgressHeartbeatInterval) + defer ticker.Stop() + for { + select { + case <-ticker.C: + report(monitor.Snapshot()) + case <-heartbeatDone: + return + } + } + }() + } + commandErr := runHostLoggedCommand(ctx, logPath, io.MultiWriter(transcript.Stdout, stdoutProgress), io.MultiWriter(transcript.Stderr, stderrProgress), name, args...) + close(heartbeatDone) + heartbeat.Wait() + stdoutProgress.Flush() + stderrProgress.Flush() + if interactive { + fmt.Fprintln(dockerPullProgressConsole) + } + if commandErr == nil { + snapshot := monitor.Snapshot() + completion := prefix + " complete" + if m.Config.Provider.Type == "docker-sandboxes" { + completion = "Docker Sandboxes template Buildx phase complete; finalizing evidence and importing the template next" + } + m.logger().Info(completion, append(attributes, "elapsed", snapshot.Elapsed.Round(time.Second))...) + } + return commandErr +} + +func buildxDockerArchiveDestination(args []string) string { + const prefix = "type=docker,dest=" + for index, argument := range args { + if argument == "--output" && index+1 < len(args) && strings.HasPrefix(args[index+1], prefix) { + return strings.TrimPrefix(args[index+1], prefix) + } + if strings.HasPrefix(argument, "--output="+prefix) { + return strings.TrimPrefix(argument, "--output="+prefix) + } + } + return "" +} + +func stringSliceContains(values []string, wanted string) bool { + for _, value := range values { + if value == wanted { + return true + } + } + return false +} + func (m *Manager) retentionPolicy() logging.RetentionPolicy { days := func(value int) time.Duration { return time.Duration(value) * 24 * time.Hour } return logging.RetentionPolicy{ diff --git a/internal/pool/logging_test.go b/internal/pool/logging_test.go new file mode 100644 index 0000000..f6a7c1d --- /dev/null +++ b/internal/pool/logging_test.go @@ -0,0 +1,80 @@ +package pool + +import ( + "bytes" + "errors" + "strings" + "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/logging" +) + +func TestPoolLifecycleConsoleGuidance(t *testing.T) { + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: t.TempDir(), + ManagerSinks: logging.SinkConsole, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + manager := Manager{Logging: runtime} + + manager.logPoolRunning("Docker Sandboxes pool") + manager.logReplacementReady("Docker Sandboxes pool", "epar-test-002") + cleanupCalled := false + if err := manager.cleanupPoolWithStatus("owned runner resources", func() error { + cleanupCalled = true + return nil + }); err != nil { + t.Fatal(err) + } + if !cleanupCalled { + t.Fatal("cleanup callback was not called") + } + + output := console.String() + for _, expected := range []string{ + "Docker Sandboxes pool is running.", + "Press Ctrl-C once to stop, then wait for cleanup to finish before closing this window.", + "Replacement runner epar-test-002 is online; Docker Sandboxes pool is ready for the next job.", + "Stopping EPAR pool. Cleaning up owned runner resources.", + "Please wait; do not press Ctrl-C again or close this window.", + "Cleanup complete. EPAR can now exit safely.", + } { + if !strings.Contains(output, expected) { + t.Fatalf("lifecycle console output does not contain %q: %q", expected, output) + } + } + if strings.Index(output, "Stopping EPAR pool.") > strings.Index(output, "Cleanup complete.") { + t.Fatalf("cleanup completion preceded cleanup start: %q", output) + } +} + +func TestPoolLifecycleConsoleReportsIncompleteCleanup(t *testing.T) { + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: t.TempDir(), + ManagerSinks: logging.SinkConsole, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatal(err) + } + defer runtime.Close() + manager := Manager{Logging: runtime} + cleanupErr := errors.New("remote state unavailable") + + err = manager.cleanupPoolWithStatus("owned runner resources", func() error { return cleanupErr }) + if !errors.Is(err, cleanupErr) { + t.Fatalf("cleanup error = %v, want %v", err, cleanupErr) + } + output := console.String() + if !strings.Contains(output, "Cleanup did not fully complete.") || !strings.Contains(output, "will reconcile it on the next run") || strings.Contains(output, "EPAR can now exit safely") { + t.Fatalf("incomplete cleanup console output = %q", output) + } +} diff --git a/internal/pool/manager.go b/internal/pool/manager.go index af81fb5..26482d5 100644 --- a/internal/pool/manager.go +++ b/internal/pool/manager.go @@ -2,6 +2,7 @@ package pool import ( "context" + "encoding/json" "errors" "fmt" "math" @@ -21,30 +22,53 @@ import ( gh "github.com/solutionforest/ephemeral-action-runner/internal/github" "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" "github.com/solutionforest/ephemeral-action-runner/internal/logging" + poolstate "github.com/solutionforest/ephemeral-action-runner/internal/pool/state" "github.com/solutionforest/ephemeral-action-runner/internal/provider" ) type Manager struct { - Config config.Config - Provider provider.Provider - GitHub GitHubClient - ProjectRoot string - ConfigPath string - DryRun bool - Logging *logging.Runtime - startupTiming *startupTiming - transcriptMu sync.Mutex - transcripts map[string]*logging.Transcript + Config config.Config + Provider provider.Provider + Lifecycle provider.Lifecycle + PolicyManager provider.PolicyManager + Storage provider.StorageContribution + LifecycleState *poolstate.Store + LifecycleStateEnabled bool + GitHub GitHubClient + ProjectRoot string + ConfigPath string + DryRun bool + Logging *logging.Runtime + AllowInsufficientStorage bool + StorageOverrideCommand string + AutomaticImageLifecycle bool + // AcknowledgeFailedDiagnostics permits the explicit cleanup command to + // dispose an exact retained sandbox after the operator has captured the + // durable failed-diagnostics evidence. Normal startup and automatic cleanup + // never set this override. + AcknowledgeFailedDiagnostics bool + startupTiming *startupTiming + transcriptMu sync.Mutex + transcripts map[string]*logging.Transcript hostTrustResolver func(context.Context) (hosttrust.Snapshot, error) + buildTrustResolver func(context.Context) (hosttrust.Snapshot, error) hostTrustImageEnsurer func(context.Context) error hostTrustImageMu sync.Mutex + imageEnsureMu sync.Mutex + imageEnsured bool now func() time.Time randomFloat64 func() float64 } +func (m *Manager) ConfigureStorageAdmissionOverride(allow bool, command string) { + m.AllowInsufficientStorage = allow + m.StorageOverrideCommand = command +} + type GitHubClient interface { OrganizationURL() string + EvaluateRunnerGroupPolicy(ctx context.Context, configuredGroup string, policy config.RunnerGroupSecurityConfig) (gh.RunnerGroupPolicyResult, error) RegistrationToken(ctx context.Context) (gh.RegistrationToken, error) ListRunners(ctx context.Context) ([]gh.Runner, error) RunnerByName(ctx context.Context, name string) (gh.Runner, bool, error) @@ -80,14 +104,23 @@ const ( LifecycleCleanupPending LifecyclePhase = "cleanup-pending" ) +const ( + runnerProcessRunningSentinel = "EPAR_RUNNER_PROCESS=running" + runnerProcessStoppedSentinel = "EPAR_RUNNER_PROCESS=stopped" + runnerProcessInactiveReason = "actions runner process is confirmed inactive" + runnerConfirmedInactiveCheckLimit = 2 +) + type ProvisionedInstance struct { Name string IP string LogPath string GuestLogPath string RunnerID int64 + ProviderID string HostTrustGeneration string Phase LifecyclePhase + ProviderOwned bool } var runtimeValidationRetryDelay = 5 * time.Second @@ -101,11 +134,19 @@ const ( ) func (m *Manager) Verify(ctx context.Context, opts VerifyOptions) error { + if opts.RegisterOnly { + if err := m.PreflightRunnerGroup(ctx); err != nil { + return err + } + } poolLock, err := m.AcquirePoolControllerLock() if err != nil { return err } defer poolLock.Close() + if err := m.recoverInterruptedProvisionLeases(ctx); err != nil { + return err + } controllerLock, err := m.acquireHostTrustControllerLock() if err != nil { return err @@ -113,6 +154,16 @@ func (m *Manager) Verify(ctx context.Context, opts VerifyOptions) error { if controllerLock != nil { defer controllerLock.Close() } + stopStorageLease, err := m.startStorageCatalogControllerLease() + if err != nil { + return err + } + defer stopStorageLease() + if m.AutomaticImageLifecycle { + if err := m.EnsureImage(ctx); err != nil { + return fmt.Errorf("ensure current provider artifact before verification: %w", err) + } + } opts.Instances = m.requestedInstances(opts.Instances) names := RunnerNames(m.Config.Pool.NamePrefix, opts.Instances, time.Now()) m.logger().Info("verifying instances", "provider", m.Config.Provider.Type, "operation", "verify", "instances", opts.Instances, "instanceNames", strings.Join(names, ", "), "sourceImage", m.Config.Provider.SourceImage) @@ -174,6 +225,11 @@ func (m *Manager) Verify(ctx context.Context, opts VerifyOptions) error { } func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { + if opts.Register { + if err := m.PreflightRunnerGroup(ctx); err != nil { + return err + } + } if !opts.PoolLockHeld { controllerLock, err := m.AcquirePoolControllerLock() if err != nil { @@ -181,6 +237,9 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { } defer controllerLock.Close() } + if err := m.recoverInterruptedProvisionLeases(ctx); err != nil { + return err + } if !opts.HostTrustLockHeld { controllerLock, err := m.AcquireHostTrustControllerLock() if err != nil { @@ -190,15 +249,26 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { defer controllerLock.Close() } } + stopStorageLease, err := m.startStorageCatalogControllerLease() + if err != nil { + return err + } + defer stopStorageLease() + if m.AutomaticImageLifecycle { + if err := m.EnsureImage(ctx); err != nil { + return fmt.Errorf("ensure current provider artifact before pool startup: %w", err) + } + } opts.Instances = m.requestedInstances(opts.Instances) if opts.MonitorInterval <= 0 { opts.MonitorInterval = 15 * time.Second } if ctx.Err() != nil { if opts.KeepOnExit { + m.infof("Stopping EPAR pool. --keep-on-exit is enabled, so owned runner resources will remain running.\n") return nil } - return m.cleanupWithFreshContext() + return m.cleanupPoolWithStatus("owned GitHub runner registrations and provider instances", m.cleanupWithFreshContext) } active, err := m.reconcilePhysicalPool(ctx, nil, opts.Register) if err != nil { @@ -212,9 +282,10 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { poolTrustGeneration := "" cleanup := func() error { if opts.KeepOnExit { + m.infof("Stopping EPAR pool. --keep-on-exit is enabled, so owned runner resources will remain running.\n") return nil } - return m.cleanupWithFreshContext() + return m.cleanupPoolWithStatus("owned GitHub runner registrations and provider instances", m.cleanupWithFreshContext) } leaseAdd, stopLeaseKeeper := m.startHostTrustLeaseKeeper(ctx) for len(active) < opts.Instances { @@ -246,7 +317,7 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { } stopLeaseKeeper() if !opts.Register || (!opts.ReplaceCompleted && !m.hostTrustEnabled()) { - m.infof("pool is running; press Ctrl-C to stop") + m.logPoolRunning("EPAR pool") if !m.Config.Logging.RetentionEnabled { <-ctx.Done() return cleanup() @@ -262,7 +333,8 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { } } } - m.infof("pool supervisor is running; monitoring every %s; press Ctrl-C to stop\n", opts.MonitorInterval) + m.infof("Pool supervisor is monitoring every %s.\n", opts.MonitorInterval) + m.logPoolRunning("EPAR pool") tickInterval := opts.MonitorInterval if m.hostTrustEnabled() && tickInterval > hostTrustRefreshInterval { tickInterval = hostTrustRefreshInterval @@ -273,7 +345,11 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { nextRetention := time.Now().Add(time.Duration(m.Config.Logging.RetentionIntervalMinutes) * time.Minute) nextHostTrustCollection := time.Time{} var currentHostTrust hosttrust.Snapshot + hostTrustBusyHandoff := make(map[string]bool) + confirmedInactiveChecks := make(map[string]int) + imageMaintenanceIdleChecks := make(map[string]int) retry := replacementRetryState{} + imageMaintenancePending := false for { select { case <-ctx.Done(): @@ -285,6 +361,38 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { m.pruneLogsBestEffort() nextRetention = time.Now().Add(time.Duration(m.Config.Logging.RetentionIntervalMinutes) * time.Minute) } + if m.AutomaticImageLifecycle && !imageMaintenancePending && m.Config.Image.UpdateFrequency != config.ImageUpdateFrequencyManual { + check, checkErr := m.CheckRemoteImageUpdate(ctx, now) + if checkErr != nil { + m.warnf("scheduled image update check failed; the current verified pool remains available: %v\n", checkErr) + } else if check.Changed { + if !m.Config.Runner.Ephemeral { + _ = m.DeferPendingImageUpdate("runner.ephemeral=false; the update will be applied on the next EPAR startup") + m.warnf("image update is available but this pool uses persistent runners; restart EPAR to apply it without an assignment race\n") + } else { + imageMaintenancePending = true + m.infof("image or Actions runner update is available; draining the ephemeral pool before artifact provisioning\n") + } + } + } + if imageMaintenancePending { + remaining, drainErr := m.drainPoolForImageUpdate(ctx, active, imageMaintenanceIdleChecks) + if drainErr != nil { + m.warnf("scheduled image maintenance drain warning; retrying without creating replacements: %v\n", drainErr) + continue + } + if remaining > 0 { + continue + } + m.infof("scheduled image maintenance drain complete; building and activating the verified replacement artifact\n") + if updateErr := m.ApplyPendingImageUpdate(ctx, now); updateErr != nil { + m.warnf("scheduled image update failed; restoring pool capacity with the previous verified generation: %v\n", updateErr) + } else { + m.infof("scheduled image update activated; restoring pool capacity\n") + } + imageMaintenancePending = false + clear(imageMaintenanceIdleChecks) + } trustRetired := 0 trustCapacityReady := true if m.hostTrustEnabled() { @@ -303,7 +411,7 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { // receive a mismatching lease so no subsequent job can start. currentHostTrust = current if !dependencyCooldown { - trustRetired += m.reconcileHostTrustRunners(ctx, active, current) + trustRetired += m.reconcileHostTrustRunners(ctx, active, current, hostTrustBusyHandoff) } m.infof("host trust generation changed (%s -> %s); building replacement image\n", emptyDash(poolTrustGeneration), current.Generation) ready = false @@ -339,7 +447,7 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { } } if currentHostTrust.Generation != "" && !dependencyCooldown { - trustRetired += m.reconcileHostTrustRunners(ctx, active, currentHostTrust) + trustRetired += m.reconcileHostTrustRunners(ctx, active, currentHostTrust, hostTrustBusyHandoff) } } if dependencyCooldown { @@ -355,18 +463,29 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { for name, vm := range active { alive, reason, err := m.runnerAlive(ctx, vm) if err != nil { - m.logger().Warn("liveness check failed", "provider", m.Config.Provider.Type, "instance", name, "operation", "liveness-check", "error", err) + recordRunnerLiveness(confirmedInactiveChecks, name, alive, reason, err) + m.warnf("[%s] runner health is temporarily unknown; keeping the runner and retrying: %v\n", name, err) continue } if alive { + recordRunnerLiveness(confirmedInactiveChecks, name, alive, reason, nil) continue } + confirmedCount, retire := recordRunnerLiveness(confirmedInactiveChecks, name, alive, reason, nil) + if !retire { + m.warnf("[%s] runner process is confirmed inactive (%d/%d); EPAR will verify once more before cleanup\n", name, confirmedCount, runnerConfirmedInactiveCheckLimit) + continue + } + if reason == runnerProcessInactiveReason { + m.captureRunnerReadinessDiagnostics(name, vm.GuestLogPath) + } m.infof("[%s] runner is finished or unhealthy: %s\n", name, reason) if err := m.retireInstance(context.Background(), vm, reason); err != nil { m.warnf("[%s] retirement warning: %v\n", name, err) continue } delete(active, name) + delete(confirmedInactiveChecks, name) } } var reconcileErr error @@ -450,11 +569,51 @@ func (m *Manager) RunPool(ctx context.Context, opts RunOptions) error { poolTrustGeneration = vm.HostTrustGeneration } m.infof("%s online at %s providerLog=%s guestLog=%s\n", vm.Name, vm.IP, vm.LogPath, vm.GuestLogPath) + m.logReplacementReady("EPAR pool", vm.Name) } } } } +func (m *Manager) drainPoolForImageUpdate(ctx context.Context, active map[string]ProvisionedInstance, confirmedIdle map[string]int) (int, error) { + for name, vm := range active { + if m.GitHub != nil { + runner, found, err := m.GitHub.RunnerByName(ctx, name) + if err != nil { + delete(confirmedIdle, name) + return len(active), fmt.Errorf("inspect GitHub runner %s before maintenance: %w", name, err) + } + if found { + vm.RunnerID = runner.ID + active[name] = vm + if err := m.recordLifecycleJobObservation(ctx, runner); err != nil { + delete(confirmedIdle, name) + return len(active), fmt.Errorf("record GitHub job state for %s before maintenance: %w", name, err) + } + } + if found && runner.Busy { + delete(confirmedIdle, name) + m.infof("[%s] scheduled image maintenance is waiting for the active job to finish\n", name) + continue + } + if found { + confirmedIdle[name]++ + if confirmedIdle[name] < 2 { + m.infof("[%s] scheduled image maintenance observed the runner idle; confirming it remains unassigned before retirement\n", name) + continue + } + } + } + m.infof("[%s] retiring idle runner for scheduled image maintenance\n", name) + if err := m.retireInstance(context.Background(), vm, "scheduled image and Actions runner update"); err != nil { + return len(active), err + } + delete(active, name) + delete(confirmedIdle, name) + } + return len(active), nil +} + func currentHostTrustCapacity(active map[string]ProvisionedInstance, generation string) int { capacity := 0 for _, instance := range active { @@ -486,7 +645,7 @@ func (m *Manager) reconcilePhysicalPool(ctx context.Context, known map[string]Pr } if !register { for name, vm := range reconciled { - if vm.Phase != LifecycleCleanupPending { + if vm.ProviderOwned && vm.Phase != LifecycleCleanupPending { vm.Phase = LifecycleReady reconciled[name] = vm } @@ -513,7 +672,17 @@ func (m *Manager) reconcilePhysicalPool(ctx context.Context, known map[string]Pr } } for name, vm := range reconciled { + if !vm.ProviderOwned { + delete(remoteByName, name) + vm.Phase = LifecycleQuarantined + reconciled[name] = vm + continue + } if vm.Phase == LifecycleCleanupPending { + // The local cleanup path already matched this exact lifecycle + // record. Keep its remote identity attached to the protected + // record instead of treating it as an orphan below. + delete(remoteByName, name) continue } runner, found := remoteByName[name] @@ -527,6 +696,11 @@ func (m *Manager) reconcilePhysicalPool(ctx context.Context, known map[string]Pr } } if !found { + if err := m.recordLifecycleRemoteAbsence(ctx, name); err != nil { + vm.Phase = LifecycleQuarantined + reconciled[name] = vm + return reconciled, fmt.Errorf("record GitHub runner absence for %s: %w", name, err) + } if err := m.deleteLocalInstance(context.Background(), vm); err != nil { vm.Phase = LifecycleCleanupPending reconciled[name] = vm @@ -538,6 +712,11 @@ func (m *Manager) reconcilePhysicalPool(ctx context.Context, known map[string]Pr } delete(remoteByName, name) vm.RunnerID = runner.ID + if err := m.recordLifecycleJobObservation(ctx, runner); err != nil { + vm.Phase = LifecycleQuarantined + reconciled[name] = vm + return reconciled, fmt.Errorf("record GitHub job phase for %s: %w", name, err) + } if runner.Status == "online" { vm.Phase = LifecycleReady reconciled[name] = vm @@ -559,7 +738,13 @@ func (m *Manager) reconcilePhysicalPool(ctx context.Context, known map[string]Pr continue } alive, _, processErr := m.runnerProcessAlive(ctx, vm) - if processErr == nil && alive { + if processErr != nil { + vm.Phase = LifecycleQuarantined + reconciled[name] = vm + m.warnf("[%s] reconciliation could not verify the Actions runner process; preserving the exact instance in quarantine: %v\n", name, processErr) + continue + } + if alive { vm.Phase = LifecycleQuarantined reconciled[name] = vm continue @@ -573,6 +758,14 @@ func (m *Manager) reconcilePhysicalPool(ctx context.Context, known map[string]Pr } } for _, runner := range remoteByName { + owned, ownershipErr := m.lifecycleOwnsRunner(ctx, runner.Name, runner.ID) + if ownershipErr != nil { + return reconciled, ownershipErr + } + if !owned { + m.warnf("reconciliation: quarantined unowned GitHub runner %s id=%d; prefix-only resources are report-only\n", runner.Name, runner.ID) + continue + } if err := m.deleteRemoteRunner(context.Background(), runner); err != nil { return reconciled, err } @@ -623,17 +816,36 @@ func (m *Manager) reconcileLocalInventory(known map[string]ProvisionedInstance) } func (m *Manager) reconcileLocalInventoryWithContext(ctx context.Context, known map[string]ProvisionedInstance) (map[string]ProvisionedInstance, error) { - locals, err := m.Provider.List(ctx) + locals, err := m.inventoryProvider(ctx) if err != nil { return known, err } reconciled := make(map[string]ProvisionedInstance) - for _, local := range locals { + for _, item := range locals { + local := item.Instance if !HasPrefix(local.Name, m.Config.Pool.NamePrefix) { continue } vm := m.reconciledInstance(known, local.Name) - if !localInstanceStopped(local.State) { + vm.ProviderID = local.ProviderID + owned, ownershipErr := m.lifecycleOwns(ctx, local.Name, local.ProviderID) + if ownershipErr != nil { + return known, fmt.Errorf("verify lifecycle ownership for %s: %w", local.Name, ownershipErr) + } + vm.ProviderOwned = owned + if !owned { + providerID := local.ProviderID + if providerID == "" { + providerID = "unidentified:" + local.Name + } + if reportErr := m.reportUnknownLifecycle(ctx, local.Name, providerID, item.Source, item.State); reportErr != nil { + return known, fmt.Errorf("quarantine unowned provider instance %s: %w", local.Name, reportErr) + } + vm.Phase = LifecycleQuarantined + reconciled[local.Name] = vm + continue + } + if !localInstanceStopped(item.State) { reconciled[local.Name] = vm continue } @@ -651,10 +863,11 @@ func (m *Manager) reconciledInstance(known map[string]ProvisionedInstance, name return vm } return ProvisionedInstance{ - Name: name, - LogPath: m.instanceLogPath(name, "."+m.Config.Provider.Type+".log"), - GuestLogPath: m.instanceLogPath(name, ".guest.log"), - Phase: LifecycleQuarantined, + Name: name, + LogPath: m.instanceLogPath(name, "."+m.Config.Provider.Type+".log"), + GuestLogPath: m.instanceLogPath(name, ".guest.log"), + Phase: LifecycleQuarantined, + ProviderOwned: m.LifecycleState == nil, } } @@ -668,13 +881,40 @@ func localInstanceStopped(state string) bool { } func (m *Manager) deleteLocalInstance(ctx context.Context, vm ProvisionedInstance) error { + if m.LifecycleState != nil { + record, err := m.LifecycleState.Read(ctx, vm.Name) + if err != nil { + return err + } + inventory, err := m.inventoryProvider(ctx) + if err != nil { + return err + } + return m.cleanupLifecycleRecord(ctx, record, inventoryByName(inventory)[vm.Name]) + } cleanupCtx, cancel := context.WithTimeout(ctx, cleanupTimeout) defer cancel() + if vm.ProviderID == "" && m.Lifecycle == nil && m.Provider != nil { + stopCtx, stopCancel := context.WithTimeout(cleanupCtx, 60*time.Second) + _ = m.Provider.Stop(stopCtx, vm.Name) + stopCancel() + deleteCtx, deleteCancel := context.WithTimeout(cleanupCtx, 60*time.Second) + err := m.Provider.Delete(deleteCtx, vm.Name) + deleteCancel() + return err + } + instance, err := m.providerInstance(cleanupCtx, vm.Name) + if err != nil { + return err + } + if vm.ProviderID != "" && instance.ProviderID != vm.ProviderID { + return fmt.Errorf("same-name provider instance id=%s does not match expected id=%s; refusing deletion", instance.ProviderID, vm.ProviderID) + } stopCtx, stopCancel := context.WithTimeout(cleanupCtx, 60*time.Second) - _ = m.Provider.Stop(stopCtx, vm.Name) + _ = m.stopProviderInstance(stopCtx, instance) stopCancel() deleteCtx, deleteCancel := context.WithTimeout(cleanupCtx, 60*time.Second) - err := m.Provider.Delete(deleteCtx, vm.Name) + err = m.deleteProviderInstance(deleteCtx, instance) deleteCancel() if err != nil { return err @@ -853,6 +1093,9 @@ func (m *Manager) ProvisionPool(ctx context.Context, instances int, register boo return nil, err } defer poolLock.Close() + if err := m.recoverInterruptedProvisionLeases(ctx); err != nil { + return nil, err + } hostTrustLock, err := m.AcquireHostTrustControllerLock() if err != nil { return nil, err @@ -889,25 +1132,43 @@ func (m *Manager) Cleanup(ctx context.Context) error { return err } defer poolLock.Close() + if err := m.recoverInterruptedProvisionLeases(ctx); err != nil { + return err + } return m.cleanupUnlocked(ctx) } func (m *Manager) cleanupUnlocked(ctx context.Context) error { + if m.LifecycleState != nil { + return m.cleanupOwnedLifecycle(ctx) + } + if m.Lifecycle == nil && m.Provider != nil { + return m.cleanupLegacyTestProvider(ctx) + } var firstErr error - vms, err := m.Provider.List(ctx) + items, err := m.inventoryProvider(ctx) if err != nil { firstErr = err } - for _, vm := range vms { + for _, item := range items { + vm := item.Instance if !HasPrefix(vm.Name, m.Config.Pool.NamePrefix) { continue } + if vm.ProviderID == "" { + m.warnf("cleanup: provider instance %s has no immutable identity; leaving it report-only\n", vm.Name) + continue + } + if m.DryRun { + m.infof("[dry-run] cleanup would delete exact provider instance %s id=%s\n", vm.Name, vm.ProviderID) + continue + } m.infof("cleanup: deleting instance %s\n", vm.Name) stopCtx, stopCancel := context.WithTimeout(ctx, 60*time.Second) - _ = m.Provider.Stop(stopCtx, vm.Name) + _ = m.stopProviderInstance(stopCtx, vm) stopCancel() deleteCtx, deleteCancel := context.WithTimeout(ctx, 60*time.Second) - deleteErr := m.Provider.Delete(deleteCtx, vm.Name) + deleteErr := m.deleteProviderInstance(deleteCtx, vm) if deleteErr != nil && firstErr == nil { firstErr = deleteErr } @@ -937,16 +1198,85 @@ func (m *Manager) cleanupUnlocked(ctx context.Context) error { return firstErr } +// cleanupLegacyTestProvider keeps the old in-memory test seam isolated from +// production. Every registry-constructed manager has Lifecycle and durable +// state, so real cleanup always uses immutable provider and GitHub identities. +func (m *Manager) cleanupLegacyTestProvider(ctx context.Context) error { + var firstErr error + vms, err := m.Provider.List(ctx) + if err != nil { + firstErr = err + } + for _, vm := range vms { + if !HasPrefix(vm.Name, m.Config.Pool.NamePrefix) { + continue + } + m.infof("cleanup: deleting instance %s\n", vm.Name) + stopCtx, stopCancel := context.WithTimeout(ctx, 60*time.Second) + _ = m.Provider.Stop(stopCtx, vm.Name) + stopCancel() + deleteCtx, deleteCancel := context.WithTimeout(ctx, 60*time.Second) + deleteErr := m.Provider.Delete(deleteCtx, vm.Name) + deleteCancel() + if deleteErr != nil && firstErr == nil { + firstErr = deleteErr + } + } + if m.GitHub != nil { + deleteCtx, cancel := context.WithTimeout(ctx, 60*time.Second) + defer cancel() + _, err := m.GitHub.DeleteRunnersByPrefix(deleteCtx, m.Config.Pool.NamePrefix) + if err != nil && firstErr == nil { + firstErr = err + } + } + return firstErr +} + func (m *Manager) Status(ctx context.Context) (string, error) { var b strings.Builder - vms, err := m.Provider.List(ctx) + updateStatus, updateErr := m.ImageUpdatePolicyStatus() + if updateErr != nil { + if m.Config.Image.UpdateFrequency == config.ImageUpdateFrequencyManual { + fmt.Fprintf(&b, "Image updates:\n policy=manual\tstate=unavailable\terror=%s\n", updateErr) + } else { + fmt.Fprintf(&b, "Image updates:\n policy=%s at %s local\tstate=unavailable\terror=%s\n", m.Config.Image.UpdateFrequency, m.Config.Image.UpdateTime, updateErr) + } + } else { + if updateStatus.Frequency == config.ImageUpdateFrequencyManual { + fmt.Fprintf(&b, "Image updates:\n policy=manual") + } else { + fmt.Fprintf(&b, "Image updates:\n policy=%s at %s local", updateStatus.Frequency, updateStatus.UpdateTime) + } + if !updateStatus.LastSuccessfulCheckAt.IsZero() { + fmt.Fprintf(&b, "\tlast=%s", updateStatus.LastSuccessfulCheckAt.In(time.Local).Format("2006-01-02 15:04 MST")) + } + if !updateStatus.NextEligibleAt.IsZero() { + fmt.Fprintf(&b, "\tnext=%s", updateStatus.NextEligibleAt.In(time.Local).Format("2006-01-02 15:04 MST")) + } + if !updateStatus.NextRetryAt.IsZero() { + fmt.Fprintf(&b, "\tretry=%s", updateStatus.NextRetryAt.In(time.Local).Format("2006-01-02 15:04 MST")) + } + if updateStatus.Pending { + fmt.Fprintf(&b, "\tpending=%s", updateStatus.PendingIdentity) + } + b.WriteString("\n") + if updateStatus.DeferredReason != "" { + fmt.Fprintf(&b, " deferred: %s\n", updateStatus.DeferredReason) + } + if updateStatus.LastError != "" { + fmt.Fprintf(&b, " last error: %s\n", updateStatus.LastError) + } + } + items, err := m.inventoryProvider(ctx) if err != nil { return "", err } b.WriteString("Instances:\n") - for _, vm := range vms { + for _, item := range items { + vm := item.Instance if HasPrefix(vm.Name, m.Config.Pool.NamePrefix) { - fmt.Fprintf(&b, " %s\t%s\n", vm.Name, vm.State) + fmt.Fprintf(&b, " %s\t%s\tid=%s\n", vm.Name, item.State, emptyDash(vm.ProviderID)) } } if m.GitHub != nil { @@ -967,77 +1297,97 @@ func (m *Manager) Status(ctx context.Context) (string, error) { var errHostTrustImageMismatch = errors.New("runner image host trust generation does not match current host trust") func (m *Manager) provisionOne(ctx context.Context, name string, register, allowBusy bool) (ProvisionedInstance, error) { - const attempts = 3 - var lastErr error - for attempt := 1; attempt <= attempts; attempt++ { - vm, err := m.provisionOneAttempt(ctx, name, register, allowBusy) - if err == nil || !errors.Is(err, errHostTrustImageMismatch) { - return vm, err - } - lastErr = err - if isPhysicalPhase(vm.Phase) { - _ = m.retireInstance(context.Background(), vm, "discarding stale host-trust image generation") - } - if attempt == attempts { - break - } - m.infof("[%s] host trust changed before runner publication; rebuilding image (attempt %d/%d)\n", name, attempt+1, attempts) - if err := m.ensureHostTrustImage(ctx); err != nil { - return vm, fmt.Errorf("rebuild image after host trust changed during provisioning: %w", err) - } - } - return ProvisionedInstance{Name: name}, fmt.Errorf("provision runner after %d host trust image stabilization attempts: %w", attempts, lastErr) + return m.provisionOneAttempt(ctx, name, register, allowBusy) } -func (m *Manager) deleteRemoteRunnerByNameBestEffort(name string) { - if m.GitHub == nil { - return +func (m *Manager) provisionOneAttempt(ctx context.Context, name string, register, allowBusy bool) (vm ProvisionedInstance, err error) { + logPath := m.instanceLogPath(name, "."+m.Config.Provider.Type+".log") + guestLogPath := m.instanceLogPath(name, ".guest.log") + vm = ProvisionedInstance{Name: name, LogPath: logPath, GuestLogPath: guestLogPath, ProviderOwned: true} + if register && m.GitHub != nil && !m.DryRun && m.LifecycleState != nil { + if runner, found, lookupErr := m.GitHub.RunnerByName(ctx, name); lookupErr != nil { + return vm, fmt.Errorf("verify exact GitHub runner name is unallocated: %w", lookupErr) + } else if found { + return vm, fmt.Errorf("GitHub runner name %q is already allocated to id=%d", name, runner.ID) + } } - ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second) - defer cancel() - runner, found, err := m.GitHub.RunnerByName(ctx, name) - if err != nil { - m.warnf("[%s] deferred exact-name GitHub reconciliation after rollback: %v\n", name, err) - return + if err := m.preflightStorage("instance-create", m.instanceCreateExpansion()); err != nil { + return vm, err } - if !found { - return + if err := m.reserveLifecycle(ctx, name); err != nil { + return vm, err } - if err := m.GitHub.DeleteRunnerIfExists(ctx, runner.ID); err != nil { - m.warnf("[%s] deferred exact-name GitHub runner deletion after rollback: %v\n", name, err) + vm.Phase = LifecycleProvisioning + if err := m.acquireLifecycleLease(ctx, name, "provision", "controller", 2*time.Hour); err != nil { + return vm, fmt.Errorf("acquire provisioning lifecycle lease: %w", err) } -} - -func (m *Manager) provisionOneAttempt(ctx context.Context, name string, register, allowBusy bool) (vm ProvisionedInstance, err error) { - logPath := m.instanceLogPath(name, "."+m.Config.Provider.Type+".log") - guestLogPath := m.instanceLogPath(name, ".guest.log") - vm = ProvisionedInstance{Name: name, LogPath: logPath, GuestLogPath: guestLogPath, Phase: LifecycleProvisioning} - localMayExist := false configureAttempted := false listenerMayBeRunning := false defer func() { + m.releaseLifecycleLease(context.Background(), name, "provision", "controller") if err == nil { vm.Phase = LifecycleReady return } - if !localMayExist { - vm.Phase = "" - return + remoteKnownAbsent := !configureAttempted + if configureAttempted && m.GitHub != nil { + runner, found, lookupErr := m.GitHub.RunnerByName(context.Background(), name) + if lookupErr != nil { + m.quarantineLifecycle(context.Background(), name, fmt.Errorf("%w; exact GitHub registration lookup failed: %v", err, lookupErr)) + vm.Phase = LifecycleQuarantined + return + } + remoteKnownAbsent = !found + if found { + vm.RunnerID = runner.ID + if recordErr := m.recordLifecycleRegistered(context.Background(), name, runner.ID); recordErr != nil { + m.quarantineLifecycle(context.Background(), name, fmt.Errorf("%w; exact GitHub runner id=%d could not be recorded: %v", err, runner.ID, recordErr)) + vm.Phase = LifecycleQuarantined + return + } + } } if listenerMayBeRunning { + m.quarantineLifecycle(context.Background(), name, err) vm.Phase = LifecycleQuarantined return } - cleanupErr := m.deleteLocalInstance(context.Background(), vm) + if m.LifecycleState == nil { + if vm.RunnerID != 0 && m.GitHub != nil { + if deleteErr := m.GitHub.DeleteRunnerIfExists(context.Background(), vm.RunnerID); deleteErr != nil { + vm.Phase = LifecycleCleanupPending + err = errors.Join(err, fmt.Errorf("rollback exact GitHub runner id=%d: %w", vm.RunnerID, deleteErr)) + return + } + } + cleanupErr := m.deleteLocalInstance(context.Background(), vm) + if cleanupErr != nil { + vm.Phase = LifecycleCleanupPending + err = errors.Join(err, fmt.Errorf("rollback local instance %s: %w", name, cleanupErr)) + } else { + vm.Phase = "" + } + return + } + record, recordErr := m.LifecycleState.Read(context.Background(), name) + if recordErr != nil { + vm.Phase = LifecycleCleanupPending + err = errors.Join(err, fmt.Errorf("read lifecycle for rollback: %w", recordErr)) + return + } + inventory, inventoryErr := m.inventoryProvider(context.Background()) + if inventoryErr != nil { + m.quarantineLifecycle(context.Background(), name, errors.Join(err, inventoryErr)) + vm.Phase = LifecycleQuarantined + return + } + cleanupErr := m.cleanupLifecycleRecordWithRemoteAbsence(context.Background(), record, inventoryByName(inventory)[name], remoteKnownAbsent) if cleanupErr != nil { vm.Phase = LifecycleCleanupPending err = errors.Join(err, fmt.Errorf("rollback local instance %s: %w", name, cleanupErr)) return } vm.Phase = "" - if configureAttempted { - m.deleteRemoteRunnerByNameBestEffort(name) - } }() var trustSnapshot hosttrust.Snapshot if m.hostTrustEnabled() { @@ -1052,10 +1402,56 @@ func (m *Manager) provisionOneAttempt(ctx context.Context, name string, register return vm, err } m.logger().Info("cloning instance", "provider", m.Config.Provider.Type, "instance", name, "operation", "clone", "sourceImage", m.Config.Provider.SourceImage, "logPath", logPath) - localMayExist = true - if err := m.timeFirstInstanceStage(name, "instance_container_create", func() error { - return m.Provider.Clone(ctx, m.Config.Provider.SourceImage, name) - }); err != nil { + var created provider.Instance + createStageErr := m.timeFirstInstanceStage(name, "instance_container_create", func() error { + var createErr error + created, createErr = m.createProviderInstance(ctx, name) + return createErr + }) + if created.Name != "" || created.ProviderID != "" || created.ReceiptVersion != "" || len(created.Receipt) != 0 { + if created.Name != name || created.ProviderID == "" { + identityErr := fmt.Errorf("provider create returned an incomplete immutable identity for %q", name) + if createStageErr != nil { + return vm, errors.Join(createStageErr, identityErr) + } + return vm, identityErr + } + if created.ReceiptVersion == "" || len(created.Receipt) == 0 { + receiptErr := fmt.Errorf("provider create returned an incomplete versioned receipt for %q", name) + if createStageErr != nil { + return vm, errors.Join(createStageErr, receiptErr) + } + return vm, receiptErr + } + var providerReceipt map[string]any + if json.Unmarshal(created.Receipt, &providerReceipt) != nil || providerReceipt == nil { + receiptErr := fmt.Errorf("provider create returned an invalid versioned receipt for %q", name) + if createStageErr != nil { + return vm, errors.Join(createStageErr, receiptErr) + } + return vm, receiptErr + } + vm.ProviderID = created.ProviderID + if recordErr := m.recordLifecycleCreated(context.WithoutCancel(ctx), created); recordErr != nil { + if createStageErr != nil { + return vm, errors.Join(createStageErr, recordErr) + } + return vm, recordErr + } + } + if createStageErr != nil { + return vm, createStageErr + } + if created.Name != name || created.ProviderID == "" { + return vm, fmt.Errorf("provider create returned no immutable identity for %q", name) + } + if err := m.recordLifecycleValidationIntent(ctx, name); err != nil { + return vm, fmt.Errorf("record runtime validation intent: %w", err) + } + if err := m.applyProviderNetworkPolicy(ctx, created); err != nil { + return vm, fmt.Errorf("apply provider network policy: %w", err) + } + if err := m.verifyProviderAdmission(ctx, created); err != nil { return vm, err } m.logger().Info("starting instance", "provider", m.Config.Provider.Type, "instance", name, "operation", "start", "logPath", logPath) @@ -1064,23 +1460,37 @@ func (m *Manager) provisionOneAttempt(ctx context.Context, name string, register if startOptionsErr != nil { return startOptionsErr } - _, err := m.Provider.Start(ctx, name, startOptions) + _, err := m.startProviderInstance(ctx, created, startOptions) return err }); err != nil { return vm, err } - ip, err := m.Provider.IP(ctx, name, m.Config.Timeouts.BootSeconds) + ip, available, err := m.providerAddress(ctx, created, m.Config.Timeouts.BootSeconds) if err != nil { return vm, err } vm.IP = ip - m.logger().Info("instance reachable", "provider", m.Config.Provider.Type, "instance", name, "operation", "wait-reachable", "address", ip) + if available { + m.logger().Info("instance reachable", "provider", m.Config.Provider.Type, "instance", name, "operation", "wait-reachable", "address", ip) + } else { + m.logger().Info("instance uses delegated provider execution", "provider", m.Config.Provider.Type, "instance", name, "operation", "wait-reachable") + } + if m.hostTrustEnabled() { + trustSnapshot, err = m.resolveHostTrust(ctx) + if err != nil { + return vm, fmt.Errorf("refresh host trust before runtime installation: %w", err) + } + if err := m.installHostTrustRuntime(ctx, name, trustSnapshot); err != nil { + return vm, err + } + vm.HostTrustGeneration = trustSnapshot.Generation + } m.logger().Info("validating runner runtime", "provider", m.Config.Provider.Type, "instance", name, "operation", "validate-runtime", "stage", "start") if err := m.timeFirstInstanceStage(name, "runtime_validation", func() error { if err := m.configureDockerRegistryMirrors(ctx, name); err != nil { return err } - return m.validateRuntimeWithRetry(ctx, name, guestLogPath) + return m.verifyProviderRuntimeWithRetry(ctx, created, guestLogPath) }); err != nil { return vm, err } @@ -1095,7 +1505,16 @@ func (m *Manager) provisionOneAttempt(ctx context.Context, name string, register return vm, fmt.Errorf("%w: refresh host trust after runtime validation: %v", errHostTrustImageMismatch, err) } if err := validateHostTrustMarkerAgainstSnapshot(marker, currentTrust); err != nil { - return vm, fmt.Errorf("%w: %v", errHostTrustImageMismatch, err) + if installErr := m.installHostTrustRuntime(ctx, name, currentTrust); installErr != nil { + return vm, fmt.Errorf("%w: %v; runtime refresh failed: %v", errHostTrustImageMismatch, err, installErr) + } + marker, err = m.readInstanceHostTrustMarker(ctx, name) + if err != nil { + return vm, fmt.Errorf("%w: read refreshed marker: %v", errHostTrustImageMismatch, err) + } + if err := validateHostTrustMarkerAgainstSnapshot(marker, currentTrust); err != nil { + return vm, fmt.Errorf("%w: refreshed runtime marker: %v", errHostTrustImageMismatch, err) + } } // Track the immutable generation read from the cloned image, not merely // the pre-clone snapshot. This prevents a trust-store change racing image @@ -1103,10 +1522,19 @@ func (m *Manager) provisionOneAttempt(ctx context.Context, name string, register vm.HostTrustGeneration = marker.Generation trustSnapshot = currentTrust } + if err := m.recordLifecycleValidated(ctx, name); err != nil { + return vm, fmt.Errorf("record validated runtime: %w", err) + } if register { + if err := m.recordLifecycleRegistrationIntent(ctx, name); err != nil { + return vm, fmt.Errorf("record GitHub registration intent: %w", err) + } if err := m.issueHostTrustLease(ctx, name, trustSnapshot); err != nil { return vm, fmt.Errorf("issue host trust lease: %w", err) } + if err := m.verifyProviderAdmission(ctx, created); err != nil { + return vm, err + } if m.GitHub == nil { if m.DryRun { m.infof("[dry-run] would register GitHub runner %s with labels %s\n", name, strings.Join(m.Config.Runner.Labels, ",")) @@ -1114,6 +1542,9 @@ func (m *Manager) provisionOneAttempt(ctx context.Context, name string, register } return vm, fmt.Errorf("github client is required for registration") } + if err := m.PreflightRunnerGroup(ctx); err != nil { + return vm, err + } var ( token gh.RegistrationToken runner gh.Runner @@ -1154,6 +1585,9 @@ func (m *Manager) provisionOneAttempt(ctx context.Context, name string, register return vm, err } vm.RunnerID = runner.ID + if err := m.recordLifecycleRegistered(ctx, name, runner.ID); err != nil { + return vm, fmt.Errorf("record exact GitHub runner identity: %w", err) + } m.infof("[%s] GitHub runner %s id=%d busy=%t\n", name, readiness, runner.ID, runner.Busy) m.finishFirstRunnerReady(name) } else { @@ -1162,8 +1596,43 @@ func (m *Manager) provisionOneAttempt(ctx context.Context, name string, register return vm, nil } +func (m *Manager) PreflightRunnerGroup(ctx context.Context) error { + if m.DryRun { + m.infof("[dry-run] would verify GitHub runner-group security policy before registration\n") + return nil + } + if m.GitHub == nil { + return fmt.Errorf("runner-group security preflight requires a GitHub client") + } + policy := m.Config.Security.RunnerGroup + result, err := m.GitHub.EvaluateRunnerGroupPolicy(ctx, m.Config.Runner.Group, policy) + if err != nil { + message := fmt.Sprintf("runner-group security preflight could not read GitHub policy: %v", err) + if policy.Enforcement == config.RunnerGroupEnforcementWarn { + m.warnf("warning: %s; continuing because security.runnerGroup.enforcement is warn\n", message) + return nil + } + return fmt.Errorf("%s", message) + } + for _, advisory := range result.Advisories { + m.warnf("warning: runner-group security advisory: %s\n", advisory) + } + if len(result.Violations) > 0 { + message := strings.Join(result.Violations, "; ") + if policy.Enforcement == config.RunnerGroupEnforcementWarn { + m.warnf("warning: runner-group security policy violation: %s; continuing because security.runnerGroup.enforcement is warn\n", message) + return nil + } + return fmt.Errorf("runner-group security preflight failed: %s", message) + } + if result.Resolved { + m.infof("runner-group security preflight passed for %q\n", result.Group.Name) + } + return nil +} + func (m *Manager) startupInstanceStartStage() string { - if m.Config.Provider.Type == "docker-dind" { + if m.Config.Provider.Type == "docker-container" { return "instance_start_and_inner_docker_ready" } return "instance_start_and_provider_ready" @@ -1205,6 +1674,15 @@ func (m *Manager) waitRunnerReadyAndHealthy(ctx context.Context, vm ProvisionedI case result := <-resultCh: return result.runner, result.err case <-ticker.C: + instance, instanceErr := m.providerInstance(waitCtx, vm.Name) + if instanceErr != nil { + cancel() + return gh.Runner{}, instanceErr + } + if err := m.verifyProviderAdmission(waitCtx, instance); err != nil { + cancel() + return gh.Runner{}, err + } if m.hostTrustEnabled() && !time.Now().Before(nextLeaseRefresh) { current, err := m.resolveHostTrust(waitCtx) if err != nil { @@ -1268,6 +1746,15 @@ func (m *Manager) captureRunnerReadinessDiagnostics(name, guestLogPath string) { } func (m *Manager) runnerAlive(ctx context.Context, vm ProvisionedInstance) (bool, string, error) { + if _, globalAdmission := m.Lifecycle.(provider.AdmissionVerifier); globalAdmission { + instance, err := m.providerInstance(ctx, vm.Name) + if err != nil { + return false, "provider instance identity is unavailable", err + } + if err := m.verifyProviderAdmission(ctx, instance); err != nil { + return false, "provider admission changed", err + } + } if m.GitHub != nil { runner, found, err := m.GitHub.RunnerByName(ctx, vm.Name) if err != nil { @@ -1276,6 +1763,9 @@ func (m *Manager) runnerAlive(ctx context.Context, vm ProvisionedInstance) (bool } } else { if !found { + if err := m.recordLifecycleRemoteAbsence(ctx, vm.Name); err != nil { + return false, "GitHub runner record is gone", fmt.Errorf("record GitHub runner absence: %w", err) + } return false, "GitHub runner record is gone", nil } if runner.Busy { @@ -1289,27 +1779,80 @@ func (m *Manager) runnerAlive(ctx context.Context, vm ProvisionedInstance) (bool return m.runnerProcessAlive(ctx, vm) } +func recordRunnerLiveness(confirmedInactive map[string]int, name string, alive bool, reason string, err error) (int, bool) { + if err != nil || alive { + delete(confirmedInactive, name) + return 0, false + } + if reason != runnerProcessInactiveReason { + delete(confirmedInactive, name) + return 0, true + } + confirmedInactive[name]++ + return confirmedInactive[name], confirmedInactive[name] >= runnerConfirmedInactiveCheckLimit +} + func (m *Manager) runnerProcessAlive(ctx context.Context, vm ProvisionedInstance) (bool, string, error) { - if err := m.checkRunnerProcess(ctx, vm.Name); err != nil { - return false, "actions runner process is no longer active", nil + running, err := m.probeRunnerProcess(ctx, vm.Name) + if err != nil { + return true, "runner process health could not be measured", err + } + if !running { + return false, runnerProcessInactiveReason, nil } return true, "", nil } func (m *Manager) checkRunnerProcess(ctx context.Context, name string) error { + running, err := m.probeRunnerProcess(ctx, name) + if err != nil { + return err + } + if !running { + return fmt.Errorf(runnerProcessInactiveReason) + } + return nil +} + +func (m *Manager) probeRunnerProcess(ctx context.Context, name string) (bool, error) { checkCtx, cancel := context.WithTimeout(ctx, 30*time.Second) defer cancel() - _, err := m.execGuest(checkCtx, name, provider.ShellCommand("if test -x /opt/epar/check-runner.sh; then sudo bash /opt/epar/check-runner.sh; else systemctl is-active --quiet actions-runner.service; fi"), provider.ExecOptions{}) - return err + script := fmt.Sprintf("if test -x /opt/epar/check-runner.sh; then if sudo bash /opt/epar/check-runner.sh; then printf '%%s\\n' %s; else printf '%%s\\n' %s; fi; elif systemctl is-active --quiet actions-runner.service; then printf '%%s\\n' %s; else printf '%%s\\n' %s; fi", shellQuote(runnerProcessRunningSentinel), shellQuote(runnerProcessStoppedSentinel), shellQuote(runnerProcessRunningSentinel), shellQuote(runnerProcessStoppedSentinel)) + result, err := m.execGuest(checkCtx, name, provider.ShellCommand(script), provider.ExecOptions{SuppressTranscript: true}) + if err != nil { + return false, fmt.Errorf("execute runner process health probe: %w", err) + } + switch strings.TrimSpace(result.Stdout) { + case runnerProcessRunningSentinel: + return true, nil + case runnerProcessStoppedSentinel: + return false, nil + default: + return false, fmt.Errorf("runner process health probe returned an unsupported response") + } } func isTransientGitHubLivenessError(err error) bool { var httpErr *gh.HTTPError - return errors.As(err, &httpErr) && httpErr.StatusCode >= http.StatusInternalServerError + return errors.As(err, &httpErr) && (httpErr.StatusCode == http.StatusTooManyRequests || httpErr.StatusCode >= http.StatusInternalServerError) } func (m *Manager) retireInstance(ctx context.Context, vm ProvisionedInstance, reason string) error { + if m.LifecycleState != nil && !vm.ProviderOwned { + return fmt.Errorf("refusing to retire unowned provider instance %q; prefix-only resources are report-only", vm.Name) + } m.infof("[%s] retiring instance: %s\n", vm.Name, reason) + if m.LifecycleState != nil { + record, err := m.LifecycleState.Read(ctx, vm.Name) + if err != nil { + return err + } + inventory, err := m.inventoryProvider(ctx) + if err != nil { + return err + } + return m.cleanupLifecycleRecord(ctx, record, inventoryByName(inventory)[vm.Name]) + } var firstErr error if m.GitHub != nil && vm.RunnerID != 0 { deleteCtx, cancel := context.WithTimeout(ctx, 60*time.Second) @@ -1319,11 +1862,32 @@ func (m *Manager) retireInstance(ctx context.Context, vm ProvisionedInstance, re } cancel() } + if vm.ProviderID == "" && m.Lifecycle == nil && m.Provider != nil { + stopCtx, stopCancel := context.WithTimeout(ctx, 60*time.Second) + _ = m.Provider.Stop(stopCtx, vm.Name) + stopCancel() + deleteCtx, deleteCancel := context.WithTimeout(ctx, 60*time.Second) + deleteErr := m.Provider.Delete(deleteCtx, vm.Name) + deleteCancel() + if deleteErr == nil { + if releaseErr := m.releaseInstanceTranscripts(vm); releaseErr != nil { + m.logger().Warn("instance transcript close failed after retirement", "provider", m.Config.Provider.Type, "instance", vm.Name, "operation", "retire", "error", releaseErr) + } + } + return deleteErr + } + instance, err := m.providerInstance(ctx, vm.Name) + if err != nil { + return err + } + if vm.ProviderID != "" && instance.ProviderID != vm.ProviderID { + return fmt.Errorf("same-name provider instance id=%s does not match expected id=%s; refusing retirement", instance.ProviderID, vm.ProviderID) + } stopCtx, stopCancel := context.WithTimeout(ctx, 60*time.Second) - _ = m.Provider.Stop(stopCtx, vm.Name) + _ = m.stopProviderInstance(stopCtx, instance) stopCancel() deleteCtx, deleteCancel := context.WithTimeout(ctx, 60*time.Second) - deleteErr := m.Provider.Delete(deleteCtx, vm.Name) + deleteErr := m.deleteProviderInstance(deleteCtx, instance) if deleteErr != nil && firstErr == nil { firstErr = deleteErr } @@ -1341,11 +1905,11 @@ func (m *Manager) validateRuntime(ctx context.Context, name string) error { return err } -func (m *Manager) validateRuntimeWithRetry(ctx context.Context, name, guestLogPath string) error { +func (m *Manager) verifyProviderRuntimeWithRetry(ctx context.Context, instance provider.Instance, guestLogPath string) error { const attempts = 2 var lastErr error for attempt := 1; attempt <= attempts; attempt++ { - err := m.validateRuntime(ctx, name) + err := m.verifyProviderRuntime(ctx, instance) if err == nil { return nil } @@ -1356,8 +1920,8 @@ func (m *Manager) validateRuntimeWithRetry(ctx context.Context, name, guestLogPa if attempt == attempts { break } - m.warnf("[%s] runtime validation attempt %d/%d failed: %v\n", name, attempt, attempts, err) - m.infof("[%s] retrying runtime validation in %s; guest log: %s\n", name, runtimeValidationRetryDelay, guestLogPath) + m.warnf("[%s] runtime validation attempt %d/%d failed: %v\n", instance.Name, attempt, attempts, err) + m.infof("[%s] retrying runtime validation in %s; guest log: %s\n", instance.Name, runtimeValidationRetryDelay, guestLogPath) select { case <-ctx.Done(): return ctx.Err() @@ -1377,7 +1941,7 @@ func (m *Manager) configureDockerRegistryMirrors(ctx context.Context, name strin if err != nil { return fmt.Errorf("read Docker daemon configuration script %s: %w", hostPath, err) } - if err := provider.CopyText(ctx, m.Provider, name, "/opt/epar/configure-docker-daemon.sh", "0755", guestText(content)); err != nil { + if err := m.copyTextGuest(ctx, name, "/opt/epar/configure-docker-daemon.sh", "0755", guestText(content), false); err != nil { return err } _, err = m.execGuest(ctx, name, []string{"sudo", "-E", "bash", "/opt/epar/configure-docker-daemon.sh"}, provider.ExecOptions{ @@ -1393,17 +1957,32 @@ func (m *Manager) execGuest(ctx context.Context, name string, cmd []string, opts if timeout <= 0 { timeout = 15 * time.Minute } - if opts.LogPath == "" { - opts.LogPath = m.instanceLogPath(name, ".guest.log") - } - transcript, err := m.transcript(opts.LogPath, name, transcriptComponent(opts.LogPath)) - if err != nil { - return provider.ExecResult{}, err + if !opts.SuppressTranscript { + if opts.LogPath == "" { + opts.LogPath = m.instanceLogPath(name, ".guest.log") + } + transcript, err := m.transcript(opts.LogPath, name, transcriptComponent(opts.LogPath)) + if err != nil { + return provider.ExecResult{}, err + } + opts.Stdout = transcript.Stdout + opts.Stderr = transcript.Stderr } - opts.Stdout = transcript.Stdout - opts.Stderr = transcript.Stderr cctx, cancel := context.WithTimeout(ctx, timeout) defer cancel() + if m.Lifecycle != nil { + instance, err := m.providerInstance(cctx, name) + if err != nil { + return provider.ExecResult{}, err + } + providerOpts := opts + providerOpts.LogPath = "" + providerOpts.Env = nil + return m.Lifecycle.Exec(cctx, instance, provider.EnvCommand(opts.Env, cmd), providerOpts) + } + if m.Provider == nil { + return provider.ExecResult{}, fmt.Errorf("provider lifecycle is required") + } return m.Provider.Exec(cctx, name, cmd, opts) } diff --git a/internal/pool/manager_test.go b/internal/pool/manager_test.go index 2e0c076..3d1ffb7 100644 --- a/internal/pool/manager_test.go +++ b/internal/pool/manager_test.go @@ -19,6 +19,7 @@ import ( gh "github.com/solutionforest/ephemeral-action-runner/internal/github" "github.com/solutionforest/ephemeral-action-runner/internal/hosttrust" "github.com/solutionforest/ephemeral-action-runner/internal/logging" + poolstate "github.com/solutionforest/ephemeral-action-runner/internal/pool/state" "github.com/solutionforest/ephemeral-action-runner/internal/provider" ) @@ -42,6 +43,106 @@ func TestRunnerAliveKeepsBusyGitHubRunnerWithoutServiceCheck(t *testing.T) { } } +func TestImageMaintenanceDrainRequiresConsecutiveIdleObservations(t *testing.T) { + host := &fakeProvider{} + github := &fakeGitHub{ + runner: gh.Runner{ID: 42, Name: "epar-test-1", Status: "online"}, + found: true, + } + manager := Manager{Provider: host, GitHub: github} + active := map[string]ProvisionedInstance{ + "epar-test-1": {Name: "epar-test-1", RunnerID: 42}, + } + confirmedIdle := make(map[string]int) + + remaining, err := manager.drainPoolForImageUpdate(context.Background(), active, confirmedIdle) + if err != nil { + t.Fatal(err) + } + if remaining != 1 || atomic.LoadInt32(&github.deleteCalls) != 0 || atomic.LoadInt32(&host.deleteCalls) != 0 { + t.Fatalf("first idle observation retired runner: remaining=%d remoteDeletes=%d localDeletes=%d", remaining, github.deleteCalls, host.deleteCalls) + } + + github.runner.Busy = true + remaining, err = manager.drainPoolForImageUpdate(context.Background(), active, confirmedIdle) + if err != nil { + t.Fatal(err) + } + if remaining != 1 || confirmedIdle["epar-test-1"] != 0 { + t.Fatalf("busy observation did not reset idle confirmation: remaining=%d checks=%d", remaining, confirmedIdle["epar-test-1"]) + } + + github.runner.Busy = false + if remaining, err = manager.drainPoolForImageUpdate(context.Background(), active, confirmedIdle); err != nil || remaining != 1 { + t.Fatalf("idle confirmation after busy = remaining %d, error %v", remaining, err) + } + if remaining, err = manager.drainPoolForImageUpdate(context.Background(), active, confirmedIdle); err != nil || remaining != 0 { + t.Fatalf("second consecutive idle confirmation = remaining %d, error %v", remaining, err) + } + if atomic.LoadInt32(&github.deleteCalls) != 1 || atomic.LoadInt32(&host.deleteCalls) != 1 { + t.Fatalf("confirmed idle retirement calls = remote %d local %d, want 1 each", github.deleteCalls, host.deleteCalls) + } +} + +func TestRunnerGroupPreflightEnforcesAndWarns(t *testing.T) { + violation := gh.RunnerGroupPolicyResult{Violations: []string{"runner group allows public repositories"}} + for _, test := range []struct { + name string + enforcement string + result gh.RunnerGroupPolicyResult + apiErr error + wantErr bool + }{ + {name: "enforce violation", enforcement: config.RunnerGroupEnforcementEnforce, result: violation, wantErr: true}, + {name: "warn violation", enforcement: config.RunnerGroupEnforcementWarn, result: violation}, + {name: "enforce API failure", enforcement: config.RunnerGroupEnforcementEnforce, apiErr: errors.New("forbidden"), wantErr: true}, + {name: "warn API failure", enforcement: config.RunnerGroupEnforcementWarn, apiErr: errors.New("forbidden")}, + {name: "enforce allowed", enforcement: config.RunnerGroupEnforcementEnforce, result: gh.RunnerGroupPolicyResult{Resolved: true, Group: gh.RunnerGroup{Name: "restricted"}}}, + } { + t.Run(test.name, func(t *testing.T) { + cfg := config.Default() + cfg.Runner.Group = "restricted" + cfg.Security.RunnerGroup.Enforcement = test.enforcement + github := &fakeGitHub{policyResult: test.result, policyErr: test.apiErr} + manager := Manager{Config: cfg, GitHub: github} + err := manager.PreflightRunnerGroup(context.Background()) + if (err != nil) != test.wantErr { + t.Fatalf("PreflightRunnerGroup() error = %v, wantErr=%t", err, test.wantErr) + } + if got := atomic.LoadInt32(&github.policyCalls); got != 1 { + t.Fatalf("policy calls = %d, want 1", got) + } + if got := atomic.LoadInt32(&github.registrationCalls); got != 0 { + t.Fatalf("registration-token calls = %d, want 0", got) + } + if got := atomic.LoadInt32(&github.deleteCalls); got != 0 { + t.Fatalf("runner delete calls = %d, want 0", got) + } + }) + } +} + +func TestRegistrationPreflightRejectsBeforeTokenRequest(t *testing.T) { + provider := &fakeProvider{ip: "127.0.0.1"} + github := &fakeGitHub{policyResult: gh.RunnerGroupPolicyResult{Violations: []string{"unsafe group"}}} + manager := newRegisteredTestManager(t, provider, github) + manager.Config.Security = config.Default().Security + manager.Config.Security.RunnerGroup.Enforcement = config.RunnerGroupEnforcementEnforce + _, err := manager.provisionOne(context.Background(), "epar-test-policy", true, false) + if err == nil || !strings.Contains(err.Error(), "unsafe group") { + t.Fatalf("provisionOne() error = %v, want policy rejection", err) + } + if got := atomic.LoadInt32(&github.policyCalls); got != 1 { + t.Fatalf("policy calls = %d, want 1", got) + } + if got := atomic.LoadInt32(&github.registrationCalls); got != 0 { + t.Fatalf("registration-token calls = %d, want 0", got) + } + if got := atomic.LoadInt32(&github.deleteCalls); got != 0 { + t.Fatalf("remote runner delete calls = %d, want 0", got) + } +} + func TestRetiredInstanceTranscriptsBecomeRetentionEligibleWhileLiveInstanceStaysProtected(t *testing.T) { root := t.TempDir() runtime, err := logging.NewRuntime(logging.Options{Directory: root, TranscriptSinks: logging.SinkFile}) @@ -52,14 +153,14 @@ func TestRetiredInstanceTranscriptsBecomeRetentionEligibleWhileLiveInstanceStays manager := Manager{ Config: config.Config{ Logging: config.LoggingConfig{Directory: root}, - Provider: config.ProviderConfig{Type: "docker-dind"}, + Provider: config.ProviderConfig{Type: "docker-container"}, }, ProjectRoot: root, Logging: runtime, } retired := ProvisionedInstance{ Name: "retired-runner", - LogPath: filepath.Join(root, "instances", "retired-runner.docker-dind.log"), + LogPath: filepath.Join(root, "instances", "retired-runner.docker-container.log"), GuestLogPath: filepath.Join(root, "instances", "retired-runner.guest.log"), } livePath := filepath.Join(root, "instances", "live-runner.guest.log") @@ -115,13 +216,13 @@ func TestRetirementSuccessIsNotReversedByTranscriptCloseFailure(t *testing.T) { manager := Manager{ Config: config.Config{ Logging: config.LoggingConfig{Directory: root}, - Provider: config.ProviderConfig{Type: "docker-dind"}, + Provider: config.ProviderConfig{Type: "docker-container"}, }, Provider: provider, ProjectRoot: root, Logging: runtime, } - vm := ProvisionedInstance{Name: "retired-runner", LogPath: filepath.Join(root, "instances", "retired-runner.docker-dind.log")} + vm := ProvisionedInstance{Name: "retired-runner", LogPath: filepath.Join(root, "instances", "retired-runner.docker-container.log")} transcript, err := manager.transcript(vm.LogPath, vm.Name, "provider") if err != nil { t.Fatal(err) @@ -158,7 +259,12 @@ func TestRetirementSuccessIsNotReversedByTranscriptCloseFailure(t *testing.T) { } func TestRunnerAliveRetiresIdleRunnerWhenServiceIsInactive(t *testing.T) { - provider := &fakeProvider{execErr: errors.New("inactive")} + provider := &fakeProvider{execFunc: func(_ context.Context, _ string, command []string, _ provider.ExecOptions) (provider.ExecResult, error) { + if strings.Contains(strings.Join(command, " "), runnerProcessRunningSentinel) { + return provider.ExecResult{Stdout: runnerProcessStoppedSentinel + "\n"}, nil + } + return provider.ExecResult{}, nil + }} github := &fakeGitHub{ runner: gh.Runner{Name: "epar-test-1", Status: "online", Busy: false}, found: true, @@ -172,12 +278,66 @@ func TestRunnerAliveRetiresIdleRunnerWhenServiceIsInactive(t *testing.T) { if alive { t.Fatal("runnerAlive() alive = true, want false") } - if reason != "actions runner process is no longer active" { + if reason != runnerProcessInactiveReason { t.Fatalf("reason = %q", reason) } if got := atomic.LoadInt32(&provider.execCalls); got != 1 { t.Fatalf("service check ran %d time(s), want 1", got) } + provider.mu.Lock() + defer provider.mu.Unlock() + if len(provider.execOptions) != 1 || !provider.execOptions[0].SuppressTranscript { + t.Fatal("machine-readable health probe was not excluded from the guest transcript") + } +} + +func TestRunnerAlivePreservesRunnerWhenProcessProbeIsUnavailable(t *testing.T) { + provider := &fakeProvider{execErr: context.DeadlineExceeded} + github := &fakeGitHub{ + runner: gh.Runner{Name: "epar-test-1", Status: "online", Busy: false}, + found: true, + } + manager := Manager{Provider: provider, GitHub: github} + + alive, reason, err := manager.runnerAlive(context.Background(), ProvisionedInstance{Name: "epar-test-1"}) + if err == nil { + t.Fatal("runnerAlive() error = nil, want unavailable process-probe error") + } + if !alive { + t.Fatalf("runnerAlive() alive = false, reason = %q; an unavailable probe must preserve the runner", reason) + } + if got := atomic.LoadInt32(&provider.deleteCalls); got != 0 { + t.Fatalf("provider delete calls = %d, want 0", got) + } +} + +func TestRunnerProcessProbeRejectsUnsupportedOutput(t *testing.T) { + provider := &fakeProvider{execFunc: func(context.Context, string, []string, provider.ExecOptions) (provider.ExecResult, error) { + return provider.ExecResult{Stdout: "unexpected\n"}, nil + }} + manager := Manager{Provider: provider} + if _, err := manager.probeRunnerProcess(context.Background(), "epar-test-1"); err == nil || !strings.Contains(err.Error(), "unsupported response") { + t.Fatalf("probeRunnerProcess() error = %v, want unsupported-response error", err) + } +} + +func TestRunnerLivenessRequiresConsecutiveConfirmedInactiveProbes(t *testing.T) { + counts := make(map[string]int) + if count, retire := recordRunnerLiveness(counts, "runner-1", false, runnerProcessInactiveReason, nil); count != 1 || retire { + t.Fatalf("first confirmed inactive probe = count %d retire %t, want 1 false", count, retire) + } + if count, retire := recordRunnerLiveness(counts, "runner-1", true, "", context.DeadlineExceeded); count != 0 || retire { + t.Fatalf("unavailable probe = count %d retire %t, want 0 false", count, retire) + } + if count, retire := recordRunnerLiveness(counts, "runner-1", false, runnerProcessInactiveReason, nil); count != 1 || retire { + t.Fatalf("first probe after uncertainty = count %d retire %t, want 1 false", count, retire) + } + if count, retire := recordRunnerLiveness(counts, "runner-1", false, runnerProcessInactiveReason, nil); count != 2 || !retire { + t.Fatalf("second consecutive confirmed inactive probe = count %d retire %t, want 2 true", count, retire) + } + if count, retire := recordRunnerLiveness(counts, "runner-2", false, "GitHub runner record is gone", nil); count != 0 || !retire { + t.Fatalf("authoritative remote absence = count %d retire %t, want 0 true", count, retire) + } } func TestRunnerAliveFallsBackToServiceCheckWhenGitHubLivenessHasServerError(t *testing.T) { @@ -283,6 +443,7 @@ func TestRunPoolReplacesCompletedRunnerAfterBusyProvisioning(t *testing.T) { Pool: config.PoolConfig{Instances: 1, NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}, Runner: config.RunnerConfig{Labels: []string{"self-hosted"}, Ephemeral: true}, + Security: config.Default().Security, }, Provider: provider, GitHub: github, @@ -324,7 +485,7 @@ func TestRunPoolAddsCurrentTrustCapacityWhileOldGenerationDrains(t *testing.T) { } manager := Manager{ Config: config.Config{ - Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-dind"}, + Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-container"}, Pool: config.PoolConfig{Instances: 1, NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}, Runner: config.RunnerConfig{Labels: []string{"self-hosted"}, Ephemeral: true}, @@ -336,7 +497,6 @@ func TestRunPoolAddsCurrentTrustCapacityWhileOldGenerationDrains(t *testing.T) { GitHub: github, ProjectRoot: t.TempDir(), } - var resolveCalls int32 snapshot := func(generation string) hosttrust.Snapshot { return hosttrust.Snapshot{ Generation: generation, HostOS: "linux", Scopes: []string{"system"}, @@ -345,7 +505,7 @@ func TestRunPoolAddsCurrentTrustCapacityWhileOldGenerationDrains(t *testing.T) { } } manager.hostTrustResolver = func(context.Context) (hosttrust.Snapshot, error) { - if atomic.AddInt32(&resolveCalls, 1) <= 2 { + if atomic.LoadInt32(&github.waitOnlineCalls)+atomic.LoadInt32(&github.waitOnlineIdleCalls) == 0 { return snapshot("g1"), nil } return snapshot("g2"), nil @@ -393,7 +553,7 @@ func TestVerifyUsesIdleReadiness(t *testing.T) { } manager := Manager{ Config: config.Config{ - Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-dind"}, + Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-container"}, Pool: config.PoolConfig{Instances: 1, NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}, Runner: config.RunnerConfig{Labels: []string{"self-hosted"}, Ephemeral: true}, @@ -453,7 +613,7 @@ func TestProvisionOneRetriesTransientRuntimeValidationFailure(t *testing.T) { } manager := Manager{ Config: config.Config{ - Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-dind"}, + Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-container"}, Pool: config.PoolConfig{Instances: 1, NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}, Timeouts: config.TimeoutConfig{CommandSeconds: 5}, @@ -481,7 +641,7 @@ func TestVerifyCleanupUsesFreshContextAfterCancellation(t *testing.T) { } manager := Manager{ Config: config.Config{ - Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-dind"}, + Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-container"}, Pool: config.PoolConfig{Instances: 1, NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}, Timeouts: config.TimeoutConfig{CommandSeconds: 5}, @@ -512,7 +672,7 @@ func TestRunPoolCleanupUsesFreshContextAfterCancellation(t *testing.T) { } manager := Manager{ Config: config.Config{ - Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-dind"}, + Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-container"}, Pool: config.PoolConfig{Instances: 1, NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}, Timeouts: config.TimeoutConfig{CommandSeconds: 5}, @@ -547,7 +707,7 @@ func TestProvisionOnePassesRunnerRegistrationControlsWithoutPrivateKey(t *testin GitHub: config.GitHubConfig{PrivateKeyPath: "/secret/app.pem"}, Provider: config.ProviderConfig{ SourceImage: "image", - Type: "docker-dind", + Type: "docker-container", }, Pool: config.PoolConfig{ Instances: 1, @@ -713,6 +873,91 @@ func TestProvisionOneCapturesReadinessTimeoutAndPreservesCause(t *testing.T) { } } +func TestRunPoolCancellationAfterListenerStartCleansExactRunner(t *testing.T) { + state, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + ctx, cancel := context.WithCancel(context.Background()) + var deleted atomic.Bool + var lookupCalls atomic.Int32 + fake := &fakeProvider{ip: "127.0.0.1"} + fake.execFunc = func(_ context.Context, _ string, command []string, _ provider.ExecOptions) (provider.ExecResult, error) { + if strings.Contains(strings.Join(command, " "), "run-runner.sh") { + cancel() + } + return provider.ExecResult{}, nil + } + github := &fakeGitHub{ + runnerByNameFunc: func(_ context.Context, name string) (gh.Runner, bool, error) { + if lookupCalls.Add(1) == 1 || deleted.Load() { + return gh.Runner{}, false, nil + } + return gh.Runner{Name: name, ID: 718}, true, nil + }, + deleteFunc: func(context.Context, int64) error { + deleted.Store(true) + return nil + }, + waitFunc: func(ctx context.Context, _ string, _ time.Duration) (gh.Runner, error) { + <-ctx.Done() + return gh.Runner{}, ctx.Err() + }, + } + manager := newRegisteredTestManager(t, fake, github) + manager.Lifecycle = provider.AdaptLegacy(fake) + manager.LifecycleState = state + + if err := manager.RunPool(ctx, RunOptions{Instances: 1, Register: true, ReplaceCompleted: true, PoolLockHeld: true, HostTrustLockHeld: true}); err != nil { + t.Fatalf("RunPool() cancellation error = %v, want exact cleanup", err) + } + records, err := state.List(context.Background()) + if err != nil { + t.Fatal(err) + } + if len(records) != 1 { + t.Fatalf("lifecycle records = %#v, want one tombstoned candidate", records) + } + record := records[0] + if record.Phase != poolstate.PhaseTombstoned || record.GitHub.RunnerID != 718 { + t.Fatalf("lifecycle after cancellation = phase %q runner id %d, want tombstoned with id 718", record.Phase, record.GitHub.RunnerID) + } + if got := atomic.LoadInt32(&github.deleteCalls); got != 1 { + t.Fatalf("remote delete calls = %d, want 1", got) + } + github.mu.Lock() + defer github.mu.Unlock() + if len(github.deletedIDs) != 1 || github.deletedIDs[0] != 718 { + t.Fatalf("deleted remote IDs = %v, want [718]", github.deletedIDs) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 1 { + t.Fatalf("local delete calls = %d, want 1", got) + } +} + +func TestCommonLifecycleOwnsGuestTranscriptPath(t *testing.T) { + fake := &fakeProvider{instances: []provider.Instance{{Name: "epar-test-1", State: "running"}}} + manager := newRegisteredTestManager(t, fake, nil) + manager.Lifecycle = provider.AdaptLegacy(fake) + + if _, err := manager.execGuest(context.Background(), "epar-test-1", []string{"true"}, provider.ExecOptions{Env: map[string]string{"EPAR_TEST": "value"}}); err != nil { + t.Fatal(err) + } + if got := fake.logPathFor("true"); got != "" { + t.Fatalf("provider received common host transcript path %q", got) + } + fake.mu.Lock() + command := fake.commands[len(fake.commands)-1] + options := fake.execOptions[len(fake.execOptions)-1] + fake.mu.Unlock() + if len(options.Env) != 0 { + t.Fatalf("provider received ambient host environment: %#v", options.Env) + } + if !strings.Contains(command, "env EPAR_TEST=value true") { + t.Fatalf("provider command = %q, want explicit guest environment", command) + } +} + func TestProvisionOneReadinessSucceedsWhileRunnerProcessStaysHealthy(t *testing.T) { oldInterval := runnerReadinessHealthCheckInterval runnerReadinessHealthCheckInterval = time.Millisecond @@ -799,6 +1044,59 @@ func TestProvisioningFailureRollbackBoundary(t *testing.T) { } } +func TestProvisioningRecordsPartialCreateIdentityBeforeExactRollback(t *testing.T) { + const name = "epar-test-partial-create" + partial := provider.Instance{ + Name: name, + ProviderID: "fake:partial-create-id", + Source: "image", + State: "running", + ReceiptVersion: "v1", + Receipt: json.RawMessage(`{"providerId":"fake:partial-create-id","source":"image"}`), + } + fake := &fakeProvider{} + baseLifecycle := provider.AdaptLegacy(fake) + lifecycle := &partialCreateLifecycle{ + Lifecycle: baseLifecycle, + create: func() (provider.Instance, error) { + fake.mu.Lock() + fake.instances = append(fake.instances, partial) + fake.mu.Unlock() + return partial, errors.New("post-create verification failed") + }, + } + manager := newRegisteredTestManager(t, fake, nil) + manager.Lifecycle = lifecycle + store, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + manager.LifecycleState = store + + if _, err := manager.provisionOne(context.Background(), name, false, false); err == nil || !strings.Contains(err.Error(), "post-create verification failed") { + t.Fatalf("provisionOne() error = %v", err) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 1 { + t.Fatalf("exact provider delete calls = %d, want 1", got) + } + record, err := store.Read(context.Background(), name) + if err != nil { + t.Fatal(err) + } + if record.ProviderID != partial.ProviderID || record.Phase != poolstate.PhaseTombstoned { + t.Fatalf("lifecycle record = %#v, want exact provider identity tombstoned", record) + } +} + +type partialCreateLifecycle struct { + provider.Lifecycle + create func() (provider.Instance, error) +} + +func (l *partialCreateLifecycle) Create(context.Context, provider.CreateRequest) (provider.Instance, error) { + return l.create() +} + func TestConfigureFailureDeletesExactLocalAndRemoteCandidate(t *testing.T) { p := &fakeProvider{ip: "127.0.0.1"} p.execFunc = func(_ context.Context, _ string, command []string, _ provider.ExecOptions) (provider.ExecResult, error) { @@ -1184,7 +1482,7 @@ func newRegisteredTestManager(t *testing.T, provider provider.Provider, github G t.Helper() return Manager{ Config: config.Config{ - Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-dind"}, + Provider: config.ProviderConfig{SourceImage: "image", Type: "docker-container"}, Pool: config.PoolConfig{Instances: 1, NamePrefix: "epar-test"}, Logging: config.LoggingConfig{Directory: t.TempDir()}, Runner: config.RunnerConfig{Labels: []string{"self-hosted"}, Ephemeral: true}, @@ -1226,7 +1524,7 @@ type fakeProvider struct { func (p *fakeProvider) Clone(_ context.Context, source, name string) error { atomic.AddInt32(&p.cloneCalls, 1) p.mu.Lock() - p.instances = append(p.instances, provider.Instance{Name: name, Source: source, State: "running"}) + p.instances = append(p.instances, provider.Instance{Name: name, ProviderID: "fake:" + name, Source: source, State: "running"}) if len(p.instances) > int(atomic.LoadInt32(&p.maxInventory)) { atomic.StoreInt32(&p.maxInventory, int32(len(p.instances))) } @@ -1260,6 +1558,9 @@ func (p *fakeProvider) Exec(ctx context.Context, name string, command []string, if int(call) <= len(p.execErrs) { return provider.ExecResult{}, p.execErrs[call-1] } + if p.execErr == nil && strings.Contains(commandText, runnerProcessRunningSentinel) { + return provider.ExecResult{Stdout: runnerProcessRunningSentinel + "\n"}, nil + } return provider.ExecResult{}, p.execErr } @@ -1326,17 +1627,25 @@ func (p *fakeProvider) List(ctx context.Context) ([]provider.Instance, error) { } p.mu.Lock() defer p.mu.Unlock() - return append([]provider.Instance(nil), p.instances...), p.listErr + result := append([]provider.Instance(nil), p.instances...) + for i := range result { + if result[i].ProviderID == "" { + result[i].ProviderID = "fake:" + result[i].Name + } + } + return result, p.listErr } type fakeGitHub struct { runner gh.Runner + runnerByNameFunc func(context.Context, string) (gh.Runner, bool, error) waitRunner gh.Runner waitErr error waitFunc func(context.Context, string, time.Duration) (gh.Runner, error) found bool runnerErr error deleteErr error + deleteFunc func(context.Context, int64) error waitOnlineCalls int32 waitOnlineIdleCalls int32 listRunners []gh.Runner @@ -1344,6 +1653,10 @@ type fakeGitHub struct { listFunc func(context.Context) ([]gh.Runner, error) registrationErr error registrationToken string + policyResult gh.RunnerGroupPolicyResult + policyErr error + policyCalls int32 + registrationCalls int32 deleteCalls int32 deletedIDs []int64 mu sync.Mutex @@ -1355,7 +1668,13 @@ func (g *fakeGitHub) OrganizationURL() string { return "https://github.test/example" } +func (g *fakeGitHub) EvaluateRunnerGroupPolicy(context.Context, string, config.RunnerGroupSecurityConfig) (gh.RunnerGroupPolicyResult, error) { + atomic.AddInt32(&g.policyCalls, 1) + return g.policyResult, g.policyErr +} + func (g *fakeGitHub) RegistrationToken(context.Context) (gh.RegistrationToken, error) { + atomic.AddInt32(&g.registrationCalls, 1) token := g.registrationToken if token == "" { token = "token" @@ -1371,8 +1690,11 @@ func (g *fakeGitHub) ListRunners(ctx context.Context) ([]gh.Runner, error) { return append([]gh.Runner(nil), g.listRunners...), g.listErr } -func (g *fakeGitHub) RunnerByName(context.Context, string) (gh.Runner, bool, error) { +func (g *fakeGitHub) RunnerByName(ctx context.Context, name string) (gh.Runner, bool, error) { atomic.AddInt32(&g.runnerByNameCalls, 1) + if g.runnerByNameFunc != nil { + return g.runnerByNameFunc(ctx, name) + } return g.runner, g.found, g.runnerErr } @@ -1399,11 +1721,14 @@ func (g *fakeGitHub) waitReady(ctx context.Context, name string, timeout time.Du return g.runner, nil } -func (g *fakeGitHub) DeleteRunnerIfExists(_ context.Context, id int64) error { +func (g *fakeGitHub) DeleteRunnerIfExists(ctx context.Context, id int64) error { atomic.AddInt32(&g.deleteCalls, 1) g.mu.Lock() g.deletedIDs = append(g.deletedIDs, id) g.mu.Unlock() + if g.deleteFunc != nil { + return g.deleteFunc(ctx, id) + } return g.deleteErr } diff --git a/internal/pool/provider_lifecycle.go b/internal/pool/provider_lifecycle.go new file mode 100644 index 0000000..31c701e --- /dev/null +++ b/internal/pool/provider_lifecycle.go @@ -0,0 +1,178 @@ +package pool + +import ( + "context" + "errors" + "fmt" + "path/filepath" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + poolstate "github.com/solutionforest/ephemeral-action-runner/internal/pool/state" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +func (m *Manager) copyTextGuest(ctx context.Context, name, path, mode, content string, atomic bool) error { + staging := "/tmp/epar-copy" + script := fmt.Sprintf("cat > %s && if command -v sudo >/dev/null 2>&1; then sudo install -m %s %s %s; else install -m %s %s %s; fi && rm -f %s", shellQuote(staging), shellQuote(mode), shellQuote(staging), shellQuote(path), shellQuote(mode), shellQuote(staging), shellQuote(path), shellQuote(staging)) + if atomic { + temporary := path + ".tmp" + script = fmt.Sprintf("cat > %s && if command -v sudo >/dev/null 2>&1; then sudo install -m %s %s %s && sudo mv -f %s %s; else install -m %s %s %s && mv -f %s %s; fi && rm -f %s", shellQuote(staging), shellQuote(mode), shellQuote(staging), shellQuote(temporary), shellQuote(temporary), shellQuote(path), shellQuote(mode), shellQuote(staging), shellQuote(temporary), shellQuote(temporary), shellQuote(path), shellQuote(staging)) + } + _, err := m.execGuest(ctx, name, provider.ShellCommand(script), provider.ExecOptions{Stdin: content}) + return err +} + +func (m *Manager) createProviderInstance(ctx context.Context, name string) (provider.Instance, error) { + lifecycle := m.providerLifecycle() + if lifecycle == nil { + return provider.Instance{}, fmt.Errorf("provider lifecycle is required") + } + return lifecycle.Create(ctx, provider.CreateRequest{ + Name: name, + Source: m.Config.Provider.SourceImage, + StagingPath: filepath.Join(config.ProjectPath(m.ProjectRoot, m.Config.DockerSandboxes.StagingRoot), name), + CPUs: m.Config.DockerSandboxes.CPUs, + Memory: m.Config.DockerSandboxes.Memory, + RootDisk: m.Config.DockerSandboxes.RootDisk, + DockerDisk: m.Config.DockerSandboxes.DockerDisk, + }) +} + +func (m *Manager) startProviderInstance(ctx context.Context, instance provider.Instance, opts provider.StartOptions) (*provider.RunningProcess, error) { + lifecycle := m.providerLifecycle() + if lifecycle == nil { + return nil, fmt.Errorf("provider lifecycle is required") + } + return lifecycle.Start(ctx, instance, opts) +} + +func (m *Manager) providerAddress(ctx context.Context, instance provider.Instance, waitSeconds int) (string, bool, error) { + lifecycle := m.providerLifecycle() + if lifecycle == nil { + return "", false, fmt.Errorf("provider lifecycle is required") + } + return lifecycle.Address(ctx, instance, waitSeconds) +} + +func (m *Manager) verifyProviderRuntime(ctx context.Context, instance provider.Instance) error { + if m.Lifecycle == nil { + if m.Provider == nil { + return fmt.Errorf("provider lifecycle is required") + } + return m.validateRuntime(ctx, instance.Name) + } + lifecycle := m.providerLifecycle() + if lifecycle == nil { + return fmt.Errorf("provider lifecycle is required") + } + info, err := lifecycle.VerifyRuntime(ctx, instance) + if err != nil { + return err + } + if !info.Ready { + return fmt.Errorf("provider runtime for %q is not ready", instance.Name) + } + return nil +} + +func (m *Manager) verifyProviderAdmission(ctx context.Context, instance provider.Instance) error { + if verifier, ok := m.Lifecycle.(provider.AdmissionVerifier); ok { + if err := verifier.VerifyAdmission(ctx); err != nil { + return fmt.Errorf("provider-wide admission failed: %w", err) + } + } + if verifier, ok := m.Lifecycle.(provider.InstanceAdmissionVerifier); ok { + if err := verifier.VerifyInstanceAdmission(ctx, instance); err != nil { + return fmt.Errorf("instance admission failed: %w", err) + } + } + return nil +} + +func (m *Manager) applyProviderNetworkPolicy(ctx context.Context, instance provider.Instance) error { + if m.PolicyManager == nil { + return nil + } + var rules []provider.NetworkPolicyRule + if m.Config.DockerSandboxes.NetworkBaseline == config.DockerSandboxesNetworkBaselineOpen { + rules = append(rules, + provider.NetworkPolicyRule{Name: "epar-public-egress", Decision: provider.NetworkPolicyAllow, Resources: []string{"**"}}, + provider.NetworkPolicyRule{Name: "epar-host-alias-guardrails", Decision: provider.NetworkPolicyDeny, Resources: config.DockerSandboxesOpenDefaultDenyResources()}, + ) + } + if len(m.Config.DockerSandboxes.AdditionalAllow) != 0 { + rules = append(rules, provider.NetworkPolicyRule{Name: "epar-additional-allow", Decision: provider.NetworkPolicyAllow, Resources: append([]string(nil), m.Config.DockerSandboxes.AdditionalAllow...)}) + } + if len(m.Config.DockerSandboxes.AdditionalDeny) != 0 { + rules = append(rules, provider.NetworkPolicyRule{Name: "epar-additional-deny", Decision: provider.NetworkPolicyDeny, Resources: append([]string(nil), m.Config.DockerSandboxes.AdditionalDeny...)}) + } + if err := m.PolicyManager.ApplyNetworkPolicy(ctx, instance, rules); err != nil { + return err + } + _, err := m.PolicyManager.ReadNetworkPolicy(ctx, instance) + return err +} + +func (m *Manager) inventoryProvider(ctx context.Context) ([]provider.InventoryItem, error) { + lifecycle := m.providerLifecycle() + if lifecycle == nil { + return nil, fmt.Errorf("provider lifecycle is required") + } + return lifecycle.Inventory(ctx) +} + +func (m *Manager) providerInstance(ctx context.Context, name string) (provider.Instance, error) { + if m.LifecycleState != nil { + record, err := m.LifecycleState.Read(ctx, name) + if err == nil { + if record.ProviderType != m.Config.Provider.Type || record.ProviderID == "" { + return provider.Instance{}, fmt.Errorf("lifecycle record for %q has no exact provider identity", name) + } + return provider.Instance{ + Name: name, + ProviderID: record.ProviderID, + ReceiptVersion: record.Receipt.Version, + Receipt: append([]byte(nil), record.Receipt.Payload...), + }, nil + } + if !errors.Is(err, poolstate.ErrNotFound) { + return provider.Instance{}, err + } + } + items, err := m.inventoryProvider(ctx) + if err != nil { + return provider.Instance{}, err + } + for _, item := range items { + if item.Instance.Name == name { + return item.Instance, nil + } + } + return provider.Instance{}, fmt.Errorf("provider instance %q is missing", name) +} + +func (m *Manager) stopProviderInstance(ctx context.Context, instance provider.Instance) error { + lifecycle := m.providerLifecycle() + if lifecycle == nil { + return fmt.Errorf("provider lifecycle is required") + } + return lifecycle.Stop(ctx, instance) +} + +func (m *Manager) deleteProviderInstance(ctx context.Context, instance provider.Instance) error { + lifecycle := m.providerLifecycle() + if lifecycle == nil { + return fmt.Errorf("provider lifecycle is required") + } + return lifecycle.Delete(ctx, instance) +} + +func (m *Manager) providerLifecycle() provider.Lifecycle { + if m.Lifecycle != nil { + return m.Lifecycle + } + if m.Provider == nil { + return nil + } + return provider.AdaptLegacy(m.Provider, m.DryRun) +} diff --git a/internal/pool/runner_script_test.go b/internal/pool/runner_script_test.go index cef619a..2fb79eb 100644 --- a/internal/pool/runner_script_test.go +++ b/internal/pool/runner_script_test.go @@ -431,7 +431,10 @@ func TestHostTrustGenerationHookAcceptsCurrentLease(t *testing.T) { t.Skip("the guest hook requires the Linux image's python3 runtime") } marker := `{"schemaVersion":1,"generation":"g1","hostOS":"windows","mode":"overlay","scopes":["system","user"]}` - lease := fmt.Sprintf(`{"schemaVersion":1,"generation":"g1","hostOS":"windows","mode":"overlay","scopes":["system","user"],"expiresAt":%q}`, time.Now().Add(time.Minute).UTC().Format(time.RFC3339Nano)) + // The host test invokes macOS's system Python, whose fromisoformat support is + // older than the Python shipped in the Linux runner image. Whole-second + // RFC3339 still exercises the production timezone and expiry checks. + lease := fmt.Sprintf(`{"schemaVersion":1,"generation":"g1","hostOS":"windows","mode":"overlay","scopes":["system","user"],"expiresAt":%q}`, time.Now().Add(time.Minute).UTC().Format(time.RFC3339)) output, err := runHostTrustGenerationHook(t, marker, lease) if err != nil { t.Fatalf("current host-trust lease rejected: %v\n%s", err, output) @@ -441,6 +444,58 @@ func TestHostTrustGenerationHookAcceptsCurrentLease(t *testing.T) { } } +func TestHostTrustGenerationHookProductionPathsCannotBeRedirectedByWorkflowEnvironment(t *testing.T) { + paths := []string{ + filepath.Join("..", "..", "scripts", "guest", "ubuntu", "check-host-trust-generation.sh"), + filepath.Join("..", "..", "templates", "docker-sandboxes", "guest", "check-host-trust-generation.sh"), + } + for _, path := range paths { + content, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + for _, forbidden := range []string{"EPAR_HOST_TRUST_MARKER", "EPAR_HOST_TRUST_LEASE"} { + if strings.Contains(string(content), forbidden) { + t.Fatalf("%s permits workflow-controlled path override %q", path, forbidden) + } + } + for _, required := range []string{ + "marker=\"/opt/epar/host-trust-generation.json\"", + "lease=\"/run/epar/host-trust-lease.json\"", + "/usr/bin/env -i PATH=/usr/bin:/bin LANG=C.UTF-8 /usr/bin/python3 -I -S -", + } { + if !strings.Contains(string(content), required) { + t.Fatalf("%s omitted fixed production invariant %q", path, required) + } + } + } +} + +func TestDockerSandboxesRunnerRequiresExplicitDisabledOrOverlayTrustPolicy(t *testing.T) { + path := filepath.Join("..", "..", "templates", "docker-sandboxes", "guest", "run-runner.sh") + content, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + text := string(content) + for _, required := range []string{ + "[[ -s /opt/epar/host-trust-generation.json ]]", + "ACTIONS_RUNNER_HOOK_JOB_STARTED=/opt/epar/check-host-trust-generation.sh", + "PATH=/opt/epar/hook-bin:", + `if mode == "disabled":`, + `elif mode == "overlay":`, + `raise SystemExit(f"EPAR runner trust policy: unknown mode {mode!r}")`, + `if [[ "${trust_mode}" == "overlay" ]]; then`, + } { + if !strings.Contains(text, required) { + t.Fatalf("Docker Sandboxes runner omitted trust-policy invariant %q", required) + } + } + if strings.Contains(text, `if [[ -s /opt/epar/host-trust-generation.json ]]`) { + t.Fatal("Docker Sandboxes runner accepts a missing policy marker") + } +} + func TestHostTrustGenerationHookRejectsMismatchAndExpiry(t *testing.T) { if runtime.GOOS == "windows" { t.Skip("the guest hook requires the Linux image's python3 runtime") @@ -449,8 +504,8 @@ func TestHostTrustGenerationHookRejectsMismatchAndExpiry(t *testing.T) { for _, tc := range []struct { name, lease, want string }{ - {name: "generation mismatch", lease: fmt.Sprintf(`{"generation":"g2","hostOS":"linux","mode":"overlay","scopes":["system"],"expiresAt":%q}`, time.Now().Add(time.Minute).UTC().Format(time.RFC3339Nano)), want: "generation mismatch"}, - {name: "expired", lease: fmt.Sprintf(`{"generation":"g1","hostOS":"linux","mode":"overlay","scopes":["system"],"expiresAt":%q}`, time.Now().Add(-time.Minute).UTC().Format(time.RFC3339Nano)), want: "lease expired"}, + {name: "generation mismatch", lease: fmt.Sprintf(`{"generation":"g2","hostOS":"linux","mode":"overlay","scopes":["system"],"expiresAt":%q}`, time.Now().Add(time.Minute).UTC().Format(time.RFC3339)), want: "generation mismatch"}, + {name: "expired", lease: fmt.Sprintf(`{"generation":"g1","hostOS":"linux","mode":"overlay","scopes":["system"],"expiresAt":%q}`, time.Now().Add(-time.Minute).UTC().Format(time.RFC3339)), want: "lease expired"}, } { t.Run(tc.name, func(t *testing.T) { output, err := runHostTrustGenerationHook(t, marker, tc.lease) @@ -479,8 +534,8 @@ func runHostTrustGenerationHook(t *testing.T, marker, lease string) (string, err if err := os.WriteFile(leasePath, []byte(lease), 0644); err != nil { t.Fatal(err) } - cmd := exec.Command(gitBashForRunnerScriptTest(t), bashPath(hookPath)) - cmd.Env = append(os.Environ(), "EPAR_HOST_TRUST_MARKER="+bashPath(markerPath), "EPAR_HOST_TRUST_LEASE="+bashPath(leasePath)) + cmd := exec.Command(gitBashForRunnerScriptTest(t), bashPath(hookPath), bashPath(markerPath), bashPath(leasePath)) + cmd.Env = append(os.Environ(), "EPAR_HOST_TRUST_MARKER=/workflow/forged-marker.json", "EPAR_HOST_TRUST_LEASE=/workflow/forged-lease.json") output, runErr := cmd.CombinedOutput() return string(output), runErr } diff --git a/internal/pool/startup_timing.go b/internal/pool/startup_timing.go index adb065d..5013af7 100644 --- a/internal/pool/startup_timing.go +++ b/internal/pool/startup_timing.go @@ -2,6 +2,7 @@ package pool import ( "encoding/json" + "fmt" "log/slog" "os" "path/filepath" @@ -39,7 +40,7 @@ type startupTiming struct { logger *slog.Logger } -// StartStartupTiming records the initial Docker-DinD or WSL start path. +// StartStartupTiming records the initial Docker Container or WSL start path. func (m *Manager) StartStartupTiming() (string, error) { provider := m.Config.Provider.Type if !supportsStartupTiming(provider) { @@ -151,14 +152,28 @@ func (t *startupTiming) finish(err error) { } else { t.stages = append(t.stages, startupTimingStage{name: "total_startup", elapsed: elapsed}) t.eventLocked("total_startup", "completed", elapsed, nil) - for _, stage := range t.stages { - t.logger.Info("startup timing", "provider", t.provider, "operation", "startup", "stage", stage.name, "duration", stage.elapsed.Round(time.Millisecond), "logPath", t.path) - } + t.logger.Info(startupTimingSummary(t.provider, t.stages, t.path), "provider", t.provider, "operation", "startup", "stages", slog.GroupValue(startupTimingStageAttributes(t.stages)...), "logPath", t.path) } t.closed = true _ = t.file.Close() } +func startupTimingSummary(provider string, stages []startupTimingStage, path string) string { + parts := make([]string, 0, len(stages)) + for _, stage := range stages { + parts = append(parts, fmt.Sprintf("%s=%s", stage.name, stage.elapsed.Round(time.Millisecond))) + } + return fmt.Sprintf("%s startup timing: %s; log: %s", startupTimingLabel(provider), strings.Join(parts, ", "), path) +} + +func startupTimingStageAttributes(stages []startupTimingStage) []slog.Attr { + attributes := make([]slog.Attr, 0, len(stages)) + for _, stage := range stages { + attributes = append(attributes, slog.Duration(stage.name, stage.elapsed.Round(time.Millisecond))) + } + return attributes +} + func (t *startupTiming) eventLocked(stage, outcome string, elapsed time.Duration, err error) { event := startupTimingEvent{ Timestamp: time.Now().UTC().Format(time.RFC3339Nano), @@ -174,13 +189,13 @@ func (t *startupTiming) eventLocked(stage, outcome string, elapsed time.Duration } func supportsStartupTiming(provider string) bool { - return provider == "docker-dind" || provider == "wsl" + return provider == "docker-container" || provider == "wsl" } func startupTimingLabel(provider string) string { switch provider { - case "docker-dind": - return "DinD" + case "docker-container": + return "Docker Container" case "wsl": return "WSL" default: diff --git a/internal/pool/startup_timing_test.go b/internal/pool/startup_timing_test.go index 10f98bb..4ecd2c3 100644 --- a/internal/pool/startup_timing_test.go +++ b/internal/pool/startup_timing_test.go @@ -1,9 +1,16 @@ package pool import ( + "bytes" + "encoding/json" "errors" + "os" + "path/filepath" "strings" "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/logging" ) func TestSanitizeTimingErrorRedactsSecretAssignments(t *testing.T) { @@ -16,3 +23,68 @@ func TestSanitizeTimingErrorRedactsSecretAssignments(t *testing.T) { t.Fatalf("sanitizeTimingError did not retain sanitized keys: %q", got) } } + +func TestStartupTimingWritesOneReadableConsoleSummaryAndStructuredRecord(t *testing.T) { + root := t.TempDir() + logDirectory := filepath.Join(root, "logs") + var console bytes.Buffer + runtime, err := logging.NewRuntime(logging.Options{ + Directory: logDirectory, + ManagerSinks: logging.SinkBoth, + ManagerFileFormat: logging.FormatJSON, + TranscriptSinks: logging.SinkNone, + Stdout: &console, + Stderr: &console, + }) + if err != nil { + t.Fatalf("create logging runtime: %v", err) + } + manager := Manager{ + Config: config.Config{Provider: config.ProviderConfig{Type: "docker-container"}, Logging: config.LoggingConfig{Directory: "logs"}}, + ProjectRoot: root, + Logging: runtime, + } + path, err := manager.StartStartupTiming() + if err != nil { + t.Fatalf("start startup timing: %v", err) + } + for _, stage := range []string{"source_image_pull", "instance_container_create"} { + if err := manager.timeStartupStage(stage, func() error { return nil }); err != nil { + t.Fatalf("measure %s: %v", stage, err) + } + } + manager.FinishStartupTiming(nil) + if err := runtime.Close(); err != nil { + t.Fatalf("close logging runtime: %v", err) + } + + output := console.String() + if count := strings.Count(output, "\n"); count != 1 { + t.Fatalf("console emitted %d timing records, want 1: %q", count, output) + } + for _, want := range []string{"Docker Container startup timing:", "source_image_pull=", "instance_container_create=", "total_startup=", "log: " + path} { + if !strings.Contains(output, want) { + t.Fatalf("console summary missing %q: %q", want, output) + } + } + + content, err := os.ReadFile(filepath.Join(logDirectory, logging.ManagerFilename)) + if err != nil { + t.Fatalf("read structured manager log: %v", err) + } + var record map[string]any + if err := json.Unmarshal(content, &record); err != nil { + t.Fatalf("decode structured manager record: %v\n%s", err, content) + } + message, ok := record["msg"].(string) + if !ok || !strings.Contains(message, "Docker Container startup timing:") || !strings.Contains(message, "source_image_pull=") || !strings.Contains(message, "instance_container_create=") || !strings.Contains(message, "total_startup=") || !strings.Contains(message, "log: "+path) { + t.Fatalf("unexpected structured message: %#v", record["msg"]) + } + if record["provider"] != "docker-container" || record["operation"] != "startup" || record["logPath"] != path { + t.Fatalf("structured record missing context: %#v", record) + } + stages, ok := record["stages"].(map[string]any) + if !ok || stages["source_image_pull"] == nil || stages["instance_container_create"] == nil || stages["total_startup"] == nil { + t.Fatalf("structured record missing stage durations: %#v", record["stages"]) + } +} diff --git a/internal/pool/state/doc.go b/internal/pool/state/doc.go new file mode 100644 index 0000000..b971144 --- /dev/null +++ b/internal/pool/state/doc.go @@ -0,0 +1,20 @@ +// Package state provides a provider-neutral, durable ownership ledger for pool +// instances. It records intents before side effects and only permits exact-name +// records to progress through the lifecycle. Provider-specific details belong +// in Receipt, an explicitly versioned opaque JSON object. +// +// Unknown provider inventory is stored as a Discovery. Discoveries are +// quarantine/report-only: they cannot be converted into an owned Record or +// deleted through this package. +// +// The primary transition table is: +// +// reserved -> creating -> created -> validating -> standby -> registering +// -> ready -> busy -> draining -> fencing -> fenced +// -> remote-reconciling -> remote-absent -> local-removing +// -> local-absent -> tombstoned +// +// Quarantine is terminal for normal allocation and registration. It can only +// proceed through fencing and exact cleanup. Cleanup failures become +// cleanup-pending and resume only at fencing. +package state diff --git a/internal/pool/state/store.go b/internal/pool/state/store.go new file mode 100644 index 0000000..25c98ba --- /dev/null +++ b/internal/pool/state/store.go @@ -0,0 +1,616 @@ +package state + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "sort" + "sync" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/filelock" +) + +const snapshotFilename = "pool-lifecycle-v1.json" +const lockFilename = ".pool-lifecycle.lock" + +type diskState struct { + SchemaVersion int `json:"schemaVersion"` + Generation uint64 `json:"generation"` + Records []Record `json:"records"` + Discoveries []Discovery `json:"discoveries"` + Checksum string `json:"checksum"` +} + +type checksumState struct { + SchemaVersion int `json:"schemaVersion"` + Generation uint64 `json:"generation"` + Records []Record `json:"records"` + Discoveries []Discovery `json:"discoveries"` +} + +// Store serializes state across goroutines and cooperating controller processes. +type Store struct { + directory string + path string + lockPath string + mu sync.Mutex + now func() time.Time + fault func(string) error // test-only crash boundary injection +} + +func Open(directory string) (*Store, error) { + if directory == "" { + return nil, fmt.Errorf("%w: state directory is empty", ErrInvalidRecord) + } + if err := os.MkdirAll(directory, 0o700); err != nil { + return nil, fmt.Errorf("create lifecycle state directory: %w", err) + } + directory, err := filepath.Abs(directory) + if err != nil { + return nil, fmt.Errorf("resolve lifecycle state directory: %w", err) + } + store := &Store{directory: directory, path: filepath.Join(directory, snapshotFilename), lockPath: filepath.Join(directory, lockFilename), now: time.Now} + if err := store.withLock(context.Background(), func() error { _, err := store.load(); return err }); err != nil { + return nil, err + } + return store, nil +} + +func (s *Store) Path() string { return s.path } + +func (s *Store) Reserve(ctx context.Context, spec CreateSpec) (Record, error) { + if err := validateName("instance name", spec.Name); err != nil { + return Record{}, err + } + if err := validateName("provider type", spec.ProviderType); err != nil { + return Record{}, err + } + if err := validateName("GitHub runner name", spec.GitHub.ExactName); err != nil { + return Record{}, err + } + var result Record + err := s.mutate(ctx, func(state *diskState, now time.Time) error { + if _, found := findRecord(state.Records, spec.Name); found { + return fmt.Errorf("%w: %s", ErrAlreadyExists, spec.Name) + } + state.Generation++ + result = Record{Name: spec.Name, ProviderType: spec.ProviderType, GitHub: spec.GitHub, Phase: PhaseReserved, Generation: state.Generation, CreatedAt: now, UpdatedAt: now} + state.Records = append(state.Records, result) + sort.Slice(state.Records, func(i, j int) bool { return state.Records[i].Name < state.Records[j].Name }) + return nil + }) + return cloneRecord(result), err +} + +func (s *Store) Transition(ctx context.Context, name string, transition Transition) (Record, error) { + if err := validateName("instance name", name); err != nil { + return Record{}, err + } + var result Record + err := s.mutate(ctx, func(state *diskState, now time.Time) error { + index, found := findRecord(state.Records, name) + if !found { + return fmt.Errorf("%w: %s", ErrNotFound, name) + } + record := cloneRecord(state.Records[index]) + if err := applyTransition(&record, transition, now); err != nil { + return err + } + state.Generation++ + record.Generation, record.UpdatedAt = state.Generation, now + state.Records[index], result = record, record + return nil + }) + return cloneRecord(result), err +} + +func (s *Store) AcquireLease(ctx context.Context, name string, lease Lease) (Record, error) { + if err := validateName("instance name", name); err != nil { + return Record{}, err + } + if err := validateName("lease purpose", lease.Purpose); err != nil { + return Record{}, err + } + if err := validateName("lease holder", lease.Holder); err != nil { + return Record{}, err + } + var result Record + err := s.mutate(ctx, func(state *diskState, now time.Time) error { + index, found := findRecord(state.Records, name) + if !found { + return fmt.Errorf("%w: %s", ErrNotFound, name) + } + record := cloneRecord(state.Records[index]) + if record.Phase == PhaseTombstoned || !lease.ExpiresAt.After(now) { + return fmt.Errorf("%w: cannot acquire lease", ErrInvalidTransition) + } + retained := record.Leases[:0] + for _, existing := range record.Leases { + if existing.ExpiresAt.After(now) && !(existing.Purpose == lease.Purpose && existing.Holder == lease.Holder) { + retained = append(retained, existing) + } + } + record.Leases = append(retained, Lease{Purpose: lease.Purpose, Holder: lease.Holder, ExpiresAt: lease.ExpiresAt.UTC()}) + state.Generation++ + record.Generation, record.UpdatedAt = state.Generation, now + state.Records[index], result = record, record + return nil + }) + return cloneRecord(result), err +} + +func (s *Store) ReleaseLease(ctx context.Context, name, purpose, holder string) (Record, error) { + if err := validateName("instance name", name); err != nil { + return Record{}, err + } + if err := validateName("lease purpose", purpose); err != nil { + return Record{}, err + } + if err := validateName("lease holder", holder); err != nil { + return Record{}, err + } + var result Record + err := s.mutate(ctx, func(state *diskState, now time.Time) error { + index, found := findRecord(state.Records, name) + if !found { + return fmt.Errorf("%w: %s", ErrNotFound, name) + } + record := cloneRecord(state.Records[index]) + retained := record.Leases[:0] + for _, lease := range record.Leases { + if !(lease.Purpose == purpose && lease.Holder == holder) && lease.ExpiresAt.After(now) { + retained = append(retained, lease) + } + } + record.Leases = retained + state.Generation++ + record.Generation, record.UpdatedAt = state.Generation, now + state.Records[index], result = record, record + return nil + }) + return cloneRecord(result), err +} + +// ReportUnknown records inventory that was not created by this state store. +// It is intentionally not exposed to Transition or cleanup operations. +func (s *Store) ReportUnknown(ctx context.Context, discovery Discovery) (Discovery, error) { + if err := validateName("provider type", discovery.ProviderType); err != nil { + return Discovery{}, err + } + if err := validateName("provider id", discovery.ProviderID); err != nil { + return Discovery{}, err + } + if err := validateName("discovered instance name", discovery.ExactName); err != nil { + return Discovery{}, err + } + if err := validateReceipt(discovery.Receipt); err != nil { + return Discovery{}, err + } + err := s.mutate(ctx, func(state *diskState, now time.Time) error { + discovery.ObservedAt = now + state.Generation++ + for i, existing := range state.Discoveries { + if existing.ProviderType == discovery.ProviderType && existing.ProviderID == discovery.ProviderID { + state.Discoveries[i] = discovery + return nil + } + } + state.Discoveries = append(state.Discoveries, discovery) + sort.Slice(state.Discoveries, func(i, j int) bool { return discoveryKey(state.Discoveries[i]) < discoveryKey(state.Discoveries[j]) }) + return nil + }) + return discovery, err +} + +func (s *Store) Read(ctx context.Context, name string) (Record, error) { + if err := validateName("instance name", name); err != nil { + return Record{}, err + } + var result Record + err := s.withLock(ctx, func() error { + state, err := s.load() + if err != nil { + return err + } + index, found := findRecord(state.Records, name) + if !found { + return fmt.Errorf("%w: %s", ErrNotFound, name) + } + result = cloneRecord(state.Records[index]) + return nil + }) + return result, err +} +func (s *Store) List(ctx context.Context) ([]Record, error) { + var result []Record + err := s.withLock(ctx, func() error { + state, err := s.load() + if err != nil { + return err + } + result = make([]Record, len(state.Records)) + for i := range state.Records { + result[i] = cloneRecord(state.Records[i]) + } + return nil + }) + return result, err +} +func (s *Store) Discoveries(ctx context.Context) ([]Discovery, error) { + var result []Discovery + err := s.withLock(ctx, func() error { + state, err := s.load() + if err != nil { + return err + } + result = append([]Discovery(nil), state.Discoveries...) + return nil + }) + return result, err +} + +func (s *Store) mutate(ctx context.Context, operation func(*diskState, time.Time) error) error { + return s.withLock(ctx, func() error { + state, err := s.load() + if err != nil { + return err + } + if err := operation(&state, s.now().UTC()); err != nil { + return err + } + if err := validateState(state); err != nil { + return err + } + return s.save(state) + }) +} +func (s *Store) withLock(ctx context.Context, operation func() error) error { + s.mu.Lock() + defer s.mu.Unlock() + for { + lock, err := filelock.Acquire(s.lockPath) + if err == nil { + defer lock.Close() + return operation() + } + if !errors.Is(err, filelock.ErrLocked) { + return err + } + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(10 * time.Millisecond): + } + } +} + +func (s *Store) load() (diskState, error) { + data, err := os.ReadFile(s.path) + if os.IsNotExist(err) { + return diskState{SchemaVersion: SchemaVersion, Records: []Record{}, Discoveries: []Discovery{}}, nil + } + if err != nil { + return diskState{}, err + } + var state diskState + if err := json.Unmarshal(data, &state); err != nil { + return diskState{}, fmt.Errorf("%w: decode: %v", ErrCorrupt, err) + } + if state.SchemaVersion != SchemaVersion { + return diskState{}, fmt.Errorf("%w: schema %d", ErrCorrupt, state.SchemaVersion) + } + sum, err := checksum(state) + if err != nil || state.Checksum != hex.EncodeToString(sum) { + return diskState{}, fmt.Errorf("%w: checksum", ErrCorrupt) + } + if err := validateState(state); err != nil { + return diskState{}, fmt.Errorf("%w: %v", ErrCorrupt, err) + } + return state, nil +} +func (s *Store) save(state diskState) error { + sum, err := checksum(state) + if err != nil { + return err + } + state.Checksum = hex.EncodeToString(sum) + data, err := json.MarshalIndent(state, "", " ") + if err != nil { + return err + } + data = append(data, '\n') + if s.fault != nil { + if err := s.fault("before-temp"); err != nil { + return err + } + } + file, err := os.CreateTemp(s.directory, ".pool-lifecycle-*.tmp") + if err != nil { + return err + } + temp := file.Name() + committed := false + defer func() { + _ = file.Close() + if !committed { + _ = os.Remove(temp) + } + }() + if _, err := file.Write(data); err != nil { + return err + } + if err := file.Sync(); err != nil { + return err + } + if err := file.Close(); err != nil { + return err + } + if s.fault != nil { + if err := s.fault("before-rename"); err != nil { + return err + } + } + if err := os.Rename(temp, s.path); err != nil { + return err + } + committed = true + return nil +} +func checksum(state diskState) ([]byte, error) { + encoded, err := json.Marshal(checksumState{SchemaVersion: state.SchemaVersion, Generation: state.Generation, Records: state.Records, Discoveries: state.Discoveries}) + if err != nil { + return nil, err + } + result := sha256.Sum256(encoded) + return result[:], nil +} + +func findRecord(records []Record, name string) (int, bool) { + index := sort.Search(len(records), func(i int) bool { return records[i].Name >= name }) + return index, index < len(records) && records[index].Name == name +} +func discoveryKey(discovery Discovery) string { + return discovery.ProviderType + "\x00" + discovery.ProviderID +} +func timePtr(value time.Time) *time.Time { copy := value; return © } + +func applyTransition(record *Record, transition Transition, now time.Time) error { + if record.Phase == PhaseTombstoned { + return invalid(record, transition.Action) + } + switch transition.Action { + case ActionCreateIntent: + return move(record, PhaseReserved, PhaseCreating, transition.Action) + case ActionAbandonCreate: + identitylessQuarantine := record.Phase == PhaseQuarantined && record.ProviderID == "" && emptyReceipt(record.Receipt) + if record.Phase != PhaseReserved && record.Phase != PhaseCreating && !identitylessQuarantine { + return invalid(record, transition.Action) + } + if record.ProviderID != "" || !emptyReceipt(record.Receipt) || len(activeLeases(record.Leases, now)) != 0 { + return invalid(record, transition.Action) + } + record.Cleanup.RemoteVerifyAt = timePtr(now) + record.Cleanup.RemoteAbsentAt = timePtr(now) + record.Cleanup.LocalRemoveIntentAt = timePtr(now) + record.Cleanup.LocalAbsentAt = timePtr(now) + record.Phase = PhaseTombstoned + record.TombstonedAt = timePtr(now) + case ActionCreated: + if record.Phase != PhaseCreating || validateName("provider id", transition.ProviderID) != nil || validateReceipt(transition.Receipt) != nil { + return invalid(record, transition.Action) + } + record.ProviderID, record.Receipt, record.Phase = transition.ProviderID, transition.Receipt, PhaseCreated + case ActionValidateIntent: + return move(record, PhaseCreated, PhaseValidating, transition.Action) + case ActionValidated: + return move(record, PhaseValidating, PhaseStandby, transition.Action) + case ActionRegisterIntent: + return move(record, PhaseStandby, PhaseRegistering, transition.Action) + case ActionRegistered: + if record.Phase != PhaseRegistering || transition.RunnerID <= 0 { + return invalid(record, transition.Action) + } + record.GitHub.RunnerID, record.Phase = transition.RunnerID, PhaseReady + case ActionJobStarted: + return move(record, PhaseReady, PhaseBusy, transition.Action) + case ActionJobFinished: + return move(record, PhaseBusy, PhaseDraining, transition.Action) + case ActionQuarantine: + if transition.Reason == "" || !quarantineAllowed(record.Phase) { + return invalid(record, transition.Action) + } + record.Phase, record.Quarantine = PhaseQuarantined, &Quarantine{Reason: transition.Reason, ReportedAt: now} + case ActionFenceIntent: + if !fenceAllowed(record.Phase) { + return invalid(record, transition.Action) + } + record.Phase, record.Cleanup.FenceIntentAt = PhaseFencing, timePtr(now) + case ActionFenced: + return move(record, PhaseFencing, PhaseFenced, transition.Action) + case ActionVerifyRemoteIntent: + if record.Phase != PhaseFenced { + return invalid(record, transition.Action) + } + record.Phase, record.Cleanup.RemoteVerifyAt = PhaseRemoteReconciling, timePtr(now) + case ActionRemoteAbsent: + if record.Phase != PhaseRemoteReconciling { + return invalid(record, transition.Action) + } + record.Phase, record.Cleanup.RemoteAbsentAt = PhaseRemoteAbsent, timePtr(now) + case ActionRemoveLocalIntent: + if record.Phase != PhaseRemoteAbsent { + return invalid(record, transition.Action) + } + record.Phase, record.Cleanup.LocalRemoveIntentAt = PhaseLocalRemoving, timePtr(now) + case ActionLocalAbsent: + if record.Phase != PhaseLocalRemoving { + return invalid(record, transition.Action) + } + record.Phase, record.Cleanup.LocalAbsentAt = PhaseLocalAbsent, timePtr(now) + case ActionCleanupPending: + if record.Phase != PhaseFencing && record.Phase != PhaseRemoteReconciling && record.Phase != PhaseLocalRemoving { + return invalid(record, transition.Action) + } + record.Phase = PhaseCleanupPending + case ActionResumeCleanup: + if record.Phase != PhaseCleanupPending { + return invalid(record, transition.Action) + } + switch { + case record.Cleanup.LocalAbsentAt != nil: + record.Phase = PhaseLocalAbsent + case record.Cleanup.LocalRemoveIntentAt != nil: + record.Phase = PhaseLocalRemoving + case record.Cleanup.RemoteAbsentAt != nil: + record.Phase = PhaseRemoteAbsent + case record.Cleanup.RemoteVerifyAt != nil: + record.Phase = PhaseRemoteReconciling + default: + record.Phase = PhaseFencing + } + case ActionTombstone: + if record.Phase != PhaseLocalAbsent || record.Cleanup.RemoteAbsentAt == nil || record.Cleanup.LocalAbsentAt == nil || len(activeLeases(record.Leases, now)) != 0 { + return invalid(record, transition.Action) + } + record.Phase, record.TombstonedAt = PhaseTombstoned, timePtr(now) + default: + return fmt.Errorf("%w: unknown action %q", ErrInvalidTransition, transition.Action) + } + return nil +} + +func move(record *Record, from, to Phase, action Action) error { + if record.Phase != from { + return invalid(record, action) + } + record.Phase = to + return nil +} +func invalid(record *Record, action Action) error { + return fmt.Errorf("%w: %s from %s", ErrInvalidTransition, action, record.Phase) +} +func fenceAllowed(phase Phase) bool { + switch phase { + case PhaseCreated, PhaseValidating, PhaseStandby, PhaseRegistering, PhaseReady, PhaseBusy, PhaseDraining, PhaseQuarantined: + return true + default: + return false + } +} +func quarantineAllowed(phase Phase) bool { + switch phase { + case PhaseReserved, PhaseCreating, PhaseCreated, PhaseValidating, PhaseStandby, PhaseRegistering, PhaseReady, PhaseBusy, PhaseDraining: + return true + default: + return false + } +} +func activeLeases(leases []Lease, now time.Time) []Lease { + result := make([]Lease, 0, len(leases)) + for _, lease := range leases { + if lease.ExpiresAt.After(now) { + result = append(result, lease) + } + } + return result +} + +func validateState(state diskState) error { + if state.SchemaVersion != SchemaVersion { + return fmt.Errorf("%w: schema", ErrInvalidRecord) + } + for i := range state.Records { + record := state.Records[i] + if err := validateRecord(record); err != nil { + return err + } + if i > 0 && state.Records[i-1].Name >= record.Name { + return fmt.Errorf("%w: records are not sorted", ErrInvalidRecord) + } + } + for i, discovery := range state.Discoveries { + if err := validateDiscovery(discovery); err != nil { + return err + } + if i > 0 && discoveryKey(state.Discoveries[i-1]) >= discoveryKey(discovery) { + return fmt.Errorf("%w: discoveries are not sorted", ErrInvalidRecord) + } + } + return nil +} +func validateRecord(record Record) error { + if err := validateName("instance name", record.Name); err != nil { + return err + } + if err := validateName("provider type", record.ProviderType); err != nil { + return err + } + if err := validateName("GitHub runner name", record.GitHub.ExactName); err != nil { + return err + } + if !validPhase(record.Phase) || record.Generation == 0 || record.CreatedAt.IsZero() || record.UpdatedAt.IsZero() || record.UpdatedAt.Before(record.CreatedAt) { + return fmt.Errorf("%w: record phase, generation, or time", ErrInvalidRecord) + } + if record.Phase != PhaseReserved && record.Phase != PhaseCreating && !(record.Phase == PhaseQuarantined && record.ProviderID == "" && emptyReceipt(record.Receipt)) && !(record.Phase == PhaseTombstoned && record.ProviderID == "" && emptyReceipt(record.Receipt)) { + if err := validateName("provider id", record.ProviderID); err != nil { + return err + } + if err := validateReceipt(record.Receipt); err != nil { + return err + } + } + for _, lease := range record.Leases { + if err := validateName("lease purpose", lease.Purpose); err != nil { + return err + } + if err := validateName("lease holder", lease.Holder); err != nil { + return err + } + if lease.ExpiresAt.IsZero() { + return fmt.Errorf("%w: lease expiry", ErrInvalidRecord) + } + } + if record.Quarantine != nil && (record.Quarantine.Reason == "" || record.Quarantine.ReportedAt.IsZero()) { + return fmt.Errorf("%w: quarantine", ErrInvalidRecord) + } + if record.Phase == PhaseTombstoned && (record.TombstonedAt == nil || record.Cleanup.RemoteAbsentAt == nil || record.Cleanup.LocalAbsentAt == nil) { + return fmt.Errorf("%w: tombstone needs exact absence", ErrInvalidRecord) + } + return nil +} + +func emptyReceipt(receipt Receipt) bool { + return receipt.Version == "" && (len(receipt.Payload) == 0 || string(receipt.Payload) == "null") +} +func validateDiscovery(discovery Discovery) error { + if err := validateName("provider type", discovery.ProviderType); err != nil { + return err + } + if err := validateName("provider id", discovery.ProviderID); err != nil { + return err + } + if err := validateName("discovered instance name", discovery.ExactName); err != nil { + return err + } + if err := validateReceipt(discovery.Receipt); err != nil { + return err + } + if discovery.ObservedAt.IsZero() { + return fmt.Errorf("%w: discovery time", ErrInvalidRecord) + } + return nil +} +func validPhase(phase Phase) bool { + switch phase { + case PhaseReserved, PhaseCreating, PhaseCreated, PhaseValidating, PhaseStandby, PhaseRegistering, PhaseReady, PhaseBusy, PhaseDraining, PhaseQuarantined, PhaseCleanupPending, PhaseFencing, PhaseFenced, PhaseRemoteReconciling, PhaseRemoteAbsent, PhaseLocalRemoving, PhaseLocalAbsent, PhaseTombstoned: + return true + default: + return false + } +} diff --git a/internal/pool/state/store_test.go b/internal/pool/state/store_test.go new file mode 100644 index 0000000..f4a21f0 --- /dev/null +++ b/internal/pool/state/store_test.go @@ -0,0 +1,316 @@ +package state + +import ( + "context" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "reflect" + "sync" + "testing" + "time" +) + +func receipt(t *testing.T) Receipt { + t.Helper() + payload, err := json.Marshal(map[string]string{"external": "exact-id"}) + if err != nil { + t.Fatal(err) + } + return Receipt{Version: "v1", Payload: payload} +} +func reserve(t *testing.T, store *Store, name string) Record { + t.Helper() + record, err := store.Reserve(context.Background(), CreateSpec{Name: name, ProviderType: "test-provider", GitHub: GitHubIdentity{ExactName: name + "-runner"}}) + if err != nil { + t.Fatal(err) + } + return record +} + +func TestLifecycleAllowsOnlyDocumentedTransitions(t *testing.T) { + store, err := Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + record := reserve(t, store, "instance-a") + sequence := []Transition{{Action: ActionCreateIntent}, {Action: ActionCreated, ProviderID: "provider:1", Receipt: receipt(t)}, {Action: ActionValidateIntent}, {Action: ActionValidated}, {Action: ActionRegisterIntent}, {Action: ActionRegistered, RunnerID: 42}, {Action: ActionJobStarted}, {Action: ActionJobFinished}, {Action: ActionFenceIntent}, {Action: ActionFenced}, {Action: ActionVerifyRemoteIntent}, {Action: ActionRemoteAbsent}, {Action: ActionRemoveLocalIntent}, {Action: ActionLocalAbsent}, {Action: ActionTombstone}} + for _, transition := range sequence { + record, err = store.Transition(context.Background(), record.Name, transition) + if err != nil { + t.Fatalf("%s from %s: %v", transition.Action, record.Phase, err) + } + } + if record.Phase != PhaseTombstoned { + t.Fatalf("phase = %s", record.Phase) + } + if _, err := store.Transition(context.Background(), record.Name, Transition{Action: ActionJobStarted}); !errors.Is(err, ErrInvalidTransition) { + t.Fatalf("terminal transition error = %v", err) + } +} + +func TestForbiddenTransitionsAndQuarantineResume(t *testing.T) { + store, err := Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + record := reserve(t, store, "instance-b") + for _, transition := range []Transition{{Action: ActionValidated}, {Action: ActionRegistered, RunnerID: 1}, {Action: ActionTombstone}} { + if _, err := store.Transition(context.Background(), record.Name, transition); !errors.Is(err, ErrInvalidTransition) { + t.Fatalf("%s error = %v", transition.Action, err) + } + } + if _, err := store.Transition(context.Background(), record.Name, Transition{Action: ActionQuarantine, Reason: "create outcome is uncertain"}); err != nil { + t.Fatalf("quarantine uncertain create = %v", err) + } + record = reserve(t, store, "instance-b2") + if _, err := store.Transition(context.Background(), record.Name, Transition{Action: ActionCreateIntent}); err != nil { + t.Fatal(err) + } + record, err = store.Transition(context.Background(), record.Name, Transition{Action: ActionCreated, ProviderID: "provider:quarantine", Receipt: receipt(t)}) + if err != nil { + t.Fatal(err) + } + record, err = store.Transition(context.Background(), record.Name, Transition{Action: ActionQuarantine, Reason: "provider inventory differed"}) + if err != nil { + t.Fatal(err) + } + for _, transition := range []Transition{{Action: ActionFenceIntent}, {Action: ActionCleanupPending}, {Action: ActionResumeCleanup}} { + record, err = store.Transition(context.Background(), record.Name, transition) + if err != nil { + t.Fatalf("%s: %v", transition.Action, err) + } + } + if record.Phase != PhaseFencing { + t.Fatalf("phase = %s", record.Phase) + } +} + +func TestAbandonCreateRequiresNoProviderIdentityAndRecordsExactAbsence(t *testing.T) { + store, err := Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + for _, name := range []string{"reserved-instance", "creating-instance"} { + record := reserve(t, store, name) + if name == "creating-instance" { + record, err = store.Transition(context.Background(), name, Transition{Action: ActionCreateIntent}) + if err != nil { + t.Fatal(err) + } + } + record, err = store.Transition(context.Background(), name, Transition{Action: ActionAbandonCreate}) + if err != nil { + t.Fatalf("abandon %s: %v", name, err) + } + if record.Phase != PhaseTombstoned || record.Cleanup.RemoteAbsentAt == nil || record.Cleanup.LocalAbsentAt == nil { + t.Fatalf("abandoned record = %#v", record) + } + } +} + +func TestAbandonCreateAllowsOnlyIdentitylessQuarantine(t *testing.T) { + store, err := Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + record := reserve(t, store, "quarantined-create") + record, err = store.Transition(context.Background(), record.Name, Transition{Action: ActionQuarantine, Reason: "create outcome was uncertain"}) + if err != nil { + t.Fatal(err) + } + record, err = store.Transition(context.Background(), record.Name, Transition{Action: ActionAbandonCreate}) + if err != nil { + t.Fatalf("abandon identityless quarantine: %v", err) + } + if record.Phase != PhaseTombstoned || record.Cleanup.RemoteAbsentAt == nil || record.Cleanup.LocalAbsentAt == nil { + t.Fatalf("abandoned quarantine = %#v", record) + } + + identified := reserve(t, store, "identified-quarantine") + if _, err := store.Transition(context.Background(), identified.Name, Transition{Action: ActionCreateIntent}); err != nil { + t.Fatal(err) + } + identified, err = store.Transition(context.Background(), identified.Name, Transition{Action: ActionCreated, ProviderID: "provider:identified", Receipt: receipt(t)}) + if err != nil { + t.Fatal(err) + } + identified, err = store.Transition(context.Background(), identified.Name, Transition{Action: ActionQuarantine, Reason: "identified quarantine"}) + if err != nil { + t.Fatal(err) + } + if _, err := store.Transition(context.Background(), identified.Name, Transition{Action: ActionAbandonCreate}); !errors.Is(err, ErrInvalidTransition) { + t.Fatalf("identified quarantine abandon error = %v, want invalid transition", err) + } +} + +func TestProviderReceiptAndUnknownDiscoveryRoundTrip(t *testing.T) { + store, err := Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + record := reserve(t, store, "instance-c") + if _, err = store.Transition(context.Background(), record.Name, Transition{Action: ActionCreateIntent}); err != nil { + t.Fatal(err) + } + want := receipt(t) + if _, err = store.Transition(context.Background(), record.Name, Transition{Action: ActionCreated, ProviderID: "provider:2", Receipt: want}); err != nil { + t.Fatal(err) + } + if _, err = store.ReportUnknown(context.Background(), Discovery{ProviderType: "test-provider", ProviderID: "foreign:1", ExactName: "foreign-instance", Receipt: want}); err != nil { + t.Fatal(err) + } + reopened, err := Open(filepath.Dir(store.Path())) + if err != nil { + t.Fatal(err) + } + got, err := reopened.Read(context.Background(), record.Name) + if err != nil { + t.Fatal(err) + } + if !jsonEqual(got.Receipt.Payload, want.Payload) { + t.Fatalf("receipt = %s", got.Receipt.Payload) + } + discoveries, err := reopened.Discoveries(context.Background()) + if err != nil || len(discoveries) != 1 || discoveries[0].ExactName != "foreign-instance" { + t.Fatalf("discoveries = %#v err=%v", discoveries, err) + } + if _, err = reopened.Transition(context.Background(), "foreign-instance", Transition{Action: ActionFenceIntent}); !errors.Is(err, ErrNotFound) { + t.Fatalf("unknown resource became owned: %v", err) + } +} + +func jsonEqual(left, right json.RawMessage) bool { + var leftValue any + var rightValue any + if json.Unmarshal(left, &leftValue) != nil || json.Unmarshal(right, &rightValue) != nil { + return false + } + return reflect.DeepEqual(leftValue, rightValue) +} + +func TestLeaseExpiryAndTombstoneProtection(t *testing.T) { + store, err := Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + fixed := time.Now().UTC() + store.now = func() time.Time { return fixed } + record := reserve(t, store, "instance-d") + if _, err = store.AcquireLease(context.Background(), record.Name, Lease{Purpose: "cleanup", Holder: "controller", ExpiresAt: fixed.Add(time.Hour)}); err != nil { + t.Fatal(err) + } + for _, transition := range []Transition{{Action: ActionCreateIntent}, {Action: ActionCreated, ProviderID: "provider:3", Receipt: receipt(t)}, {Action: ActionFenceIntent}, {Action: ActionFenced}, {Action: ActionVerifyRemoteIntent}, {Action: ActionRemoteAbsent}, {Action: ActionRemoveLocalIntent}, {Action: ActionLocalAbsent}} { + if _, err = store.Transition(context.Background(), record.Name, transition); err != nil { + t.Fatalf("%s: %v", transition.Action, err) + } + } + if _, err = store.Transition(context.Background(), record.Name, Transition{Action: ActionTombstone}); !errors.Is(err, ErrInvalidTransition) { + t.Fatalf("tombstone with lease = %v", err) + } + if _, err = store.ReleaseLease(context.Background(), record.Name, "cleanup", "controller"); err != nil { + t.Fatal(err) + } + if _, err = store.AcquireLease(context.Background(), record.Name, Lease{Purpose: "audit", Holder: "controller", ExpiresAt: fixed.Add(time.Hour)}); err != nil { + t.Fatal(err) + } + fixed = fixed.Add(2 * time.Hour) + if _, err = store.Transition(context.Background(), record.Name, Transition{Action: ActionTombstone}); err != nil { + t.Fatal(err) + } +} + +func TestPartialWriteKeepsPriorSnapshotAndCorruptionFailsClosed(t *testing.T) { + directory := t.TempDir() + store, err := Open(directory) + if err != nil { + t.Fatal(err) + } + reserve(t, store, "instance-e") + store.fault = func(point string) error { + if point == "before-rename" { + return errors.New("simulated crash") + } + return nil + } + if _, err = store.Reserve(context.Background(), CreateSpec{Name: "instance-f", ProviderType: "test-provider", GitHub: GitHubIdentity{ExactName: "instance-f-runner"}}); err == nil { + t.Fatal("faulted write succeeded") + } + store.fault = nil + reopened, err := Open(directory) + if err != nil { + t.Fatal(err) + } + records, err := reopened.List(context.Background()) + if err != nil || len(records) != 1 || records[0].Name != "instance-e" { + t.Fatalf("recovery records=%#v err=%v", records, err) + } + if err := os.WriteFile(reopened.Path(), []byte("{"), 0o600); err != nil { + t.Fatal(err) + } + if _, err := Open(directory); !errors.Is(err, ErrCorrupt) { + t.Fatalf("corrupt open error = %v", err) + } +} + +func TestRejectsUnknownPhaseFromDisk(t *testing.T) { + directory := t.TempDir() + store, err := Open(directory) + if err != nil { + t.Fatal(err) + } + reserve(t, store, "instance-g") + state, err := store.load() + if err != nil { + t.Fatal(err) + } + state.Records[0].Phase = "invented" + sum, err := checksum(state) + if err != nil { + t.Fatal(err) + } + state.Checksum = hex.EncodeToString(sum) + data, err := json.Marshal(state) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(store.Path(), data, 0o600); err != nil { + t.Fatal(err) + } + if _, err := Open(directory); !errors.Is(err, ErrCorrupt) { + t.Fatalf("unknown phase error = %v", err) + } +} + +func TestConcurrentReservationsAreSerialized(t *testing.T) { + store, err := Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + const count = 12 + var wg sync.WaitGroup + errorsCh := make(chan error, count) + for index := 0; index < count; index++ { + index := index + wg.Add(1) + go func() { + defer wg.Done() + _, err := store.Reserve(context.Background(), CreateSpec{Name: fmt.Sprintf("instance-%d", index), ProviderType: "test-provider", GitHub: GitHubIdentity{ExactName: fmt.Sprintf("runner-%d", index)}}) + errorsCh <- err + }() + } + wg.Wait() + close(errorsCh) + for err := range errorsCh { + if err != nil { + t.Fatal(err) + } + } + records, err := store.List(context.Background()) + if err != nil || len(records) != count { + t.Fatalf("records=%d err=%v", len(records), err) + } +} diff --git a/internal/pool/state/types.go b/internal/pool/state/types.go new file mode 100644 index 0000000..e853f0b --- /dev/null +++ b/internal/pool/state/types.go @@ -0,0 +1,172 @@ +package state + +import ( + "encoding/json" + "errors" + "fmt" + "strings" + "time" +) + +const SchemaVersion = 1 + +var ( + ErrAlreadyExists = errors.New("pool lifecycle record already exists") + ErrNotFound = errors.New("pool lifecycle record not found") + ErrInvalidRecord = errors.New("invalid pool lifecycle record") + ErrInvalidTransition = errors.New("invalid pool lifecycle transition") + ErrCorrupt = errors.New("pool lifecycle state is corrupt") +) + +type Phase string + +const ( + PhaseReserved Phase = "reserved" + PhaseCreating Phase = "creating" + PhaseCreated Phase = "created" + PhaseValidating Phase = "validating" + PhaseStandby Phase = "standby" + PhaseRegistering Phase = "registering" + PhaseReady Phase = "ready" + PhaseBusy Phase = "busy" + PhaseDraining Phase = "draining" + PhaseQuarantined Phase = "quarantined" + PhaseCleanupPending Phase = "cleanup-pending" + PhaseFencing Phase = "fencing" + PhaseFenced Phase = "fenced" + PhaseRemoteReconciling Phase = "remote-reconciling" + PhaseRemoteAbsent Phase = "remote-absent" + PhaseLocalRemoving Phase = "local-removing" + PhaseLocalAbsent Phase = "local-absent" + PhaseTombstoned Phase = "tombstoned" +) + +type Action string + +const ( + ActionCreateIntent Action = "create-intent" + ActionAbandonCreate Action = "abandon-create" + ActionCreated Action = "created" + ActionValidateIntent Action = "validate-intent" + ActionValidated Action = "validated" + ActionRegisterIntent Action = "register-intent" + ActionRegistered Action = "registered" + ActionJobStarted Action = "job-started" + ActionJobFinished Action = "job-finished" + ActionQuarantine Action = "quarantine" + ActionFenceIntent Action = "fence-intent" + ActionFenced Action = "fenced" + ActionVerifyRemoteIntent Action = "verify-remote-absent-intent" + ActionRemoteAbsent Action = "remote-absent" + ActionRemoveLocalIntent Action = "remove-local-intent" + ActionLocalAbsent Action = "local-absent" + ActionCleanupPending Action = "cleanup-pending" + ActionResumeCleanup Action = "resume-cleanup" + ActionTombstone Action = "tombstone" +) + +// Receipt is provider-owned, versioned state. It is intentionally opaque to +// the shared lifecycle package and must not contain credentials. +type Receipt struct { + Version string `json:"version"` + Payload json.RawMessage `json:"payload"` +} + +type GitHubIdentity struct { + ExactName string `json:"exactName"` + RunnerID int64 `json:"runnerId,omitempty"` +} + +type Lease struct { + Purpose string `json:"purpose"` + Holder string `json:"holder"` + ExpiresAt time.Time `json:"expiresAt"` +} + +type Quarantine struct { + Reason string `json:"reason"` + ReportedAt time.Time `json:"reportedAt"` +} + +type Cleanup struct { + FenceIntentAt *time.Time `json:"fenceIntentAt,omitempty"` + RemoteVerifyAt *time.Time `json:"remoteVerifyAt,omitempty"` + RemoteAbsentAt *time.Time `json:"remoteAbsentAt,omitempty"` + LocalRemoveIntentAt *time.Time `json:"localRemoveIntentAt,omitempty"` + LocalAbsentAt *time.Time `json:"localAbsentAt,omitempty"` +} + +type Record struct { + Name string `json:"name"` + ProviderType string `json:"providerType"` + ProviderID string `json:"providerId,omitempty"` + Receipt Receipt `json:"receipt"` + GitHub GitHubIdentity `json:"github"` + Phase Phase `json:"phase"` + Leases []Lease `json:"leases"` + Quarantine *Quarantine `json:"quarantine,omitempty"` + Cleanup Cleanup `json:"cleanup"` + Generation uint64 `json:"generation"` + CreatedAt time.Time `json:"createdAt"` + UpdatedAt time.Time `json:"updatedAt"` + TombstonedAt *time.Time `json:"tombstonedAt,omitempty"` +} + +type CreateSpec struct { + Name string + ProviderType string + GitHub GitHubIdentity +} + +type Transition struct { + Action Action + ProviderID string + Receipt Receipt + RunnerID int64 + Reason string +} + +// Discovery stores an unowned provider resource. It is never a cleanup target. +type Discovery struct { + ProviderType string `json:"providerType"` + ProviderID string `json:"providerId"` + ExactName string `json:"exactName"` + Receipt Receipt `json:"receipt"` + ObservedAt time.Time `json:"observedAt"` +} + +func validateName(label, value string) error { + if len(value) < 1 || len(value) > 128 || strings.TrimSpace(value) != value { + return fmt.Errorf("%w: invalid %s", ErrInvalidRecord, label) + } + for _, r := range value { + if !(r == '-' || r == '_' || r == '.' || r == ':' || r >= 'a' && r <= 'z' || r >= 'A' && r <= 'Z' || r >= '0' && r <= '9') { + return fmt.Errorf("%w: invalid %s", ErrInvalidRecord, label) + } + } + return nil +} + +func validateReceipt(receipt Receipt) error { + if err := validateName("receipt version", receipt.Version); err != nil { + return err + } + if len(receipt.Payload) == 0 || !json.Valid(receipt.Payload) { + return fmt.Errorf("%w: receipt payload must be valid JSON", ErrInvalidRecord) + } + var object map[string]json.RawMessage + if err := json.Unmarshal(receipt.Payload, &object); err != nil || object == nil { + return fmt.Errorf("%w: receipt payload must be a JSON object", ErrInvalidRecord) + } + return nil +} + +func cloneRecord(record Record) Record { + record.Receipt.Payload = append(json.RawMessage(nil), record.Receipt.Payload...) + record.Leases = append([]Lease(nil), record.Leases...) + if record.Quarantine != nil { + copy := *record.Quarantine + record.Quarantine = © + } + return record +} diff --git a/internal/pool/storage_catalog_lease.go b/internal/pool/storage_catalog_lease.go new file mode 100644 index 0000000..5af4635 --- /dev/null +++ b/internal/pool/storage_catalog_lease.go @@ -0,0 +1,100 @@ +package pool + +import ( + "context" + "fmt" + "path/filepath" + "strings" + "sync" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + storagecatalog "github.com/solutionforest/ephemeral-action-runner/internal/storage/catalog" +) + +const ( + storageCatalogControllerLeaseLifetime = 2 * time.Minute + storageCatalogControllerLeaseRefresh = 30 * time.Second +) + +func (m *Manager) startStorageCatalogControllerLease() (func(), error) { + if !m.AutomaticImageLifecycle { + return func() {}, nil + } + store, err := storagecatalog.Open("") + if err != nil { + return nil, err + } + projectRoot := strings.TrimSpace(m.ProjectRoot) + if projectRoot == "" { + projectRoot = "." + } + configPath := strings.TrimSpace(m.ConfigPath) + if configPath == "" { + configPath = filepath.Join(projectRoot, ".local", "config.yml") + } + configuredLimit := strings.TrimSpace(m.Config.Storage.BuildCacheLimit) + if configuredLimit == "" { + configuredLimit = "20GiB" + } + limit, err := config.ParseByteSize(configuredLimit) + if err != nil { + return nil, err + } + refresh := func(now time.Time) error { + _, updateErr := store.WithLock(now, func(value *storagecatalog.Catalog) error { + record, registerErr := storagecatalog.RegisterConfig(value, projectRoot, configPath, now) + if registerErr != nil { + return registerErr + } + for index := range value.Configs { + if value.Configs[index].ID == record.ID { + value.Configs[index].BuildCacheLimitBytes = uint64(limit) + break + } + } + return storagecatalog.RefreshControllerLease(value, record.ID, now.Add(storageCatalogControllerLeaseLifetime)) + }) + return updateErr + } + if err := refresh(m.currentTime()); err != nil { + return nil, fmt.Errorf("register EPAR storage controller lease: %w", err) + } + leaseContext, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { + defer close(done) + ticker := time.NewTicker(storageCatalogControllerLeaseRefresh) + defer ticker.Stop() + for { + select { + case <-leaseContext.Done(): + return + case <-ticker.C: + if err := refresh(m.currentTime()); err != nil { + m.warnf("EPAR storage controller lease refresh warning: %v\n", err) + } + } + } + }() + var once sync.Once + stop := func() { + once.Do(func() { + cancel() + <-done + now := m.currentTime() + _, releaseErr := store.WithLock(now, func(value *storagecatalog.Catalog) error { + configID, idErr := storagecatalog.ConfigID(projectRoot, configPath) + if idErr != nil { + return idErr + } + storagecatalog.ReleaseControllerLease(value, configID) + return nil + }) + if releaseErr != nil { + m.warnf("EPAR storage controller lease release warning: %v\n", releaseErr) + } + }) + } + return stop, nil +} diff --git a/internal/pool/storage_preflight.go b/internal/pool/storage_preflight.go new file mode 100644 index 0000000..3dfb02b --- /dev/null +++ b/internal/pool/storage_preflight.go @@ -0,0 +1,157 @@ +package pool + +import ( + "context" + "fmt" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/invocation" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +const ( + instanceCreateExpansionBytes = 10 * storage.GiB +) + +func (m *Manager) preflightStorage(operation string, peakBytes uint64) error { + return m.preflightStorageAttempt(operation, peakBytes, false) +} + +func (m *Manager) preflightStorageAttempt(operation string, peakBytes uint64, housekeepingRetried bool) error { + if !housekeepingRetried && m.AutomaticImageLifecycle && operation == "instance-create" { + pending, pendingErr := m.imageCoordinator().StorageCleanupPending() + if pendingErr != nil { + m.warnf("EPAR cleanup-pending check before %s was deferred: %v\n", operation, pendingErr) + } else if pending { + if cleanupErr := m.imageCoordinator().HousekeepStorage(context.Background()); cleanupErr != nil { + m.warnf("EPAR cleanup-pending retry before %s was deferred: %v\n", operation, cleanupErr) + } + housekeepingRetried = true + } + } + minimumFree, err := config.EffectiveMinimumFreeBytes(m.Config) + if err != nil { + return err + } + if m.Storage != nil { + snapshot, err := m.Storage.StorageSnapshot(context.Background(), provider.StorageRequest{ + Operation: operation, + Now: m.currentTime(), + PeakBytes: peakBytes, + MinimumFreeBytes: minimumFree, + }) + if err != nil { + measurementErr := fmt.Errorf("provider storage surface cannot be measured before %s: %w\n\nInspect storage with:\n %s", operation, err, invocation.Command("storage", "status", "--provider", m.Config.Provider.Type)) + if m.AllowInsufficientStorage { + m.warnStorageOverride(operation, measurementErr) + return nil + } + return measurementErr + } + if len(snapshot.Surfaces) == 0 || len(snapshot.Requirements) == 0 { + return fmt.Errorf("provider %q returned no required storage surface for %s", m.Config.Provider.Type, operation) + } + surfaces := make(map[string]storage.Surface, len(snapshot.Surfaces)) + for _, surface := range snapshot.Surfaces { + surfaces[surface.ID] = surface + } + for _, requirement := range snapshot.Requirements { + surface, found := surfaces[requirement.SurfaceID] + if !found { + return fmt.Errorf("provider storage requirement %q references unknown surface %q", requirement.ID, requirement.SurfaceID) + } + check, err := storage.EvaluateCapacity(surface, requirement) + if err != nil { + return fmt.Errorf("evaluate storage capacity before %s: %w", operation, err) + } + if check.Status != storage.CapacityReady { + if !housekeepingRetried && m.AutomaticImageLifecycle && operation == "instance-create" { + if cleanupErr := m.imageCoordinator().HousekeepStorage(context.Background()); cleanupErr != nil { + m.warnf("EPAR storage housekeeping retry before %s was deferred: %v\n", operation, cleanupErr) + } + return m.preflightStorageAttempt(operation, peakBytes, true) + } + admissionErr := storageAdmissionError(operation, surface, requirement, check, m.Config.Provider.Type, m.StorageOverrideCommand) + if m.AllowInsufficientStorage { + m.warnStorageOverride(operation, admissionErr) + continue + } + return admissionErr + } + } + return nil + } + capacity, err := storage.ProbeFilesystemCapacity(m.ProjectRoot, m.currentTime()) + if err != nil { + measurementErr := fmt.Errorf("storage surface %q cannot be measured before %s: %w\n\nInspect storage with:\n %s", m.ProjectRoot, operation, err, invocation.Command("storage", "status", "--provider", m.Config.Provider.Type)) + if m.AllowInsufficientStorage { + m.warnStorageOverride(operation, measurementErr) + return nil + } + return measurementErr + } + surface := storage.Surface{ + ID: "project", + Provider: m.Config.Provider.Type, + Kind: storage.SurfaceHostFilesystem, + Location: m.ProjectRoot, + Classification: "physical", + Confidence: "authoritative-filesystem-probe", + AdmissionAuthoritative: true, + Capacity: capacity, + } + requirement := storage.Requirement{ + ID: operation, + Provider: m.Config.Provider.Type, + SurfaceID: surface.ID, + PeakBytes: peakBytes, + MinimumFreeBytes: minimumFree, + } + check, err := storage.EvaluateCapacity(surface, requirement) + if err != nil { + return fmt.Errorf("evaluate storage capacity before %s: %w", operation, err) + } + if check.Status != storage.CapacityReady { + if !housekeepingRetried && m.AutomaticImageLifecycle && operation == "instance-create" { + if cleanupErr := m.imageCoordinator().HousekeepStorage(context.Background()); cleanupErr != nil { + m.warnf("EPAR storage housekeeping retry before %s was deferred: %v\n", operation, cleanupErr) + } + return m.preflightStorageAttempt(operation, peakBytes, true) + } + admissionErr := storageAdmissionError(operation, surface, requirement, check, m.Config.Provider.Type, m.StorageOverrideCommand) + if m.AllowInsufficientStorage { + m.warnStorageOverride(operation, admissionErr) + return nil + } + return admissionErr + } + return nil +} + +func (m *Manager) instanceCreateExpansion() uint64 { + return uint64(instanceCreateExpansionBytes) +} + +func storageAdmissionError(operation string, surface storage.Surface, requirement storage.Requirement, check storage.CapacityCheck, providerType, overrideCommand string) error { + action := "complete " + strings.ReplaceAll(operation, "-", " ") + if operation == "instance-create" { + action = "initialize the runner" + } + err := storage.CapacityAdmissionError(action, surface, requirement, check, invocation.Command("storage", "prune", "--provider", providerType)) + if overrideCommand != "" { + return fmt.Errorf("%w\n\nContinue this invocation despite the storage risk with:\n %s", err, overrideCommand) + } + return err +} + +func (m *Manager) warnStorageOverride(operation string, err error) { + m.warnf("\n*** STORAGE SAFETY OVERRIDE ACTIVE ***\n%s\nContinuing %s because --allow-insufficient-storage was explicitly supplied for this invocation.\n\n", err, strings.ReplaceAll(operation, "-", " ")) +} + +// PreflightProviderStorage lets provider-side controllers apply the same +// fail-closed reserve rule to an exact operation before provider side effects. +func (m *Manager) PreflightProviderStorage(operation string, peakBytes uint64) error { + return m.preflightStorage(operation, peakBytes) +} diff --git a/internal/pool/storage_preflight_test.go b/internal/pool/storage_preflight_test.go new file mode 100644 index 0000000..7c1cb9a --- /dev/null +++ b/internal/pool/storage_preflight_test.go @@ -0,0 +1,109 @@ +package pool + +import ( + "context" + "strings" + "sync/atomic" + "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + poolstate "github.com/solutionforest/ephemeral-action-runner/internal/pool/state" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +type insufficientStorage struct{} + +func (insufficientStorage) StorageSnapshot(context.Context, provider.StorageRequest) (provider.StorageSnapshot, error) { + return provider.StorageSnapshot{ + Surfaces: []storage.Surface{{ + ID: "provider-backing", + Provider: "docker-sandboxes", + Kind: storage.SurfaceHostFilesystem, + Location: "test-backing", + Capacity: storage.Capacity{Known: true, TotalBytes: 100 * storage.GiB, AvailableBytes: 30 * storage.GiB}, + }}, + Requirements: []storage.Requirement{{ + ID: "instance-create-provider-backing", + Provider: "docker-sandboxes", + SurfaceID: "provider-backing", + PeakBytes: 20 * storage.GiB, + MinimumFreeBytes: 20 * storage.GiB, + }}, + }, nil +} + +func TestRunPoolCapacityRejectionDoesNotCleanupUncreatedInstance(t *testing.T) { + t.Setenv("EPAR_INVOCATION", "start") + state, err := poolstate.Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + fake := &fakeProvider{} + manager := Manager{ + Config: config.Config{ + Provider: config.ProviderConfig{Type: "docker-sandboxes"}, + Pool: config.PoolConfig{Instances: 1, NamePrefix: "epar-capacity"}, + Logging: config.LoggingConfig{Directory: t.TempDir()}, + Storage: config.StorageConfig{MinimumFree: "20GiB"}, + }, + Provider: fake, + Lifecycle: provider.AdaptLegacy(fake), + LifecycleState: state, + Storage: insufficientStorage{}, + ProjectRoot: t.TempDir(), + } + + err = manager.RunPool(context.Background(), RunOptions{Instances: 1, PoolLockHeld: true, HostTrustLockHeld: true}) + if err == nil || !strings.Contains(err.Error(), "not enough disk space to initialize the runner") { + t.Fatalf("RunPool() error = %v, want capacity rejection", err) + } + for _, want := range []string{ + "Storage location: test-backing", + "Available: 30.00 GiB", + "Estimated operation growth: 20.00 GiB", + "Free-space reserve: 20.00 GiB", + "Required before starting: 40.00 GiB", + "Additional space needed: 10.00 GiB", + "./start storage prune --provider docker-sandboxes", + } { + if !strings.Contains(err.Error(), want) { + t.Errorf("RunPool() error = %q, want %q", err, want) + } + } + if strings.Contains(err.Error(), "32212254720") { + t.Fatalf("RunPool() error contains raw byte count: %v", err) + } + if strings.Contains(err.Error(), "lifecycle record not found") { + t.Fatalf("RunPool() appended cleanup error for an uncreated instance: %v", err) + } + if got := atomic.LoadInt32(&fake.stopCalls); got != 0 { + t.Fatalf("provider stop calls = %d, want 0", got) + } + if got := atomic.LoadInt32(&fake.deleteCalls); got != 0 { + t.Fatalf("provider delete calls = %d, want 0", got) + } + records, err := state.List(context.Background()) + if err != nil { + t.Fatal(err) + } + if len(records) != 0 { + t.Fatalf("lifecycle records = %+v, want none before capacity admission", records) + } +} + +func TestStorageOverrideContinuesOnlyStorageAdmission(t *testing.T) { + manager := Manager{ + Config: config.Config{ + Provider: config.ProviderConfig{Type: "docker-sandboxes"}, + Storage: config.StorageConfig{MinimumFree: "1GiB"}, + }, + Storage: insufficientStorage{}, + AllowInsufficientStorage: true, + StorageOverrideCommand: "./start --allow-insufficient-storage", + ProjectRoot: t.TempDir(), + } + if err := manager.preflightStorage("instance-create", 20*storage.GiB); err != nil { + t.Fatalf("storage override rejected storage-only admission: %v", err) + } +} diff --git a/internal/pool/trusted_ca_test.go b/internal/pool/trusted_ca_test.go index c6e10dc..1017024 100644 --- a/internal/pool/trusted_ca_test.go +++ b/internal/pool/trusted_ca_test.go @@ -18,7 +18,7 @@ import ( "github.com/solutionforest/ephemeral-action-runner/internal/config" ) -func TestDockerDindBuildContextInstallsTrustedCABeforeNetworkSteps(t *testing.T) { +func TestDockerContainerBuildContextInstallsTrustedCABeforeNetworkSteps(t *testing.T) { root := t.TempDir() for _, dir := range []string{ filepath.Join(root, "scripts", "guest", "ubuntu"), @@ -38,7 +38,7 @@ func TestDockerDindBuildContextInstallsTrustedCABeforeNetworkSteps(t *testing.T) ProjectRoot: root, } buildContext := t.TempDir() - if err := manager.prepareDockerDindBuildContext(buildContext, t.TempDir(), `{"hash":"test"}`+"\n"); err != nil { + if err := manager.prepareDockerContainerBuildContext(buildContext, t.TempDir(), `{"hash":"test"}`+"\n"); err != nil { t.Fatal(err) } @@ -172,7 +172,7 @@ func TestTrustedCACertificateDigestInvalidatesImageManifest(t *testing.T) { }, ProjectRoot: root, } - manifest, err := manager.desiredImageManifest(context.Background()) + manifest, err := manager.desiredLocalImageManifest(context.Background()) if err != nil { t.Fatal(err) } @@ -184,7 +184,7 @@ func TestTrustedCACertificateDigestInvalidatesImageManifest(t *testing.T) { t.Fatal(err) } writeTestCACertificate(t, certificatePath, "Enterprise Root Two") - manifest, err = manager.desiredImageManifest(context.Background()) + manifest, err = manager.desiredLocalImageManifest(context.Background()) if err != nil { t.Fatal(err) } diff --git a/internal/provider/dockerdind/docker_dind.go b/internal/provider/dockercontainer/docker_container.go similarity index 90% rename from internal/provider/dockerdind/docker_dind.go rename to internal/provider/dockercontainer/docker_container.go index 0beb08b..a219e21 100644 --- a/internal/provider/dockerdind/docker_dind.go +++ b/internal/provider/dockercontainer/docker_container.go @@ -1,4 +1,4 @@ -package dockerdind +package dockercontainer import ( "bytes" @@ -14,7 +14,7 @@ import ( const ( labelManaged = "epar.managed=true" - labelProvider = "epar.provider=docker-dind" + labelProvider = "epar.provider=docker-container" ) type Provider struct { @@ -47,7 +47,7 @@ func (p *Provider) Clone(ctx context.Context, source, name string) error { func (p *Provider) Start(ctx context.Context, name string, opts provider.StartOptions) (*provider.RunningProcess, error) { if opts.Network != "" && opts.Network != "default" { - return nil, fmt.Errorf("unsupported docker-dind network mode %q", opts.Network) + return nil, fmt.Errorf("unsupported docker-container network mode %q", opts.Network) } if _, err := p.run(ctx, nil, "start", name); err != nil { return nil, err @@ -60,7 +60,7 @@ func (p *Provider) Start(ctx context.Context, name string, opts provider.StartOp func (p *Provider) Exec(ctx context.Context, name string, command []string, opts provider.ExecOptions) (provider.ExecResult, error) { args := []string{"exec"} - if opts.Stdin != "" { + if opts.Stdin != "" || opts.StdinReader != nil { args = append(args, "-i") } for key, value := range opts.Env { @@ -69,7 +69,9 @@ func (p *Provider) Exec(ctx context.Context, name string, command []string, opts args = append(args, name) args = append(args, command...) var stdin io.Reader - if opts.Stdin != "" { + if opts.StdinReader != nil { + stdin = opts.StdinReader + } else if opts.Stdin != "" { stdin = strings.NewReader(opts.Stdin) } return p.runWithSensitiveLog(ctx, stdin, opts.LogPath, opts.Stdout, opts.Stderr, opts.SensitiveValues, args...) @@ -98,7 +100,7 @@ func (p *Provider) IP(ctx context.Context, name string, waitSeconds int) (string if lastErr != nil { return "", lastErr } - return "", fmt.Errorf("docker-dind container %q did not report an IP within %d seconds", name, waitSeconds) + return "", fmt.Errorf("Docker Container %q did not report an IP within %d seconds", name, waitSeconds) } select { case <-ctx.Done(): @@ -125,7 +127,7 @@ func (p *Provider) Delete(ctx context.Context, name string) error { } func (p *Provider) List(ctx context.Context) ([]provider.Instance, error) { - result, err := p.run(ctx, nil, "ps", "-a", "--filter", "label="+labelProvider, "--format", "{{.Names}}\t{{.Image}}\t{{.Status}}") + result, err := p.run(ctx, nil, "ps", "-a", "--no-trunc", "--filter", "label="+labelProvider, "--format", "{{.ID}}\t{{.Names}}\t{{.Image}}\t{{.Status}}") if err != nil { return nil, err } @@ -135,11 +137,11 @@ func (p *Provider) List(ctx context.Context) ([]provider.Instance, error) { if line == "" { continue } - fields := strings.SplitN(line, "\t", 3) - if len(fields) < 3 { + fields := strings.SplitN(line, "\t", 4) + if len(fields) < 4 || strings.TrimSpace(fields[0]) == "" { continue } - out = append(out, provider.Instance{Name: fields[0], Source: fields[1], State: fields[2]}) + out = append(out, provider.Instance{ProviderID: "docker:" + fields[0], Name: fields[1], Source: fields[2], State: fields[3]}) } return out, nil } diff --git a/internal/provider/dockerdind/docker_dind_test.go b/internal/provider/dockercontainer/docker_container_test.go similarity index 87% rename from internal/provider/dockerdind/docker_dind_test.go rename to internal/provider/dockercontainer/docker_container_test.go index 306bbf9..649ba17 100644 --- a/internal/provider/dockerdind/docker_dind_test.go +++ b/internal/provider/dockercontainer/docker_container_test.go @@ -1,4 +1,4 @@ -package dockerdind +package dockercontainer import ( "bytes" @@ -42,18 +42,18 @@ func TestCreateArgsUsePrivilegedWithoutHostSocketOrPorts(t *testing.T) { "HTTPS_PROXY": "http://proxy.example.test:3128", "HTTP_PROXY": "http://proxy.example.test:3128", }, true) - args := p.createArgs("runner-image", "epar-dind-1") + args := p.createArgs("runner-image", "epar-docker-container-1") joined := strings.Join(args, " ") for _, want := range []string{ "create", "--platform linux/arm64", - "--name epar-dind-1", + "--name epar-docker-container-1", "--privileged", "--add-host host.docker.internal:host-gateway", "--env HTTP_PROXY=http://proxy.example.test:3128", "--env HTTPS_PROXY=http://proxy.example.test:3128", "--env NO_PROXY=localhost,127.0.0.1", - "--label epar.provider=docker-dind", + "--label epar.provider=docker-container", "runner-image", } { if !strings.Contains(joined, want) { @@ -79,14 +79,14 @@ func TestExecArgsPreserveEnvAndStdin(t *testing.T) { gotStdin = stdin != nil return provider.ExecResult{}, nil } - _, err := p.Exec(context.Background(), "epar-dind-1", []string{"bash", "-lc", "echo ok"}, provider.ExecOptions{ + _, err := p.Exec(context.Background(), "epar-docker-container-1", []string{"bash", "-lc", "echo ok"}, provider.ExecOptions{ Stdin: "input", Env: map[string]string{"A": "B"}, }) if err != nil { t.Fatal(err) } - want := []string{"exec", "-i", "-e", "A=B", "epar-dind-1", "bash", "-lc", "echo ok"} + want := []string{"exec", "-i", "-e", "A=B", "epar-docker-container-1", "bash", "-lc", "echo ok"} if !reflect.DeepEqual(gotArgs, want) { t.Fatalf("exec args = %#v, want %#v", gotArgs, want) } @@ -135,16 +135,16 @@ func TestExecRedactsSecretAssignmentsWithoutSensitiveValues(t *testing.T) { func TestListParsesDockerPSOutput(t *testing.T) { p := New("docker", "", false) p.runCommand = func(_ context.Context, _ io.Reader, _ string, _, _ io.Writer, args ...string) (provider.ExecResult, error) { - if strings.Join(args, " ") != "ps -a --filter label=epar.provider=docker-dind --format {{.Names}}\t{{.Image}}\t{{.Status}}" { + if strings.Join(args, " ") != "ps -a --no-trunc --filter label=epar.provider=docker-container --format {{.ID}}\t{{.Names}}\t{{.Image}}\t{{.Status}}" { t.Fatalf("unexpected list args: %#v", args) } - return provider.ExecResult{Stdout: "epar-dind-1\trunner-image\tUp 2 minutes\n"}, nil + return provider.ExecResult{Stdout: "0123456789abcdef\tepar-docker-container-1\trunner-image\tUp 2 minutes\n"}, nil } instances, err := p.List(context.Background()) if err != nil { t.Fatal(err) } - want := []provider.Instance{{Name: "epar-dind-1", Source: "runner-image", State: "Up 2 minutes"}} + want := []provider.Instance{{Name: "epar-docker-container-1", ProviderID: "docker:0123456789abcdef", Source: "runner-image", State: "Up 2 minutes"}} if !reflect.DeepEqual(instances, want) { t.Fatalf("instances = %#v, want %#v", instances, want) } @@ -152,7 +152,7 @@ func TestListParsesDockerPSOutput(t *testing.T) { func TestDryRunIPReturnsPlaceholder(t *testing.T) { p := New("docker", "", true) - ip, err := p.IP(context.Background(), "epar-dind-1", 1) + ip, err := p.IP(context.Background(), "epar-docker-container-1", 1) if err != nil { t.Fatal(err) } diff --git a/internal/provider/dockersandboxes/capacity/capacity.go b/internal/provider/dockersandboxes/capacity/capacity.go new file mode 100644 index 0000000..87d46bd --- /dev/null +++ b/internal/provider/dockersandboxes/capacity/capacity.go @@ -0,0 +1,174 @@ +// Package capacity derives Docker Sandboxes resource floors and evaluates +// host admission without performing external side effects. +package capacity + +import ( + "errors" + "fmt" + "math" +) + +const ( + GiB uint64 = 1 << 30 + MinimumRootDisk = 20 * GiB + RootWritableHeadroom = 20 * GiB + CustomizationAllowance = 5 * GiB + DefaultDockerDisk = 50 * GiB + MinimumDockerDisk = 1 * GiB + MinimumHostFreeSpace = 1 * GiB + rootRoundingQuantum = 10 * GiB +) + +// Requirements contains minimum sizes derived from one measured template and +// workload. These values are evidence inputs to a promotion manifest, not +// guesses made at provisioning time. +type Requirements struct { + RootDisk uint64 + DockerDisk uint64 + MinHostFreeSpace uint64 +} + +type HostSpace struct { + AvailableBytes uint64 + TotalBytes uint64 +} + +// HostWatermark returns the configured fixed physical free-space reserve. The +// backing volume size is deliberately not used: a percentage reserve produces +// nonsensical requirements on large volumes. +func HostWatermark(configured, backingVolumeSize uint64) (uint64, error) { + _ = backingVolumeSize + if configured == 0 { + return MinimumHostFreeSpace, nil + } + return configured, nil +} + +// Derive calculates the resource floors required by the Docker Sandboxes +// promotion plan. The 25 percent additions use ceiling arithmetic so integer +// truncation can never weaken a floor. +func Derive(measuredRootPeak, representativeDockerPeak, backingVolumeSize uint64) (Requirements, error) { + if measuredRootPeak == 0 { + return Requirements{}, errors.New("measured root peak must be greater than zero") + } + if representativeDockerPeak == 0 { + return Requirements{}, errors.New("representative Docker peak must be greater than zero") + } + rootDisk, err := DeriveRootDisk(measuredRootPeak, RootWritableHeadroom) + if err != nil { + return Requirements{}, fmt.Errorf("derive root disk: %w", err) + } + _ = representativeDockerPeak + dockerDisk := DefaultDockerDisk + hostWatermark, err := HostWatermark(0, backingVolumeSize) + if err != nil { + return Requirements{}, fmt.Errorf("derive host watermark: %w", err) + } + return Requirements{RootDisk: rootDisk, DockerDisk: dockerDisk, MinHostFreeSpace: hostWatermark}, nil +} + +// DeriveRootDisk turns the expanded source-image estimate and writable +// headroom into the total root-disk capacity passed to Docker Sandboxes. +func DeriveRootDisk(measuredRootPeak, writableHeadroom uint64) (uint64, error) { + if measuredRootPeak == 0 { + return 0, errors.New("measured root peak must be greater than zero") + } + if writableHeadroom < RootWritableHeadroom { + return 0, fmt.Errorf("writable root headroom must be at least %d", RootWritableHeadroom) + } + if measuredRootPeak > math.MaxUint64-CustomizationAllowance { + return 0, errors.New("customization allowance overflows uint64") + } + rootWithAllowance := measuredRootPeak + CustomizationAllowance + if rootWithAllowance > math.MaxUint64-writableHeadroom { + return 0, errors.New("writable headroom overflows uint64") + } + rootWithMargin := rootWithAllowance + writableHeadroom + rootDisk, err := roundUp(rootWithMargin, rootRoundingQuantum) + if err != nil { + return 0, err + } + if rootDisk < MinimumRootDisk { + rootDisk = MinimumRootDisk + } + return rootDisk, nil +} + +func addPercentAndHeadroom(measured, headroom uint64) (uint64, error) { + quarter := ceilDiv(measured, 4) + if measured > math.MaxUint64-quarter { + return 0, errors.New("25 percent margin overflows uint64") + } + withMargin := measured + quarter + if withMargin > math.MaxUint64-headroom { + return 0, errors.New("headroom addition overflows uint64") + } + return withMargin + headroom, nil +} + +func roundUp(value, quantum uint64) (uint64, error) { + if quantum == 0 { + return 0, errors.New("rounding quantum must be greater than zero") + } + remainder := value % quantum + if remainder == 0 { + return value, nil + } + increment := quantum - remainder + if value > math.MaxUint64-increment { + return 0, errors.New("rounding overflows uint64") + } + return value + increment, nil +} + +func ceilDiv(value, divisor uint64) uint64 { + return value/divisor + boolToUint64(value%divisor != 0) +} + +func boolToUint64(value bool) uint64 { + if value { + return 1 + } + return 0 +} + +// Admission is a point-in-time host capacity decision. ReservedBytes includes +// all live and uncertain ledger reservations; uncertainty must never release a +// reservation. +type Admission struct { + HostFreeBytes uint64 + ReservedBytes uint64 + RequestedBytes uint64 + MinHostFreeSpace uint64 + ActiveCreates int + MaxConcurrentCreates int +} + +// Check rejects an admission when either the create-concurrency ceiling or the +// post-reservation physical free-space watermark would be crossed. +func (a Admission) Check() error { + if a.MaxConcurrentCreates <= 0 { + return errors.New("max concurrent creates must be greater than zero") + } + if a.ActiveCreates < 0 { + return errors.New("active creates must not be negative") + } + if a.ActiveCreates >= a.MaxConcurrentCreates { + return fmt.Errorf("create concurrency exhausted: %d active, limit %d", a.ActiveCreates, a.MaxConcurrentCreates) + } + if a.MinHostFreeSpace < MinimumHostFreeSpace { + return fmt.Errorf("host free-space watermark %d is below the hard minimum %d", a.MinHostFreeSpace, MinimumHostFreeSpace) + } + if a.ReservedBytes > a.HostFreeBytes { + return errors.New("existing reservations exceed reported host free space") + } + available := a.HostFreeBytes - a.ReservedBytes + if a.RequestedBytes > available { + return fmt.Errorf("requested reservation %d exceeds unreserved host free space %d", a.RequestedBytes, available) + } + remaining := available - a.RequestedBytes + if remaining < a.MinHostFreeSpace { + return fmt.Errorf("requested reservation would leave %d bytes, below watermark %d", remaining, a.MinHostFreeSpace) + } + return nil +} diff --git a/internal/provider/dockersandboxes/capacity/capacity_test.go b/internal/provider/dockersandboxes/capacity/capacity_test.go new file mode 100644 index 0000000..85c2bd0 --- /dev/null +++ b/internal/provider/dockersandboxes/capacity/capacity_test.go @@ -0,0 +1,140 @@ +package capacity + +import ( + "math" + "testing" +) + +func TestDerivePromotionRequirementsFromMeasuredGuestUsage(t *testing.T) { + requirements, err := Derive(72*GiB, 80*GiB, 2_000*GiB) + if err != nil { + t.Fatal(err) + } + if got, want := requirements.RootDisk, 100*GiB; got != want { + t.Fatalf("root disk = %d, want %d", got, want) + } + if got, want := requirements.DockerDisk, DefaultDockerDisk; got != want { + t.Fatalf("Docker disk = %d, want %d", got, want) + } + if got, want := requirements.MinHostFreeSpace, MinimumHostFreeSpace; got != want { + t.Fatalf("host watermark = %d, want %d", got, want) + } +} + +func TestDeriveRootDiskSeparatesMeasuredPeakFromHeadroom(t *testing.T) { + got, err := DeriveRootDisk(324_780_032, RootWritableHeadroom) + if err != nil { + t.Fatal(err) + } + if want := uint64(30 * GiB); got != want { + t.Fatalf("root disk = %d, want %d", got, want) + } + for _, test := range []struct { + name string + peak uint64 + headroom uint64 + }{ + {name: "missing peak", headroom: RootWritableHeadroom}, + {name: "headroom below preview floor", peak: GiB, headroom: GiB}, + {name: "overflow", peak: math.MaxUint64, headroom: RootWritableHeadroom}, + } { + t.Run(test.name, func(t *testing.T) { + if _, err := DeriveRootDisk(test.peak, test.headroom); err == nil { + t.Fatal("DeriveRootDisk accepted invalid input") + } + }) + } +} + +func TestDeriveEnforcesAbsoluteFloors(t *testing.T) { + requirements, err := Derive(1*GiB, 1*GiB, 100*GiB) + if err != nil { + t.Fatal(err) + } + if got, want := requirements.RootDisk, 30*GiB; got != want { + t.Fatalf("root disk = %d, want %d", got, want) + } + if got, want := requirements.DockerDisk, DefaultDockerDisk; got != want { + t.Fatalf("Docker disk = %d, want %d", got, want) + } + if got, want := requirements.MinHostFreeSpace, MinimumHostFreeSpace; got != want { + t.Fatalf("host watermark = %d, want %d", got, want) + } +} + +func TestDeriveRejectsMissingAndOverflowingEvidence(t *testing.T) { + for _, test := range []struct { + name string + template, peak, disk uint64 + }{ + {name: "template", template: 0, peak: GiB, disk: GiB}, + {name: "peak", template: GiB, peak: 0, disk: GiB}, + {name: "overflow", template: math.MaxUint64, peak: GiB, disk: GiB}, + } { + t.Run(test.name, func(t *testing.T) { + if _, err := Derive(test.template, test.peak, test.disk); err == nil { + t.Fatal("Derive accepted invalid evidence") + } + }) + } +} + +func TestHostWatermarkUsesFixedConfiguredReserve(t *testing.T) { + tests := []struct { + name string + configured uint64 + volume uint64 + want uint64 + }{ + {name: "two terabyte volume uses fixed default", volume: 2_000 * GiB, want: 1 * GiB}, + {name: "twenty terabyte volume uses fixed default", volume: 20_000 * GiB, want: 1 * GiB}, + {name: "configured strengthening", configured: 250 * GiB, volume: 2_000 * GiB, want: 250 * GiB}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + got, err := HostWatermark(test.configured, test.volume) + if err != nil { + t.Fatal(err) + } + if got != test.want { + t.Fatalf("watermark = %d, want %d", got, test.want) + } + }) + } + if got, err := HostWatermark(0, 0); err != nil || got != MinimumHostFreeSpace { + t.Fatalf("HostWatermark(0, 0) = %d, %v; want fixed default", got, err) + } +} + +func TestAdmissionAccountsForReservationsAndWatermark(t *testing.T) { + valid := Admission{ + HostFreeBytes: 500 * GiB, + ReservedBytes: 100 * GiB, + RequestedBytes: 120 * GiB, + MinHostFreeSpace: 200 * GiB, + ActiveCreates: 1, + MaxConcurrentCreates: 2, + } + if err := valid.Check(); err != nil { + t.Fatalf("valid admission rejected: %v", err) + } + + tests := []struct { + name string + mutate func(*Admission) + }{ + {name: "concurrency", mutate: func(a *Admission) { a.ActiveCreates = 2 }}, + {name: "weak watermark", mutate: func(a *Admission) { a.MinHostFreeSpace = 0 }}, + {name: "uncertain reservations", mutate: func(a *Admission) { a.ReservedBytes = 450 * GiB }}, + {name: "post-reservation watermark", mutate: func(a *Admission) { a.RequestedBytes = 250 * GiB }}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + candidate := valid + test.mutate(&candidate) + if err := candidate.Check(); err == nil { + t.Fatal("admission unexpectedly passed") + } + }) + } +} diff --git a/internal/provider/dockersandboxes/capacity/storage_darwin.go b/internal/provider/dockersandboxes/capacity/storage_darwin.go new file mode 100644 index 0000000..0df8c26 --- /dev/null +++ b/internal/provider/dockersandboxes/capacity/storage_darwin.go @@ -0,0 +1,22 @@ +//go:build darwin + +package capacity + +import ( + "errors" + "os" + "path/filepath" +) + +// DockerSandboxesStorageRoot returns the documented macOS state root that +// contains Docker Sandboxes' persistent sandbox data. +func DockerSandboxesStorageRoot() (string, error) { + home, err := os.UserHomeDir() + if err != nil { + return "", err + } + if !filepath.IsAbs(home) { + return "", errors.New("home directory is not an absolute path") + } + return filepath.Join(home, "Library", "Application Support", "com.docker.sandboxes"), nil +} diff --git a/internal/provider/dockersandboxes/capacity/storage_linux.go b/internal/provider/dockersandboxes/capacity/storage_linux.go new file mode 100644 index 0000000..a013bdd --- /dev/null +++ b/internal/provider/dockersandboxes/capacity/storage_linux.go @@ -0,0 +1,26 @@ +//go:build linux + +package capacity + +import ( + "errors" + "os" + "path/filepath" +) + +// DockerSandboxesStorageRoot returns the documented XDG state root that +// contains Docker Sandboxes' persistent sandbox data. +func DockerSandboxesStorageRoot() (string, error) { + stateHome := os.Getenv("XDG_STATE_HOME") + if stateHome == "" { + home, err := os.UserHomeDir() + if err != nil { + return "", err + } + stateHome = filepath.Join(home, ".local", "state") + } + if !filepath.IsAbs(stateHome) { + return "", errors.New("XDG state home is not an absolute path") + } + return filepath.Join(stateHome, "sandboxes"), nil +} diff --git a/internal/provider/dockersandboxes/capacity/storage_linux_test.go b/internal/provider/dockersandboxes/capacity/storage_linux_test.go new file mode 100644 index 0000000..1812c1a --- /dev/null +++ b/internal/provider/dockersandboxes/capacity/storage_linux_test.go @@ -0,0 +1,27 @@ +//go:build linux + +package capacity + +import ( + "path/filepath" + "testing" +) + +func TestDockerSandboxesStorageRootUsesXDGStateHome(t *testing.T) { + stateHome := t.TempDir() + t.Setenv("XDG_STATE_HOME", stateHome) + got, err := DockerSandboxesStorageRoot() + if err != nil { + t.Fatal(err) + } + if want := filepath.Join(stateHome, "sandboxes"); got != want { + t.Fatalf("storage root = %q, want %q", got, want) + } +} + +func TestDockerSandboxesStorageRootRejectsRelativeXDGStateHome(t *testing.T) { + t.Setenv("XDG_STATE_HOME", "relative-state") + if _, err := DockerSandboxesStorageRoot(); err == nil { + t.Fatal("relative XDG_STATE_HOME was accepted") + } +} diff --git a/internal/provider/dockersandboxes/capacity/storage_other.go b/internal/provider/dockersandboxes/capacity/storage_other.go new file mode 100644 index 0000000..71a5b39 --- /dev/null +++ b/internal/provider/dockersandboxes/capacity/storage_other.go @@ -0,0 +1,12 @@ +//go:build !windows && !linux && !darwin + +package capacity + +import ( + "fmt" + "runtime" +) + +func DockerSandboxesStorageRoot() (string, error) { + return "", fmt.Errorf("Docker Sandboxes storage discovery is unsupported on %s", runtime.GOOS) +} diff --git a/internal/provider/dockersandboxes/capacity/storage_windows.go b/internal/provider/dockersandboxes/capacity/storage_windows.go new file mode 100644 index 0000000..b84afe0 --- /dev/null +++ b/internal/provider/dockersandboxes/capacity/storage_windows.go @@ -0,0 +1,22 @@ +//go:build windows + +package capacity + +import ( + "errors" + "os" + "path/filepath" +) + +// DockerSandboxesStorageRoot returns the documented Windows root that contains +// Docker Sandboxes' persistent state, cache, configuration, and sandbox data. +func DockerSandboxesStorageRoot() (string, error) { + localAppData := os.Getenv("LOCALAPPDATA") + if localAppData == "" { + return "", errors.New("LOCALAPPDATA is unavailable") + } + if !filepath.IsAbs(localAppData) { + return "", errors.New("LOCALAPPDATA is not an absolute path") + } + return filepath.Join(localAppData, "DockerSandboxes"), nil +} diff --git a/internal/provider/dockersandboxes/doc.go b/internal/provider/dockersandboxes/doc.go new file mode 100644 index 0000000..1347be5 --- /dev/null +++ b/internal/provider/dockersandboxes/doc.go @@ -0,0 +1,7 @@ +// Package dockersandboxes implements Docker Sandboxes host integration. +// +// Provider-owned capacity, staging, ownership, policy, and certification logic +// lives in the child packages below this directory. Pool-owned GitHub +// registration, readiness, replacement, status, and cleanup orchestration +// remains in internal/pool. +package dockersandboxes diff --git a/internal/provider/dockersandboxes/json_parsing.go b/internal/provider/dockersandboxes/json_parsing.go new file mode 100644 index 0000000..d9c9b5d --- /dev/null +++ b/internal/provider/dockersandboxes/json_parsing.go @@ -0,0 +1,267 @@ +package dockersandboxes + +import ( + "bytes" + "encoding/json" + "fmt" + "io" + "regexp" + "strconv" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +var cachedTemplateIDPattern = regexp.MustCompile(`^[a-f0-9]{12}$`) + +type cachedTemplate struct { + ID string + Repository string + Tag string + Flavor string + CreatedAt time.Time + Size int64 +} + +func parseTemplateInventory(data []byte) ([]cachedTemplate, error) { + var wrapper map[string]json.RawMessage + if err := decodeStrictJSON(data, &wrapper); err != nil { + return nil, fmt.Errorf("docker sandbox template inventory returned an unsupported json schema") + } + rawImages, ok := wrapper["images"] + if !ok || bytes.Equal(bytes.TrimSpace(rawImages), []byte("null")) { + return nil, fmt.Errorf("docker sandbox template inventory returned an unsupported json schema") + } + var records []map[string]json.RawMessage + if err := json.Unmarshal(rawImages, &records); err != nil { + return nil, fmt.Errorf("docker sandbox template inventory returned an unsupported json schema") + } + images := make([]cachedTemplate, 0, len(records)) + seenIDs := make(map[string]struct{}, len(records)) + seenReferences := make(map[string]struct{}, len(records)) + for _, record := range records { + id, idErr := requiredJSONString(record, "id") + repository, repositoryErr := requiredJSONString(record, "repository") + tag, tagErr := requiredJSONString(record, "tag") + flavor, flavorErr := optionalJSONString(record, "flavor") + createdAtText, createdAtErr := requiredJSONString(record, "created_at") + createdAt, timestampErr := time.Parse(time.RFC3339, createdAtText) + var size json.Number + rawSize, sizePresent := record["size"] + sizeErr := json.Unmarshal(rawSize, &size) + sizeValue, integerErr := strconv.ParseInt(size.String(), 10, 64) + if idErr != nil || !cachedTemplateIDPattern.MatchString(id) || repositoryErr != nil || tagErr != nil || flavorErr != nil || createdAtErr != nil || timestampErr != nil || !sizePresent || len(rawSize) == 0 || rawSize[0] < '0' || rawSize[0] > '9' || sizeErr != nil || integerErr != nil || sizeValue <= 0 { + return nil, fmt.Errorf("docker sandbox template inventory returned an unsupported image schema") + } + if !templatePattern.MatchString(repository) || !profilePattern.MatchString(tag) || flavor != "" && !profilePattern.MatchString(flavor) { + return nil, fmt.Errorf("docker sandbox template inventory returned an invalid image identity") + } + reference := repository + ":" + tag + if _, duplicate := seenIDs[id]; duplicate { + return nil, fmt.Errorf("docker sandbox template inventory returned a duplicate image id") + } + if _, duplicate := seenReferences[reference]; duplicate { + return nil, fmt.Errorf("docker sandbox template inventory returned a duplicate image reference") + } + seenIDs[id] = struct{}{} + seenReferences[reference] = struct{}{} + images = append(images, cachedTemplate{ID: id, Repository: repository, Tag: tag, Flavor: flavor, CreatedAt: createdAt, Size: sizeValue}) + } + return images, nil +} + +func parseInventory(data []byte) ([]provider.InventoryItem, error) { + decoder := json.NewDecoder(bytes.NewReader(data)) + decoder.UseNumber() + var wrapper map[string]json.RawMessage + if err := decoder.Decode(&wrapper); err != nil { + return nil, fmt.Errorf("docker sandboxes inventory returned an unsupported json schema") + } + if err := requireJSONEOF(decoder); err != nil { + return nil, fmt.Errorf("docker sandboxes inventory returned an unsupported json schema") + } + var records []map[string]json.RawMessage + rawSandboxes, ok := wrapper["sandboxes"] + if !ok || bytes.Equal(bytes.TrimSpace(rawSandboxes), []byte("null")) || json.Unmarshal(rawSandboxes, &records) != nil { + return nil, fmt.Errorf("docker sandboxes inventory returned an unsupported json schema") + } + items := make([]provider.InventoryItem, 0, len(records)) + seenNames := make(map[string]struct{}, len(records)) + seenIDs := make(map[string]struct{}, len(records)) + for _, record := range records { + id, err := requiredJSONString(record, "id") + if err != nil || !providerIDPattern.MatchString(id) { + return nil, fmt.Errorf("docker sandboxes inventory omitted a valid stable id") + } + name, err := requiredJSONString(record, "name") + if err != nil || !sandboxNamePattern.MatchString(name) { + return nil, fmt.Errorf("docker sandboxes inventory contained an invalid name") + } + status, err := requiredJSONString(record, "status") + if err != nil || strings.TrimSpace(status) == "" { + return nil, fmt.Errorf("docker sandboxes inventory omitted status") + } + agent, err := optionalJSONString(record, "agent") + if err != nil { + return nil, fmt.Errorf("docker sandboxes inventory returned an unsupported agent field") + } + workspaces, err := requiredStringArray(record, "workspaces") + if err != nil || len(workspaces) == 0 { + return nil, fmt.Errorf("docker sandboxes inventory omitted workspaces") + } + if _, duplicate := seenNames[name]; duplicate { + return nil, fmt.Errorf("docker sandboxes inventory returned a duplicate name") + } + if _, duplicate := seenIDs[id]; duplicate { + return nil, fmt.Errorf("docker sandboxes inventory returned a duplicate stable id") + } + seenNames[name] = struct{}{} + seenIDs[id] = struct{}{} + instance := provider.Instance{Name: name, ProviderID: id, Source: agent, State: status} + items = append(items, provider.InventoryItem{Instance: instance, State: status, Source: agent, Workspaces: workspaces}) + } + return items, nil +} + +func requiredStringArray(record map[string]json.RawMessage, key string) ([]string, error) { + raw, ok := record[key] + if !ok { + return nil, fmt.Errorf("missing field") + } + var values []string + if err := json.Unmarshal(raw, &values); err != nil || len(values) == 0 { + return nil, fmt.Errorf("invalid field") + } + seen := make(map[string]struct{}, len(values)) + for _, value := range values { + if strings.TrimSpace(value) == "" || strings.TrimSpace(value) != value || strings.ContainsRune(value, 0) { + return nil, fmt.Errorf("invalid field") + } + if _, duplicate := seen[value]; duplicate { + return nil, fmt.Errorf("duplicate field value") + } + seen[value] = struct{}{} + } + return values, nil +} + +func parseDaemonStatus(data []byte) (state string, healthy bool, err error) { + var record map[string]json.RawMessage + if err := decodeStrictJSON(data, &record); err != nil { + return "", false, fmt.Errorf("docker sandboxes daemon status returned an unsupported json schema") + } + raw, ok := record["status"] + if !ok || json.Unmarshal(raw, &state) != nil || strings.TrimSpace(state) == "" { + return "", false, fmt.Errorf("docker sandboxes daemon status returned an unsupported json schema") + } + normalized := strings.ToLower(state) + return state, normalized == "running", nil +} + +func parseDiagnose(data []byte) (passed, warned, failed, skipped int, err error) { + var record map[string]json.RawMessage + if err := decodeStrictJSON(data, &record); err != nil { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics returned an unsupported json schema") + } + version, versionErr := requiredJSONString(record, "version") + if versionErr != nil || version != "1.0" { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics returned an unsupported schema version") + } + rawChecks, ok := record["checks"] + if !ok { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics returned an unsupported json schema") + } + var checks []map[string]json.RawMessage + if err := json.Unmarshal(rawChecks, &checks); err != nil { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics returned an unsupported json schema") + } + for _, check := range checks { + if _, fieldErr := requiredJSONString(check, "name"); fieldErr != nil { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics returned an unsupported check schema") + } + for _, field := range []string{"message", "detail", "hint"} { + if _, fieldErr := requiredStringField(check, field); fieldErr != nil { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics returned an unsupported check schema") + } + } + status, statusErr := requiredJSONString(check, "status") + if statusErr != nil { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics returned an unsupported check schema") + } + switch strings.ToLower(status) { + case "pass": + passed++ + case "warn": + warned++ + case "fail": + failed++ + case "skip": + skipped++ + default: + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics returned an unknown check status") + } + } + var summary map[string]json.RawMessage + rawSummary, ok := record["summary"] + if !ok || json.Unmarshal(rawSummary, &summary) != nil { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics summary did not match its checks") + } + summaryCounts := make([]int, 4) + for index, key := range []string{"pass", "warn", "fail", "skip"} { + rawCount, present := summary[key] + if !present || json.Unmarshal(rawCount, &summaryCounts[index]) != nil || summaryCounts[index] < 0 { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics summary did not match its checks") + } + } + if summaryCounts[0] != passed || summaryCounts[1] != warned || summaryCounts[2] != failed || summaryCounts[3] != skipped { + return 0, 0, 0, 0, fmt.Errorf("docker sandboxes diagnostics summary did not match its checks") + } + return passed, warned, failed, skipped, nil +} + +func requiredJSONString(record map[string]json.RawMessage, key string) (string, error) { + raw, ok := record[key] + if !ok { + return "", fmt.Errorf("missing field") + } + var value string + if err := json.Unmarshal(raw, &value); err != nil || value == "" { + return "", fmt.Errorf("invalid field") + } + return value, nil +} + +func optionalJSONString(record map[string]json.RawMessage, key string) (string, error) { + raw, ok := record[key] + if !ok || string(raw) == "null" { + return "", nil + } + var value string + if err := json.Unmarshal(raw, &value); err != nil { + return "", err + } + return value, nil +} + +func requiredStringField(record map[string]json.RawMessage, key string) (string, error) { + raw, ok := record[key] + if !ok { + return "", fmt.Errorf("missing field") + } + var value string + if err := json.Unmarshal(raw, &value); err != nil { + return "", fmt.Errorf("invalid field") + } + return value, nil +} + +func requireJSONEOF(decoder *json.Decoder) error { + var extra any + if err := decoder.Decode(&extra); err == nil { + return fmt.Errorf("unexpected trailing json") + } else if err != io.EOF { + return err + } + return nil +} diff --git a/internal/provider/dockersandboxes/live_test.go b/internal/provider/dockersandboxes/live_test.go new file mode 100644 index 0000000..2fce8bf --- /dev/null +++ b/internal/provider/dockersandboxes/live_test.go @@ -0,0 +1,292 @@ +package dockersandboxes + +import ( + "context" + "crypto/rand" + "encoding/hex" + "errors" + "fmt" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + sandboxpolicy "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/policy" + sandboxfs "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/staging" +) + +func TestLiveRunnerTemplateIsolation(t *testing.T) { + if os.Getenv("EPAR_LIVE_DOCKER_SANDBOXES") != "1" { + t.Skip("set EPAR_LIVE_DOCKER_SANDBOXES=1 to run the destructive live Docker Sandboxes proof") + } + template := os.Getenv("EPAR_LIVE_DOCKER_SANDBOXES_TEMPLATE") + digest := os.Getenv("EPAR_LIVE_DOCKER_SANDBOXES_TEMPLATE_DIGEST") + stagingRoot := os.Getenv("EPAR_LIVE_DOCKER_SANDBOXES_STAGING_ROOT") + if template == "" || digest == "" || stagingRoot == "" { + t.Fatal("live Docker Sandboxes template, digest, and absolute staging root are required") + } + staging, err := sandboxfs.Open(stagingRoot) + if err != nil { + t.Fatal(err) + } + name := fmt.Sprintf("epar-live-%d", time.Now().UnixNano()) + ownedStaging, err := staging.CreateOwned(name) + if err != nil { + t.Fatal(err) + } + path := ownedStaging.Path + rootDisk := os.Getenv("EPAR_LIVE_DOCKER_SANDBOXES_ROOT_SIZE") + if rootDisk == "" { + rootDisk = "30GiB" + } + dockerDisk := os.Getenv("EPAR_LIVE_DOCKER_SANDBOXES_DOCKER_SIZE") + if dockerDisk == "" { + dockerDisk = "100GiB" + } + p := New("sbx") + ctx, cancel := context.WithTimeout(context.Background(), 20*time.Minute) + defer cancel() + createStarted := time.Now() + instance, err := p.Create(ctx, provider.CreateRequest{ + Name: name, + Template: template, + TemplateDigest: digest, + StagingPath: path, + CPUs: 4, + Memory: "8GiB", + RootDisk: rootDisk, + DockerDisk: dockerDisk, + }) + if err != nil { + t.Fatal(err) + } + t.Logf("cached sandbox create duration=%s", time.Since(createStarted)) + defer func() { + cleanupCtx, cleanupCancel := context.WithTimeout(context.Background(), 5*time.Minute) + defer cleanupCancel() + cleanupStarted := time.Now() + if stopErr := p.Stop(cleanupCtx, instance); stopErr != nil { + t.Errorf("stop exact live sandbox: %v", stopErr) + } + if deleteErr := p.Delete(cleanupCtx, instance); deleteErr != nil { + t.Errorf("delete exact live sandbox: %v", deleteErr) + } + if _, verifyErr := staging.VerifyOwnedEmpty(name, ownedStaging.Identity); verifyErr != nil { + t.Errorf("verify live staging remains empty: %v", verifyErr) + return + } + if removeErr := staging.RemoveEmptyOwned(name, ownedStaging.Identity); removeErr != nil { + t.Errorf("remove exact live staging: %v", removeErr) + } + t.Logf("exact stop, force-remove, and staging absence duration=%s", time.Since(cleanupStarted)) + }() + if _, err := p.Start(ctx, instance, provider.StartOptions{}); err != nil { + t.Fatal(err) + } + runtimeInfo, err := p.VerifyRuntime(ctx, instance) + if err != nil || !runtimeInfo.Ready || runtimeInfo.Runtime != "docker" || runtimeInfo.Version == "" { + t.Fatalf("private Docker runtime = %#v, error = %v", runtimeInfo, err) + } + daemonManagement, err := p.Exec(ctx, instance, []string{"bash", "-lc", `set -euo pipefail +mapfile -t pids < <(pgrep -x dockerd) +[[ "${#pids[@]}" == "1" ]] +ps -o pid=,ppid=,args= -p "${pids[0]}" +printf 'pid1=' +tr '\0' ' ' /dev/null 2>&1 || true + docker image rm "$image" >/dev/null 2>&1 || true + rm -rf -- "$workdir" +} +trap cleanup EXIT +mkdir -p "$workdir/build" +printf 'FROM scratch\nLABEL org.opencontainers.image.title="EPAR Docker Sandboxes Buildx probe"\n' >"$workdir/build/Dockerfile" +cat >"$workdir/compose.yml" <<'YAML' +services: + probe: + image: docker.io/library/alpine@sha256:14358309a308569c32bdc37e2e0e9694be33a9d99e68afb0f5ff33cc1f695dce + command: ["sleep", "120"] +YAML +docker buildx version +docker buildx inspect default +docker buildx build --builder default --load --tag "$image" "$workdir/build" +docker image inspect "$image" >/dev/null +docker compose version +docker compose --project-name "$project" --file "$workdir/compose.yml" up --detach +[[ "$(docker compose --project-name "$project" --file "$workdir/compose.yml" ps --status running --services)" == "probe" ]] +docker compose --project-name "$project" --file "$workdir/compose.yml" down --remove-orphans --volumes +trap - EXIT +docker image rm "$image" >/dev/null +rm -rf -- "$workdir" +`, "--", name + "-compose"}, provider.ExecOptions{}) + if err != nil { + t.Fatalf("representative Buildx and Compose workload failed: %v\n%s", err, representativeResult.Stderr) + } + t.Logf("representative Buildx and Compose workload duration=%s\n%s", time.Since(representativeStarted), strings.TrimSpace(representativeResult.Stdout)) + diskUsageAfter, err := p.Exec(ctx, instance, []string{"df", "-B1", "--output=used,target", "/", "/var/lib/docker"}, provider.ExecOptions{}) + if err != nil { + t.Fatal(err) + } + t.Logf("sandbox disk usage after representative nested workload:\n%s", strings.TrimSpace(diskUsageAfter.Stdout)) + if _, err := os.Lstat(filepath.Join(path, ".runner")); err == nil { + t.Fatal("runner registration state escaped into the canonical host staging directory") + } else if !os.IsNotExist(err) { + t.Fatal(err) + } +} + +func verifyAuthenticatedRegistryLifecycle(ctx context.Context, p *Provider, instance provider.Instance) error { + registryImage := strings.TrimSpace(os.Getenv("EPAR_LIVE_DOCKER_SANDBOXES_REGISTRY_IMAGE")) + htpasswdImage := strings.TrimSpace(os.Getenv("EPAR_LIVE_DOCKER_SANDBOXES_HTPASSWD_IMAGE")) + if !strings.Contains(registryImage, "@sha256:") || !strings.Contains(htpasswdImage, "@sha256:") { + return errors.New("EPAR_LIVE_DOCKER_SANDBOXES_REGISTRY_IMAGE and EPAR_LIVE_DOCKER_SANDBOXES_HTPASSWD_IMAGE must be immutable digest references") + } + username, err := randomLiveCredential("epar-") + if err != nil { + return err + } + password, err := randomLiveCredential("") + if err != nil { + return err + } + script := `set -euo pipefail +registry_image="$1" +htpasswd_image="$2" +IFS= read -r username +IFS= read -r password +auth_dir="$(mktemp -d)" +registry_name="epar-auth-registry-$$" +registry_ref="" +probe_image="" +cleanup() { + if [[ -n "${registry_ref}" ]]; then docker logout "${registry_ref}" >/dev/null 2>&1 || true; fi + docker rm -f "${registry_name}" >/dev/null 2>&1 || true + if [[ -n "${probe_image}" ]]; then docker image rm "${probe_image}" >/dev/null 2>&1 || true; fi + rm -rf -- "${auth_dir}" +} +trap cleanup EXIT +printf '%s\n' "${password}" | docker run --rm -i -v "${auth_dir}:/auth" "${htpasswd_image}" htpasswd -B -i -c /auth/htpasswd "${username}" >/dev/null +docker run --name "${registry_name}" --detach --publish 127.0.0.1::5000 --env REGISTRY_AUTH=htpasswd --env REGISTRY_AUTH_HTPASSWD_REALM='EPAR live proof' --env REGISTRY_AUTH_HTPASSWD_PATH=/auth/htpasswd --volume "${auth_dir}:/auth:ro" "${registry_image}" >/dev/null +registry_port="$(docker port "${registry_name}" 5000/tcp | awk -F: 'NR == 1 { print $NF }')" +[[ "${registry_port}" =~ ^[0-9]+$ ]] +registry_ref="127.0.0.1:${registry_port}" +probe_image="${registry_ref}/epar/private-pull-proof:latest" +login_succeeded=false +for attempt in $(seq 1 30); do + if printf '%s\n' "${password}" | docker login --username "${username}" --password-stdin "${registry_ref}" >/dev/null 2>&1; then login_succeeded=true; break; fi + if [[ "${attempt}" == 30 ]]; then echo 'authenticated registry did not become ready' >&2; exit 1; fi + sleep 1 +done +[[ "${login_succeeded}" == true ]] +python3 - "${DOCKER_CONFIG}/config.json" "${registry_ref}" <<'PY' +import json +import pathlib +import sys +config = json.loads(pathlib.Path(sys.argv[1]).read_text(encoding="utf-8")) +if sys.argv[2] not in config.get("auths", {}): + raise SystemExit("login did not create the expected Docker auth entry") +PY +docker pull docker.io/library/alpine@sha256:14358309a308569c32bdc37e2e0e9694be33a9d99e68afb0f5ff33cc1f695dce >/dev/null +docker tag docker.io/library/alpine@sha256:14358309a308569c32bdc37e2e0e9694be33a9d99e68afb0f5ff33cc1f695dce "${probe_image}" +docker push "${probe_image}" >/dev/null +docker image rm "${probe_image}" >/dev/null +docker pull "${probe_image}" >/dev/null +docker logout "${registry_ref}" >/dev/null +python3 - "${DOCKER_CONFIG}/config.json" "${registry_ref}" <<'PY' +import json +import pathlib +import sys +path = pathlib.Path(sys.argv[1]) +config = json.loads(path.read_text(encoding="utf-8")) if path.exists() else {} +if sys.argv[2] in config.get("auths", {}): + raise SystemExit("Docker auth entry survived logout") +PY +printf 'authenticated registry login, separate-command pull, and credential cleanup passed\n' +` + result, err := p.Exec(ctx, instance, []string{"bash", "-lc", script, "--", registryImage, htpasswdImage}, provider.ExecOptions{ + Stdin: username + "\n" + password + "\n", + SensitiveValues: []string{username, password}, + }) + if err != nil { + return fmt.Errorf("authenticated local registry proof: %w: %s", err, strings.TrimSpace(result.Stderr)) + } + return nil +} + +func randomLiveCredential(prefix string) (string, error) { + value := make([]byte, 24) + if _, err := rand.Read(value); err != nil { + return "", fmt.Errorf("generate live registry credential: %w", err) + } + return prefix + hex.EncodeToString(value), nil +} diff --git a/internal/provider/dockersandboxes/network_policy.go b/internal/provider/dockersandboxes/network_policy.go new file mode 100644 index 0000000..3f78c91 --- /dev/null +++ b/internal/provider/dockersandboxes/network_policy.go @@ -0,0 +1,373 @@ +package dockersandboxes + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "slices" + "sort" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +func (p *Provider) ApplyNetworkPolicy(ctx context.Context, instance provider.Instance, rules []provider.NetworkPolicyRule) error { + if len(rules) == 0 { + return nil + } + present, err := p.assertIdentity(ctx, instance) + if err != nil { + return err + } + if !present { + return fmt.Errorf("docker sandbox is missing") + } + for _, rule := range rules { + if err := validateNetworkRule(rule, false); err != nil { + return err + } + args := []string{"policy", string(rule.Decision), "network", "--sandbox", instance.Name, strings.Join(rule.Resources, ",")} + result, runErr := p.run(ctx, commandRequest{args: args, operation: "apply docker sandbox network policy"}) + if runErr != nil && !strings.Contains(strings.ToLower(result.Stdout+"\n"+result.Stderr), "already covered") { + return runErr + } + } + actual, err := p.readNetworkPolicyVerified(ctx, instance) + if err != nil { + return err + } + for _, expected := range rules { + if !containsSandboxPolicyRule(actual, expected, instance.Name) { + return fmt.Errorf("docker sandbox network policy readback did not contain an applied rule") + } + } + return nil +} + +func (p *Provider) ReadNetworkPolicy(ctx context.Context, instance provider.Instance) ([]provider.NetworkPolicyRule, error) { + present, err := p.assertIdentity(ctx, instance) + if err != nil { + return nil, err + } + if !present { + return nil, fmt.Errorf("docker sandbox is missing") + } + return p.readNetworkPolicyVerified(ctx, instance) +} + +// ReadGlobalNetworkPolicy returns only the provider-wide policy baseline. It +// never changes policy and preserves the attribution supplied by Docker +// Sandboxes for every returned rule. +func (p *Provider) ReadGlobalNetworkPolicy(ctx context.Context) ([]provider.NetworkPolicyRule, error) { + result, err := p.run(ctx, commandRequest{ + args: []string{"policy", "ls", "--include-inactive", "--json"}, + operation: "read docker sandboxes global network policy", + outputLimit: diagnosticOutputLimit, + }) + if err != nil { + return nil, err + } + return parseGlobalNetworkPolicy([]byte(result.Stdout)) +} + +func (p *Provider) readNetworkPolicyVerified(ctx context.Context, instance provider.Instance) ([]provider.NetworkPolicyRule, error) { + result, err := p.run(ctx, commandRequest{ + args: []string{"policy", "ls", instance.Name, "--include-inactive", "--json"}, + operation: "read docker sandbox network policy", + }) + if err != nil { + return nil, err + } + return parseNetworkPolicy([]byte(result.Stdout), instance.Name) +} + +func (p *Provider) RemoveNetworkPolicy(ctx context.Context, instance provider.Instance, rules []provider.NetworkPolicyRule) error { + if len(rules) == 0 { + return nil + } + present, err := p.assertIdentity(ctx, instance) + if err != nil || !present { + return err + } + actual, err := p.readNetworkPolicyVerified(ctx, instance) + if err != nil { + return err + } + seenIDs := make(map[string]struct{}, len(rules)) + for _, rule := range rules { + if err := validateNetworkRule(rule, true); err != nil { + return err + } + if _, duplicate := seenIDs[rule.ID]; duplicate { + continue + } + seenIDs[rule.ID] = struct{}{} + matched, found := findPolicyRuleID(actual, rule.ID) + if !found { + continue + } + if !isRemovableSandboxPolicyRule(matched, instance.Name) { + return fmt.Errorf("refusing to remove a network policy rule that is not an editable local rule scoped to this sandbox") + } + if !sameStablePolicyRuleIdentity(matched, rule) { + return fmt.Errorf("refusing to remove a network policy rule whose stable identity changed") + } + result, runErr := p.run(ctx, commandRequest{ + args: []string{"policy", "rm", "network", "--sandbox", instance.Name, "--id", rule.ID}, + operation: "remove docker sandbox network policy", + }) + if runErr != nil && !isMissingPolicyRule(result.Stdout+"\n"+result.Stderr+"\n"+runErr.Error()) { + return runErr + } + } + remaining, err := p.readNetworkPolicyVerified(ctx, instance) + if err != nil { + return err + } + for _, removed := range rules { + if containsPolicyRuleID(remaining, removed.ID) { + return fmt.Errorf("docker sandbox network policy rule remained after exact removal") + } + } + return nil +} + +func parseGlobalNetworkPolicy(data []byte) ([]provider.NetworkPolicyRule, error) { + rules, err := parseNetworkPolicy(data, "") + if err != nil { + return nil, err + } + global := make([]provider.NetworkPolicyRule, 0, len(rules)) + for _, rule := range rules { + if rule.Scope == "global" && rule.AppliesTo == "all" { + global = append(global, rule) + } + } + return global, nil +} + +func parseNetworkPolicy(data []byte, sandboxName string) ([]provider.NetworkPolicyRule, error) { + decoder := json.NewDecoder(bytes.NewReader(data)) + var raw json.RawMessage + if err := decoder.Decode(&raw); err != nil || requireJSONEOF(decoder) != nil { + return nil, fmt.Errorf("docker sandbox network policy returned an unsupported json schema") + } + var records []map[string]json.RawMessage + var wrapper map[string]json.RawMessage + if err := json.Unmarshal(raw, &wrapper); err != nil { + return nil, fmt.Errorf("docker sandbox network policy returned an unsupported json schema") + } + rulesJSON, ok := wrapper["rules"] + if !ok || bytes.Equal(bytes.TrimSpace(rulesJSON), []byte("null")) || json.Unmarshal(rulesJSON, &records) != nil { + return nil, fmt.Errorf("docker sandbox network policy returned an unsupported json schema") + } + + out := make([]provider.NetworkPolicyRule, 0, len(records)) + seenIDs := make(map[string]struct{}, len(records)) + for _, record := range records { + ruleType, err := requiredJSONString(record, "resource_type") + if err != nil { + return nil, fmt.Errorf("docker sandbox network policy rule omitted its type") + } + id, err := requiredJSONString(record, "id") + if err != nil || !providerIDPattern.MatchString(id) { + return nil, fmt.Errorf("docker sandbox network policy rule omitted a valid id") + } + decisionText, err := requiredJSONString(record, "decision") + if err != nil { + return nil, fmt.Errorf("docker sandbox network policy rule omitted its decision") + } + decision := provider.NetworkPolicyDecision(strings.ToLower(decisionText)) + if decision != provider.NetworkPolicyAllow && decision != provider.NetworkPolicyDeny { + return nil, fmt.Errorf("docker sandbox network policy rule used an unknown decision") + } + name, err := requiredJSONString(record, "name") + if err != nil { + return nil, fmt.Errorf("docker sandbox network policy rule omitted its name") + } + policyID, err := requiredJSONString(record, "policy_id") + if err != nil { + return nil, fmt.Errorf("docker sandbox network policy rule omitted its policy id") + } + scope, err := requiredJSONString(record, "scope") + if err != nil { + return nil, fmt.Errorf("docker sandbox network policy rule omitted its scope") + } + appliesTo, err := requiredJSONString(record, "applies_to") + if err != nil { + return nil, fmt.Errorf("docker sandbox network policy rule omitted its target") + } + resources, err := policyRuleResources(record) + if err != nil { + return nil, err + } + status, active, err := policyRuleStatus(record) + if err != nil { + return nil, err + } + origin, err := requiredJSONString(record, "origin") + if err != nil { + return nil, fmt.Errorf("docker sandbox network policy rule omitted its origin") + } + var editable bool + if rawEditable, ok := record["editable"]; !ok || json.Unmarshal(rawEditable, &editable) != nil { + return nil, fmt.Errorf("docker sandbox network policy rule omitted editable state") + } + if ruleType == "network" { + for _, resource := range resources { + if err := validateReadbackPolicyResource(resource); err != nil { + return nil, err + } + } + } + if _, duplicate := seenIDs[id]; duplicate { + return nil, fmt.Errorf("docker sandbox network policy returned a duplicate rule id") + } + seenIDs[id] = struct{}{} + if scope != "global" { + sandboxID, sandboxIDErr := requiredJSONString(record, "sandbox_id") + if sandboxIDErr != nil || scope != "sandbox:"+sandboxID || appliesTo != scope { + return nil, fmt.Errorf("docker sandbox policy returned an unsupported sandbox attribution") + } + } + relevant, targetErr := isRelevantPolicyTarget(scope, appliesTo, sandboxName) + if targetErr != nil { + return nil, targetErr + } + if !relevant { + continue + } + out = append(out, provider.NetworkPolicyRule{ + ID: id, + Name: name, + PolicyID: policyID, + Scope: scope, + AppliesTo: appliesTo, + ResourceType: ruleType, + Resources: append([]string(nil), resources...), + Decision: decision, + Origin: origin, + Status: status, + Editable: editable, + Active: active, + }) + } + return out, nil +} + +func validateReadbackPolicyResource(resource string) error { + if resource == "" || resource != strings.TrimSpace(resource) || len(resource) > 2048 || strings.ContainsAny(resource, "\x00\r\n\t") { + return fmt.Errorf("docker sandbox network policy returned an invalid resource") + } + return nil +} + +func policyRuleResources(record map[string]json.RawMessage) ([]string, error) { + raw := record["resources"] + if len(raw) == 0 { + return nil, fmt.Errorf("docker sandbox network policy rule omitted its resource") + } + var many []string + if json.Unmarshal(raw, &many) != nil || len(many) == 0 { + return nil, fmt.Errorf("docker sandbox network policy rule returned an invalid resource") + } + return many, nil +} + +func isSandboxPolicyTarget(scope, appliesTo, sandboxName string) bool { + target := "sandbox:" + sandboxName + return scope == target && appliesTo == target +} + +func isRelevantPolicyTarget(scope, appliesTo, sandboxName string) (bool, error) { + switch scope { + case "global": + if appliesTo != "all" { + return false, fmt.Errorf("docker sandbox policy returned an unsupported global target") + } + return true, nil + default: + if !strings.HasPrefix(scope, "sandbox:") || appliesTo != scope { + return false, fmt.Errorf("docker sandbox policy returned an unsupported sandbox target") + } + targetName := strings.TrimPrefix(scope, "sandbox:") + if !sandboxNamePattern.MatchString(targetName) { + return false, fmt.Errorf("docker sandbox policy returned an unsupported sandbox target") + } + return targetName == sandboxName, nil + } +} + +func policyRuleStatus(record map[string]json.RawMessage) (string, bool, error) { + status, err := requiredJSONString(record, "status") + if err != nil { + return "", false, fmt.Errorf("docker sandbox network policy rule omitted status") + } + switch strings.ToLower(status) { + case "active": + return status, true, nil + case "inactive": + return status, false, nil + default: + return "", false, fmt.Errorf("docker sandbox network policy rule returned unknown status") + } +} + +func containsSandboxPolicyRule(rules []provider.NetworkPolicyRule, expected provider.NetworkPolicyRule, sandboxName string) bool { + for _, expectedResource := range expected.Resources { + found := false + for _, rule := range rules { + if rule.Decision != expected.Decision || !isSandboxPolicyTarget(rule.Scope, rule.AppliesTo, sandboxName) { + continue + } + for _, actualResource := range rule.Resources { + if actualResource == expectedResource { + found = true + break + } + } + if found { + break + } + } + if !found { + return false + } + } + return true +} + +func containsPolicyRuleID(rules []provider.NetworkPolicyRule, id string) bool { + _, found := findPolicyRuleID(rules, id) + return found +} + +func findPolicyRuleID(rules []provider.NetworkPolicyRule, id string) (provider.NetworkPolicyRule, bool) { + for _, rule := range rules { + if rule.ID == id { + return rule, true + } + } + return provider.NetworkPolicyRule{}, false +} + +func isRemovableSandboxPolicyRule(rule provider.NetworkPolicyRule, sandboxName string) bool { + return rule.ResourceType == "network" && isSandboxPolicyTarget(rule.Scope, rule.AppliesTo, sandboxName) && strings.EqualFold(rule.Origin, "scoped") && rule.Editable +} + +func sameStablePolicyRuleIdentity(actual, expected provider.NetworkPolicyRule) bool { + if actual.ID != expected.ID || actual.PolicyID != expected.PolicyID || actual.Scope != expected.Scope || actual.AppliesTo != expected.AppliesTo || actual.ResourceType != expected.ResourceType || actual.Decision != expected.Decision || !strings.EqualFold(actual.Origin, expected.Origin) || actual.Editable != expected.Editable { + return false + } + actualResources := append([]string(nil), actual.Resources...) + expectedResources := append([]string(nil), expected.Resources...) + sort.Strings(actualResources) + sort.Strings(expectedResources) + return slices.Equal(actualResources, expectedResources) +} + +func isMissingPolicyRule(text string) bool { + text = strings.ToLower(text) + return strings.Contains(text, "rule not found") || strings.Contains(text, "no matching rule") || strings.Contains(text, "status 404") +} diff --git a/internal/provider/dockersandboxes/policy/policy.go b/internal/provider/dockersandboxes/policy/policy.go new file mode 100644 index 0000000..2c53e50 --- /dev/null +++ b/internal/provider/dockersandboxes/policy/policy.go @@ -0,0 +1,262 @@ +// Package policy canonicalizes and verifies the complete effective +// Docker Sandboxes policy before a runner registration token is requested. +package policy + +import ( + "crypto/sha256" + "encoding/hex" + "encoding/json" + "fmt" + "sort" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +type canonicalRule struct { + ID string `json:"id"` + Name string `json:"name"` + PolicyID string `json:"policyId"` + Scope string `json:"scope"` + AppliesTo string `json:"appliesTo"` + ResourceType string `json:"resourceType"` + Resources []string `json:"resources"` + Decision provider.NetworkPolicyDecision `json:"decision"` + Origin string `json:"origin"` + Status string `json:"status"` + Editable bool `json:"editable"` + Active bool `json:"active"` +} + +// Fingerprint hashes all supplied rules and attribution fields in a stable +// order. It intentionally includes automatic and non-editable rules. +func Fingerprint(rules []provider.NetworkPolicyRule) (string, error) { + canonical := make([]canonicalRule, len(rules)) + for index, rule := range rules { + if err := validateAttributedRule(rule); err != nil { + return "", err + } + resources := append([]string(nil), rule.Resources...) + sort.Strings(resources) + canonical[index] = canonicalRule{ + ID: rule.ID, + Name: rule.Name, + PolicyID: rule.PolicyID, + Scope: rule.Scope, + AppliesTo: rule.AppliesTo, + ResourceType: rule.ResourceType, + Resources: resources, + Decision: rule.Decision, + Origin: rule.Origin, + Status: rule.Status, + Editable: rule.Editable, + Active: rule.Active, + } + } + sort.Slice(canonical, func(left, right int) bool { + leftJSON, _ := json.Marshal(canonical[left]) + rightJSON, _ := json.Marshal(canonical[right]) + return string(leftJSON) < string(rightJSON) + }) + encoded, err := json.Marshal(canonical) + if err != nil { + return "", fmt.Errorf("encode canonical Docker Sandboxes policy: %w", err) + } + digest := sha256.Sum256(encoded) + return "sha256:" + hex.EncodeToString(digest[:]), nil +} + +// VerifyBaseline requires a complete active global baseline plus only the +// exact built-in rule that Docker Sandboxes may attach to a shell sandbox. The +// global content fingerprint remains stable across sandbox names and +// provider-generated rule IDs; the built-in rule is verified structurally. +func VerifyBaseline(expectedFingerprint, sandboxName string, rules []provider.NetworkPolicyRule) error { + global, scoped, err := partition(sandboxName, rules) + if err != nil { + return err + } + if err := verifyBalancedGlobal(global); err != nil { + return err + } + if err := verifyExpectedBuiltins(sandboxName, scoped); err != nil { + return err + } + actual, err := Fingerprint(global) + if err != nil { + return err + } + if actual != expectedFingerprint { + return fmt.Errorf("Docker Sandboxes balanced policy fingerprint mismatch: got %s, want %s", actual, expectedFingerprint) + } + return nil +} + +// VerifyEffective proves that the global baseline is unchanged and that every +// exact sandbox-scoped policy resource is active, scoped, editable, network-only, +// and represented by the configured allow/deny sets with no extras. +func VerifyEffective(expectedBaseline, sandboxName string, rules []provider.NetworkPolicyRule, allow, deny []string) error { + global, scoped, err := partition(sandboxName, rules) + if err != nil { + return err + } + if err := verifyBalancedGlobal(global); err != nil { + return err + } + actualBaseline, err := Fingerprint(global) + if err != nil { + return err + } + if actualBaseline != expectedBaseline { + return fmt.Errorf("Docker Sandboxes global policy changed after scoped policy application: got %s, want %s", actualBaseline, expectedBaseline) + } + want := make(map[string]provider.NetworkPolicyDecision, len(allow)+len(deny)) + for _, resource := range allow { + want[resource] = provider.NetworkPolicyAllow + } + for _, resource := range deny { + want[resource] = provider.NetworkPolicyDeny + } + seen := make(map[string]provider.NetworkPolicyDecision, len(want)) + builtinCount := 0 + for _, rule := range scoped { + if IsExpectedBuiltinRule(sandboxName, rule) { + builtinCount++ + continue + } + if rule.ResourceType != "network" || !strings.EqualFold(rule.Origin, "scoped") || !rule.Editable { + return fmt.Errorf("Docker Sandboxes policy contained an unmanaged sandbox-scoped rule %q", rule.ID) + } + for _, resource := range rule.Resources { + expectedDecision, exists := want[resource] + if !exists || expectedDecision != rule.Decision { + return fmt.Errorf("Docker Sandboxes policy contained unexpected sandbox-scoped resource %q", resource) + } + if previous, duplicate := seen[resource]; duplicate && previous != rule.Decision { + return fmt.Errorf("Docker Sandboxes policy contained conflicting decisions for %q", resource) + } + seen[resource] = rule.Decision + } + } + if builtinCount > 1 { + return fmt.Errorf("Docker Sandboxes policy contained %d duplicate built-in shell rules", builtinCount) + } + for resource, decision := range want { + if actual, exists := seen[resource]; !exists || actual != decision { + return fmt.Errorf("Docker Sandboxes policy did not activate configured %s resource %q", decision, resource) + } + } + return nil +} + +func verifyBalancedGlobal(rules []provider.NetworkPolicyRule) error { + networkAllowFound := false + for _, rule := range rules { + if rule.ResourceType != "network" { + continue + } + for _, resource := range rule.Resources { + host := resource + if separator := strings.LastIndex(resource, ":"); separator >= 0 { + host = resource[:separator] + } + if host == "*" || host == "**" { + if rule.Decision == provider.NetworkPolicyDeny { + return fmt.Errorf("Docker Sandboxes global policy is locked down rather than balanced") + } + if rule.Decision == provider.NetworkPolicyAllow { + return fmt.Errorf("Docker Sandboxes global policy is open rather than balanced") + } + } + if rule.Decision == provider.NetworkPolicyAllow { + networkAllowFound = true + } + } + } + if !networkAllowFound { + return fmt.Errorf("Docker Sandboxes global policy is locked down rather than balanced") + } + return nil +} + +func partition(sandboxName string, rules []provider.NetworkPolicyRule) (global, scoped []provider.NetworkPolicyRule, err error) { + if strings.TrimSpace(sandboxName) == "" { + return nil, nil, fmt.Errorf("Docker Sandboxes policy sandbox name is required") + } + seenIDs := make(map[string]struct{}, len(rules)) + for _, rule := range rules { + if err := validateAttributedRule(rule); err != nil { + return nil, nil, err + } + if _, duplicate := seenIDs[rule.ID]; duplicate { + return nil, nil, fmt.Errorf("Docker Sandboxes policy contained duplicate rule id %q", rule.ID) + } + seenIDs[rule.ID] = struct{}{} + if !rule.Active || !strings.EqualFold(rule.Status, "active") { + return nil, nil, fmt.Errorf("Docker Sandboxes policy rule %q is inactive", rule.ID) + } + switch { + case rule.Scope == "global" && rule.AppliesTo == "all": + global = append(global, rule) + case isExactSandboxScope(rule, sandboxName): + scoped = append(scoped, rule) + default: + return nil, nil, fmt.Errorf("Docker Sandboxes policy contained rule %q with unexpected scope %q and target %q", rule.ID, rule.Scope, rule.AppliesTo) + } + } + if len(global) == 0 { + return nil, nil, fmt.Errorf("Docker Sandboxes policy omitted its global baseline") + } + return global, scoped, nil +} + +func validateAttributedRule(rule provider.NetworkPolicyRule) error { + for key, value := range map[string]string{ + "id": rule.ID, "name": rule.Name, "policy id": rule.PolicyID, "scope": rule.Scope, "target": rule.AppliesTo, "resource type": rule.ResourceType, "origin": rule.Origin, "status": rule.Status, + } { + if strings.TrimSpace(value) == "" { + return fmt.Errorf("Docker Sandboxes policy rule omitted %s", key) + } + } + if rule.Decision != provider.NetworkPolicyAllow && rule.Decision != provider.NetworkPolicyDeny { + return fmt.Errorf("Docker Sandboxes policy rule %q used unsupported decision %q", rule.ID, rule.Decision) + } + if len(rule.Resources) == 0 { + return fmt.Errorf("Docker Sandboxes policy rule %q omitted resources", rule.ID) + } + for _, resource := range rule.Resources { + if strings.TrimSpace(resource) == "" { + return fmt.Errorf("Docker Sandboxes policy rule %q contained an empty resource", rule.ID) + } + } + return nil +} + +func isExactSandboxScope(rule provider.NetworkPolicyRule, sandboxName string) bool { + target := "sandbox:" + sandboxName + return rule.Scope == target && rule.AppliesTo == target +} + +// IsExpectedBuiltinRule identifies the exact non-editable shell-kit egress +// rule observed in Docker Sandboxes. It deliberately checks every stable +// semantic field while allowing provider-generated IDs to vary. +func IsExpectedBuiltinRule(sandboxName string, rule provider.NetworkPolicyRule) bool { + return isExactSandboxScope(rule, sandboxName) && + rule.Name == "kit:"+sandboxName && + rule.ResourceType == "network" && + rule.Decision == provider.NetworkPolicyAllow && + len(rule.Resources) == 1 && rule.Resources[0] == "openrouter.ai" && + strings.EqualFold(rule.Origin, "scoped") && + strings.EqualFold(rule.Status, "active") && rule.Active && !rule.Editable +} + +func verifyExpectedBuiltins(sandboxName string, rules []provider.NetworkPolicyRule) error { + if len(rules) > 1 { + return fmt.Errorf("Docker Sandboxes policy contained %d unexpected pre-existing sandbox-scoped rules", len(rules)) + } + for _, rule := range rules { + if !IsExpectedBuiltinRule(sandboxName, rule) { + return fmt.Errorf("Docker Sandboxes policy contained unexpected pre-existing sandbox-scoped rule %q", rule.ID) + } + } + return nil +} diff --git a/internal/provider/dockersandboxes/policy/policy_test.go b/internal/provider/dockersandboxes/policy/policy_test.go new file mode 100644 index 0000000..fb5e7bf --- /dev/null +++ b/internal/provider/dockersandboxes/policy/policy_test.go @@ -0,0 +1,128 @@ +package policy + +import ( + "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +func TestFingerprintIsOrderIndependentAndIncludesAttribution(t *testing.T) { + rules := baselineRules() + first, err := Fingerprint(rules) + if err != nil { + t.Fatal(err) + } + reversed := []provider.NetworkPolicyRule{rules[1], rules[0]} + second, err := Fingerprint(reversed) + if err != nil { + t.Fatal(err) + } + if first != second { + t.Fatalf("fingerprint changed with rule order: %s != %s", first, second) + } + changed := append([]provider.NetworkPolicyRule(nil), rules...) + changed[0].Origin = "organization" + third, err := Fingerprint(changed) + if err != nil { + t.Fatal(err) + } + if first == third { + t.Fatal("fingerprint ignored policy attribution") + } +} + +func TestVerifyBaselineRejectsUnexpectedOrInactiveRules(t *testing.T) { + rules := baselineRules() + fingerprint, err := Fingerprint(rules) + if err != nil { + t.Fatal(err) + } + if err := VerifyBaseline(fingerprint, "epar-1", rules); err != nil { + t.Fatalf("valid baseline rejected: %v", err) + } + withBuiltin := append(append([]provider.NetworkPolicyRule(nil), rules...), builtinRule("epar-1")) + if err := VerifyBaseline(fingerprint, "epar-1", withBuiltin); err != nil { + t.Fatalf("exact shell-kit baseline rejected: %v", err) + } + unexpected := append(append([]provider.NetworkPolicyRule(nil), rules...), scopedRule("epar-1", "rule-1", provider.NetworkPolicyAllow, "example.test")) + if err := VerifyBaseline(fingerprint, "epar-1", unexpected); err == nil { + t.Fatal("pre-existing scoped rule accepted") + } + inactive := append([]provider.NetworkPolicyRule(nil), rules...) + inactive[0].Active = false + inactive[0].Status = "inactive" + if err := VerifyBaseline(fingerprint, "epar-1", inactive); err == nil { + t.Fatal("inactive baseline rule accepted") + } +} + +func TestVerifyBaselineRejectsOpenAndLockedDownGlobalModes(t *testing.T) { + open := baselineRules() + open[0].Resources = append(open[0].Resources, "**") + openFingerprint, err := Fingerprint(open) + if err != nil { + t.Fatal(err) + } + if err := VerifyBaseline(openFingerprint, "epar-1", open); err == nil { + t.Fatal("open global policy accepted as balanced") + } + + lockedDown := baselineRules()[1:] + lockedDownFingerprint, err := Fingerprint(lockedDown) + if err != nil { + t.Fatal(err) + } + if err := VerifyBaseline(lockedDownFingerprint, "epar-1", lockedDown); err == nil { + t.Fatal("locked-down global policy accepted as balanced") + } + + denyAll := baselineRules() + denyAll = append(denyAll, provider.NetworkPolicyRule{ID: "deny-all", Name: "deny-all", PolicyID: "local-policy", Scope: "global", AppliesTo: "all", ResourceType: "network", Resources: []string{"**"}, Decision: provider.NetworkPolicyDeny, Origin: "local", Status: "active", Editable: true, Active: true}) + denyAllFingerprint, err := Fingerprint(denyAll) + if err != nil { + t.Fatal(err) + } + if err := VerifyBaseline(denyAllFingerprint, "epar-1", denyAll); err == nil { + t.Fatal("bounded allow plus universal deny accepted as balanced") + } +} + +func TestVerifyEffectiveRequiresExactScopedResources(t *testing.T) { + global := baselineRules() + fingerprint, err := Fingerprint(global) + if err != nil { + t.Fatal(err) + } + effective := append(append([]provider.NetworkPolicyRule(nil), global...), + builtinRule("epar-1"), + scopedRule("epar-1", "allow-1", provider.NetworkPolicyAllow, "api.example.test"), + scopedRule("epar-1", "deny-1", provider.NetworkPolicyDeny, "telemetry.example.test"), + ) + if err := VerifyEffective(fingerprint, "epar-1", effective, []string{"api.example.test"}, []string{"telemetry.example.test"}); err != nil { + t.Fatalf("valid effective policy rejected: %v", err) + } + extra := append(append([]provider.NetworkPolicyRule(nil), effective...), scopedRule("epar-1", "extra", provider.NetworkPolicyAllow, "unexpected.example.test")) + if err := VerifyEffective(fingerprint, "epar-1", extra, []string{"api.example.test"}, []string{"telemetry.example.test"}); err == nil { + t.Fatal("unexpected scoped rule accepted") + } + if err := VerifyEffective(fingerprint, "epar-1", effective[:len(effective)-1], []string{"api.example.test"}, []string{"telemetry.example.test"}); err == nil { + t.Fatal("missing configured rule accepted") + } +} + +func baselineRules() []provider.NetworkPolicyRule { + return []provider.NetworkPolicyRule{ + {ID: "network", Name: "network", PolicyID: "local-policy", Scope: "global", AppliesTo: "all", ResourceType: "network", Resources: []string{"github.com:443", "**.github.com:443"}, Decision: provider.NetworkPolicyAllow, Origin: "local", Status: "active", Editable: true, Active: true}, + {ID: "filesystem", Name: "filesystem", PolicyID: "local-policy", Scope: "global", AppliesTo: "all", ResourceType: "filesystem:read", Resources: []string{"**"}, Decision: provider.NetworkPolicyAllow, Origin: "local", Status: "active", Editable: false, Active: true}, + } +} + +func scopedRule(sandboxName, id string, decision provider.NetworkPolicyDecision, resource string) provider.NetworkPolicyRule { + target := "sandbox:" + sandboxName + return provider.NetworkPolicyRule{ID: id, Name: id, PolicyID: "local-policy", Scope: target, AppliesTo: target, ResourceType: "network", Resources: []string{resource}, Decision: decision, Origin: "scoped", Status: "active", Editable: true, Active: true} +} + +func builtinRule(sandboxName string) provider.NetworkPolicyRule { + target := "sandbox:" + sandboxName + return provider.NetworkPolicyRule{ID: "kit-rule", Name: "kit:" + sandboxName, PolicyID: "kit-policy", Scope: target, AppliesTo: target, ResourceType: "network", Resources: []string{"openrouter.ai"}, Decision: provider.NetworkPolicyAllow, Origin: "scoped", Status: "active", Editable: false, Active: true} +} diff --git a/internal/provider/dockersandboxes/process_other.go b/internal/provider/dockersandboxes/process_other.go new file mode 100644 index 0000000..f4fcf03 --- /dev/null +++ b/internal/provider/dockersandboxes/process_other.go @@ -0,0 +1,7 @@ +//go:build !windows + +package dockersandboxes + +import "os/exec" + +func isolateKeepaliveProcess(*exec.Cmd) {} diff --git a/internal/provider/dockersandboxes/process_windows.go b/internal/provider/dockersandboxes/process_windows.go new file mode 100644 index 0000000..93c9007 --- /dev/null +++ b/internal/provider/dockersandboxes/process_windows.go @@ -0,0 +1,15 @@ +//go:build windows + +package dockersandboxes + +import ( + "os/exec" + "syscall" +) + +func isolateKeepaliveProcess(command *exec.Cmd) { + command.SysProcAttr = &syscall.SysProcAttr{ + HideWindow: true, + NoInheritHandles: true, + } +} diff --git a/internal/provider/dockersandboxes/promotion/preflight.go b/internal/provider/dockersandboxes/promotion/preflight.go new file mode 100644 index 0000000..81ebea0 --- /dev/null +++ b/internal/provider/dockersandboxes/promotion/preflight.go @@ -0,0 +1,480 @@ +package promotion + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "fmt" + "io" + "os" + "os/exec" + "regexp" + "strconv" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + sandboxcapacity "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/capacity" + sandboxpolicy "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/policy" +) + +const ( + DisableEnvironment = "EPAR_DISABLE_DOCKER_SANDBOXES" + preflightOutputLimit = 256 << 10 +) + +var ( + sandboxScopePattern = regexp.MustCompile(`^sandbox:[A-Za-z0-9][A-Za-z0-9._-]{0,127}$`) +) + +type Failure struct { + Gate string + Detail string + Resolution string +} + +type PreflightResult struct { + Failures []Failure +} + +type HostSpace = sandboxcapacity.HostSpace + +func (result PreflightResult) Passed() bool { + return len(result.Failures) == 0 +} + +type PreflightOptions struct { + ProjectRoot string + StorageRoot string + NativeController bool + ControllerRevision string + RunSBX func(context.Context, []string) ([]byte, error) + HostSpace func(string) (HostSpace, error) + CheckVirtualization func() error +} + +func LocalPreflight(ctx context.Context, record Record, projectRoot string, nativeController bool, controllerRevision string) PreflightResult { + if record.Platform != CurrentPlatform() { + return PreflightResult{Failures: []Failure{{ + Gate: "promoted platform", + Detail: fmt.Sprintf("promotion record targets %s, but the native controller is running on %s", record.Platform, CurrentPlatform()), + Resolution: "Explicitly choose another provider; never reuse a Docker Sandboxes promotion record across platforms.", + }}} + } + if os.Getenv(DisableEnvironment) == "1" { + return PreflightResult{Failures: []Failure{{ + Gate: "operator kill switch", + Detail: DisableEnvironment + "=1 disables Docker Sandboxes admission and automatic selection", + Resolution: "Unset the kill switch only after the Docker Sandboxes issue is resolved, or explicitly choose another provider.", + }}} + } + storageRoot, err := sandboxcapacity.DockerSandboxesStorageRoot() + if err != nil { + return PreflightResult{Failures: []Failure{{ + Gate: "resource availability", + Detail: fmt.Sprintf("cannot locate Docker Sandboxes provider storage: %v", err), + Resolution: "Use another explicitly selected provider until Docker Sandboxes storage can be located for capacity admission.", + }}} + } + return RunPreflight(ctx, record, PreflightOptions{ + ProjectRoot: projectRoot, + StorageRoot: storageRoot, + NativeController: nativeController, + ControllerRevision: controllerRevision, + RunSBX: runSBXCommand, + HostSpace: sandboxHostSpace, + CheckVirtualization: sandboxVirtualizationAvailable, + }) +} + +func RunPreflight(ctx context.Context, record Record, opts PreflightOptions) PreflightResult { + var result PreflightResult + add := func(gate, detail, resolution string) { + result.Failures = append(result.Failures, Failure{Gate: gate, Detail: detail, Resolution: resolution}) + } + if err := Validate(record); err != nil { + add("promotion record", err.Error(), "Use Docker Container or another explicitly selected provider until a valid platform promotion record is embedded.") + return result + } + if !validSHA256(opts.ControllerRevision) { + add("controller revision", "the running native controller does not report an exact clean source/build sha256 identity", "Rebuild the native controller from the exact clean promoted source tree with scripts/build-native-controller, then rerun setup.") + } else if opts.ControllerRevision != record.EPARRevision { + add("controller revision", fmt.Sprintf("the running native controller identity is %s, but the promotion record requires %s", opts.ControllerRevision, record.EPARRevision), "Rebuild and run the exact promoted native controller revision, or use another explicitly selected provider.") + } + if !opts.NativeController { + add("native controller", "EPAR is running inside the legacy controller container, where Docker Sandboxes host state is unavailable.", "Run EPAR through ./start or scripts/build-native-controller so the controller executes on the host.") + } + if opts.CheckVirtualization == nil { + add("virtualization", "the native virtualization check is unavailable", "Use Docker Container or another provider until EPAR can verify host virtualization.") + } else if err := opts.CheckVirtualization(); err != nil { + add("virtualization", err.Error(), "Enable the platform virtualization facility and ensure the current host user can access it, then rerun setup.") + } + if opts.StorageRoot == "" { + add("resource availability", "the Docker Sandboxes provider-storage path is unavailable", "Use another provider until EPAR can locate the actual Docker Sandboxes storage volume.") + } else if opts.HostSpace == nil { + add("resource availability", "the host free-space check is unavailable", "Use Docker Container or another provider until EPAR can verify host capacity.") + } else { + space, err := opts.HostSpace(opts.StorageRoot) + required, overflow := requiredHostFreeBytes(record, space.TotalBytes) + switch { + case err != nil: + add("resource availability", fmt.Sprintf("cannot read free space for Docker Sandboxes provider storage %s: %v", opts.StorageRoot, err), "Make the Docker Sandboxes storage filesystem available to the native controller, then rerun setup.") + case overflow: + add("resource availability", "the promoted disk reservation overflows the supported byte range", "Use Docker Container and report the invalid promotion record.") + case space.AvailableBytes < required: + add("resource availability", fmt.Sprintf("Docker Sandboxes storage free space is %d bytes; the configured fixed reserve requires at least %d bytes on the %d-byte provider-storage volume", space.AvailableBytes, required, space.TotalBytes), "Free space on the Docker Sandboxes provider-storage volume, then rerun setup.") + } + } + if opts.RunSBX == nil { + add("sbx command", "the Docker Sandboxes command runner is unavailable", "Install sbx on the native host, then run sbx diagnose --output json and review the hints for any failed checks.") + return result + } + + daemonOutput, daemonErr := opts.RunSBX(ctx, []string{"daemon", "status", "--json"}) + if daemonErr != nil { + add("daemon health", daemonErr.Error(), "Start or repair the Docker Sandboxes daemon, confirm sbx daemon status --json reports running, then rerun setup.") + } else if err := verifyDaemonRunning(daemonOutput); err != nil { + add("daemon health", err.Error(), "Start or repair the Docker Sandboxes daemon, confirm sbx daemon status --json reports running, then rerun setup.") + } + + diagnoseOutput, diagnoseErr := opts.RunSBX(ctx, []string{"diagnose", "--output", "json"}) + if diagnoseErr != nil { + add("daemon diagnostics", diagnoseErr.Error(), "Run sbx diagnose --output json and review the hints for any failed checks, then rerun setup.") + } else { + checks, err := parseDiagnostics(diagnoseOutput) + if err != nil { + add("daemon diagnostics", err.Error(), "Run sbx diagnose --output json and review the hints for any failed checks, then rerun setup.") + } else { + passed, failed := diagnosticPassAndFailureCounts(checks) + if failed != 0 { + add("daemon diagnostics", fmt.Sprintf("diagnostics reported %d failed check(s)", failed), "Run sbx diagnose --output json and review the hints for each failed check, then rerun setup.") + } else if passed == 0 { + add("daemon diagnostics", "diagnostics reported no passing checks", "Run sbx diagnose --output json and review its check details, then rerun setup.") + } + } + } + + templateOutput, templateErr := opts.RunSBX(ctx, []string{"template", "ls", "--json"}) + if templateErr != nil { + add("promoted template", templateErr.Error(), fmt.Sprintf("Build and load the exact promoted template %s, then rerun setup.", record.Template)) + } else if err := verifyPromotedTemplate(templateOutput, record.Template, record.TemplateCacheID); err != nil { + add("promoted template", err.Error(), fmt.Sprintf("Build and load the exact promoted template %s with cache ID %s, then rerun setup.", record.Template, record.TemplateCacheID)) + } + + policyOutput, policyErr := opts.RunSBX(ctx, []string{"policy", "ls", "--include-inactive", "--json"}) + if policyErr != nil { + add("promoted policy", policyErr.Error(), "Restore the exact promoted Docker Sandboxes global policy and rerun setup.") + } else if err := verifyPromotedPolicy(policyOutput, record.PolicyFingerprint); err != nil { + add("promoted policy", err.Error(), "Restore the exact promoted Docker Sandboxes global policy and rerun setup.") + } + return result +} + +func requiredHostFreeBytes(record Record, backingVolumeSize uint64) (uint64, bool) { + watermark, err := sandboxcapacity.HostWatermark(record.MinHostFreeSpaceBytes, backingVolumeSize) + if err != nil { + return 0, true + } + return watermark, false +} + +func runSBXCommand(ctx context.Context, args []string) ([]byte, error) { + if len(args) == 0 { + return nil, errors.New("refusing to invoke sbx without a subcommand") + } + if args[0] == "tui" || args[0] == "reset" { + return nil, fmt.Errorf("refusing to invoke forbidden sbx subcommand %q", args[0]) + } + command := exec.CommandContext(ctx, "sbx", args...) + command.Env = sandboxCommandEnvironment() + stdout := &preflightBuffer{limit: preflightOutputLimit} + stderr := &preflightBuffer{limit: preflightOutputLimit} + command.Stdout = stdout + command.Stderr = stderr + err := command.Run() + if stdout.overflow || stderr.overflow { + err = errors.Join(err, errors.New("sbx preflight output limit exceeded")) + } + if err != nil { + if text := strings.TrimSpace(stderr.String()); text != "" { + return nil, fmt.Errorf("sbx %s failed: %w: %s", args[0], err, text) + } + return nil, fmt.Errorf("sbx %s failed: %w", args[0], err) + } + return append([]byte(nil), stdout.Bytes()...), nil +} + +func sandboxCommandEnvironment() []string { + environment := make([]string, 0, len(os.Environ())) + for _, item := range os.Environ() { + key, _, _ := strings.Cut(item, "=") + if strings.HasPrefix(strings.ToUpper(key), "DOCKER_SANDBOXES_") { + continue + } + environment = append(environment, item) + } + return environment +} + +type preflightBuffer struct { + bytes.Buffer + limit int + overflow bool +} + +func (buffer *preflightBuffer) Write(data []byte) (int, error) { + remaining := buffer.limit - buffer.Len() + if remaining <= 0 { + buffer.overflow = true + return len(data), nil + } + if len(data) > remaining { + _, _ = buffer.Buffer.Write(data[:remaining]) + buffer.overflow = true + return len(data), nil + } + return buffer.Buffer.Write(data) +} + +func decodeStrictJSON(data []byte, value any) error { + decoder := json.NewDecoder(bytes.NewReader(data)) + decoder.UseNumber() + if err := decoder.Decode(value); err != nil { + return err + } + var extra any + if err := decoder.Decode(&extra); err == nil { + return errors.New("unexpected trailing JSON value") + } else if !errors.Is(err, io.EOF) { + return err + } + return nil +} + +func verifyDaemonRunning(data []byte) error { + var record map[string]json.RawMessage + if err := decodeStrictJSON(data, &record); err != nil { + return errors.New("Docker Sandboxes daemon status returned unsupported JSON") + } + var status string + if raw, ok := record["status"]; !ok || json.Unmarshal(raw, &status) != nil || !strings.EqualFold(strings.TrimSpace(status), "running") { + return fmt.Errorf("Docker Sandboxes daemon status is not running") + } + return nil +} + +type diagnosticCheck struct { + Name string + Status string +} + +func parseDiagnostics(data []byte) ([]diagnosticCheck, error) { + var record map[string]json.RawMessage + if err := decodeStrictJSON(data, &record); err != nil { + return nil, errors.New("Docker Sandboxes diagnostics returned unsupported JSON") + } + var version string + if raw, ok := record["version"]; !ok || json.Unmarshal(raw, &version) != nil || version != "1.0" { + return nil, errors.New("Docker Sandboxes diagnostics returned an unsupported schema version") + } + var rawChecks []map[string]json.RawMessage + if raw, ok := record["checks"]; !ok || json.Unmarshal(raw, &rawChecks) != nil || len(rawChecks) == 0 { + return nil, errors.New("Docker Sandboxes diagnostics omitted its checks") + } + checks := make([]diagnosticCheck, 0, len(rawChecks)) + counts := map[string]int{"pass": 0, "warn": 0, "fail": 0, "skip": 0} + for _, rawCheck := range rawChecks { + name, err := requiredString(rawCheck, "name", false) + if err != nil { + return nil, errors.New("Docker Sandboxes diagnostics returned an unsupported check schema") + } + status, err := requiredString(rawCheck, "status", false) + if err != nil { + return nil, errors.New("Docker Sandboxes diagnostics returned an unsupported check schema") + } + for _, field := range []string{"message", "detail", "hint"} { + if _, err := requiredString(rawCheck, field, true); err != nil { + return nil, errors.New("Docker Sandboxes diagnostics returned an unsupported check schema") + } + } + status = strings.ToLower(status) + if _, ok := counts[status]; !ok { + return nil, errors.New("Docker Sandboxes diagnostics returned an unknown check status") + } + counts[status]++ + checks = append(checks, diagnosticCheck{Name: name, Status: status}) + } + var summary map[string]json.RawMessage + if raw, ok := record["summary"]; !ok || json.Unmarshal(raw, &summary) != nil { + return nil, errors.New("Docker Sandboxes diagnostic summary is missing") + } + for _, key := range []string{"pass", "warn", "fail", "skip"} { + var count int + if raw, ok := summary[key]; !ok || json.Unmarshal(raw, &count) != nil || count != counts[key] { + return nil, errors.New("Docker Sandboxes diagnostic summary did not match its checks") + } + } + return checks, nil +} + +func diagnosticPassAndFailureCounts(checks []diagnosticCheck) (passed, failed int) { + for _, check := range checks { + switch check.Status { + case "pass": + passed++ + case "fail": + failed++ + } + } + return passed, failed +} + +func verifyPromotedTemplate(data []byte, reference, cacheID string) error { + repository, tag, err := splitTemplateReference(reference) + if err != nil { + return err + } + if !validTemplateCacheID(cacheID) { + return errors.New("promoted Docker Sandboxes template cache ID is invalid") + } + var wrapper map[string]json.RawMessage + if err := decodeStrictJSON(data, &wrapper); err != nil { + return errors.New("Docker Sandboxes template inventory returned unsupported JSON") + } + var images []map[string]json.RawMessage + if raw, ok := wrapper["images"]; !ok || json.Unmarshal(raw, &images) != nil { + return errors.New("Docker Sandboxes template inventory omitted images") + } + for _, image := range images { + actualRepository, repositoryErr := requiredString(image, "repository", false) + actualTag, tagErr := requiredString(image, "tag", false) + actualID, idErr := requiredString(image, "id", false) + createdAt, createdAtErr := requiredString(image, "created_at", false) + var size json.Number + sizeErr := json.Unmarshal(image["size"], &size) + sizeValue, integerErr := strconv.ParseInt(size.String(), 10, 64) + if repositoryErr != nil || tagErr != nil || idErr != nil || !validTemplateCacheID(actualID) || createdAtErr != nil || sizeErr != nil || integerErr != nil || sizeValue <= 0 { + return errors.New("Docker Sandboxes template inventory returned an unsupported image schema") + } + if _, err := time.Parse(time.RFC3339, createdAt); err != nil { + return errors.New("Docker Sandboxes template inventory returned an invalid creation time") + } + if actualRepository == repository && actualTag == tag { + if actualID != cacheID { + return fmt.Errorf("cached template cache ID %s does not match promoted cache ID %s", actualID, cacheID) + } + return nil + } + } + return fmt.Errorf("promoted template %s is not present in the Docker Sandboxes cache", reference) +} + +func splitTemplateReference(reference string) (string, string, error) { + separator := strings.LastIndex(reference, ":") + if separator <= strings.LastIndex(reference, "/") || separator == len(reference)-1 || strings.Contains(reference, "@") { + return "", "", errors.New("promoted Docker Sandboxes template must be an exact repository:tag reference") + } + repository, tag := reference[:separator], reference[separator+1:] + if repository == "" || tag == "" { + return "", "", errors.New("promoted Docker Sandboxes template must be an exact repository:tag reference") + } + if !strings.Contains(repository, "/") { + repository = "docker.io/library/" + repository + } else { + first := strings.SplitN(repository, "/", 2)[0] + if first != "localhost" && !strings.ContainsAny(first, ".:") { + repository = "docker.io/" + repository + } + } + return repository, tag, nil +} + +func verifyPromotedPolicy(data []byte, expected string) error { + rules, err := parseGlobalPolicy(data) + if err != nil { + return err + } + actual, err := sandboxpolicy.Fingerprint(rules) + if err != nil { + return fmt.Errorf("fingerprint Docker Sandboxes global policy: %w", err) + } + if actual != expected { + return fmt.Errorf("Docker Sandboxes global policy fingerprint is %s, want %s", actual, expected) + } + return nil +} + +func parseGlobalPolicy(data []byte) ([]provider.NetworkPolicyRule, error) { + var wrapper map[string]json.RawMessage + if err := decodeStrictJSON(data, &wrapper); err != nil { + return nil, errors.New("Docker Sandboxes policy inventory returned unsupported JSON") + } + var records []map[string]json.RawMessage + if raw, ok := wrapper["rules"]; !ok || json.Unmarshal(raw, &records) != nil { + return nil, errors.New("Docker Sandboxes policy inventory omitted rules") + } + var global []provider.NetworkPolicyRule + seen := make(map[string]struct{}, len(records)) + for _, record := range records { + id, idErr := requiredString(record, "id", false) + name, nameErr := requiredString(record, "name", false) + policyID, policyErr := requiredString(record, "policy_id", false) + scope, scopeErr := requiredString(record, "scope", false) + appliesTo, targetErr := requiredString(record, "applies_to", false) + resourceType, typeErr := requiredString(record, "resource_type", false) + decisionText, decisionErr := requiredString(record, "decision", false) + origin, originErr := requiredString(record, "origin", false) + status, statusErr := requiredString(record, "status", false) + var resources []string + resourcesErr := json.Unmarshal(record["resources"], &resources) + var editable bool + editableErr := json.Unmarshal(record["editable"], &editable) + if idErr != nil || nameErr != nil || policyErr != nil || scopeErr != nil || targetErr != nil || typeErr != nil || decisionErr != nil || originErr != nil || statusErr != nil || resourcesErr != nil || len(resources) == 0 || editableErr != nil { + return nil, errors.New("Docker Sandboxes policy inventory returned an unsupported rule schema") + } + if _, duplicate := seen[id]; duplicate { + return nil, fmt.Errorf("Docker Sandboxes policy inventory returned duplicate rule id %q", id) + } + seen[id] = struct{}{} + decision := provider.NetworkPolicyDecision(strings.ToLower(decisionText)) + if decision != provider.NetworkPolicyAllow && decision != provider.NetworkPolicyDeny { + return nil, fmt.Errorf("Docker Sandboxes policy rule %q has unsupported decision %q", id, decisionText) + } + if scope != "global" { + if !sandboxScopePattern.MatchString(scope) || appliesTo != scope { + return nil, fmt.Errorf("Docker Sandboxes policy rule %q has unsupported scope", id) + } + if !strings.EqualFold(status, "active") { + return nil, fmt.Errorf("Docker Sandboxes sandbox-scoped policy rule %q is not active", id) + } + continue + } + if appliesTo != "all" { + return nil, fmt.Errorf("Docker Sandboxes global policy rule %q has unsupported target %q", id, appliesTo) + } + if !strings.EqualFold(status, "active") { + return nil, fmt.Errorf("Docker Sandboxes global policy rule %q is not active", id) + } + global = append(global, provider.NetworkPolicyRule{ + ID: id, Name: name, PolicyID: policyID, Scope: scope, AppliesTo: appliesTo, ResourceType: resourceType, Resources: resources, + Decision: decision, Origin: origin, Status: status, Editable: editable, Active: true, + }) + } + if len(global) == 0 { + return nil, errors.New("Docker Sandboxes policy inventory omitted its global baseline") + } + return global, nil +} + +func requiredString(record map[string]json.RawMessage, key string, allowEmpty bool) (string, error) { + raw, ok := record[key] + if !ok { + return "", errors.New("missing field") + } + var value string + if json.Unmarshal(raw, &value) != nil || !allowEmpty && strings.TrimSpace(value) == "" { + return "", errors.New("invalid field") + } + return value, nil +} diff --git a/internal/provider/dockersandboxes/promotion/preflight_test.go b/internal/provider/dockersandboxes/promotion/preflight_test.go new file mode 100644 index 0000000..04413b0 --- /dev/null +++ b/internal/provider/dockersandboxes/promotion/preflight_test.go @@ -0,0 +1,357 @@ +package promotion + +import ( + "context" + "errors" + "fmt" + "slices" + "strings" + "testing" + "time" + + sandboxpolicy "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/policy" +) + +func TestRunPreflightPassesEveryIndependentGateWithExactReadOnlyArgv(t *testing.T) { + record, outputs := validPreflightFixture(t) + var commands [][]string + storageRoot := t.TempDir() + result := RunPreflight(context.Background(), record, PreflightOptions{ + ProjectRoot: t.TempDir(), + StorageRoot: storageRoot, + NativeController: true, + ControllerRevision: record.EPARRevision, + RunSBX: func(_ context.Context, args []string) ([]byte, error) { + commands = append(commands, append([]string(nil), args...)) + output, ok := outputs[strings.Join(args, "\x00")] + if !ok { + return nil, fmt.Errorf("unexpected command %#v", args) + } + return append([]byte(nil), output...), nil + }, + HostSpace: func(path string) (HostSpace, error) { + if path != storageRoot { + t.Fatalf("capacity path = %q, want provider storage %q", path, storageRoot) + } + required, overflow := requiredHostFreeBytes(record, 500<<30) + if overflow { + t.Fatal("valid fixture resource total overflowed") + } + return HostSpace{AvailableBytes: required, TotalBytes: 500 << 30}, nil + }, + CheckVirtualization: func() error { return nil }, + }) + if !result.Passed() { + t.Fatalf("preflight failures = %+v", result.Failures) + } + want := [][]string{ + {"daemon", "status", "--json"}, + {"diagnose", "--output", "json"}, + {"template", "ls", "--json"}, + {"policy", "ls", "--include-inactive", "--json"}, + } + if !slices.EqualFunc(commands, want, slices.Equal[[]string]) { + t.Fatalf("commands = %#v, want %#v", commands, want) + } + for _, args := range commands { + if len(args) == 0 || args[0] == "tui" || args[0] == "reset" { + t.Fatalf("unsafe Docker Sandboxes argv invoked: %#v", args) + } + } +} + +func TestRunPreflightFailsClosedForEveryAdmissionGate(t *testing.T) { + tests := []struct { + name string + gate string + edit func(*Record, map[string][]byte, *PreflightOptions) + }{ + { + name: "unknown controller revision", + gate: "controller revision", + edit: func(_ *Record, _ map[string][]byte, opts *PreflightOptions) { + opts.ControllerRevision = "unknown" + }, + }, + { + name: "stale controller revision", + gate: "controller revision", + edit: func(_ *Record, _ map[string][]byte, opts *PreflightOptions) { + opts.ControllerRevision = "sha256:" + strings.Repeat("8", 64) + }, + }, + { + name: "native controller", + gate: "native controller", + edit: func(_ *Record, _ map[string][]byte, opts *PreflightOptions) { opts.NativeController = false }, + }, + { + name: "virtualization", + gate: "virtualization", + edit: func(_ *Record, _ map[string][]byte, opts *PreflightOptions) { + opts.CheckVirtualization = func() error { return errors.New("hardware virtualization unavailable") } + }, + }, + { + name: "resource availability", + gate: "resource availability", + edit: func(_ *Record, _ map[string][]byte, opts *PreflightOptions) { + opts.HostSpace = func(string) (HostSpace, error) { return HostSpace{AvailableBytes: 1, TotalBytes: 500 << 30}, nil } + }, + }, + { + name: "provider storage path", + gate: "resource availability", + edit: func(_ *Record, _ map[string][]byte, opts *PreflightOptions) { + opts.StorageRoot = "" + }, + }, + { + name: "daemon", + gate: "daemon health", + edit: func(_ *Record, outputs map[string][]byte, _ *PreflightOptions) { + outputs["daemon\x00status\x00--json"] = []byte(`{"status":"stopped"}`) + }, + }, + { + name: "authentication", + gate: "daemon diagnostics", + edit: func(_ *Record, outputs map[string][]byte, _ *PreflightOptions) { + outputs["diagnose\x00--output\x00json"] = []byte(diagnosticsFixture("Authentication", "fail")) + }, + }, + { + name: "diagnostics without a pass", + gate: "daemon diagnostics", + edit: func(_ *Record, outputs map[string][]byte, _ *PreflightOptions) { + outputs["diagnose\x00--output\x00json"] = []byte(`{"version":"1.0","checks":[{"name":"Optional integration","status":"skip","message":"","detail":"","hint":""}],"summary":{"pass":0,"warn":0,"fail":0,"skip":1}}`) + }, + }, + { + name: "template", + gate: "promoted template", + edit: func(_ *Record, outputs map[string][]byte, _ *PreflightOptions) { + outputs["template\x00ls\x00--json"] = []byte(`{"images":[{"id":"bbbbbbbbbbbb","repository":"docker.io/library/epar-template","tag":"promoted","flavor":"shell","created_at":"2026-07-23T00:00:00Z","size":1024}]}`) + }, + }, + { + name: "policy", + gate: "promoted policy", + edit: func(record *Record, _ map[string][]byte, _ *PreflightOptions) { + record.PolicyFingerprint = "sha256:" + strings.Repeat("9", 64) + }, + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + record, outputs := validPreflightFixture(t) + required, overflow := requiredHostFreeBytes(record, 500<<30) + if overflow { + t.Fatal("valid fixture resource total overflowed") + } + opts := PreflightOptions{ + ProjectRoot: t.TempDir(), + StorageRoot: t.TempDir(), + NativeController: true, + ControllerRevision: record.EPARRevision, + RunSBX: fixtureCommandRunner(outputs), + HostSpace: func(string) (HostSpace, error) { + return HostSpace{AvailableBytes: required, TotalBytes: 500 << 30}, nil + }, + CheckVirtualization: func() error { return nil }, + } + test.edit(&record, outputs, &opts) + result := RunPreflight(context.Background(), record, opts) + if result.Passed() { + t.Fatal("preflight unexpectedly passed") + } + found := false + for _, failure := range result.Failures { + if failure.Gate == test.gate && failure.Detail != "" && failure.Resolution != "" { + found = true + break + } + } + if !found { + t.Fatalf("preflight failures = %+v, want actionable %q failure", result.Failures, test.gate) + } + }) + } +} + +func TestRunPreflightAcceptsDiagnosticWarningsAndSkips(t *testing.T) { + for _, status := range []string{"warn", "skip"} { + t.Run(status, func(t *testing.T) { + record, outputs := validPreflightFixture(t) + outputs["diagnose\x00--output\x00json"] = []byte(diagnosticsFixture("Storage directories", status)) + required, _ := requiredHostFreeBytes(record, 500<<30) + result := RunPreflight(context.Background(), record, PreflightOptions{ + ProjectRoot: t.TempDir(), + StorageRoot: t.TempDir(), + NativeController: true, + ControllerRevision: record.EPARRevision, + RunSBX: fixtureCommandRunner(outputs), + HostSpace: func(string) (HostSpace, error) { + return HostSpace{AvailableBytes: required, TotalBytes: 500 << 30}, nil + }, + CheckVirtualization: func() error { return nil }, + }) + if !result.Passed() { + t.Fatalf("preflight rejected diagnostic %s: %+v", status, result.Failures) + } + }) + } +} + +func TestRunPreflightDiagnosticFailureExplainsHowToInspectHints(t *testing.T) { + record, outputs := validPreflightFixture(t) + outputs["diagnose\x00--output\x00json"] = []byte(diagnosticsFixture("Daemon", "fail")) + required, _ := requiredHostFreeBytes(record, 500<<30) + result := RunPreflight(context.Background(), record, PreflightOptions{ + ProjectRoot: t.TempDir(), + StorageRoot: t.TempDir(), + NativeController: true, + ControllerRevision: record.EPARRevision, + RunSBX: fixtureCommandRunner(outputs), + HostSpace: func(string) (HostSpace, error) { + return HostSpace{AvailableBytes: required, TotalBytes: 500 << 30}, nil + }, + CheckVirtualization: func() error { return nil }, + }) + for _, failure := range result.Failures { + if failure.Gate == "daemon diagnostics" && strings.Contains(failure.Resolution, "sbx diagnose --output json") && strings.Contains(failure.Resolution, "hints") { + return + } + } + t.Fatalf("preflight diagnostic failure omitted command and hint remediation: %+v", result.Failures) +} + +func TestRunPreflightDoesNotInferVirtualizationFromDiagnostics(t *testing.T) { + record, outputs := validPreflightFixture(t) + required, _ := requiredHostFreeBytes(record, 500<<30) + result := RunPreflight(context.Background(), record, PreflightOptions{ + ProjectRoot: t.TempDir(), + StorageRoot: t.TempDir(), + NativeController: true, + ControllerRevision: record.EPARRevision, + RunSBX: fixtureCommandRunner(outputs), + HostSpace: func(string) (HostSpace, error) { + return HostSpace{AvailableBytes: required, TotalBytes: 500 << 30}, nil + }, + CheckVirtualization: func() error { return errors.New("independent host virtualization proof failed") }, + }) + if result.Passed() { + t.Fatal("preflight passed using diagnostics that contain no virtualization check") + } + for _, failure := range result.Failures { + if failure.Gate == "virtualization" && strings.Contains(failure.Detail, "independent host") { + return + } + } + t.Fatalf("preflight failures = %+v, want independent virtualization failure", result.Failures) +} + +func TestLocalPreflightRejectsCrossPlatformRecordAndKillSwitchBeforeCommands(t *testing.T) { + record, _ := validPreflightFixture(t) + if record.Platform == CurrentPlatform() { + record.Platform = DarwinARM64 + if record.Platform == CurrentPlatform() { + record.Platform = LinuxAMD64 + } + } + result := LocalPreflight(context.Background(), record, t.TempDir(), true, record.EPARRevision) + if result.Passed() || len(result.Failures) != 1 || result.Failures[0].Gate != "promoted platform" { + t.Fatalf("cross-platform preflight result = %+v", result) + } + + record.Platform = CurrentPlatform() + t.Setenv(DisableEnvironment, "1") + result = LocalPreflight(context.Background(), record, t.TempDir(), true, record.EPARRevision) + if result.Passed() || len(result.Failures) != 1 || result.Failures[0].Gate != "operator kill switch" { + t.Fatalf("kill-switch preflight result = %+v", result) + } +} + +func fixtureCommandRunner(outputs map[string][]byte) func(context.Context, []string) ([]byte, error) { + return func(_ context.Context, args []string) ([]byte, error) { + output, ok := outputs[strings.Join(args, "\x00")] + if !ok { + return nil, fmt.Errorf("unexpected command %#v", args) + } + return append([]byte(nil), output...), nil + } +} + +func validPreflightFixture(t *testing.T) (Record, map[string][]byte) { + t.Helper() + policyJSON := []byte(`{"rules":[{"id":"global-1","name":"automatic baseline","policy_id":"local-policy","scope":"global","applies_to":"all","resource_type":"network","decision":"allow","resources":["api.github.com"],"origin":"local","status":"active","editable":true}]}`) + rules, err := parseGlobalPolicy(policyJSON) + if err != nil { + t.Fatal(err) + } + policyFingerprint, err := sandboxpolicy.Fingerprint(rules) + if err != nil { + t.Fatal(err) + } + digest := func(character string) string { return "sha256:" + strings.Repeat(character, 64) } + record := Record{ + Platform: WindowsAMD64, + EPARRevision: digest("1"), + Template: "epar-template:promoted", + TemplateDigest: digest("a"), + TemplateCacheID: strings.Repeat("a", 12), + TemplateMetadataDigest: digest("b"), + TemplateArchiveDigest: digest("b"), + PolicyFingerprint: policyFingerprint, + EvidenceDigest: digest("c"), + SBOMDigest: digest("d"), + ProvenanceDigest: digest("e"), + SoftwareInventoryDigest: digest("f"), + VerifiedAt: time.Date(2026, 7, 23, 0, 0, 0, 0, time.UTC), + Verifier: "independent-test-verifier", + Gates: GateResults{Local: true, Functional: true, Recovery: true, Security: true, Policy: true, Cleanup: true, SecretScanning: true, ConcurrentProvisioning: true, IndependentSecurityReview: true}, + RootDiskBytes: 120 << 30, + DockerDiskBytes: 100 << 30, + MinHostFreeSpaceBytes: 50 << 30, + ReliabilityJobs: 25, + ReliabilityDuration: 2 * time.Hour, + CachedCreateP95: 30 * time.Second, + QueueToOnlineP95: 90 * time.Second, + ForceRemoveP95: 60 * time.Second, + BuildxComposeSlowdownPct: 10, + } + outputs := map[string][]byte{ + "daemon\x00status\x00--json": []byte(`{"status":"running","socket":"test","logs":"test"}`), + "diagnose\x00--output\x00json": []byte(diagnosticsFixture("", "")), + "template\x00ls\x00--json": []byte(`{"images":[{"id":"aaaaaaaaaaaa","repository":"docker.io/library/epar-template","tag":"promoted","flavor":"shell","created_at":"2026-07-23T00:00:00Z","size":1024}]}`), + "policy\x00ls\x00--include-inactive\x00--json": policyJSON, + } + return record, outputs +} + +func diagnosticsFixture(changedCheck, changedStatus string) string { + names := []string{"CLI binary", "CLI invocation", "Daemon", "Daemon diagnostics", "Runtime compatibility", "Storage directories", "Directory permissions", "Socket", "Authentication"} + checks := make([]string, 0, len(names)) + passed := 0 + failed := 0 + warned := 0 + skipped := 0 + for _, name := range names { + status := "pass" + if name == changedCheck { + status = changedStatus + } + switch status { + case "pass": + passed++ + case "fail": + failed++ + case "warn": + warned++ + case "skip": + skipped++ + } + checks = append(checks, fmt.Sprintf(`{"name":%q,"status":%q,"message":"ok","detail":"","hint":""}`, name, status)) + } + return fmt.Sprintf(`{"version":"1.0","checks":[%s],"summary":{"pass":%d,"warn":%d,"fail":%d,"skip":%d}}`, strings.Join(checks, ","), passed, warned, failed, skipped) +} diff --git a/internal/provider/dockersandboxes/promotion/promotion.go b/internal/provider/dockersandboxes/promotion/promotion.go new file mode 100644 index 0000000..ec350e1 --- /dev/null +++ b/internal/provider/dockersandboxes/promotion/promotion.go @@ -0,0 +1,183 @@ +// Package promotion records independently certified Docker Sandboxes +// platform decisions. Wizard default selection is based on local capabilities. +package promotion + +import ( + "fmt" + "runtime" + "strings" + "time" +) + +type Platform string + +const ( + WindowsAMD64 Platform = "windows/amd64" + DarwinARM64 Platform = "darwin/arm64" + LinuxAMD64 Platform = "linux/amd64" +) + +type Record struct { + Platform Platform + EPARRevision string + Template string + TemplateDigest string + TemplateCacheID string + TemplateMetadataDigest string + TemplateArchiveDigest string + PolicyFingerprint string + EvidenceDigest string + SBOMDigest string + ProvenanceDigest string + SoftwareInventoryDigest string + VerifiedAt time.Time + Verifier string + Gates GateResults + RootDiskBytes uint64 + DockerDiskBytes uint64 + MinHostFreeSpaceBytes uint64 + ReliabilityJobs int + ReliabilityDuration time.Duration + CachedCreateP95 time.Duration + QueueToOnlineP95 time.Duration + ForceRemoveP95 time.Duration + BuildxComposeSlowdownPct float64 +} + +// GateResults records the non-performance, non-soak promotion gates. Every +// field is non-waivable; the evidence digest binds the detailed transcripts. +type GateResults struct { + Local bool + Functional bool + Recovery bool + Security bool + Policy bool + Cleanup bool + SecretScanning bool + ConcurrentProvisioning bool + IndependentSecurityReview bool +} + +// embeddedRecords deliberately starts empty. A record represents the stronger +// independently reviewed certification for one exact source and artifact +// identity; it is not required for operator-accepted first-run default status. +var embeddedRecords = map[Platform]Record{} + +func CurrentPlatform() Platform { + return Platform(runtime.GOOS + "/" + runtime.GOARCH) +} + +func Lookup(platform Platform) (Record, bool) { + record, ok := embeddedRecords[platform] + return record, ok +} + +func Validate(record Record) error { + switch record.Platform { + case WindowsAMD64, DarwinARM64, LinuxAMD64: + default: + return fmt.Errorf("unsupported Docker Sandboxes promotion platform %q", record.Platform) + } + for key, value := range map[string]string{ + "EPAR revision": record.EPARRevision, + "template": record.Template, + "template digest": record.TemplateDigest, + "template cache ID": record.TemplateCacheID, + "template metadata": record.TemplateMetadataDigest, + "template archive": record.TemplateArchiveDigest, + "policy fingerprint": record.PolicyFingerprint, + "evidence digest": record.EvidenceDigest, + "SBOM digest": record.SBOMDigest, + "provenance digest": record.ProvenanceDigest, + "inventory digest": record.SoftwareInventoryDigest, + "verifier": record.Verifier, + } { + if strings.TrimSpace(value) == "" { + return fmt.Errorf("Docker Sandboxes promotion record %s is required", key) + } + } + if !validSHA256(record.EPARRevision) { + return fmt.Errorf("Docker Sandboxes promotion record EPAR revision must be an exact clean source/build sha256 identity") + } + for key, digest := range map[string]string{ + "template digest": record.TemplateDigest, + "template metadata digest": record.TemplateMetadataDigest, + "template archive digest": record.TemplateArchiveDigest, + "policy fingerprint": record.PolicyFingerprint, + "evidence digest": record.EvidenceDigest, + "SBOM digest": record.SBOMDigest, + "provenance digest": record.ProvenanceDigest, + "inventory digest": record.SoftwareInventoryDigest, + } { + if !validSHA256(digest) { + return fmt.Errorf("Docker Sandboxes promotion record %s must be sha256:<64 lowercase hex>", key) + } + } + if !validTemplateCacheID(record.TemplateCacheID) { + return fmt.Errorf("Docker Sandboxes promotion record template cache ID must be exactly 12 lowercase hexadecimal characters") + } + if record.TemplateCacheID != strings.TrimPrefix(record.TemplateDigest, "sha256:")[:12] { + return fmt.Errorf("Docker Sandboxes promotion record template cache ID must match the first 12 hexadecimal characters of the full template identity") + } + if record.VerifiedAt.IsZero() { + return fmt.Errorf("Docker Sandboxes promotion record verification time is required") + } + for gate, passed := range map[string]bool{ + "local": record.Gates.Local, + "functional": record.Gates.Functional, + "recovery": record.Gates.Recovery, + "security": record.Gates.Security, + "policy": record.Gates.Policy, + "cleanup": record.Gates.Cleanup, + "secret scanning": record.Gates.SecretScanning, + "concurrent provisioning": record.Gates.ConcurrentProvisioning, + "independent security review": record.Gates.IndependentSecurityReview, + } { + if !passed { + return fmt.Errorf("Docker Sandboxes promotion record %s gate did not pass", gate) + } + } + if record.RootDiskBytes < 20<<30 || record.DockerDiskBytes < 1<<30 || record.MinHostFreeSpaceBytes < 1<<30 { + return fmt.Errorf("Docker Sandboxes promotion record resource floors are incomplete") + } + if record.ReliabilityJobs < 25 || record.ReliabilityDuration < 2*time.Hour { + return fmt.Errorf("Docker Sandboxes promotion record reliability gate is incomplete") + } + if record.CachedCreateP95 <= 0 || record.CachedCreateP95 > 60*time.Second { + return fmt.Errorf("Docker Sandboxes promotion record cached-create p95 failed") + } + if record.QueueToOnlineP95 <= 0 || record.QueueToOnlineP95 > 180*time.Second { + return fmt.Errorf("Docker Sandboxes promotion record queue-to-online p95 failed") + } + if record.ForceRemoveP95 <= 0 || record.ForceRemoveP95 > 120*time.Second { + return fmt.Errorf("Docker Sandboxes promotion record force-remove p95 failed") + } + if record.BuildxComposeSlowdownPct < 0 || record.BuildxComposeSlowdownPct > 25 { + return fmt.Errorf("Docker Sandboxes promotion record Buildx/Compose slowdown failed") + } + return nil +} + +func validSHA256(value string) bool { + if len(value) != len("sha256:")+64 || !strings.HasPrefix(value, "sha256:") { + return false + } + for _, character := range value[len("sha256:"):] { + if !((character >= '0' && character <= '9') || (character >= 'a' && character <= 'f')) { + return false + } + } + return true +} + +func validTemplateCacheID(value string) bool { + if len(value) != 12 { + return false + } + for _, character := range value { + if !((character >= '0' && character <= '9') || (character >= 'a' && character <= 'f')) { + return false + } + } + return true +} diff --git a/internal/provider/dockersandboxes/promotion/promotion_test.go b/internal/provider/dockersandboxes/promotion/promotion_test.go new file mode 100644 index 0000000..6fb5c76 --- /dev/null +++ b/internal/provider/dockersandboxes/promotion/promotion_test.go @@ -0,0 +1,92 @@ +package promotion + +import ( + "strings" + "testing" + "time" +) + +func TestNoPlatformIsPromotedWithoutEmbeddedEvidence(t *testing.T) { + for _, platform := range []Platform{WindowsAMD64, DarwinARM64, LinuxAMD64} { + if _, promoted := Lookup(platform); promoted { + t.Fatalf("%s unexpectedly has a Docker Sandboxes promotion record", platform) + } + } +} + +func TestValidateCompleteRecord(t *testing.T) { + record := validRecord() + if err := Validate(record); err != nil { + t.Fatalf("valid promotion record rejected: %v", err) + } +} + +func TestValidateRejectsEveryNonWaivableGate(t *testing.T) { + tests := []struct { + name string + mutate func(*Record) + }{ + {name: "unknown platform", mutate: func(record *Record) { record.Platform = "plan9/amd64" }}, + {name: "unknown EPAR revision", mutate: func(record *Record) { record.EPARRevision = "unknown" }}, + {name: "wrong template cache ID", mutate: func(record *Record) { record.TemplateCacheID = "bbbbbbbbbbbb" }}, + {name: "unverified", mutate: func(record *Record) { record.Verifier = "" }}, + {name: "weak Docker disk", mutate: func(record *Record) { record.DockerDiskBytes = 512 << 20 }}, + {name: "too few jobs", mutate: func(record *Record) { record.ReliabilityJobs = 24 }}, + {name: "short soak", mutate: func(record *Record) { record.ReliabilityDuration = 119 * time.Minute }}, + {name: "slow create", mutate: func(record *Record) { record.CachedCreateP95 = 61 * time.Second }}, + {name: "slow online", mutate: func(record *Record) { record.QueueToOnlineP95 = 181 * time.Second }}, + {name: "slow remove", mutate: func(record *Record) { record.ForceRemoveP95 = 121 * time.Second }}, + {name: "slow workload", mutate: func(record *Record) { record.BuildxComposeSlowdownPct = 25.01 }}, + {name: "missing artifact digest", mutate: func(record *Record) { record.SBOMDigest = "" }}, + {name: "failed recovery gate", mutate: func(record *Record) { record.Gates.Recovery = false }}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + record := validRecord() + test.mutate(&record) + if err := Validate(record); err == nil { + t.Fatal("invalid promotion record accepted") + } + }) + } +} + +func validRecord() Record { + digest := "sha256:" + strings.Repeat("a", 64) + return Record{ + Platform: WindowsAMD64, + EPARRevision: digest, + Template: "epar-docker-sandboxes-catthehacker-full:version", + TemplateDigest: digest, + TemplateCacheID: strings.Repeat("a", 12), + TemplateMetadataDigest: digest, + TemplateArchiveDigest: digest, + PolicyFingerprint: digest, + EvidenceDigest: digest, + SBOMDigest: digest, + ProvenanceDigest: digest, + SoftwareInventoryDigest: digest, + VerifiedAt: time.Date(2026, 7, 23, 0, 0, 0, 0, time.UTC), + Verifier: "independent-security-verifier", + Gates: GateResults{ + Local: true, + Functional: true, + Recovery: true, + Security: true, + Policy: true, + Cleanup: true, + SecretScanning: true, + ConcurrentProvisioning: true, + IndependentSecurityReview: true, + }, + RootDiskBytes: 120 << 30, + DockerDiskBytes: 100 << 30, + MinHostFreeSpaceBytes: 50 << 30, + ReliabilityJobs: 25, + ReliabilityDuration: 2 * time.Hour, + CachedCreateP95: 60 * time.Second, + QueueToOnlineP95: 180 * time.Second, + ForceRemoveP95: 120 * time.Second, + BuildxComposeSlowdownPct: 25, + } +} diff --git a/internal/provider/dockersandboxes/promotion/space_unix.go b/internal/provider/dockersandboxes/promotion/space_unix.go new file mode 100644 index 0000000..f6847a7 --- /dev/null +++ b/internal/provider/dockersandboxes/promotion/space_unix.go @@ -0,0 +1,49 @@ +//go:build !windows + +package promotion + +import ( + "fmt" + "math" + "os" + "os/exec" + "runtime" + "strings" + "syscall" +) + +func sandboxHostSpace(path string) (HostSpace, error) { + var stats syscall.Statfs_t + if err := syscall.Statfs(path, &stats); err != nil { + return HostSpace{}, err + } + blockSize := uint64(stats.Bsize) + available := uint64(stats.Bavail) + total := uint64(stats.Blocks) + if blockSize != 0 && (available > math.MaxUint64/blockSize || total > math.MaxUint64/blockSize) { + return HostSpace{}, fmt.Errorf("filesystem space result overflow") + } + return HostSpace{AvailableBytes: available * blockSize, TotalBytes: total * blockSize}, nil +} + +func sandboxVirtualizationAvailable() error { + switch runtime.GOOS { + case "linux": + file, err := os.OpenFile("/dev/kvm", os.O_RDWR, 0) + if err != nil { + return fmt.Errorf("open /dev/kvm read/write: %w", err) + } + return file.Close() + case "darwin": + output, err := exec.Command("/usr/sbin/sysctl", "-n", "kern.hv_support").Output() + if err != nil { + return fmt.Errorf("query kern.hv_support: %w", err) + } + if strings.TrimSpace(string(output)) != "1" { + return fmt.Errorf("kern.hv_support did not report 1") + } + return nil + default: + return fmt.Errorf("unsupported native virtualization platform %s", runtime.GOOS) + } +} diff --git a/internal/provider/dockersandboxes/promotion/space_windows.go b/internal/provider/dockersandboxes/promotion/space_windows.go new file mode 100644 index 0000000..8b16ca7 --- /dev/null +++ b/internal/provider/dockersandboxes/promotion/space_windows.go @@ -0,0 +1,33 @@ +//go:build windows + +package promotion + +import ( + "errors" + "fmt" + + "golang.org/x/sys/windows" +) + +const processorFeatureVirtualizationFirmwareEnabled = 21 + +func sandboxHostSpace(path string) (HostSpace, error) { + pathPointer, err := windows.UTF16PtrFromString(path) + if err != nil { + return HostSpace{}, err + } + var available uint64 + var total uint64 + var free uint64 + if err := windows.GetDiskFreeSpaceEx(pathPointer, &available, &total, &free); err != nil { + return HostSpace{}, fmt.Errorf("GetDiskFreeSpaceEx: %w", err) + } + return HostSpace{AvailableBytes: available, TotalBytes: total}, nil +} + +func sandboxVirtualizationAvailable() error { + if !windows.IsProcessorFeaturePresent(processorFeatureVirtualizationFirmwareEnabled) { + return errors.New("Windows did not report firmware virtualization as enabled and available to the operating system") + } + return nil +} diff --git a/internal/provider/dockersandboxes/provider.go b/internal/provider/dockersandboxes/provider.go new file mode 100644 index 0000000..5283030 --- /dev/null +++ b/internal/provider/dockersandboxes/provider.go @@ -0,0 +1,1039 @@ +package dockersandboxes + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "fmt" + "io" + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" + "sync" + "sync/atomic" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes/staging" +) + +const ( + defaultOutputLimit = 8 << 20 + diagnosticOutputLimit = 256 << 10 + commandWaitDelay = 5 * time.Second + keepaliveStartupDelay = 500 * time.Millisecond +) + +const directWorkspaceVerificationScript = `set -euo pipefail +if test -n "${SSH_AUTH_SOCK:-}" || test -n "${SSH_AUTH_SOCK_GATEWAY:-}" || test -n "${SSH_AGENT_PID:-}" || test -e /run/ssh-agent.sock || test -L /run/ssh-agent.sock; then + echo "Docker Sandboxes exposed host SSH-agent forwarding; stop the daemon and restart it with SSH_AUTH_SOCK, SSH_AUTH_SOCK_GATEWAY, and SSH_AGENT_PID unset" >&2 + exit 1 +fi +workspace="$(pwd -P)" +test -n "${workspace}" +source_options="$(findmnt -T "${workspace}" -n -o OPTIONS)" +case ",${source_options}," in + *,rw,*) ;; + *) echo "dedicated host staging workspace is not read-write" >&2; exit 1 ;; +esac +test ! -e .git +test -z "$(find . -mindepth 1 -maxdepth 1 -print -quit)" +test ! -e /run/sandbox/source +! pgrep -x git-daemon >/dev/null` + +const runtimeVerificationScript = `set -euo pipefail +test -x /opt/epar/verify-template.sh +test -s /opt/epar/helpers.sha256 +cd /opt/epar +sha256sum -c helpers.sha256 >/dev/null +/opt/epar/verify-template.sh >/dev/null +docker info --format '{{json .ServerVersion}}'` + +type Provider struct { + Binary string + + runCommand runCommandFunc + activeMu sync.RWMutex + activeTemplate provider.TemplateArtifact + dryRun bool +} + +type instanceReceipt struct { + SchemaVersion int `json:"schemaVersion"` + StagingPath string `json:"stagingPath"` + StagingIdentity string `json:"stagingIdentity"` + Template string `json:"template"` + TemplateDigest string `json:"templateDigest"` +} + +// CachedTemplate is one image retained in the Docker Sandboxes template cache. +// CacheID is the provider's short cache identity, not a content digest. +type CachedTemplate struct { + Reference string + CacheID string + CreatedAt time.Time + SizeBytes int64 +} + +// HostReadiness is the validated machine-readable summary returned by +// `sbx diagnose --output json`. +type HostReadiness struct { + ChecksPassed int + ChecksWarned int + ChecksFailed int + ChecksSkipped int +} + +type commandRequest struct { + args []string + stdin io.Reader + environment map[string]string + stdout io.Writer + stderr io.Writer + sensitiveValues []string + operation string + outputLimit int +} + +type runCommandFunc func(ctx context.Context, request commandRequest) (provider.ExecResult, error) + +func New(binary string) *Provider { + return NewWithDryRun(binary, false) +} + +func NewWithDryRun(binary string, dryRun bool) *Provider { + if binary == "" { + binary = "sbx" + } + return &Provider{Binary: binary, dryRun: dryRun} +} + +// StartDaemon asks Docker Sandboxes to start its host daemon in the +// background. The command is intentionally exact so onboarding cannot invoke +// other daemon mutations through this path. +func (p *Provider) StartDaemon(ctx context.Context) error { + _, err := p.run(ctx, commandRequest{ + args: []string{"daemon", "start", "--detach"}, + operation: "start docker sandboxes daemon", + outputLimit: diagnosticOutputLimit, + }) + return err +} + +// VerifyHostReadiness requires machine-readable Docker Sandboxes diagnostics +// to contain at least one passing check and no failed checks. Warnings and +// skipped checks do not make an otherwise healthy installation unavailable. +func (p *Provider) VerifyHostReadiness(ctx context.Context) (HostReadiness, error) { + readiness, err := p.readHostReadiness(ctx) + if err != nil { + return HostReadiness{}, fmt.Errorf("%w; run 'sbx diagnose --output json' and review its output", err) + } + if readiness.ChecksFailed != 0 { + return HostReadiness{}, fmt.Errorf("docker sandboxes diagnostics reported %d failed check(s); run 'sbx diagnose --output json' and review the hints for each failed check", readiness.ChecksFailed) + } + if readiness.ChecksPassed == 0 { + return HostReadiness{}, fmt.Errorf("docker sandboxes diagnostics reported no passing checks; run 'sbx diagnose --output json' and review its check details") + } + return readiness, nil +} + +func (p *Provider) Create(ctx context.Context, request provider.CreateRequest) (provider.Instance, error) { + if p.dryRun { + return provider.Instance{}, fmt.Errorf("docker-sandboxes does not support dry-run instance creation because exact sandbox and template-cache readback is required") + } + if request.Template == "" && request.TemplateDigest == "" { + p.activeMu.RLock() + active := p.activeTemplate + p.activeMu.RUnlock() + request.Template = active.Reference + request.TemplateDigest = active.Digest + if request.RootDisk == "auto" { + request.RootDisk = active.RootDisk + } + } + if err := validateCreateRequest(request); err != nil { + return provider.Instance{}, err + } + if err := p.VerifyAdmission(ctx); err != nil { + return provider.Instance{}, err + } + if err := p.verifyImportedTemplate(ctx, request.Template, request.TemplateDigest); err != nil { + return provider.Instance{}, err + } + items, err := p.inventoryVerified(ctx) + if err != nil { + return provider.Instance{}, err + } + for _, item := range items { + if item.Instance.Name == request.Name { + return provider.Instance{}, fmt.Errorf("docker sandbox name is already allocated") + } + } + var ownedStaging staging.OwnedDirectory + if p.runCommand == nil { + stagingRoot, openErr := staging.Open(filepath.Dir(request.StagingPath)) + if openErr != nil { + return provider.Instance{}, openErr + } + if filepath.Clean(request.StagingPath) != filepath.Join(stagingRoot.Root(), request.Name) { + return provider.Instance{}, fmt.Errorf("Docker Sandboxes staging path must be the exact provider-owned path for %q", request.Name) + } + ownedStaging, err = stagingRoot.CreateOwned(request.Name) + if err != nil { + return provider.Instance{}, err + } + } else { + ownedStaging = staging.OwnedDirectory{Path: request.StagingPath, Identity: "test-staging-identity"} + } + + args := []string{ + "create", + "--name", request.Name, + "--cpus", strconv.Itoa(request.CPUs), + "--memory", request.Memory, + "--template", request.Template, + } + args = append(args, "shell", request.StagingPath) + environment := make(map[string]string, 2) + if request.RootDisk != "" { + environment["DOCKER_SANDBOXES_ROOT_SIZE"] = request.RootDisk + } + if request.DockerDisk != "" { + environment["DOCKER_SANDBOXES_DOCKER_SIZE"] = request.DockerDisk + } + if _, err := p.run(ctx, commandRequest{args: args, environment: environment, operation: "create docker sandbox"}); err != nil { + return provider.Instance{}, err + } + items, err = p.inventoryVerified(ctx) + if err != nil { + return provider.Instance{}, fmt.Errorf("docker sandbox was created but identity readback failed: %w", err) + } + for _, item := range items { + if item.Instance.Name == request.Name { + if item.Instance.ProviderID == "" { + return provider.Instance{}, fmt.Errorf("docker sandbox inventory omitted the stable provider id") + } + if item.Source != "shell" || !containsExactWorkspace(item.Workspaces, request.StagingPath) { + return provider.Instance{}, fmt.Errorf("docker sandbox inventory did not bind the exact shell workspace") + } + receipt, encodeErr := json.Marshal(instanceReceipt{ + SchemaVersion: 1, + StagingPath: ownedStaging.Path, + StagingIdentity: ownedStaging.Identity, + Template: request.Template, + TemplateDigest: request.TemplateDigest, + }) + if encodeErr != nil { + return provider.Instance{}, encodeErr + } + instance := item.Instance + instance.ReceiptVersion = "v1" + instance.Receipt = receipt + if err := p.verifyNoPublishedPorts(ctx, item.Instance); err != nil { + return instance, err + } + if err := p.verifyInspection(ctx, item.Instance, &request); err != nil { + return instance, err + } + if err := p.verifyDirectWorkspace(ctx, item.Instance); err != nil { + return instance, err + } + return instance, nil + } + } + return provider.Instance{}, fmt.Errorf("docker sandbox was not present in inventory after create") +} + +// ImportTemplate performs the one exact provider mutation required after the +// shared image coordinator has built and verified a local template archive. +func (p *Provider) ImportTemplate(ctx context.Context, archivePath string) error { + if archivePath == "" || strings.ContainsRune(archivePath, 0) { + return fmt.Errorf("Docker Sandboxes template archive path is required") + } + info, err := os.Lstat(archivePath) + if err != nil { + return fmt.Errorf("inspect Docker Sandboxes template archive: %w", err) + } + if !info.Mode().IsRegular() { + return fmt.Errorf("Docker Sandboxes template archive must be a regular file") + } + if _, err := p.run(ctx, commandRequest{ + args: []string{"template", "load", archivePath}, + operation: "load exact Docker Sandboxes runner template", + outputLimit: diagnosticOutputLimit, + }); err != nil { + return err + } + return nil +} + +func (p *Provider) VerifyImportedTemplate(ctx context.Context, artifact provider.TemplateArtifact) error { + if artifact.Reference == "" || !validFullTemplateDigest(artifact.Digest) { + return fmt.Errorf("Docker Sandboxes template reference and digest are required") + } + if artifact.CacheID != strings.TrimPrefix(artifact.Digest, "sha256:")[:12] { + return fmt.Errorf("Docker Sandboxes template cache ID does not match its full digest") + } + return p.verifyImportedTemplate(ctx, artifact.Reference, artifact.Digest) +} + +func (p *Provider) ActivateTemplate(artifact provider.TemplateArtifact) error { + if artifact.Reference == "" || !validFullTemplateDigest(artifact.Digest) { + return fmt.Errorf("cannot activate an invalid Docker Sandboxes template identity") + } + if artifact.CacheID != strings.TrimPrefix(artifact.Digest, "sha256:")[:12] { + return fmt.Errorf("cannot activate a Docker Sandboxes template with a mismatched cache ID") + } + if artifact.Platform != "linux/amd64" && artifact.Platform != "linux/arm64" { + return fmt.Errorf("cannot activate a Docker Sandboxes template for unsupported platform %q", artifact.Platform) + } + if !sizePattern.MatchString(artifact.RootDisk) { + return fmt.Errorf("cannot activate a Docker Sandboxes template without a resolved root-disk size") + } + p.activeMu.Lock() + p.activeTemplate = artifact + p.activeMu.Unlock() + return nil +} + +// RemoveTemplate removes one exact imported template cache identity. The +// Docker-managed shell-docker base template is never an EPAR cleanup target. +func (p *Provider) RemoveTemplate(ctx context.Context, artifact provider.TemplateArtifact) error { + if artifact.Reference == "docker.io/docker/sandbox-templates:shell-docker" || artifact.Reference == "docker/sandbox-templates:shell-docker" { + return fmt.Errorf("refusing to remove the Docker Sandboxes shell-docker base template") + } + if artifact.CacheID == "" || len(artifact.CacheID) != 12 { + return fmt.Errorf("Docker Sandboxes template cleanup requires an exact 12-character cache ID") + } + instances, err := p.Inventory(ctx) + if err != nil { + return fmt.Errorf("verify active Docker Sandboxes before template cleanup: %w", err) + } + if len(instances) != 0 { + return fmt.Errorf("refusing template cleanup while %d Docker Sandbox instance(s) exist", len(instances)) + } + templates, err := p.CachedTemplates(ctx) + if err != nil { + return err + } + found := false + for _, item := range templates { + if item.CacheID != artifact.CacheID { + continue + } + if item.Reference != artifact.Reference { + return fmt.Errorf("Docker Sandboxes template cache identity %s now belongs to %s, not %s", artifact.CacheID, item.Reference, artifact.Reference) + } + found = true + break + } + if !found { + return nil + } + if _, err := p.run(ctx, commandRequest{ + args: []string{"template", "rm", artifact.CacheID}, + operation: "remove exact Docker Sandboxes runner template", + outputLimit: diagnosticOutputLimit, + }); err != nil { + return err + } + templates, err = p.CachedTemplates(ctx) + if err != nil { + return err + } + for _, item := range templates { + if item.CacheID == artifact.CacheID { + return fmt.Errorf("Docker Sandboxes template %s still exists after exact removal", artifact.CacheID) + } + } + return nil +} + +func (p *Provider) ObserveTemplate(ctx context.Context, artifact provider.TemplateArtifact) (bool, error) { + if artifact.Reference == "" || artifact.CacheID == "" || len(artifact.CacheID) != 12 { + return false, fmt.Errorf("Docker Sandboxes template observation requires an exact reference and 12-character cache ID") + } + templates, err := p.CachedTemplates(ctx) + if err != nil { + return false, err + } + for _, item := range templates { + if item.CacheID != artifact.CacheID { + continue + } + if item.Reference != artifact.Reference { + return false, fmt.Errorf("Docker Sandboxes template cache identity %s now belongs to %s, not %s", artifact.CacheID, item.Reference, artifact.Reference) + } + return true, nil + } + return false, nil +} + +// VerifyAdmission fail-closes on provider-wide channels Docker Sandboxes can +// inject into every sandbox. EPAR deliberately does not consume global secrets; +// repository and workflow input can never opt out of this check. +func (p *Provider) VerifyAdmission(ctx context.Context) error { + if _, err := p.VerifyHostReadiness(ctx); err != nil { + return err + } + return p.verifyNoGlobalSecrets(ctx) +} + +func (p *Provider) VerifyInstanceAdmission(ctx context.Context, instance provider.Instance) error { + present, err := p.assertIdentity(ctx, instance) + if err != nil { + return err + } + if !present { + return fmt.Errorf("docker sandbox is missing") + } + if err := p.verifyNoPublishedPorts(ctx, instance); err != nil { + return err + } + return p.verifyInspection(ctx, instance, nil) +} + +func (p *Provider) verifyInspection(ctx context.Context, instance provider.Instance, expected *provider.CreateRequest) error { + result, err := p.run(ctx, commandRequest{ + args: []string{"inspect", "--json", instance.Name}, + operation: "verify docker sandbox attached capabilities", + outputLimit: diagnosticOutputLimit, + }) + if err != nil { + return err + } + var inspection map[string]json.RawMessage + if err := decodeStrictJSON([]byte(result.Stdout), &inspection); err != nil { + return fmt.Errorf("docker sandbox inspection returned an unsupported JSON schema") + } + if stringValue(inspection["name"]) != instance.Name || stringValue(inspection["agent"]) != "shell" || strings.TrimSpace(stringValue(inspection["daemon_version"])) == "" { + return fmt.Errorf("docker sandbox inspection did not match the exact shell runtime") + } + var mcpGateway bool + if raw, ok := inspection["mcp_gateway"]; !ok || json.Unmarshal(raw, &mcpGateway) != nil || mcpGateway { + return fmt.Errorf("docker sandbox inspection reported an enabled MCP gateway") + } + if expected != nil && (stringValue(inspection["image"]) != expected.Template || stringValue(inspection["image_digest"]) != expected.TemplateDigest || stringValue(inspection["workspace"]) != expected.StagingPath) { + return fmt.Errorf("docker sandbox inspection did not bind the exact template identity and staging path") + } + for _, field := range []string{"kits", "secrets", "published_ports", "ports", "auth", "auth_mode", "docker_auth"} { + value, ok := inspection[field] + if field == "kits" && !ok { + return fmt.Errorf("docker sandbox inspection omitted required attached-capability field %q", field) + } + if ok && !emptyJSONValue(value) { + return fmt.Errorf("docker sandbox inspection reported forbidden attached capability %q", field) + } + } + return nil +} + +func stringValue(raw json.RawMessage) string { + var value string + if len(raw) == 0 || json.Unmarshal(raw, &value) != nil { + return "" + } + return value +} + +func emptyJSONValue(raw json.RawMessage) bool { + if len(raw) == 0 { + return true + } + var value any + if json.Unmarshal(raw, &value) != nil { + return false + } + switch typed := value.(type) { + case nil: + return true + case bool: + return !typed + case string: + return typed == "" + case float64: + return typed == 0 + case []any: + return len(typed) == 0 + case map[string]any: + return len(typed) == 0 + default: + return false + } +} + +func (p *Provider) verifyNoPublishedPorts(ctx context.Context, instance provider.Instance) error { + result, err := p.run(ctx, commandRequest{ + args: []string{"ports", instance.Name, "--json"}, + operation: "verify docker sandbox has no published ports", + outputLimit: diagnosticOutputLimit, + }) + if err != nil { + return err + } + var ports []map[string]json.RawMessage + if err := decodeStrictJSON([]byte(result.Stdout), &ports); err != nil || ports == nil { + return fmt.Errorf("docker sandbox published-port inventory returned an unsupported JSON schema") + } + if len(ports) != 0 { + return fmt.Errorf("docker sandbox reported a forbidden published port") + } + return nil +} + +func (p *Provider) verifyNoGlobalSecrets(ctx context.Context) error { + result, err := p.run(ctx, commandRequest{ + args: []string{"secret", "ls", "-g"}, + operation: "verify docker sandboxes global secret isolation", + outputLimit: 64 << 10, + }) + if err != nil { + return err + } + if strings.TrimSpace(strings.ReplaceAll(result.Stdout, "\r\n", "\n")) != `No secrets found for scope "(global)".` { + return fmt.Errorf("docker sandboxes global secrets are configured; EPAR refuses to expose shared registry or service credentials to workflow sandboxes") + } + return nil +} + +func (p *Provider) verifyDirectWorkspace(ctx context.Context, instance provider.Instance) error { + result, err := p.run(ctx, commandRequest{ + args: []string{"exec", "-i", instance.Name, "--", "bash", "-lc", directWorkspaceVerificationScript}, + stdin: strings.NewReader(""), + operation: "verify dedicated docker sandbox staging workspace", + }) + if err != nil { + return err + } + if strings.TrimSpace(result.Stdout) != "" { + return fmt.Errorf("dedicated docker sandbox staging verification returned unexpected output") + } + return nil +} + +func containsExactWorkspace(workspaces []string, expected string) bool { + for _, workspace := range workspaces { + if workspace == expected { + return true + } + } + return false +} + +func (p *Provider) Start(ctx context.Context, instance provider.Instance, opts provider.StartOptions) (*provider.RunningProcess, error) { + present, err := p.assertIdentity(ctx, instance) + if err != nil || !present { + if err == nil { + return nil, fmt.Errorf("docker sandbox is missing") + } + return nil, err + } + request := commandRequest{ + args: []string{"exec", "-i", instance.Name, "--", "/bin/sleep", "infinity"}, + stdin: strings.NewReader(""), + stdout: opts.Stdout, + stderr: opts.Stderr, + operation: "start docker sandbox with a managed keepalive", + } + if p.runCommand != nil { + if _, err := p.run(ctx, request); err != nil { + return nil, err + } + return &provider.RunningProcess{Name: instance.Name}, nil + } + return p.startKeepalive(ctx, instance.Name, request) +} + +func (p *Provider) startKeepalive(ctx context.Context, name string, request commandRequest) (*provider.RunningProcess, error) { + if err := validateCommandRequest(request); err != nil { + return nil, err + } + command := exec.CommandContext(ctx, p.Binary, request.args...) + isolateKeepaliveProcess(command) + command.WaitDelay = commandWaitDelay + command.Stdin = request.stdin + command.Env = childEnvironment(request.environment) + stdout := &boundedBuffer{limit: defaultOutputLimit} + stderr := &boundedBuffer{limit: defaultOutputLimit} + command.Stdout = captureWriter(stdout, request.stdout) + command.Stderr = captureWriter(stderr, request.stderr) + if err := command.Start(); err != nil { + return nil, fmt.Errorf("%s failed: %w", request.operation, err) + } + finished := make(chan error, 1) + go func() { + finished <- command.Wait() + }() + timer := time.NewTimer(keepaliveStartupDelay) + defer timer.Stop() + select { + case err := <-finished: + detail := strings.TrimSpace(stderr.String()) + if err == nil { + err = fmt.Errorf("keepalive command exited before startup completed") + } + if detail != "" { + return nil, fmt.Errorf("%s failed: %w: %s", request.operation, err, detail) + } + return nil, fmt.Errorf("%s failed: %w", request.operation, err) + case <-ctx.Done(): + return nil, ctx.Err() + case <-timer.C: + return &provider.RunningProcess{Name: name, PID: command.Process.Pid}, nil + } +} + +func (p *Provider) VerifyRuntime(ctx context.Context, instance provider.Instance) (provider.RuntimeInfo, error) { + present, err := p.assertIdentity(ctx, instance) + if err != nil { + return provider.RuntimeInfo{}, err + } + if !present { + return provider.RuntimeInfo{}, fmt.Errorf("docker sandbox is missing") + } + result, err := p.run(ctx, commandRequest{ + args: []string{"exec", "-i", instance.Name, "--", "bash", "-lc", runtimeVerificationScript}, + stdin: strings.NewReader(""), + operation: "verify docker sandbox runtime", + }) + if err != nil { + return provider.RuntimeInfo{}, err + } + var version string + if err := decodeStrictJSON([]byte(strings.TrimSpace(result.Stdout)), &version); err != nil || strings.TrimSpace(version) == "" { + return provider.RuntimeInfo{}, fmt.Errorf("docker sandbox runtime returned an unsupported verification schema") + } + return provider.RuntimeInfo{Ready: true, Runtime: "docker", Version: version}, nil +} + +func (*Provider) Address(context.Context, provider.Instance, int) (string, bool, error) { + return "", false, nil +} + +func (p *Provider) Exec(ctx context.Context, instance provider.Instance, command []string, opts provider.ExecOptions) (provider.ExecResult, error) { + if err := validateGuestCommand(command, opts); err != nil { + return provider.ExecResult{}, err + } + present, err := p.assertIdentity(ctx, instance) + if err != nil { + return provider.ExecResult{}, err + } + if !present { + return provider.ExecResult{}, fmt.Errorf("docker sandbox is missing") + } + args := make([]string, 0, len(command)+5) + args = append(args, "exec", "-i", instance.Name, "--") + args = append(args, command...) + return p.run(ctx, commandRequest{ + args: args, + stdin: execOptionsReader(opts), + stdout: opts.Stdout, + stderr: opts.Stderr, + sensitiveValues: opts.SensitiveValues, + operation: "execute in docker sandbox", + }) +} + +func execOptionsReader(opts provider.ExecOptions) io.Reader { + if opts.StdinReader != nil { + return opts.StdinReader + } + return strings.NewReader(opts.Stdin) +} + +func (p *Provider) Diagnostics(ctx context.Context, instance provider.Instance) (provider.Diagnostics, error) { + present, err := p.assertIdentity(ctx, instance) + if err != nil { + return provider.Diagnostics{}, err + } + if !present { + return provider.Diagnostics{}, fmt.Errorf("docker sandbox is missing") + } + statusResult, err := p.run(ctx, commandRequest{ + args: []string{"daemon", "status", "--json"}, + operation: "read docker sandbox daemon status", + outputLimit: diagnosticOutputLimit, + }) + if err != nil { + return provider.Diagnostics{}, err + } + daemonState, daemonHealthy, err := parseDaemonStatus([]byte(statusResult.Stdout)) + if err != nil { + return provider.Diagnostics{}, err + } + readiness, err := p.readHostReadiness(ctx) + if err != nil { + return provider.Diagnostics{}, err + } + return provider.Diagnostics{ + Healthy: daemonHealthy && readiness.ChecksPassed > 0 && readiness.ChecksFailed == 0, + DaemonState: daemonState, + ChecksPassed: readiness.ChecksPassed, + ChecksWarned: readiness.ChecksWarned, + ChecksFailed: readiness.ChecksFailed, + ChecksSkipped: readiness.ChecksSkipped, + }, nil +} + +func (p *Provider) readHostReadiness(ctx context.Context) (HostReadiness, error) { + diagnoseResult, err := p.run(ctx, commandRequest{ + args: []string{"diagnose", "--output", "json"}, + operation: "diagnose docker sandboxes", + outputLimit: diagnosticOutputLimit, + }) + if err != nil { + return HostReadiness{}, err + } + passed, warned, failed, skipped, err := parseDiagnose([]byte(diagnoseResult.Stdout)) + if err != nil { + return HostReadiness{}, err + } + return HostReadiness{ + ChecksPassed: passed, + ChecksWarned: warned, + ChecksFailed: failed, + ChecksSkipped: skipped, + }, nil +} + +func (p *Provider) Stop(ctx context.Context, instance provider.Instance) error { + present, err := p.assertIdentity(ctx, instance) + if err != nil || !present { + return err + } + result, err := p.run(ctx, commandRequest{args: []string{"stop", instance.Name}, operation: "stop docker sandbox"}) + if err != nil && isMissingSandbox(result.Stdout+"\n"+result.Stderr+"\n"+err.Error()) { + return nil + } + return err +} + +func (p *Provider) Delete(ctx context.Context, instance provider.Instance) error { + present, err := p.assertIdentity(ctx, instance) + if err != nil || !present { + return err + } + var receipt instanceReceipt + if p.runCommand == nil { + if instance.ReceiptVersion != "v1" || json.Unmarshal(instance.Receipt, &receipt) != nil || receipt.SchemaVersion != 1 || receipt.StagingPath == "" || receipt.StagingIdentity == "" { + return fmt.Errorf("refusing Docker Sandbox deletion without an exact staging ownership receipt") + } + } + result, err := p.run(ctx, commandRequest{args: []string{"rm", "--force", instance.Name}, operation: "delete docker sandbox"}) + if err != nil && isMissingSandbox(result.Stdout+"\n"+result.Stderr+"\n"+err.Error()) { + err = nil + } + if err != nil { + return err + } + if p.runCommand == nil { + stagingRoot, openErr := staging.Open(filepath.Dir(receipt.StagingPath)) + if openErr != nil { + return openErr + } + if filepath.Clean(receipt.StagingPath) != filepath.Join(stagingRoot.Root(), instance.Name) { + return fmt.Errorf("refusing Docker Sandbox staging cleanup outside the exact owned path") + } + if purgeErr := stagingRoot.PurgeOwned(instance.Name, receipt.StagingIdentity); purgeErr != nil { + return purgeErr + } + } + return nil +} + +func (p *Provider) Inventory(ctx context.Context) ([]provider.InventoryItem, error) { + return p.inventoryVerified(ctx) +} + +func (p *Provider) inventoryVerified(ctx context.Context) ([]provider.InventoryItem, error) { + result, err := p.run(ctx, commandRequest{args: []string{"ls", "--json"}, operation: "inventory docker sandboxes"}) + if err != nil { + return nil, err + } + return parseInventory([]byte(result.Stdout)) +} + +// CachedTemplates returns the strictly parsed, host-level Docker Sandboxes +// template cache inventory. It does not create, load, or otherwise mutate a +// template. +func (p *Provider) CachedTemplates(ctx context.Context) ([]CachedTemplate, error) { + result, err := p.run(ctx, commandRequest{ + args: []string{"template", "ls", "--json"}, + operation: "read docker sandbox template cache", + outputLimit: diagnosticOutputLimit, + }) + if err != nil { + return nil, err + } + images, err := parseTemplateInventory([]byte(result.Stdout)) + if err != nil { + return nil, err + } + templates := make([]CachedTemplate, 0, len(images)) + for _, image := range images { + templates = append(templates, CachedTemplate{ + Reference: image.Repository + ":" + image.Tag, + CacheID: image.ID, + CreatedAt: image.CreatedAt, + SizeBytes: image.Size, + }) + } + return templates, nil +} + +func (p *Provider) verifyImportedTemplate(ctx context.Context, reference, digest string) error { + result, err := p.run(ctx, commandRequest{args: []string{"template", "ls", "--json"}, operation: "verify cached docker sandbox template"}) + if err != nil { + return err + } + images, err := parseTemplateInventory([]byte(result.Stdout)) + if err != nil { + return err + } + repository, tag, err := splitTemplateReference(reference) + if err != nil { + return err + } + wantCacheID := strings.TrimPrefix(digest, "sha256:")[:12] + for _, image := range images { + if image.Repository == repository && image.Tag == tag { + if image.ID != wantCacheID { + return fmt.Errorf("cached Docker Sandbox template ID %s does not match imported archive identity %s", image.ID, wantCacheID) + } + return nil + } + } + return fmt.Errorf("%w: configured Docker Sandbox template was not present in the authoritative Sandbox cache", provider.ErrTemplateNotFound) +} + +func validFullTemplateDigest(value string) bool { + if len(value) != len("sha256:")+64 || !strings.HasPrefix(value, "sha256:") { + return false + } + for _, character := range value[len("sha256:"):] { + if !((character >= '0' && character <= '9') || (character >= 'a' && character <= 'f')) { + return false + } + } + return true +} + +func (p *Provider) assertIdentity(ctx context.Context, instance provider.Instance) (bool, error) { + if err := validateInstance(instance, true); err != nil { + return false, err + } + items, err := p.inventoryVerified(ctx) + if err != nil { + return false, err + } + for _, item := range items { + if item.Instance.Name != instance.Name { + continue + } + if item.Instance.ProviderID != instance.ProviderID { + return false, fmt.Errorf("docker sandbox identity changed") + } + return true, nil + } + return false, nil +} + +func (p *Provider) run(ctx context.Context, request commandRequest) (provider.ExecResult, error) { + if err := validateCommandRequest(request); err != nil { + return provider.ExecResult{}, err + } + if request.outputLimit == 0 { + request.outputLimit = defaultOutputLimit + } + if err := ctx.Err(); err != nil { + return provider.ExecResult{}, err + } + bufferedStdout, bufferedStderr, flush := provider.BufferSensitiveSinks(request.sensitiveValues, request.stdout, request.stderr) + request.stdout = bufferedStdout + request.stderr = bufferedStderr + + var result provider.ExecResult + var runErr error + if p.runCommand != nil { + result, runErr = p.runCommand(ctx, request) + } else { + result, runErr = p.runRaw(ctx, request) + } + if len(result.Stdout) > request.outputLimit || len(result.Stderr) > request.outputLimit { + runErr = errors.Join(runErr, fmt.Errorf("%s exceeded the output limit", request.operation)) + result.Stdout = truncate(result.Stdout, request.outputLimit) + result.Stderr = truncate(result.Stderr, request.outputLimit) + } + if ctxErr := ctx.Err(); ctxErr != nil { + runErr = errors.Join(ctxErr, runErr) + } + result, finishErr := provider.FinishSensitiveExecution(result, runErr, flush(), request.sensitiveValues) + if finishErr != nil { + detail := strings.TrimSpace(result.Stderr) + if detail != "" { + finishErr = fmt.Errorf("%s failed: %w: %s", request.operation, finishErr, detail) + } else { + finishErr = fmt.Errorf("%s failed: %w", request.operation, finishErr) + } + finishErr = provider.RedactError(finishErr, request.sensitiveValues...) + } + return result, finishErr +} + +func (p *Provider) runRaw(ctx context.Context, request commandRequest) (provider.ExecResult, error) { + cmd := exec.CommandContext(ctx, p.Binary, request.args...) + cmd.WaitDelay = commandWaitDelay + var cancellationKilledProcess atomic.Bool + defaultCancel := cmd.Cancel + cmd.Cancel = func() error { + err := defaultCancel() + if err == nil { + cancellationKilledProcess.Store(true) + } + return err + } + cmd.Stdin = request.stdin + cmd.Env = childEnvironment(request.environment) + stdout := &boundedBuffer{limit: request.outputLimit} + stderr := &boundedBuffer{limit: request.outputLimit} + cmd.Stdout = captureWriter(stdout, request.stdout) + cmd.Stderr = captureWriter(stderr, request.stderr) + err := cmd.Run() + if cancellationKilledProcess.Load() { + if ctxErr := ctx.Err(); ctxErr != nil { + err = ctxErr + } + } + result := provider.ExecResult{Stdout: stdout.String(), Stderr: stderr.String()} + if stdout.exceeded || stderr.exceeded { + err = errors.Join(err, fmt.Errorf("output limit exceeded")) + } + return result, err +} + +func validateCommandRequest(request commandRequest) error { + if len(request.args) == 0 { + return fmt.Errorf("refusing to invoke docker sandboxes without a subcommand") + } + if request.operation == "" { + return fmt.Errorf("docker sandboxes operation label is required") + } + switch request.args[0] { + case "version", "create", "exec", "daemon", "diagnose", "stop", "rm", "ls", "template", "policy", "inspect", "ports": + case "secret": + if len(request.args) != 3 || request.args[1] != "ls" || request.args[2] != "-g" { + return fmt.Errorf("only exact global-secret absence verification is permitted") + } + default: + return fmt.Errorf("docker sandboxes command %q is not permitted", request.args[0]) + } + if request.args[0] == "inspect" && (len(request.args) != 3 || request.args[1] != "--json" || !sandboxNamePattern.MatchString(request.args[2])) { + return fmt.Errorf("only exact machine-readable sandbox inspection is permitted") + } + if request.args[0] == "ports" && (len(request.args) != 3 || !sandboxNamePattern.MatchString(request.args[1]) || request.args[2] != "--json") { + return fmt.Errorf("only exact machine-readable published-port absence verification is permitted") + } + if request.args[0] == "daemon" && (len(request.args) != 3 || !((request.args[1] == "status" && request.args[2] == "--json") || (request.args[1] == "start" && request.args[2] == "--detach"))) { + return fmt.Errorf("only exact daemon status or detached-start operations are permitted") + } + for _, arg := range request.args { + if strings.ContainsRune(arg, 0) { + return fmt.Errorf("docker sandboxes argument contains a null byte") + } + } + for key := range request.environment { + if key != "DOCKER_SANDBOXES_ROOT_SIZE" && key != "DOCKER_SANDBOXES_DOCKER_SIZE" { + return fmt.Errorf("docker sandboxes child environment contains a forbidden override") + } + } + return nil +} + +func childEnvironment(additions map[string]string) []string { + environment := make([]string, 0, len(os.Environ())+len(additions)) + for _, item := range os.Environ() { + key, _, _ := strings.Cut(item, "=") + upperKey := strings.ToUpper(key) + if strings.HasPrefix(upperKey, "DOCKER_SANDBOXES_") || upperKey == "SSH_AUTH_SOCK" || upperKey == "SSH_AUTH_SOCK_GATEWAY" || upperKey == "SSH_AGENT_PID" { + continue + } + environment = append(environment, item) + } + for _, key := range []string{"DOCKER_SANDBOXES_ROOT_SIZE", "DOCKER_SANDBOXES_DOCKER_SIZE"} { + if value := additions[key]; value != "" { + environment = append(environment, key+"="+value) + } + } + return environment +} + +func captureWriter(capture io.Writer, sink io.Writer) io.Writer { + if sink == nil { + return capture + } + return io.MultiWriter(capture, sink) +} + +type boundedBuffer struct { + buffer bytes.Buffer + limit int + exceeded bool +} + +func (buffer *boundedBuffer) Write(data []byte) (int, error) { + written := len(data) + remaining := buffer.limit - buffer.buffer.Len() + if remaining > 0 { + if len(data) < remaining { + remaining = len(data) + } + _, _ = buffer.buffer.Write(data[:remaining]) + } + if len(data) > remaining { + buffer.exceeded = true + } + return written, nil +} + +func (buffer *boundedBuffer) String() string { return buffer.buffer.String() } + +func truncate(value string, limit int) string { + if len(value) <= limit { + return value + } + return value[:limit] +} + +func isMissingSandbox(text string) bool { + text = strings.ToLower(text) + return strings.Contains(text, "sandbox not found") || strings.Contains(text, "no such sandbox") || strings.Contains(text, "status 404") +} + +func decodeStrictJSON(data []byte, destination any) error { + decoder := json.NewDecoder(bytes.NewReader(data)) + if err := decoder.Decode(destination); err != nil { + return err + } + if decoder.More() { + return fmt.Errorf("unexpected trailing json value") + } + var trailing any + if err := decoder.Decode(&trailing); err != io.EOF { + if err == nil { + return fmt.Errorf("unexpected trailing json value") + } + return err + } + return nil +} + +var _ provider.Lifecycle = (*Provider)(nil) +var _ provider.AdmissionVerifier = (*Provider)(nil) +var _ provider.InstanceAdmissionVerifier = (*Provider)(nil) +var _ provider.PolicyManager = (*Provider)(nil) +var _ provider.TemplateArtifactRuntime = (*Provider)(nil) +var _ provider.TemplateArtifactCleaner = (*Provider)(nil) +var _ provider.TemplateArtifactObserver = (*Provider)(nil) diff --git a/internal/provider/dockersandboxes/provider_test.go b/internal/provider/dockersandboxes/provider_test.go new file mode 100644 index 0000000..7fbb202 --- /dev/null +++ b/internal/provider/dockersandboxes/provider_test.go @@ -0,0 +1,1104 @@ +package dockersandboxes + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "io" + "os" + "os/exec" + "path/filepath" + "reflect" + "runtime" + "strconv" + "strings" + "sync" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +const ( + testName = "epar-sandbox-1" + testID = "9b6dbdf3-2ef4-47cb-8f55-55b26a790c8b" + testTemplate = "docker.io/docker/sandbox-templates:shell-docker" + testDigest = "sha256:39cf20eca8610000000000000000000000000000000000000000000000000000" + emptyPortsJSON = `[]` + templateListJSON = `{"images":[{"id":"39cf20eca861","repository":"docker.io/docker/sandbox-templates","tag":"shell-docker","flavor":"shell-docker","created_at":"2026-07-22T07:13:19Z","size":599103243}]}` + healthyDiagnoseJSON = `{"version":"1.0","checks":[{"name":"daemon","status":"pass","message":"healthy","detail":"","hint":""}],"summary":{"pass":1,"warn":0,"fail":0,"skip":0}}` +) + +var ( + testWorkspace = providerTestWorkspace() + readyListJSON = `{"sandboxes":[{"id":"9b6dbdf3-2ef4-47cb-8f55-55b26a790c8b","name":"epar-sandbox-1","status":"running","workspaces":[` + strconv.Quote(testWorkspace) + `],"agent":"shell","additive_field":true}]}` + inspectionJSON = `{"name":"epar-sandbox-1","agent":"shell","kits":[],"state":"running","image":"docker.io/docker/sandbox-templates:shell-docker","image_digest":"sha256:39cf20eca8610000000000000000000000000000000000000000000000000000","workspace":` + strconv.Quote(testWorkspace) + `,"network":"epar-sandbox-1","network_policy":{"scope":"global"},"proxy":"172.17.0.1:3128","mcp_gateway":false,"sessions":0,"daemon_version":"fixture-current","daemon_uptime":"1h"}` + testInstance = provider.Instance{Name: testName, ProviderID: testID, Source: "shell", State: "running"} +) + +func providerTestWorkspace() string { + if runtime.GOOS == "windows" { + return filepath.Join(filepath.VolumeName(os.TempDir())+string(filepath.Separator), "var", "lib", "epar", "staging", "job-1") + } + return filepath.Join(string(filepath.Separator), "var", "lib", "epar", "staging", "job-1") +} + +func TestCreateDryRunFailsBeforeProviderSideEffects(t *testing.T) { + p := NewWithDryRun("sbx", true) + called := false + p.runCommand = func(context.Context, commandRequest) (provider.ExecResult, error) { + called = true + return provider.ExecResult{}, nil + } + _, err := p.Create(context.Background(), provider.CreateRequest{Name: testName}) + if err == nil || !strings.Contains(err.Error(), "does not support dry-run") { + t.Fatalf("Create() error = %v", err) + } + if called { + t.Fatal("dry-run Create invoked a provider command") + } +} + +func TestStartDaemonUsesExactDetachedCommand(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"daemon", "start", "--detach"}}, + ) + if err := p.StartDaemon(context.Background()); err != nil { + t.Fatal(err) + } + done() +} + +func TestRemoveTemplateUsesExactCacheIDAndRefusesLiveSandboxes(t *testing.T) { + artifact := provider.TemplateArtifact{ + Reference: "docker.io/library/epar-template:one", + CacheID: "aaaaaaaaaaaa", + Digest: "sha256:aaaaaaaaaaaa0000000000000000000000000000000000000000000000000000", + } + template := `{"images":[{"id":"aaaaaaaaaaaa","repository":"docker.io/library/epar-template","tag":"one","flavor":"","created_at":"2026-07-29T00:00:00Z","size":1024}]}` + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: `{"sandboxes":[]}`}}, + commandStep{args: []string{"template", "ls", "--json"}, result: provider.ExecResult{Stdout: template}}, + commandStep{args: []string{"template", "rm", "aaaaaaaaaaaa"}}, + commandStep{args: []string{"template", "ls", "--json"}, result: provider.ExecResult{Stdout: `{"images":[]}`}}, + ) + if err := p.RemoveTemplate(context.Background(), artifact); err != nil { + t.Fatal(err) + } + done() + + p, done = scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + ) + if err := p.RemoveTemplate(context.Background(), artifact); err == nil || !strings.Contains(err.Error(), "while 1 Docker Sandbox") { + t.Fatalf("RemoveTemplate() error = %v, want live Sandbox refusal", err) + } + done() +} + +func TestObserveTemplateRequiresExactReferenceAndCacheID(t *testing.T) { + artifact := provider.TemplateArtifact{Reference: "docker.io/library/epar-template:one", CacheID: "aaaaaaaaaaaa"} + p, done := scriptedProvider(t, + commandStep{args: []string{"template", "ls", "--json"}, result: provider.ExecResult{Stdout: `{"images":[{"id":"aaaaaaaaaaaa","repository":"docker.io/library/epar-template","tag":"one","flavor":"","created_at":"2026-07-29T00:00:00Z","size":1024}]}`}}, + ) + exists, err := p.ObserveTemplate(context.Background(), artifact) + if err != nil || !exists { + t.Fatalf("ObserveTemplate() = %t, %v, want true", exists, err) + } + done() +} + +type commandStep struct { + args []string + result provider.ExecResult + err error + environment map[string]string + stdin string + streamOut string + streamErr string +} + +type cancellationSignalWriter struct { + once sync.Once + started chan struct{} +} + +func (writer *cancellationSignalWriter) Write(data []byte) (int, error) { + writer.once.Do(func() { close(writer.started) }) + return len(data), nil +} + +func scriptedProvider(t *testing.T, steps ...commandStep) (*Provider, func()) { + t.Helper() + p := New("sbx-test-double") + index := 0 + p.runCommand = func(_ context.Context, request commandRequest) (provider.ExecResult, error) { + t.Helper() + if index >= len(steps) { + t.Fatalf("unexpected command: %#v", request.args) + } + step := steps[index] + index++ + if !reflect.DeepEqual(request.args, step.args) { + t.Fatalf("command %d args = %#v, want %#v", index, request.args, step.args) + } + if !reflect.DeepEqual(request.environment, step.environment) { + t.Fatalf("command %d environment = %#v, want %#v", index, request.environment, step.environment) + } + if request.stdin != nil { + data, err := io.ReadAll(request.stdin) + if err != nil { + t.Fatal(err) + } + if string(data) != step.stdin { + t.Fatalf("command %d stdin = %q, want %q", index, data, step.stdin) + } + } else if step.stdin != "" { + t.Fatalf("command %d did not receive stdin", index) + } + if request.stdout != nil { + _, _ = io.WriteString(request.stdout, step.streamOut) + } + if request.stderr != nil { + _, _ = io.WriteString(request.stderr, step.streamErr) + } + return step.result, step.err + } + return p, func() { + t.Helper() + if index != len(steps) { + t.Fatalf("executed %d commands, want %d", index, len(steps)) + } + } +} + +func TestCreateUsesHealthyDiagnosticsAndExactArgv(t *testing.T) { + wantCreate := []string{ + "create", "--name", testName, + "--cpus", "4", + "--memory", "8g", + "--template", "docker.io/docker/sandbox-templates:shell-docker", + "shell", testWorkspace, + } + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: healthyDiagnoseJSON}}, + commandStep{args: []string{"secret", "ls", "-g"}, result: provider.ExecResult{Stdout: `No secrets found for scope "(global)".`}}, + commandStep{args: []string{"template", "ls", "--json"}, result: provider.ExecResult{Stdout: templateListJSON}}, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: `{"sandboxes":[]}`}}, + commandStep{ + args: wantCreate, + environment: map[string]string{ + "DOCKER_SANDBOXES_ROOT_SIZE": "40g", + "DOCKER_SANDBOXES_DOCKER_SIZE": "60g", + }, + }, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"ports", testName, "--json"}, result: provider.ExecResult{Stdout: emptyPortsJSON}}, + commandStep{args: []string{"inspect", "--json", testName}, result: provider.ExecResult{Stdout: inspectionJSON}}, + commandStep{args: []string{"exec", "-i", testName, "--", "bash", "-lc", directWorkspaceVerificationScript}}, + ) + instance, err := p.Create(context.Background(), provider.CreateRequest{ + Name: testName, + Template: testTemplate, + TemplateDigest: testDigest, + StagingPath: testWorkspace, + CPUs: 4, + Memory: "8g", + RootDisk: "40g", + DockerDisk: "60g", + }) + if err != nil { + t.Fatal(err) + } + if instance.Name != testName || instance.ProviderID != testID { + t.Fatalf("instance = %#v", instance) + } + done() +} + +func TestCreateReturnsExactReceiptWhenPostCreateVerificationFails(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: healthyDiagnoseJSON}}, + commandStep{args: []string{"secret", "ls", "-g"}, result: provider.ExecResult{Stdout: `No secrets found for scope "(global)".`}}, + commandStep{args: []string{"template", "ls", "--json"}, result: provider.ExecResult{Stdout: templateListJSON}}, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: `{"sandboxes":[]}`}}, + commandStep{args: []string{"create", "--name", testName, "--cpus", "4", "--memory", "8g", "--template", testTemplate, "shell", testWorkspace}, environment: map[string]string{}}, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"ports", testName, "--json"}, result: provider.ExecResult{Stdout: emptyPortsJSON}}, + commandStep{args: []string{"inspect", "--json", testName}, result: provider.ExecResult{Stdout: inspectionJSON}}, + commandStep{args: []string{"exec", "-i", testName, "--", "bash", "-lc", directWorkspaceVerificationScript}, err: errors.New("workspace verification failed")}, + ) + instance, err := p.Create(context.Background(), validCreateRequest()) + if err == nil || !strings.Contains(err.Error(), "workspace verification failed") { + t.Fatalf("Create() error = %v", err) + } + if instance.Name != testName || instance.ProviderID != testID || instance.ReceiptVersion != "v1" || len(instance.Receipt) == 0 { + t.Fatalf("Create() partial instance = %#v, want exact receipted identity", instance) + } + var receipt instanceReceipt + if err := json.Unmarshal(instance.Receipt, &receipt); err != nil { + t.Fatal(err) + } + if receipt.StagingPath != testWorkspace || receipt.StagingIdentity == "" || receipt.Template != testTemplate || receipt.TemplateDigest != testDigest { + t.Fatalf("Create() receipt = %#v", receipt) + } + done() +} + +func TestSplitTemplateReferenceCanonicalizesDockerHubNames(t *testing.T) { + tests := map[string]string{ + "epar-template:version": "docker.io/library/epar-template", + "docker/sandbox-templates:shell-docker": "docker.io/docker/sandbox-templates", + "docker.io/library/epar-template:version": "docker.io/library/epar-template", + "registry.example.test/team/template:version": "registry.example.test/team/template", + "localhost:5000/team/template:version": "localhost:5000/team/template", + } + for reference, expectedRepository := range tests { + repository, _, err := splitTemplateReference(reference) + if err != nil { + t.Fatalf("splitTemplateReference(%q): %v", reference, err) + } + if repository != expectedRepository { + t.Fatalf("splitTemplateReference(%q) repository = %q, want %q", reference, repository, expectedRepository) + } + } +} + +func TestCreateFailsClosedOnCachedTemplateIdentityMismatch(t *testing.T) { + mismatch := strings.Replace(templateListJSON, "39cf20eca861", "aaaaaaaaaaaa", 1) + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: healthyDiagnoseJSON}}, + commandStep{args: []string{"secret", "ls", "-g"}, result: provider.ExecResult{Stdout: `No secrets found for scope "(global)".`}}, + commandStep{args: []string{"template", "ls", "--json"}, result: provider.ExecResult{Stdout: mismatch}}, + ) + if _, err := p.Create(context.Background(), validCreateRequest()); err == nil || !strings.Contains(err.Error(), "does not match") { + t.Fatalf("err = %v", err) + } + done() +} + +func TestCreateSucceedsWithImportedTemplateAndNoDockerStagingImage(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: healthyDiagnoseJSON}}, + commandStep{args: []string{"secret", "ls", "-g"}, result: provider.ExecResult{Stdout: `No secrets found for scope "(global)".`}}, + commandStep{args: []string{"template", "ls", "--json"}, result: provider.ExecResult{Stdout: templateListJSON}}, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: `{"sandboxes":[]}`}}, + commandStep{args: []string{"create", "--name", testName, "--cpus", "4", "--memory", "8g", "--template", testTemplate, "shell", testWorkspace}, environment: map[string]string{}}, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"ports", testName, "--json"}, result: provider.ExecResult{Stdout: emptyPortsJSON}}, + commandStep{args: []string{"inspect", "--json", testName}, result: provider.ExecResult{Stdout: inspectionJSON}}, + commandStep{args: []string{"exec", "-i", testName, "--", "bash", "-lc", directWorkspaceVerificationScript}}, + ) + if _, err := p.Create(context.Background(), validCreateRequest()); err != nil { + t.Fatal(err) + } + done() +} + +func TestCreateFailsClosedWithoutEchoingGlobalSecretMetadata(t *testing.T) { + const listedMetadata = "github registry masked-prefix masked-suffix" + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: healthyDiagnoseJSON}}, + commandStep{args: []string{"secret", "ls", "-g"}, result: provider.ExecResult{Stdout: listedMetadata}}, + ) + _, err := p.Create(context.Background(), validCreateRequest()) + if err == nil || !strings.Contains(err.Error(), "global secrets are configured") || strings.Contains(err.Error(), listedMetadata) { + t.Fatalf("global-secret preflight error = %v", err) + } + done() +} + +func TestInstanceAdmissionUsesExactInspectionAndRejectsAttachedCapabilities(t *testing.T) { + t.Run("empty", func(t *testing.T) { + p, done := identityAdmissionScript(t, commandStep{args: []string{"inspect", "--json", testName}, result: provider.ExecResult{Stdout: inspectionJSON}}) + if err := p.VerifyInstanceAdmission(context.Background(), testInstance); err != nil { + t.Fatal(err) + } + done() + }) + t.Run("different daemon version", func(t *testing.T) { + fixture := strings.Replace(inspectionJSON, `"daemon_version":"fixture-current"`, `"daemon_version":"fixture-next"`, 1) + p, done := identityAdmissionScript(t, commandStep{args: []string{"inspect", "--json", testName}, result: provider.ExecResult{Stdout: fixture}}) + if err := p.VerifyInstanceAdmission(context.Background(), testInstance); err != nil { + t.Fatal(err) + } + done() + }) + for _, mutation := range []struct { + name string + old string + new string + }{ + {name: "kit", old: `"kits":[]`, new: `"kits":[{"name":"docker-auth"}]`}, + {name: "mcp", old: `"mcp_gateway":false`, new: `"mcp_gateway":true`}, + {name: "secret", old: `"state":"running"`, new: `"state":"running","secrets":["registry"]`}, + {name: "port", old: `"state":"running"`, new: `"state":"running","published_ports":["8080:80"]`}, + {name: "auth", old: `"state":"running"`, new: `"state":"running","auth_mode":"docker-login"`}, + } { + t.Run(mutation.name, func(t *testing.T) { + fixture := strings.Replace(inspectionJSON, mutation.old, mutation.new, 1) + p, done := identityAdmissionScript(t, commandStep{args: []string{"inspect", "--json", testName}, result: provider.ExecResult{Stdout: fixture}}) + if err := p.VerifyInstanceAdmission(context.Background(), testInstance); err == nil { + t.Fatal("attached capability was accepted") + } + done() + }) + } +} + +func TestInstanceAdmissionRejectsPublishedPortInventory(t *testing.T) { + for _, test := range []struct { + name string + fixture string + message string + }{ + {name: "published", fixture: `[{"host_ip":"127.0.0.1","host_port":60002,"sandbox_port":9418,"protocol":"tcp"}]`, message: "forbidden published port"}, + {name: "null", fixture: `null`, message: "unsupported JSON schema"}, + {name: "wrapper", fixture: `{"ports":[]}`, message: "unsupported JSON schema"}, + {name: "trailing-json", fixture: `[] {}`, message: "unsupported JSON schema"}, + } { + t.Run(test.name, func(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"ports", testName, "--json"}, result: provider.ExecResult{Stdout: test.fixture}}, + ) + if err := p.VerifyInstanceAdmission(context.Background(), testInstance); err == nil || !strings.Contains(err.Error(), test.message) { + t.Fatalf("published port admission error = %v", err) + } + done() + }) + } +} + +func TestChildEnvironmentStripsHostSSHAgent(t *testing.T) { + t.Setenv("SSH_AUTH_SOCK", "/host/agent.sock") + t.Setenv("SSH_AUTH_SOCK_GATEWAY", "gateway.example.test:3129") + t.Setenv("SSH_AGENT_PID", "4242") + environment := childEnvironment(nil) + for _, item := range environment { + key, _, _ := strings.Cut(item, "=") + if strings.EqualFold(key, "SSH_AUTH_SOCK") || strings.EqualFold(key, "SSH_AUTH_SOCK_GATEWAY") || strings.EqualFold(key, "SSH_AGENT_PID") { + t.Fatalf("host SSH agent variable survived child environment filtering: %q", key) + } + } +} + +func TestDirectWorkspaceVerificationRejectsSSHAgentForwardingWithRemediation(t *testing.T) { + for _, required := range []string{"SSH_AUTH_SOCK", "SSH_AUTH_SOCK_GATEWAY", "SSH_AGENT_PID", "restart it with SSH_AUTH_SOCK"} { + if !strings.Contains(directWorkspaceVerificationScript, required) { + t.Fatalf("direct workspace verification omitted SSH-agent guardrail %q", required) + } + } + for name, environment := range map[string]string{ + "socket": "SSH_AUTH_SOCK=/tmp/host-agent.sock", + "gateway": "SSH_AUTH_SOCK_GATEWAY=gateway.example.test:3129", + "pid": "SSH_AGENT_PID=4242", + } { + t.Run(name, func(t *testing.T) { + command := exec.Command("bash", "-c", directWorkspaceVerificationScript) + command.Env = append(childEnvironment(nil), environment) + output, err := command.CombinedOutput() + if err == nil { + t.Fatal("workspace verification accepted host SSH-agent forwarding") + } + if !strings.Contains(string(output), "Docker Sandboxes exposed host SSH-agent forwarding") { + t.Fatalf("workspace verification output = %q, want actionable SSH-agent diagnostic", output) + } + }) + } +} + +func TestTemplateInventorySchemaFailsClosed(t *testing.T) { + for _, fixture := range []string{ + `[]`, + `{"templates":[]}`, + `{"images":null}`, + `{"images":[{"id":"39cf20eca861","repository":"docker.io/docker/sandbox-templates","tag":"shell-docker","flavor":"shell-docker","created_at":"not-a-time","size":599103243}]}`, + `{"images":[{"id":"39cf20eca861","repository":"docker.io/docker/sandbox-templates","tag":"shell-docker","flavor":"shell-docker","created_at":"2026-07-22T07:13:19Z","size":"599103243"}]}`, + } { + if _, err := parseTemplateInventory([]byte(fixture)); err == nil { + t.Fatalf("template schema drift was accepted: %s", fixture) + } + } +} + +func TestParseTemplateInventoryAcceptsCustomTemplateWithoutFlavor(t *testing.T) { + images, err := parseTemplateInventory([]byte(`{"images":[{"id":"f40a31d1d676","repository":"docker.io/library/epar-docker-sandboxes-catthehacker-act-22.04","tag":"20260715-amd64","created_at":"2026-07-23T04:37:43Z","size":1019288120}]}`)) + if err != nil { + t.Fatal(err) + } + if len(images) != 1 || images[0].Flavor != "" || images[0].ID != "f40a31d1d676" { + t.Fatalf("custom template inventory = %#v", images) + } +} + +func TestCachedTemplatesUsesMachineReadableInventoryWithoutVersionGate(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"template", "ls", "--json"}, result: provider.ExecResult{Stdout: templateListJSON}}, + ) + templates, err := p.CachedTemplates(context.Background()) + if err != nil { + t.Fatal(err) + } + if len(templates) != 1 { + t.Fatalf("templates = %#v", templates) + } + template := templates[0] + if template.Reference != testTemplate || template.CacheID != "39cf20eca861" || !template.CreatedAt.Equal(time.Date(2026, time.July, 22, 7, 13, 19, 0, time.UTC)) || template.SizeBytes != 599103243 { + t.Fatalf("cached template = %#v", template) + } + done() +} + +func TestCachedTemplatesFailsClosedOnMalformedInventory(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"template", "ls", "--json"}, result: provider.ExecResult{Stdout: `{"images":[{"id":"not-a-cache-id"}]}`}}, + ) + if _, err := p.CachedTemplates(context.Background()); err == nil { + t.Fatal("malformed template inventory was accepted") + } + done() +} + +func TestReadGlobalNetworkPolicyUsesGlobalOnlyReadback(t *testing.T) { + fixture := policyFixture(`[ + {"id":"global-1","name":"global","policy_id":"local","scope":"global","applies_to":"all","resource_type":"network","decision":"allow","resources":["api.example.com"],"origin":"local","status":"active","editable":true}, + {"id":"sandbox-1","name":"sandbox","policy_id":"local","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:epar-sandbox-1","resource_type":"network","decision":"deny","resources":["**"],"origin":"scoped","status":"inactive","editable":true,"sandbox_id":"epar-sandbox-1"} + ]`) + p, done := scriptedProvider(t, + commandStep{args: []string{"policy", "ls", "--include-inactive", "--json"}, result: provider.ExecResult{Stdout: fixture}}, + ) + rules, err := p.ReadGlobalNetworkPolicy(context.Background()) + if err != nil { + t.Fatal(err) + } + if len(rules) != 1 || rules[0].ID != "global-1" || rules[0].Scope != "global" || rules[0].AppliesTo != "all" || !rules[0].Active { + t.Fatalf("global policy rules = %#v", rules) + } + done() +} + +func TestReadGlobalNetworkPolicyFailsClosedOnMalformedJSON(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"policy", "ls", "--include-inactive", "--json"}, result: provider.ExecResult{Stdout: `{"rules":null}`}}, + ) + if _, err := p.ReadGlobalNetworkPolicy(context.Background()); err == nil { + t.Fatal("malformed global policy was accepted") + } + done() +} + +func TestDiagnosticsGateFailsClosedBeforeMutation(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: `{"version":"1.0","checks":[{"name":"daemon","status":"fail","message":"unhealthy","detail":"","hint":"restart the daemon"}],"summary":{"pass":0,"warn":0,"fail":1,"skip":0}}`}}, + ) + _, err := p.Create(context.Background(), validCreateRequest()) + if err == nil || !strings.Contains(err.Error(), "1 failed check") || !strings.Contains(err.Error(), "sbx diagnose --output json") || !strings.Contains(err.Error(), "hints for each failed check") { + t.Fatalf("err = %v", err) + } + done() +} + +func TestAdmissionUsesDiagnosticsWithoutReadingVersion(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: healthyDiagnoseJSON}}, + commandStep{args: []string{"secret", "ls", "-g"}, result: provider.ExecResult{Stdout: `No secrets found for scope "(global)".`}}, + ) + if err := p.VerifyAdmission(context.Background()); err != nil { + t.Fatal(err) + } + done() +} + +func TestAdmissionRechecksDiagnostics(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: healthyDiagnoseJSON}}, + commandStep{args: []string{"secret", "ls", "-g"}, result: provider.ExecResult{Stdout: `No secrets found for scope "(global)".`}}, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: `{"version":"1.0","checks":[{"name":"daemon","status":"fail","message":"unhealthy","detail":"","hint":"restart the daemon"}],"summary":{"pass":0,"warn":0,"fail":1,"skip":0}}`}}, + ) + if err := p.VerifyAdmission(context.Background()); err != nil { + t.Fatal(err) + } + if err := p.VerifyAdmission(context.Background()); err == nil || !strings.Contains(err.Error(), "1 failed check") || !strings.Contains(err.Error(), "sbx diagnose --output json") || !strings.Contains(err.Error(), "hints for each failed check") { + t.Fatalf("failed diagnostics were accepted: %v", err) + } + done() +} + +func TestInventoryParsesWrapperAndFailsClosedOnSchemaDrift(t *testing.T) { + items, err := parseInventory([]byte(readyListJSON)) + if err != nil { + t.Fatal(err) + } + if len(items) != 1 || items[0].Instance.Name != testInstance.Name || items[0].Instance.ProviderID != testInstance.ProviderID || !reflect.DeepEqual(items[0].Workspaces, []string{testWorkspace}) { + t.Fatalf("items = %#v", items) + } + for _, fixture := range []string{ + `[]`, + `{"items":[]}`, + `{"sandboxes":null}`, + `{"sandboxes":[{"name":"epar-sandbox-1","status":"running","workspaces":["/staging"]}]}`, + `{"sandboxes":[{"id":"id-1","name":"epar-sandbox-1","state":"running","workspaces":["/staging"]}]}`, + `{"sandboxes":[{"id":"id-1","name":"epar-sandbox-1","status":"running"}]}`, + `{"sandboxes":[{"id":"id-1","name":"epar-sandbox-1","status":"running","workspaces":[]}]}`, + `{"sandboxes":[{"id":"id-1","name":"epar-sandbox-1","status":"running","workspaces":["/staging","/staging"]}]}`, + } { + if _, err := parseInventory([]byte(fixture)); err == nil { + t.Fatalf("schema drift was accepted: %s", fixture) + } + } +} + +func TestLifecycleCommandsUseExactIdentityAndArgv(t *testing.T) { + t.Run("start", func(t *testing.T) { + p, done := identityScript(t, commandStep{args: []string{"exec", "-i", testName, "--", "/bin/sleep", "infinity"}}) + if _, err := p.Start(context.Background(), testInstance, provider.StartOptions{}); err != nil { + t.Fatal(err) + } + done() + }) + t.Run("verify", func(t *testing.T) { + p, done := identityScript(t, commandStep{ + args: []string{"exec", "-i", testName, "--", "bash", "-lc", runtimeVerificationScript}, + result: provider.ExecResult{Stdout: `"29.5.3"`}, + }) + info, err := p.VerifyRuntime(context.Background(), testInstance) + if err != nil || !info.Ready || info.Runtime != "docker" || info.Version != "29.5.3" { + t.Fatalf("info = %#v, err = %v", info, err) + } + done() + }) + t.Run("stop_preserves_state", func(t *testing.T) { + p, done := identityScript(t, commandStep{args: []string{"stop", testName}}) + if err := p.Stop(context.Background(), testInstance); err != nil { + t.Fatal(err) + } + done() + }) + t.Run("delete_is_exact_force_remove", func(t *testing.T) { + p, done := identityScript(t, commandStep{args: []string{"rm", "--force", testName}}) + if err := p.Delete(context.Background(), testInstance); err != nil { + t.Fatal(err) + } + done() + }) + t.Run("address_is_unavailable_without_command", func(t *testing.T) { + p, done := scriptedProvider(t) + address, available, err := p.Address(context.Background(), testInstance, 30) + if err != nil || available || address != "" { + t.Fatalf("address = %q, available = %v, err = %v", address, available, err) + } + done() + }) +} + +func TestExecPreservesGuestArgvStdinAndRedactsAllSurfaces(t *testing.T) { + const secret = "sentinel-sandbox-secret" + var stdout, stderr bytes.Buffer + p, done := identityScript(t, commandStep{ + args: []string{"exec", "-i", testName, "--", "sh", "-c", "printf '%s' \"$1\"", "sh", "semi;colon", "--privileged"}, + stdin: "payload\n", + streamOut: "stream " + secret, + streamErr: "TOKEN=" + secret, + result: provider.ExecResult{ + Stdout: "result " + secret, + Stderr: "SECRET=" + secret, + }, + err: errors.New("exit status 17: " + secret), + }) + result, err := p.Exec(context.Background(), testInstance, + []string{"sh", "-c", "printf '%s' \"$1\"", "sh", "semi;colon", "--privileged"}, + provider.ExecOptions{Stdin: "payload\n", SensitiveValues: []string{secret}, Stdout: &stdout, Stderr: &stderr}, + ) + if err == nil { + t.Fatal("expected fake execution error") + } + combined := stdout.String() + stderr.String() + result.Stdout + result.Stderr + err.Error() + if strings.Contains(combined, secret) { + t.Fatalf("secret leaked: %q", combined) + } + if !strings.Contains(combined, "[REDACTED]") { + t.Fatalf("redaction marker missing: %q", combined) + } + done() +} + +func TestExecCancellationPropagatesWithoutRealSbx(t *testing.T) { + p := New("sbx-test-double") + call := 0 + p.runCommand = func(ctx context.Context, request commandRequest) (provider.ExecResult, error) { + call++ + switch call { + case 1: + if !reflect.DeepEqual(request.args, []string{"ls", "--json"}) { + t.Fatalf("identity args = %#v", request.args) + } + return provider.ExecResult{Stdout: readyListJSON}, nil + case 2: + if !reflect.DeepEqual(request.args, []string{"exec", "-i", testName, "--", "sleep", "30"}) { + t.Fatalf("exec args = %#v", request.args) + } + <-ctx.Done() + return provider.ExecResult{}, ctx.Err() + default: + t.Fatalf("unexpected call %d", call) + return provider.ExecResult{}, nil + } + } + ctx, cancel := context.WithCancel(context.Background()) + time.AfterFunc(10*time.Millisecond, cancel) + _, err := p.Exec(ctx, testInstance, []string{"sleep", "30"}, provider.ExecOptions{}) + if !errors.Is(err, context.Canceled) { + t.Fatalf("err = %v", err) + } +} + +func TestRunRawNormalizesCommandContextKillToCancellation(t *testing.T) { + if os.Getenv("EPAR_DOCKER_SANDBOXES_RUN_RAW_HELPER") == "1" { + _, _ = os.Stdout.WriteString("ready\n") + time.Sleep(30 * time.Second) + return + } + t.Setenv("EPAR_DOCKER_SANDBOXES_RUN_RAW_HELPER", "1") + + p := New(os.Args[0]) + ctx, cancel := context.WithCancel(context.Background()) + started := make(chan struct{}) + type rawResult struct { + result provider.ExecResult + err error + } + finished := make(chan rawResult, 1) + go func() { + result, err := p.runRaw(ctx, commandRequest{ + args: []string{"-test.run=^TestRunRawNormalizesCommandContextKillToCancellation$"}, + operation: "test helper", + outputLimit: defaultOutputLimit, + stdout: &cancellationSignalWriter{started: started}, + }) + finished <- rawResult{result: result, err: err} + }() + select { + case <-started: + case <-time.After(5 * time.Second): + cancel() + t.Fatal("runRaw helper process did not start") + } + cancel() + select { + case outcome := <-finished: + if !errors.Is(outcome.err, context.Canceled) { + t.Fatalf("runRaw cancellation error = %v, want context.Canceled", outcome.err) + } + var exitErr *exec.ExitError + if errors.As(outcome.err, &exitErr) { + t.Fatalf("runRaw retained cancellation-induced process exit as a concrete error: %v", outcome.err) + } + case <-time.After(5 * time.Second): + t.Fatal("runRaw did not return after cancellation") + } +} + +func TestStopAndDeleteAreIdempotentWhenInventorySaysMissing(t *testing.T) { + for _, test := range []struct { + name string + call func(*Provider) error + }{ + {name: "stop", call: func(p *Provider) error { return p.Stop(context.Background(), testInstance) }}, + {name: "delete", call: func(p *Provider) error { return p.Delete(context.Background(), testInstance) }}, + } { + t.Run(test.name, func(t *testing.T) { + p, done := scriptedProvider(t, commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: `{"sandboxes":[]}`}}) + if err := test.call(p); err != nil { + t.Fatal(err) + } + done() + }) + } +} + +func TestIdentityMismatchFailsBeforeStateMutation(t *testing.T) { + mismatch := strings.Replace(readyListJSON, testID, "different-id", 1) + p, done := scriptedProvider(t, commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: mismatch}}) + if err := p.Delete(context.Background(), testInstance); err == nil || !strings.Contains(err.Error(), "identity changed") { + t.Fatalf("err = %v", err) + } + done() +} + +func TestDiagnosticsUsesBoundedMachineReadableCommands(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"daemon", "status", "--json"}, result: provider.ExecResult{Stdout: `{"status":"running","socket":"\\\\.\\pipe\\docker_kaname_sandboxd","logs":"C:\\logs\\daemon.log"}`}}, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: `{"version":"1.0","checks":[{"name":"daemon","status":"pass","message":"ok","detail":"","hint":""},{"name":"optional update","status":"warn","message":"available","detail":"","hint":""},{"name":"optional integration","status":"skip","message":"not configured","detail":"","hint":""}],"summary":{"pass":1,"warn":1,"fail":0,"skip":1}}`}}, + ) + diagnostics, err := p.Diagnostics(context.Background(), testInstance) + if err != nil || !diagnostics.Healthy || diagnostics.ChecksPassed != 1 || diagnostics.ChecksWarned != 1 || diagnostics.ChecksSkipped != 1 { + t.Fatalf("diagnostics = %#v, err = %v", diagnostics, err) + } + done() +} + +func TestVerifyHostReadinessAcceptsWarningsWithPassingChecksAndNoFailures(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: `{"version":"1.0","checks":[{"name":"daemon","status":"pass","message":"healthy","detail":"","hint":""},{"name":"optional update","status":"warn","message":"available","detail":"","hint":""}],"summary":{"pass":1,"warn":1,"fail":0,"skip":0}}`}}, + ) + readiness, err := p.VerifyHostReadiness(context.Background()) + if err != nil { + t.Fatal(err) + } + if readiness.ChecksPassed != 1 || readiness.ChecksWarned != 1 || readiness.ChecksFailed != 0 || readiness.ChecksSkipped != 0 { + t.Fatalf("readiness = %#v", readiness) + } + done() +} + +func TestVerifyHostReadinessRejectsFailedOrEmptyDiagnostics(t *testing.T) { + tests := []struct { + name string + fixture string + want string + }{ + { + name: "failed check", + fixture: `{"version":"1.0","checks":[{"name":"daemon","status":"fail","message":"unhealthy","detail":"","hint":""}],"summary":{"pass":0,"warn":0,"fail":1,"skip":0}}`, + want: "1 failed check", + }, + { + name: "no passing check", + fixture: `{"version":"1.0","checks":[{"name":"optional","status":"skip","message":"not applicable","detail":"","hint":""}],"summary":{"pass":0,"warn":0,"fail":0,"skip":1}}`, + want: "no passing checks", + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: test.fixture}}, + ) + if _, err := p.VerifyHostReadiness(context.Background()); err == nil || !strings.Contains(err.Error(), test.want) || !strings.Contains(err.Error(), "sbx diagnose --output json") { + t.Fatalf("VerifyHostReadiness() error = %v, want %q and diagnostic command", err, test.want) + } + done() + }) + } +} + +func TestVerifyHostReadinessRejectsCommandAndSchemaFailures(t *testing.T) { + tests := []struct { + name string + step commandStep + want string + }{ + { + name: "command failure", + step: commandStep{args: []string{"diagnose", "--output", "json"}, err: errors.New("diagnose unavailable")}, + want: "diagnose unavailable", + }, + { + name: "malformed json", + step: commandStep{args: []string{"diagnose", "--output", "json"}, result: provider.ExecResult{Stdout: `{"summary":`}}, + want: "unsupported json schema", + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + p, done := scriptedProvider(t, test.step) + if _, err := p.VerifyHostReadiness(context.Background()); err == nil || !strings.Contains(err.Error(), test.want) || !strings.Contains(err.Error(), "sbx diagnose --output json") { + t.Fatalf("VerifyHostReadiness() error = %v, want %q and diagnostic command", err, test.want) + } + done() + }) + } +} + +func TestDiagnosticsRequiresEverySummaryCount(t *testing.T) { + fixture := `{"version":"1.0","checks":[{"name":"daemon","status":"pass","message":"ok","detail":"","hint":""}],"summary":{"pass":1,"warn":0,"fail":0}}` + if _, _, _, _, err := parseDiagnose([]byte(fixture)); err == nil { + t.Fatal("diagnostics accepted a missing summary count") + } +} + +func TestDiagnosticsRejectsOversizedOutput(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"daemon", "status", "--json"}, result: provider.ExecResult{Stdout: strings.Repeat("x", diagnosticOutputLimit+1)}}, + ) + if _, err := p.Diagnostics(context.Background(), testInstance); err == nil || !strings.Contains(err.Error(), "output limit") { + t.Fatalf("err = %v", err) + } + done() +} + +func TestPolicyCommandsAreSandboxScopedAndReadBackExactRules(t *testing.T) { + globalRule := `{"id":"global-1","name":"automatic baseline","policy_id":"local-policy","scope":"global","applies_to":"all","resource_type":"network","decision":"allow","resources":["openrouter.ai"],"origin":"local","status":"active","editable":true}` + sandboxRule := `{"id":"rule-1","name":"job allowlist","policy_id":"local","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:epar-sandbox-1","resource_type":"network","decision":"allow","resources":["api.example.com","*.packages.example.com:443"],"origin":"scoped","status":"active","editable":true,"sandbox_id":"epar-sandbox-1","additive":42}` + policyJSON := policyFixture(`[` + globalRule + `,` + sandboxRule + `]`) + rule := provider.NetworkPolicyRule{Decision: provider.NetworkPolicyAllow, Resources: []string{"api.example.com", "*.packages.example.com:443"}} + t.Run("apply", func(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"policy", "allow", "network", "--sandbox", testName, "api.example.com,*.packages.example.com:443"}}, + commandStep{args: []string{"policy", "ls", testName, "--include-inactive", "--json"}, result: provider.ExecResult{Stdout: policyJSON}}, + ) + if err := p.ApplyNetworkPolicy(context.Background(), testInstance, []provider.NetworkPolicyRule{rule}); err != nil { + t.Fatal(err) + } + done() + }) + t.Run("apply_open_public_egress", func(t *testing.T) { + openRule := `{"id":"rule-open","name":"open public egress","policy_id":"local","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:epar-sandbox-1","resource_type":"network","decision":"allow","resources":["**"],"origin":"scoped","status":"active","editable":true,"sandbox_id":"epar-sandbox-1"}` + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"policy", "allow", "network", "--sandbox", testName, "**"}}, + commandStep{args: []string{"policy", "ls", testName, "--include-inactive", "--json"}, result: provider.ExecResult{Stdout: policyFixture(`[` + globalRule + `,` + openRule + `]`)}}, + ) + if err := p.ApplyNetworkPolicy(context.Background(), testInstance, []provider.NetworkPolicyRule{{ + Decision: provider.NetworkPolicyAllow, + Resources: []string{"**"}, + }}); err != nil { + t.Fatal(err) + } + done() + }) + t.Run("remove_by_stable_rule_id", func(t *testing.T) { + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"policy", "ls", testName, "--include-inactive", "--json"}, result: provider.ExecResult{Stdout: policyJSON}}, + commandStep{args: []string{"policy", "rm", "network", "--sandbox", testName, "--id", "rule-1"}}, + commandStep{args: []string{"policy", "ls", testName, "--include-inactive", "--json"}, result: provider.ExecResult{Stdout: policyFixture(`[` + globalRule + `]`)}}, + ) + remove := provider.NetworkPolicyRule{ + ID: "rule-1", + PolicyID: "local", + Scope: "sandbox:" + testName, + AppliesTo: "sandbox:" + testName, + ResourceType: "network", + Resources: append([]string(nil), rule.Resources...), + Decision: provider.NetworkPolicyAllow, + Origin: "scoped", + Editable: true, + } + if err := p.RemoveNetworkPolicy(context.Background(), testInstance, []provider.NetworkPolicyRule{remove}); err != nil { + t.Fatal(err) + } + done() + }) +} + +func TestPolicyRemovalRefusesChangedStableIdentity(t *testing.T) { + fixture := policyFixture(`[{"id":"rule-1","name":"job allowlist","policy_id":"local","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:epar-sandbox-1","resource_type":"network","decision":"allow","resources":["api.example.com"],"origin":"scoped","status":"inactive","editable":true,"sandbox_id":"epar-sandbox-1"}]`) + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"policy", "ls", testName, "--include-inactive", "--json"}, result: provider.ExecResult{Stdout: fixture}}, + ) + changed := provider.NetworkPolicyRule{ + ID: "rule-1", + PolicyID: "local", + Scope: "sandbox:" + testName, + AppliesTo: "sandbox:" + testName, + ResourceType: "network", + Resources: []string{"different.example.com"}, + Decision: provider.NetworkPolicyAllow, + Origin: "scoped", + Editable: true, + } + if err := p.RemoveNetworkPolicy(context.Background(), testInstance, []provider.NetworkPolicyRule{changed}); err == nil || !strings.Contains(err.Error(), "stable identity changed") { + t.Fatalf("err = %v", err) + } + done() +} + +func TestPolicyReadPreservesGlobalAndSandboxAttribution(t *testing.T) { + fixture := policyFixture(`[ + {"id":"global-1","name":"automatic baseline","policy_id":"local-policy","scope":"global","applies_to":"all","resource_type":"network","decision":"allow","resources":["openrouter.ai"],"origin":"local","status":"active","editable":true}, + {"id":"rule-1","name":"job allowlist","policy_id":"local","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:epar-sandbox-1","resource_type":"network","decision":"deny","resources":["**"],"origin":"scoped","status":"inactive","editable":true,"sandbox_id":"epar-sandbox-1"}, + {"id":"other-1","name":"another sandbox","policy_id":"local","scope":"sandbox:other-sandbox","applies_to":"sandbox:other-sandbox","resource_type":"network","decision":"allow","resources":["other.example.com"],"origin":"scoped","status":"active","editable":true,"sandbox_id":"other-sandbox"}, + {"id":"fs-1","name":"filesystem baseline","policy_id":"local-policy","scope":"global","applies_to":"all","resource_type":"filesystem","decision":"allow","resources":["/workspace"],"origin":"local","status":"active","editable":false} + ]`) + rules, err := parseNetworkPolicy([]byte(fixture), testName) + if err != nil { + t.Fatal(err) + } + if len(rules) != 3 { + t.Fatalf("rules = %#v", rules) + } + global := rules[0] + if global.ID != "global-1" || global.Name != "automatic baseline" || global.PolicyID != "local-policy" || global.Scope != "global" || global.AppliesTo != "all" || global.ResourceType != "network" || global.Origin != "local" || global.Status != "active" || !global.Editable || !global.Active { + t.Fatalf("global attribution was not preserved: %#v", global) + } + local := rules[1] + if local.Scope != "sandbox:"+testName || local.AppliesTo != "sandbox:"+testName || local.Origin != "scoped" || local.Status != "inactive" || local.Active { + t.Fatalf("sandbox attribution was not preserved: %#v", local) + } + filesystem := rules[2] + if filesystem.ID != "fs-1" || filesystem.ResourceType != "filesystem" || filesystem.Scope != "global" || filesystem.AppliesTo != "all" || filesystem.Editable { + t.Fatalf("filesystem attribution was not preserved: %#v", filesystem) + } +} + +func TestPolicyReadPreservesExactKitAttribution(t *testing.T) { + fixture := policyFixture(`[{"id":"kit-rule-1","name":"kit:epar-sandbox-1","policy_id":"kit-policy-1","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:epar-sandbox-1","resource_type":"network","decision":"allow","resources":["openrouter.ai"],"origin":"scoped","status":"active","editable":false,"sandbox_id":"epar-sandbox-1"}]`) + rules, err := parseNetworkPolicy([]byte(fixture), testName) + if err != nil { + t.Fatal(err) + } + if len(rules) != 1 || rules[0].Scope != "sandbox:"+testName || rules[0].AppliesTo != "sandbox:"+testName || rules[0].Origin != "scoped" || rules[0].Editable || !rules[0].Active { + t.Fatalf("kit attribution was not preserved: %#v", rules) + } +} + +func TestPolicyReadRejectsMismatchedSandboxAttribution(t *testing.T) { + for _, fixture := range []string{ + policyFixture(`[{"id":"rule-1","name":"rule","policy_id":"local","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:other-sandbox","resource_type":"network","decision":"allow","resources":["api.example.com"],"origin":"scoped","status":"active","editable":true,"sandbox_id":"epar-sandbox-1"}]`), + policyFixture(`[{"id":"rule-1","name":"rule","policy_id":"local","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:epar-sandbox-1","resource_type":"network","decision":"allow","resources":["api.example.com"],"origin":"scoped","status":"active","editable":true,"sandbox_id":"other-sandbox"}]`), + } { + if _, err := parseNetworkPolicy([]byte(fixture), testName); err == nil { + t.Fatalf("mismatched sandbox attribution was accepted: %s", fixture) + } + } +} + +func TestPolicyRemovalRefusesGlobalRule(t *testing.T) { + global := provider.NetworkPolicyRule{ID: "global-1", PolicyID: "local-policy", Scope: "global", AppliesTo: "all", ResourceType: "network", Resources: []string{"openrouter.ai"}, Decision: provider.NetworkPolicyAllow, Origin: "local", Editable: true} + fixture := policyFixture(`[{"id":"global-1","name":"automatic baseline","policy_id":"local-policy","scope":"global","applies_to":"all","resource_type":"network","decision":"allow","resources":["openrouter.ai"],"origin":"local","status":"active","editable":true}]`) + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"policy", "ls", testName, "--include-inactive", "--json"}, result: provider.ExecResult{Stdout: fixture}}, + ) + if err := p.RemoveNetworkPolicy(context.Background(), testInstance, []provider.NetworkPolicyRule{global}); err == nil || !strings.Contains(err.Error(), "refusing to remove") { + t.Fatalf("err = %v", err) + } + done() +} + +func TestPolicyRemovalRefusesFilesystemRule(t *testing.T) { + filesystem := provider.NetworkPolicyRule{ID: "fs-1", PolicyID: "local-policy", Scope: "global", AppliesTo: "all", ResourceType: "filesystem", Resources: []string{"/workspace"}, Decision: provider.NetworkPolicyAllow, Origin: "local", Editable: false} + fixture := policyFixture(`[{"id":"fs-1","name":"filesystem baseline","policy_id":"local-policy","scope":"global","applies_to":"all","resource_type":"filesystem","decision":"allow","resources":["/workspace"],"origin":"local","status":"active","editable":false}]`) + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"policy", "ls", testName, "--include-inactive", "--json"}, result: provider.ExecResult{Stdout: fixture}}, + ) + if err := p.RemoveNetworkPolicy(context.Background(), testInstance, []provider.NetworkPolicyRule{filesystem}); err == nil || !strings.Contains(err.Error(), "refusing to remove") { + t.Fatalf("err = %v", err) + } + done() +} + +func TestPolicySchemaFailsClosed(t *testing.T) { + for _, fixture := range []string{ + `[]`, + `{"policies":[]}`, + `{"rules":null}`, + policyFixture(`[{"id":"rule-1","name":"rule","policy_id":"local","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:epar-sandbox-1","resource_type":"network","decision":"permit","resources":["api.example.com"],"origin":"scoped","status":"active","editable":true,"sandbox_id":"epar-sandbox-1"}]`), + policyFixture(`[{"id":"rule-1","name":"rule","policy_id":"local","scope":"sandbox:epar-sandbox-1","applies_to":"sandbox:epar-sandbox-1","resource_type":"network","decision":"allow","resources":["api.example.com"],"origin":"scoped","status":"unknown","editable":true,"sandbox_id":"epar-sandbox-1"}]`), + } { + if _, err := parseNetworkPolicy([]byte(fixture), testName); err == nil { + t.Fatalf("policy schema drift was accepted: %s", fixture) + } + } +} + +func TestPolicyReadbackAcceptsAttributedProviderBaselinePatterns(t *testing.T) { + fixture := policyFixture(`[{"id":"default-cert-validation","name":"default-cert-validation","policy_id":"local-policy","scope":"global","applies_to":"all","resource_type":"network","decision":"allow","resources":["**.openai.com:443","crl*.digicert.com:80","**"],"origin":"local","status":"active","editable":true}]`) + rules, err := parseNetworkPolicy([]byte(fixture), testName) + if err != nil { + t.Fatal(err) + } + if len(rules) != 1 || !reflect.DeepEqual(rules[0].Resources, []string{"**.openai.com:443", "crl*.digicert.com:80", "**"}) { + t.Fatalf("provider baseline resources = %#v", rules) + } +} + +func TestInjectionCorpusIsRejectedBeforeCommandExecution(t *testing.T) { + for _, name := range []string{"-other", "../other", "other/name", "name;rm", "name\nnext", "name+other", ""} { + request := validCreateRequest() + request.Name = name + if err := validateCreateRequest(request); err == nil { + t.Fatalf("name injection accepted: %q", name) + } + } + for _, resource := range []string{"-api.example.com", "api.example.com,evil.example.com", "api.example.com --sandbox other", "api.example.com;reset", "api.example.com\n**", "https://api.example.com", "127.0.0.1", "10.0.0.0/8", "[::1]:443", "*.localhost"} { + rule := provider.NetworkPolicyRule{Decision: provider.NetworkPolicyAllow, Resources: []string{resource}} + if err := validateNetworkRule(rule, false); err == nil { + t.Fatalf("resource injection accepted: %q", resource) + } + } + openRule := provider.NetworkPolicyRule{Decision: provider.NetworkPolicyAllow, Resources: []string{"**"}} + if err := validateNetworkRule(openRule, false); err != nil { + t.Fatalf("owned sandbox-scoped open rule was rejected: %v", err) + } + denyAllRule := provider.NetworkPolicyRule{Decision: provider.NetworkPolicyDeny, Resources: []string{"**"}} + if err := validateNetworkRule(denyAllRule, false); err == nil { + t.Fatal("unbounded deny wildcard was accepted") + } +} + +func TestCommandBoundaryRejectsInteractiveAndDestructiveGlobalCommands(t *testing.T) { + for _, command := range []string{"tui", "reset", "run", "kit", "secret", "login", "logout", "setup", "ssh", "cp", "completion", "help"} { + err := validateCommandRequest(commandRequest{args: []string{command}, operation: "test forbidden command"}) + if err == nil { + t.Fatalf("forbidden Docker Sandboxes command %q was accepted", command) + } + } + for _, arguments := range [][]string{{"inspect"}, {"inspect", testName}, {"inspect", "--json", "../other"}, {"inspect", "--debug", testName}} { + if err := validateCommandRequest(commandRequest{args: arguments, operation: "test forbidden inspection"}); err == nil { + t.Fatalf("non-exact Docker Sandboxes inspection was accepted: %q", arguments) + } + } + for _, arguments := range [][]string{{"ports"}, {"ports", testName}, {"ports", "../other", "--json"}, {"ports", testName, "--json", "extra"}, {"ports", testName, "--unpublish"}} { + if err := validateCommandRequest(commandRequest{args: arguments, operation: "test forbidden port command"}); err == nil { + t.Fatalf("non-exact Docker Sandboxes published-port inspection was accepted: %q", arguments) + } + } + for _, arguments := range [][]string{{"daemon"}, {"daemon", "start"}, {"daemon", "start", "--foreground"}, {"daemon", "stop", "--detach"}, {"daemon", "status"}, {"daemon", "status", "--debug"}} { + if err := validateCommandRequest(commandRequest{args: arguments, operation: "test forbidden daemon command"}); err == nil { + t.Fatalf("non-exact Docker Sandboxes daemon command was accepted: %q", arguments) + } + } +} + +func TestExecRejectsHostPathsAndEnvironmentPassthrough(t *testing.T) { + p, done := scriptedProvider(t) + if _, err := p.Exec(context.Background(), testInstance, []string{"true"}, provider.ExecOptions{LogPath: "/host/log"}); err == nil { + t.Fatal("host log path was accepted") + } + if _, err := p.Exec(context.Background(), testInstance, []string{"true"}, provider.ExecOptions{Env: map[string]string{"TOKEN": "secret"}}); err == nil { + t.Fatal("guest environment passthrough was accepted") + } + done() +} + +func identityScript(t *testing.T, operation commandStep) (*Provider, func()) { + t.Helper() + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + operation, + ) + return p, done +} + +func identityAdmissionScript(t *testing.T, inspection commandStep) (*Provider, func()) { + t.Helper() + p, done := scriptedProvider(t, + commandStep{args: []string{"ls", "--json"}, result: provider.ExecResult{Stdout: readyListJSON}}, + commandStep{args: []string{"ports", testName, "--json"}, result: provider.ExecResult{Stdout: emptyPortsJSON}}, + inspection, + ) + return p, done +} + +func validCreateRequest() provider.CreateRequest { + return provider.CreateRequest{ + Name: testName, + Template: testTemplate, + TemplateDigest: testDigest, + StagingPath: testWorkspace, + CPUs: 4, + Memory: "8g", + } +} + +func policyFixture(rules string) string { + return `{"rules":` + rules + `}` +} diff --git a/internal/provider/dockersandboxes/staging/identity_other.go b/internal/provider/dockersandboxes/staging/identity_other.go new file mode 100644 index 0000000..f448d6a --- /dev/null +++ b/internal/provider/dockersandboxes/staging/identity_other.go @@ -0,0 +1,25 @@ +//go:build !windows + +package staging + +import ( + "fmt" + "os" + "syscall" +) + +func platformDirectoryIdentity(path string) (string, error) { + info, err := os.Lstat(path) + if err != nil { + return "", err + } + stat, ok := info.Sys().(*syscall.Stat_t) + if !ok { + return "", fmt.Errorf("filesystem did not expose a stable directory identity") + } + return fmt.Sprintf("unix:%x:%x", uint64(stat.Dev), uint64(stat.Ino)), nil +} + +func isPlatformRedirect(info os.FileInfo) bool { return info.Mode()&os.ModeSymlink != 0 } + +func platformCanonicalPathSpelling(path string) (string, error) { return path, nil } diff --git a/internal/provider/dockersandboxes/staging/identity_windows.go b/internal/provider/dockersandboxes/staging/identity_windows.go new file mode 100644 index 0000000..27c7876 --- /dev/null +++ b/internal/provider/dockersandboxes/staging/identity_windows.go @@ -0,0 +1,55 @@ +//go:build windows + +package staging + +import ( + "fmt" + "os" + "syscall" + + "golang.org/x/sys/windows" +) + +func platformDirectoryIdentity(path string) (string, error) { + pointer, err := windows.UTF16PtrFromString(path) + if err != nil { + return "", err + } + handle, err := windows.CreateFile(pointer, 0, windows.FILE_SHARE_READ|windows.FILE_SHARE_WRITE|windows.FILE_SHARE_DELETE, nil, windows.OPEN_EXISTING, windows.FILE_FLAG_BACKUP_SEMANTICS|windows.FILE_FLAG_OPEN_REPARSE_POINT, 0) + if err != nil { + return "", err + } + defer windows.CloseHandle(handle) + var information windows.ByHandleFileInformation + if err := windows.GetFileInformationByHandle(handle, &information); err != nil { + return "", err + } + return fmt.Sprintf("windows:%08x:%08x%08x", information.VolumeSerialNumber, information.FileIndexHigh, information.FileIndexLow), nil +} + +func isPlatformRedirect(info os.FileInfo) bool { + if info.Mode()&os.ModeSymlink != 0 { + return true + } + data, ok := info.Sys().(*syscall.Win32FileAttributeData) + return ok && data.FileAttributes&syscall.FILE_ATTRIBUTE_REPARSE_POINT != 0 +} + +func platformCanonicalPathSpelling(path string) (string, error) { + pointer, err := windows.UTF16PtrFromString(path) + if err != nil { + return "", err + } + size := uint32(len(path) + 1) + for { + buffer := make([]uint16, size) + length, err := windows.GetLongPathName(pointer, &buffer[0], size) + if err != nil { + return "", err + } + if length < size { + return windows.UTF16ToString(buffer[:length]), nil + } + size = length + 1 + } +} diff --git a/internal/provider/dockersandboxes/staging/permissions_other.go b/internal/provider/dockersandboxes/staging/permissions_other.go new file mode 100644 index 0000000..7c7cee8 --- /dev/null +++ b/internal/provider/dockersandboxes/staging/permissions_other.go @@ -0,0 +1,27 @@ +//go:build !windows + +package staging + +import ( + "fmt" + "os" + "syscall" +) + +func restrictPlatformPermissions(path string) error { + return os.Chmod(path, 0o700) +} + +func validatePlatformPermissions(_ string, info os.FileInfo) error { + if info.Mode().Perm()&0o077 != 0 { + return fmt.Errorf("mode %04o permits group or other access; require owner-only access", info.Mode().Perm()) + } + stat, ok := info.Sys().(*syscall.Stat_t) + if !ok { + return fmt.Errorf("filesystem did not expose staging ownership") + } + if stat.Uid != uint32(os.Geteuid()) { + return fmt.Errorf("owner uid %d does not match controller uid %d", stat.Uid, os.Geteuid()) + } + return nil +} diff --git a/internal/provider/dockersandboxes/staging/permissions_unix_test.go b/internal/provider/dockersandboxes/staging/permissions_unix_test.go new file mode 100644 index 0000000..f5e556c --- /dev/null +++ b/internal/provider/dockersandboxes/staging/permissions_unix_test.go @@ -0,0 +1,25 @@ +//go:build !windows + +package staging + +import ( + "os" + "path/filepath" + "testing" +) + +func TestRejectsForeignUnixOwnerWhenPrivileged(t *testing.T) { + if os.Geteuid() != 0 { + t.Skip("requires a privileged Unix test process") + } + root := filepath.Join(t.TempDir(), "foreign-owner") + if err := os.Mkdir(root, 0700); err != nil { + t.Fatal(err) + } + if err := os.Chown(root, 1, -1); err != nil { + t.Skipf("cannot create a foreign-owned directory: %v", err) + } + if _, err := Open(root); err == nil { + t.Fatal("foreign-owned staging root accepted") + } +} diff --git a/internal/provider/dockersandboxes/staging/permissions_windows.go b/internal/provider/dockersandboxes/staging/permissions_windows.go new file mode 100644 index 0000000..c6a7122 --- /dev/null +++ b/internal/provider/dockersandboxes/staging/permissions_windows.go @@ -0,0 +1,95 @@ +//go:build windows + +package staging + +import ( + "fmt" + "os" + "unsafe" + + "golang.org/x/sys/windows" +) + +const windowsFileAllAccess = windows.ACCESS_MASK(windows.STANDARD_RIGHTS_REQUIRED | windows.SYNCHRONIZE | 0x1ff) + +func restrictPlatformPermissions(path string) error { + user, err := windows.GetCurrentProcessToken().GetTokenUser() + if err != nil { + return fmt.Errorf("get current process user: %w", err) + } + sddl := fmt.Sprintf("D:P(A;OICI;FA;;;%s)(A;OICI;FA;;;SY)", user.User.Sid.String()) + descriptor, err := windows.SecurityDescriptorFromString(sddl) + if err != nil { + return fmt.Errorf("build private staging security descriptor: %w", err) + } + dacl, _, err := descriptor.DACL() + if err != nil { + return fmt.Errorf("read private staging DACL: %w", err) + } + information := windows.SECURITY_INFORMATION(windows.DACL_SECURITY_INFORMATION | windows.PROTECTED_DACL_SECURITY_INFORMATION) + if err := windows.SetNamedSecurityInfo(path, windows.SE_FILE_OBJECT, information, nil, nil, dacl, nil); err != nil { + return fmt.Errorf("set private staging DACL: %w", err) + } + return nil +} + +func validatePlatformPermissions(path string, _ os.FileInfo) error { + descriptor, err := windows.GetNamedSecurityInfo(path, windows.SE_FILE_OBJECT, windows.DACL_SECURITY_INFORMATION) + if err != nil { + return fmt.Errorf("read staging DACL: %w", err) + } + control, _, err := descriptor.Control() + if err != nil { + return fmt.Errorf("read staging DACL control: %w", err) + } + if control&windows.SE_DACL_PROTECTED == 0 { + return fmt.Errorf("DACL inherits access from a parent") + } + dacl, _, err := descriptor.DACL() + if err != nil { + return fmt.Errorf("read staging DACL: %w", err) + } + if dacl == nil { + return fmt.Errorf("DACL is absent") + } + user, err := windows.GetCurrentProcessToken().GetTokenUser() + if err != nil { + return fmt.Errorf("get current process user: %w", err) + } + systemSID, err := windows.CreateWellKnownSid(windows.WinLocalSystemSid) + if err != nil { + return fmt.Errorf("create SYSTEM SID: %w", err) + } + seenCurrent := false + seenSystem := false + for index := uint32(0); index < uint32(dacl.AceCount); index++ { + var ace *windows.ACCESS_ALLOWED_ACE + if err := windows.GetAce(dacl, index, &ace); err != nil { + return fmt.Errorf("read DACL access rule %d: %w", index, err) + } + if ace.Header.AceType != windows.ACCESS_ALLOWED_ACE_TYPE || + ace.Header.AceFlags != windows.OBJECT_INHERIT_ACE|windows.CONTAINER_INHERIT_ACE || + ace.Mask != windowsFileAllAccess { + return fmt.Errorf("DACL contains an unexpected access rule") + } + sid := (*windows.SID)(unsafe.Pointer(&ace.SidStart)) + switch { + case sid.Equals(user.User.Sid): + if seenCurrent { + return fmt.Errorf("DACL contains duplicate current-user access") + } + seenCurrent = true + case sid.Equals(systemSID): + if seenSystem { + return fmt.Errorf("DACL contains duplicate SYSTEM access") + } + seenSystem = true + default: + return fmt.Errorf("DACL grants an unexpected trustee") + } + } + if dacl.AceCount != 2 || !seenCurrent || !seenSystem { + return fmt.Errorf("DACL must grant full access only to the current process identity and SYSTEM") + } + return nil +} diff --git a/internal/provider/dockersandboxes/staging/permissions_windows_test.go b/internal/provider/dockersandboxes/staging/permissions_windows_test.go new file mode 100644 index 0000000..3fe82d3 --- /dev/null +++ b/internal/provider/dockersandboxes/staging/permissions_windows_test.go @@ -0,0 +1,42 @@ +//go:build windows + +package staging + +import ( + "fmt" + "path/filepath" + "testing" + + "golang.org/x/sys/windows" +) + +func TestRejectsUnexpectedWindowsDACLTrustee(t *testing.T) { + staging, err := Open(filepath.Join(t.TempDir(), "staging")) + if err != nil { + t.Fatal(err) + } + owned, err := staging.CreateOwned("weak") + if err != nil { + t.Fatal(err) + } + path := owned.Path + user, err := windows.GetCurrentProcessToken().GetTokenUser() + if err != nil { + t.Fatal(err) + } + descriptor, err := windows.SecurityDescriptorFromString(fmt.Sprintf("D:P(A;OICI;FA;;;%s)(A;OICI;FA;;;SY)(A;OICI;GR;;;WD)", user.User.Sid.String())) + if err != nil { + t.Fatal(err) + } + dacl, _, err := descriptor.DACL() + if err != nil { + t.Fatal(err) + } + information := windows.SECURITY_INFORMATION(windows.DACL_SECURITY_INFORMATION | windows.PROTECTED_DACL_SECURITY_INFORMATION) + if err := windows.SetNamedSecurityInfo(path, windows.SE_FILE_OBJECT, information, nil, nil, dacl, nil); err != nil { + t.Fatal(err) + } + if _, err := staging.VerifyOwnedEmpty("weak", owned.Identity); err == nil { + t.Fatal("staging directory with an unexpected Everyone ACE was accepted") + } +} diff --git a/internal/provider/dockersandboxes/staging/staging.go b/internal/provider/dockersandboxes/staging/staging.go new file mode 100644 index 0000000..69716b0 --- /dev/null +++ b/internal/provider/dockersandboxes/staging/staging.go @@ -0,0 +1,367 @@ +// Package staging manages the only host path exposed to a Docker Sandbox. +// +// A staging directory is always a fresh, empty, direct child of a dedicated +// root. Disposal first binds the exact filesystem object, atomically moves it +// to a deterministic quarantine child, and only then removes that private tree. +package staging + +import ( + "errors" + "fmt" + "os" + "path/filepath" + "runtime" + "strings" +) + +type Staging struct { + root string +} + +type OwnedDirectory struct { + Path string + Identity string +} + +func Open(root string) (*Staging, error) { + if strings.TrimSpace(root) == "" { + return nil, errors.New("Docker Sandboxes staging root is required") + } + absRoot, err := filepath.Abs(root) + if err != nil { + return nil, fmt.Errorf("resolve Docker Sandboxes staging root: %w", err) + } + absRoot = filepath.Clean(absRoot) + if err := rejectAlternateDataStream(absRoot); err != nil { + return nil, err + } + _, statErr := os.Lstat(absRoot) + rootExisted := statErr == nil + if statErr != nil && !os.IsNotExist(statErr) { + return nil, fmt.Errorf("inspect Docker Sandboxes staging root: %w", statErr) + } + if err := createPathWithoutRedirect(absRoot); err != nil { + return nil, fmt.Errorf("create Docker Sandboxes staging root: %w", err) + } + if !rootExisted { + if err := restrictPlatformPermissions(absRoot); err != nil { + return nil, fmt.Errorf("restrict Docker Sandboxes staging root: %w", err) + } + } + if err := validateDirectory(absRoot, false); err != nil { + return nil, fmt.Errorf("validate Docker Sandboxes staging root: %w", err) + } + canonicalRoot, err := platformCanonicalPathSpelling(absRoot) + if err != nil { + return nil, fmt.Errorf("normalize Docker Sandboxes staging root: %w", err) + } + return &Staging{root: filepath.Clean(canonicalRoot)}, nil +} + +func (s *Staging) Root() string { + return s.root +} + +// CreateOwned creates a fresh direct child and captures the stable filesystem +// object identity used to distinguish it from a later same-path replacement. +func (s *Staging) CreateOwned(name string) (OwnedDirectory, error) { + if err := validateName(name); err != nil { + return OwnedDirectory{}, err + } + path, err := s.exactPath(name) + if err != nil { + return OwnedDirectory{}, err + } + if err := os.Mkdir(path, 0700); err != nil { + if os.IsExist(err) { + return OwnedDirectory{}, fmt.Errorf("Docker Sandboxes staging directory %q already exists", path) + } + return OwnedDirectory{}, fmt.Errorf("create Docker Sandboxes staging directory %q: %w", path, err) + } + if err := restrictPlatformPermissions(path); err != nil { + _ = os.Remove(path) + return OwnedDirectory{}, fmt.Errorf("restrict Docker Sandboxes staging directory %q: %w", path, err) + } + if err := validateDirectory(path, true); err != nil { + _ = os.Remove(path) + return OwnedDirectory{}, err + } + identity, err := platformDirectoryIdentity(path) + if err != nil { + _ = os.Remove(path) + return OwnedDirectory{}, fmt.Errorf("read Docker Sandboxes staging directory identity %q: %w", path, err) + } + return OwnedDirectory{Path: path, Identity: identity}, nil +} + +func (s *Staging) verifyEmpty(name string) (string, error) { + if err := validateName(name); err != nil { + return "", err + } + path, err := s.exactPath(name) + if err != nil { + return "", err + } + if err := validateDirectory(path, true); err != nil { + return "", err + } + return path, nil +} + +func (s *Staging) VerifyOwnedEmpty(name, identity string) (string, error) { + return s.verifyOwned(name, identity, true) +} + +func (s *Staging) VerifyOwned(name, identity string) (string, error) { + return s.verifyOwned(name, identity, false) +} + +func (s *Staging) verifyOwned(name, identity string, requireEmpty bool) (string, error) { + if strings.TrimSpace(identity) == "" { + return "", fmt.Errorf("Docker Sandboxes staging directory identity is required") + } + if err := validateName(name); err != nil { + return "", err + } + path, err := s.exactPath(name) + if err != nil { + return "", err + } + if err := validateDirectory(path, requireEmpty); err != nil { + return "", err + } + actual, err := platformDirectoryIdentity(path) + if err != nil { + return "", fmt.Errorf("read Docker Sandboxes staging directory identity %q: %w", path, err) + } + if actual != identity { + return "", fmt.Errorf("Docker Sandboxes staging directory %q was replaced by a different filesystem object", path) + } + return path, nil +} + +func (s *Staging) RemoveEmptyOwned(name, identity string) error { + path, err := s.VerifyOwnedEmpty(name, identity) + if err != nil { + if os.IsNotExist(err) { + return nil + } + return err + } + if err := os.Remove(path); err != nil { + return fmt.Errorf("remove exact owned Docker Sandboxes staging directory %q: %w", path, err) + } + return nil +} + +// PurgeOwned removes a non-empty direct workspace only after the sandbox has been +// proven absent. The deterministic quarantine name makes a crash after rename +// recoverable without ever selecting a path by prefix or following a replaced +// top-level object. +func (s *Staging) PurgeOwned(name, identity string) error { + if err := validateName(name); err != nil { + return err + } + if strings.TrimSpace(identity) == "" { + return fmt.Errorf("Docker Sandboxes staging directory identity is required") + } + if err := validateDirectory(s.root, false); err != nil { + return fmt.Errorf("validate Docker Sandboxes staging root before purge: %w", err) + } + original, err := s.exactPath(name) + if err != nil { + return err + } + quarantineName := name + ".deleting" + quarantine, err := s.exactPath(quarantineName) + if err != nil { + return err + } + originalInfo, originalErr := os.Lstat(original) + quarantineInfo, quarantineErr := os.Lstat(quarantine) + if originalErr == nil && quarantineErr == nil { + return fmt.Errorf("Docker Sandboxes staging source and quarantine both exist") + } + if originalErr != nil && !os.IsNotExist(originalErr) { + return fmt.Errorf("inspect exact Docker Sandboxes staging source: %w", originalErr) + } + if quarantineErr != nil && !os.IsNotExist(quarantineErr) { + return fmt.Errorf("inspect exact Docker Sandboxes staging quarantine: %w", quarantineErr) + } + if originalErr == nil { + if !originalInfo.IsDir() { + return fmt.Errorf("Docker Sandboxes staging source is not a real directory") + } + if _, err := s.VerifyOwned(name, identity); err != nil { + return err + } + if err := os.Rename(original, quarantine); err != nil { + return fmt.Errorf("quarantine exact Docker Sandboxes staging source: %w", err) + } + } else if quarantineErr != nil { + return nil + } + if quarantineInfo != nil && !quarantineInfo.IsDir() { + return fmt.Errorf("Docker Sandboxes staging quarantine is not a real directory") + } + actual, err := platformDirectoryIdentity(quarantine) + if err != nil { + return fmt.Errorf("read quarantined Docker Sandboxes staging identity: %w", err) + } + if actual != identity { + return fmt.Errorf("Docker Sandboxes staging quarantine was replaced by a different filesystem object") + } + if err := os.RemoveAll(quarantine); err != nil { + return fmt.Errorf("purge exact quarantined Docker Sandboxes staging tree: %w", err) + } + if _, err := os.Lstat(quarantine); !os.IsNotExist(err) { + return fmt.Errorf("verify exact Docker Sandboxes staging quarantine absence: %w", err) + } + return nil +} + +func (s *Staging) exactPath(name string) (string, error) { + path := filepath.Join(s.root, name) + relative, err := filepath.Rel(s.root, path) + if err != nil || relative != name || filepath.IsAbs(relative) || strings.Contains(relative, string(filepath.Separator)) { + return "", fmt.Errorf("Docker Sandboxes staging path %q escapes its configured root", path) + } + return path, nil +} + +func validateName(name string) error { + if name == "" || name == "." || name == ".." || strings.TrimSpace(name) != name { + return fmt.Errorf("invalid Docker Sandboxes staging name %q", name) + } + for _, value := range name { + if (value >= 'a' && value <= 'z') || (value >= 'A' && value <= 'Z') || (value >= '0' && value <= '9') || value == '.' || value == '_' || value == '-' { + continue + } + return fmt.Errorf("invalid Docker Sandboxes staging name %q", name) + } + first := name[0] + if !((first >= 'a' && first <= 'z') || (first >= 'A' && first <= 'Z') || (first >= '0' && first <= '9')) { + return fmt.Errorf("invalid Docker Sandboxes staging name %q", name) + } + return nil +} + +func validateDirectory(path string, requireEmpty bool) error { + info, err := os.Lstat(path) + if err != nil { + return err + } + if err := validateNoRedirectDirectory(path); err != nil { + return err + } + if err := validatePlatformPermissions(path, info); err != nil { + return fmt.Errorf("Docker Sandboxes staging path %q has weak permissions: %w", path, err) + } + evaluated, err := filepath.EvalSymlinks(path) + if err != nil { + return fmt.Errorf("resolve Docker Sandboxes staging path %q: %w", path, err) + } + absEvaluated, err := filepath.Abs(evaluated) + if err != nil { + return fmt.Errorf("resolve canonical Docker Sandboxes staging path %q: %w", path, err) + } + absPath, err := filepath.Abs(path) + if err != nil { + return fmt.Errorf("resolve Docker Sandboxes staging path %q: %w", path, err) + } + canonicalPath, err := platformCanonicalPathSpelling(filepath.Clean(absPath)) + if err != nil { + return fmt.Errorf("normalize Docker Sandboxes staging path %q: %w", path, err) + } + if !samePath(filepath.Clean(canonicalPath), filepath.Clean(absEvaluated)) { + return fmt.Errorf("Docker Sandboxes staging path %q contains a symlink, junction, or reparse redirection", path) + } + if requireEmpty { + entries, err := os.ReadDir(path) + if err != nil { + return fmt.Errorf("read Docker Sandboxes staging path %q: %w", path, err) + } + if len(entries) != 0 { + return fmt.Errorf("Docker Sandboxes staging path %q is not empty", path) + } + } + return nil +} + +func createPathWithoutRedirect(path string) error { + missing := make([]string, 0) + cursor := path + for { + _, err := os.Lstat(cursor) + if err == nil { + if err := validateNoRedirectDirectory(cursor); err != nil { + return err + } + break + } + if !os.IsNotExist(err) { + return err + } + missing = append(missing, cursor) + parent := filepath.Dir(cursor) + if parent == cursor { + return fmt.Errorf("no existing ancestor for staging path") + } + cursor = parent + } + for index := len(missing) - 1; index >= 0; index-- { + candidate := missing[index] + if err := os.Mkdir(candidate, 0700); err != nil { + return err + } + if err := validateNoRedirectDirectory(candidate); err != nil { + return err + } + } + return nil +} + +func validateNoRedirectDirectory(path string) error { + info, err := os.Lstat(path) + if err != nil { + return err + } + if isPlatformRedirect(info) || !info.IsDir() { + return fmt.Errorf("Docker Sandboxes staging path %q is not a real directory", path) + } + evaluated, err := filepath.EvalSymlinks(path) + if err != nil { + return fmt.Errorf("resolve Docker Sandboxes staging path %q: %w", path, err) + } + absEvaluated, err := filepath.Abs(evaluated) + if err != nil { + return fmt.Errorf("resolve canonical Docker Sandboxes staging path %q: %w", path, err) + } + absPath, err := filepath.Abs(path) + if err != nil { + return fmt.Errorf("resolve Docker Sandboxes staging path %q: %w", path, err) + } + canonicalPath, err := platformCanonicalPathSpelling(filepath.Clean(absPath)) + if err != nil { + return fmt.Errorf("normalize Docker Sandboxes staging path %q: %w", path, err) + } + if !samePath(filepath.Clean(canonicalPath), filepath.Clean(absEvaluated)) { + return fmt.Errorf("Docker Sandboxes staging path %q contains a symlink, junction, or reparse redirection", path) + } + return nil +} + +func rejectAlternateDataStream(path string) error { + rest := strings.TrimPrefix(path, filepath.VolumeName(path)) + if strings.Contains(rest, ":") { + return fmt.Errorf("Docker Sandboxes staging path %q contains an alternate-data-stream separator", path) + } + return nil +} + +func samePath(left, right string) bool { + if runtime.GOOS == "windows" { + return strings.EqualFold(left, right) + } + return left == right +} diff --git a/internal/provider/dockersandboxes/staging/staging_test.go b/internal/provider/dockersandboxes/staging/staging_test.go new file mode 100644 index 0000000..b558215 --- /dev/null +++ b/internal/provider/dockersandboxes/staging/staging_test.go @@ -0,0 +1,250 @@ +package staging + +import ( + "os" + "path/filepath" + "runtime" + "strings" + "sync" + "testing" +) + +func TestCreateAndRemoveExactEmptyStagingDirectory(t *testing.T) { + staging, err := Open(filepath.Join(t.TempDir(), "staging")) + if err != nil { + t.Fatal(err) + } + owned, err := staging.CreateOwned("epar-sbx-001") + if err != nil { + t.Fatal(err) + } + path := owned.Path + if got, want := path, filepath.Join(staging.Root(), "epar-sbx-001"); got != want { + t.Fatalf("path = %q, want %q", got, want) + } + if _, err := staging.VerifyOwnedEmpty("epar-sbx-001", owned.Identity); err != nil { + t.Fatal(err) + } + if err := staging.RemoveEmptyOwned("epar-sbx-001", owned.Identity); err != nil { + t.Fatal(err) + } + if _, err := os.Stat(path); !os.IsNotExist(err) { + t.Fatalf("removed staging directory still exists: %v", err) + } + if err := staging.RemoveEmptyOwned("epar-sbx-001", owned.Identity); err != nil { + t.Fatalf("missing exact staging directory should be idempotent: %v", err) + } +} + +func TestRejectsUnsafeNamesAndAlternateDataStreams(t *testing.T) { + staging, err := Open(filepath.Join(t.TempDir(), "staging")) + if err != nil { + t.Fatal(err) + } + for _, name := range []string{"", ".", "..", "../escape", `..\\escape`, " leading", "trailing ", "name:stream", "name/subdir", "name\\subdir"} { + t.Run(strings.ReplaceAll(name, "/", "_"), func(t *testing.T) { + if _, err := staging.CreateOwned(name); err == nil { + t.Fatalf("Create(%q) succeeded", name) + } + }) + } +} + +func TestRejectsPreexistingOrNonemptyDirectory(t *testing.T) { + staging, err := Open(filepath.Join(t.TempDir(), "staging")) + if err != nil { + t.Fatal(err) + } + path := filepath.Join(staging.Root(), "existing") + if err := os.Mkdir(path, 0700); err != nil { + t.Fatal(err) + } + if _, err := staging.CreateOwned("existing"); err == nil { + t.Fatal("pre-existing staging directory accepted") + } + if err := os.WriteFile(filepath.Join(path, "sentinel"), []byte("host data"), 0600); err != nil { + t.Fatal(err) + } + if _, err := staging.verifyEmpty("existing"); err == nil { + t.Fatal("non-empty staging directory verified") + } + if _, err := os.Stat(filepath.Join(path, "sentinel")); err != nil { + t.Fatalf("sentinel was changed: %v", err) + } +} + +func TestRejectsSymlinkOrJunctionRedirection(t *testing.T) { + root := t.TempDir() + target := filepath.Join(root, "target") + if err := os.Mkdir(target, 0700); err != nil { + t.Fatal(err) + } + link := filepath.Join(root, "link") + if err := os.Symlink(target, link); err != nil { + t.Skipf("symlinks unavailable: %v", err) + } + if _, err := Open(link); err == nil { + t.Fatal("redirected staging root accepted") + } +} + +func TestOpenDoesNotCreateThroughRedirectedMissingAncestor(t *testing.T) { + root := t.TempDir() + target := filepath.Join(root, "target") + if err := os.Mkdir(target, 0700); err != nil { + t.Fatal(err) + } + redirect := filepath.Join(root, "redirect") + if err := os.Symlink(target, redirect); err != nil { + t.Skipf("symlinks unavailable: %v", err) + } + if _, err := Open(filepath.Join(redirect, "must-not-exist")); err == nil { + t.Fatal("missing staging root beneath redirect was accepted") + } + if _, err := os.Lstat(filepath.Join(target, "must-not-exist")); !os.IsNotExist(err) { + t.Fatalf("Open created content through redirected ancestor: %v", err) + } +} + +func TestRejectsWeakUnixPermissions(t *testing.T) { + if runtime.GOOS == "windows" { + t.Skip("Unix permission bits do not describe Windows ACL strength") + } + root := filepath.Join(t.TempDir(), "staging") + if err := os.Mkdir(root, 0755); err != nil { + t.Fatal(err) + } + if _, err := Open(root); err == nil { + t.Fatal("weak staging root permissions accepted") + } +} + +func TestCreatedStagingDirectoryHasStrongPlatformPermissions(t *testing.T) { + staging, err := Open(filepath.Join(t.TempDir(), "staging")) + if err != nil { + t.Fatal(err) + } + owned, err := staging.CreateOwned("private") + if err != nil { + t.Fatal(err) + } + path := owned.Path + info, err := os.Lstat(path) + if err != nil { + t.Fatal(err) + } + if err := validatePlatformPermissions(path, info); err != nil { + t.Fatalf("created staging permissions are weak: %v", err) + } +} + +func TestConcurrentCreateHasOneOwner(t *testing.T) { + staging, err := Open(filepath.Join(t.TempDir(), "staging")) + if err != nil { + t.Fatal(err) + } + const callers = 16 + results := make(chan error, callers) + var wait sync.WaitGroup + wait.Add(callers) + for index := 0; index < callers; index++ { + go func() { + defer wait.Done() + _, err := staging.CreateOwned("one-owner") + results <- err + }() + } + wait.Wait() + close(results) + successes := 0 + for err := range results { + if err == nil { + successes++ + } + } + if successes != 1 { + t.Fatalf("successful concurrent creates = %d, want 1", successes) + } +} + +func TestOwnedIdentityRejectsSamePathReplacement(t *testing.T) { + staging, err := Open(filepath.Join(t.TempDir(), "staging")) + if err != nil { + t.Fatal(err) + } + owned, err := staging.CreateOwned("replace-me") + if err != nil { + t.Fatal(err) + } + previous := filepath.Join(staging.Root(), "previous-object") + if err := os.Rename(owned.Path, previous); err != nil { + t.Fatal(err) + } + if err := os.Mkdir(owned.Path, 0700); err != nil { + t.Fatal(err) + } + if err := restrictPlatformPermissions(owned.Path); err != nil { + t.Fatal(err) + } + if _, err := staging.VerifyOwnedEmpty("replace-me", owned.Identity); err == nil || !strings.Contains(err.Error(), "replaced") { + t.Fatalf("replacement identity error = %v", err) + } + if err := staging.RemoveEmptyOwned("replace-me", owned.Identity); err == nil { + t.Fatal("same-path replacement was removed") + } + if _, err := os.Stat(owned.Path); err != nil { + t.Fatalf("same-path replacement was not preserved: %v", err) + } +} + +func TestPurgeOwnedDoesNotFollowNestedSymlink(t *testing.T) { + staging, err := Open(filepath.Join(t.TempDir(), "staging")) + if err != nil { + t.Fatal(err) + } + owned, err := staging.CreateOwned("symlink-tree") + if err != nil { + t.Fatal(err) + } + target := filepath.Join(t.TempDir(), "target") + if err := os.Mkdir(target, 0700); err != nil { + t.Fatal(err) + } + sentinel := filepath.Join(target, "preserve") + if err := os.WriteFile(sentinel, []byte("host data"), 0600); err != nil { + t.Fatal(err) + } + if err := os.Symlink(target, filepath.Join(owned.Path, "workflow-link")); err != nil { + t.Skipf("symlinks unavailable: %v", err) + } + if err := staging.PurgeOwned("symlink-tree", owned.Identity); err != nil { + t.Fatal(err) + } + if content, err := os.ReadFile(sentinel); err != nil || string(content) != "host data" { + t.Fatalf("purge followed nested symlink: content=%q err=%v", content, err) + } +} + +func TestPurgeOwnedRecoversDeterministicQuarantine(t *testing.T) { + staging, err := Open(filepath.Join(t.TempDir(), "staging")) + if err != nil { + t.Fatal(err) + } + owned, err := staging.CreateOwned("recover") + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(owned.Path, "content"), []byte("job output"), 0600); err != nil { + t.Fatal(err) + } + quarantine := filepath.Join(staging.Root(), "recover.deleting") + if err := os.Rename(owned.Path, quarantine); err != nil { + t.Fatal(err) + } + if err := staging.PurgeOwned("recover", owned.Identity); err != nil { + t.Fatal(err) + } + if _, err := os.Lstat(quarantine); !os.IsNotExist(err) { + t.Fatalf("recovered quarantine remains: %v", err) + } +} diff --git a/internal/provider/dockersandboxes/staging/staging_windows_test.go b/internal/provider/dockersandboxes/staging/staging_windows_test.go new file mode 100644 index 0000000..336b54b --- /dev/null +++ b/internal/provider/dockersandboxes/staging/staging_windows_test.go @@ -0,0 +1,76 @@ +//go:build windows + +package staging + +import ( + "os" + "path/filepath" + "strings" + "syscall" + "testing" + + "golang.org/x/sys/windows" +) + +type windowsReparseDirectoryInfo struct { + os.FileInfo +} + +func (windowsReparseDirectoryInfo) Mode() os.FileMode { + return os.ModeDir +} + +func (windowsReparseDirectoryInfo) Sys() any { + return &syscall.Win32FileAttributeData{FileAttributes: syscall.FILE_ATTRIBUTE_DIRECTORY | syscall.FILE_ATTRIBUTE_REPARSE_POINT} +} + +func TestPlatformRedirectRejectsWindowsReparseDirectory(t *testing.T) { + if !isPlatformRedirect(windowsReparseDirectoryInfo{}) { + t.Fatal("Windows reparse directory was not classified as a redirect") + } +} + +func TestOpenAcceptsWindowsShortPathAlias(t *testing.T) { + longParent := filepath.Join(t.TempDir(), "Long Directory Name") + if err := os.Mkdir(longParent, 0o700); err != nil { + t.Fatal(err) + } + if err := restrictPlatformPermissions(longParent); err != nil { + t.Fatal(err) + } + shortParent := stagingWindowsShortPath(t, longParent) + if strings.EqualFold(shortParent, longParent) { + t.Skip("filesystem did not provide a distinct short path alias") + } + staging, err := Open(filepath.Join(shortParent, "staging")) + if err != nil { + t.Fatalf("Open() rejected a Windows short path alias: %v", err) + } + want, err := platformCanonicalPathSpelling(filepath.Join(longParent, "staging")) + if err != nil { + t.Fatal(err) + } + if !strings.EqualFold(staging.Root(), want) { + t.Fatalf("Root() = %q, want canonical spelling %q", staging.Root(), want) + } +} + +func stagingWindowsShortPath(t *testing.T, path string) string { + t.Helper() + pointer, err := windows.UTF16PtrFromString(path) + if err != nil { + t.Fatal(err) + } + size := uint32(len(path) + 1) + for { + buffer := make([]uint16, size) + length, err := windows.GetShortPathName(pointer, &buffer[0], size) + if err != nil { + t.Skipf("Windows short paths unavailable: %v", err) + } + if length < size { + return windows.UTF16ToString(buffer[:length]) + } + size = length + 1 + } +} diff --git a/internal/provider/dockersandboxes/validation.go b/internal/provider/dockersandboxes/validation.go new file mode 100644 index 0000000..58b98b5 --- /dev/null +++ b/internal/provider/dockersandboxes/validation.go @@ -0,0 +1,212 @@ +package dockersandboxes + +import ( + "fmt" + "net" + "path/filepath" + "regexp" + "strconv" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +var ( + sandboxNamePattern = regexp.MustCompile(`^[a-z0-9](?:[a-z0-9.-]{0,61}[a-z0-9])?$`) + providerIDPattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$`) + sizePattern = regexp.MustCompile(`^[1-9][0-9]*(?:[kKmMgGtT](?:i?[bB])?)?$`) + templatePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._/:@+-]{0,511}$`) + templateDigestPattern = regexp.MustCompile(`^sha256:[a-f0-9]{64}$`) + profilePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$`) + hostLabelPattern = regexp.MustCompile(`^[A-Za-z0-9](?:[A-Za-z0-9-]{0,61}[A-Za-z0-9])?$`) +) + +func validateCreateRequest(request provider.CreateRequest) error { + if !sandboxNamePattern.MatchString(request.Name) { + return fmt.Errorf("invalid docker sandbox name") + } + if request.Source != "" { + return fmt.Errorf("docker sandboxes does not accept a legacy source image") + } + if request.CPUs <= 0 || request.CPUs > 256 { + return fmt.Errorf("docker sandbox cpu count must be between 1 and 256") + } + if !sizePattern.MatchString(request.Memory) { + return fmt.Errorf("invalid docker sandbox memory limit") + } + if !templatePattern.MatchString(request.Template) || strings.HasPrefix(request.Template, "-") { + return fmt.Errorf("invalid docker sandbox template") + } + if _, _, err := splitTemplateReference(request.Template); err != nil { + return err + } + if !templateDigestPattern.MatchString(request.TemplateDigest) { + return fmt.Errorf("docker sandbox template digest must be a full lowercase sha256 identity") + } + if err := validateCanonicalStagingPath(request.StagingPath); err != nil { + return err + } + if request.RootDisk != "" && !sizePattern.MatchString(request.RootDisk) { + return fmt.Errorf("invalid docker sandbox root disk size") + } + if request.DockerDisk != "" && !sizePattern.MatchString(request.DockerDisk) { + return fmt.Errorf("invalid docker sandbox docker disk size") + } + return nil +} + +func splitTemplateReference(reference string) (repository, tag string, err error) { + separator := strings.LastIndex(reference, ":") + if separator <= strings.LastIndex(reference, "/") || separator == len(reference)-1 || strings.Contains(reference, "@") { + return "", "", fmt.Errorf("docker sandbox template must be an exact repository:tag reference") + } + repository, tag = reference[:separator], reference[separator+1:] + if repository == "" || tag == "" { + return "", "", fmt.Errorf("docker sandbox template must be an exact repository:tag reference") + } + if !strings.Contains(repository, "/") { + repository = "docker.io/library/" + repository + } else { + first := strings.SplitN(repository, "/", 2)[0] + if first != "localhost" && !strings.ContainsAny(first, ".:") { + repository = "docker.io/" + repository + } + } + return repository, tag, nil +} + +func validateLocalTemplateReference(reference string) error { + if !templatePattern.MatchString(reference) || strings.HasPrefix(reference, "-") { + return fmt.Errorf("docker sandbox template must be an exact repository:tag reference") + } + _, tag, err := splitTemplateReference(reference) + if err != nil || !profilePattern.MatchString(tag) { + return fmt.Errorf("docker sandbox template must be an exact repository:tag reference") + } + return nil +} + +func validateCanonicalStagingPath(path string) error { + if path == "" || strings.ContainsRune(path, 0) || strings.ContainsAny(path, "\r\n") { + return fmt.Errorf("invalid docker sandbox staging path") + } + if !filepath.IsAbs(path) || filepath.Clean(path) != path { + return fmt.Errorf("docker sandbox staging path must be an already-validated canonical absolute path") + } + return nil +} + +func validateInstance(instance provider.Instance, requireID bool) error { + if !sandboxNamePattern.MatchString(instance.Name) { + return fmt.Errorf("invalid docker sandbox identity") + } + if requireID && !providerIDPattern.MatchString(instance.ProviderID) { + return fmt.Errorf("docker sandbox stable provider id is required") + } + if instance.ProviderID != "" && !providerIDPattern.MatchString(instance.ProviderID) { + return fmt.Errorf("invalid docker sandbox provider id") + } + return nil +} + +func validateGuestCommand(command []string, opts provider.ExecOptions) error { + if len(command) == 0 { + return fmt.Errorf("docker sandbox guest command is required") + } + for _, arg := range command { + if strings.ContainsRune(arg, 0) { + return fmt.Errorf("docker sandbox guest command contains a null byte") + } + } + if len(opts.Env) != 0 { + return fmt.Errorf("docker sandbox guest environment passthrough is not permitted") + } + if opts.LogPath != "" { + return fmt.Errorf("docker sandbox host log paths are not permitted") + } + return nil +} + +func validateNetworkRule(rule provider.NetworkPolicyRule, forRemoval bool) error { + if forRemoval { + if !providerIDPattern.MatchString(rule.ID) { + return fmt.Errorf("docker sandbox network rule id is required for exact removal") + } + if !providerIDPattern.MatchString(rule.PolicyID) || rule.Scope == "" || rule.AppliesTo == "" || rule.ResourceType == "" || rule.Origin == "" { + return fmt.Errorf("docker sandbox network rule requires its complete stable identity for exact removal") + } + if rule.Decision != provider.NetworkPolicyAllow && rule.Decision != provider.NetworkPolicyDeny { + return fmt.Errorf("invalid docker sandbox network decision") + } + if len(rule.Resources) == 0 { + return fmt.Errorf("docker sandbox network policy requires at least one resource") + } + seen := make(map[string]struct{}, len(rule.Resources)) + for _, resource := range rule.Resources { + if strings.TrimSpace(resource) == "" { + return fmt.Errorf("docker sandbox network policy contains an empty resource") + } + if _, duplicate := seen[resource]; duplicate { + return fmt.Errorf("docker sandbox network policy contains a duplicate resource") + } + seen[resource] = struct{}{} + } + return nil + } + if !forRemoval && rule.ID != "" { + return fmt.Errorf("docker sandbox network rule id must be empty when applying") + } + if !forRemoval && rule.ResourceType != "" && rule.ResourceType != "network" { + return fmt.Errorf("docker sandbox policy mutation supports only network rules") + } + if rule.Decision != provider.NetworkPolicyAllow && rule.Decision != provider.NetworkPolicyDeny { + return fmt.Errorf("invalid docker sandbox network decision") + } + if len(rule.Resources) == 0 { + return fmt.Errorf("docker sandbox network policy requires at least one resource") + } + seen := make(map[string]struct{}, len(rule.Resources)) + for _, resource := range rule.Resources { + if resource != "**" || rule.Decision != provider.NetworkPolicyAllow { + if err := validateNetworkResource(resource); err != nil { + return err + } + } + if _, duplicate := seen[resource]; duplicate { + return fmt.Errorf("docker sandbox network policy contains a duplicate resource") + } + seen[resource] = struct{}{} + } + return nil +} + +func validateNetworkResource(resource string) error { + if resource == "" || len(resource) > 512 || strings.ContainsAny(resource, "\x00\r\n\t ,\\/@?#[]") || strings.HasPrefix(resource, "-") { + return fmt.Errorf("invalid docker sandbox network resource") + } + host := resource + if candidateHost, port, ok := strings.Cut(host, ":"); ok { + if strings.Contains(port, ":") || !validPort(port) { + return fmt.Errorf("invalid docker sandbox network resource") + } + host = candidateHost + } + wildcard := strings.HasPrefix(host, "*.") + if wildcard { + host = strings.TrimPrefix(host, "*.") + } + if host == "" || len(host) > 253 || net.ParseIP(host) != nil || wildcard && !strings.Contains(host, ".") { + return fmt.Errorf("invalid docker sandbox network resource") + } + for _, label := range strings.Split(host, ".") { + if !hostLabelPattern.MatchString(label) { + return fmt.Errorf("invalid docker sandbox network resource") + } + } + return nil +} + +func validPort(value string) bool { + port, err := strconv.Atoi(value) + return err == nil && port >= 1 && port <= 65535 +} diff --git a/internal/provider/factory.go b/internal/provider/factory.go index 98b1465..96780b5 100644 --- a/internal/provider/factory.go +++ b/internal/provider/factory.go @@ -2,10 +2,37 @@ package provider import "fmt" -func SupportedTypes() []string { - return []string{"tart", "wsl", "docker-dind"} +// Descriptor is the provider-neutral registration record consumed by +// onboarding, configuration, lifecycle, image, and contract tests. A provider +// must not appear in SupportedTypes unless all required contributions exist. +type Descriptor struct { + Type string + DisplayName string + WizardSupported bool + WizardNumber string + WizardLabel string + WizardAliases []string + ConfigurationDecoder bool + ConfigurationDefaults bool + ConfigurationValidator bool + LifecycleSupported bool + StorageSupported bool + ImageMode string + GuidedArtifacts bool + WizardImageProfiles []WizardImageProfile } +type WizardImageProfile struct { + Name string + Tag string +} + +const ( + ImageModeDocker = "docker-image" + ImageModeNative = "native-image" + ImageModeTemplate = "sandbox-template" +) + func UnsupportedTypeError(providerType string) error { return fmt.Errorf("unsupported provider.type %q", providerType) } diff --git a/internal/provider/layout_test.go b/internal/provider/layout_test.go new file mode 100644 index 0000000..60cf420 --- /dev/null +++ b/internal/provider/layout_test.go @@ -0,0 +1,73 @@ +package provider + +import ( + "go/parser" + "go/token" + "os" + "path/filepath" + "runtime" + "strconv" + "strings" + "testing" +) + +func TestDockerSandboxesDomainPackagesStayUnderProviderTree(t *testing.T) { + _, filename, _, ok := runtime.Caller(0) + if !ok { + t.Fatal("could not locate provider package") + } + providerDirectory := filepath.Dir(filename) + internalDirectory := filepath.Dir(providerDirectory) + for _, name := range []string{"sandboxcapacity", "sandboxfs", "sandboxledger", "sandboxpolicy", "sandboxpromotion"} { + path := filepath.Join(internalDirectory, name) + if _, err := os.Stat(path); !os.IsNotExist(err) { + t.Fatalf("Docker Sandboxes domain package must not live directly under internal: %s", path) + } + } + for _, name := range []string{"capacity", "staging", "policy", "promotion"} { + path := filepath.Join(providerDirectory, "dockersandboxes", name) + info, err := os.Stat(path) + if err != nil { + t.Fatalf("required Docker Sandboxes domain package %s: %v", path, err) + } + if !info.IsDir() { + t.Fatalf("required Docker Sandboxes domain package is not a directory: %s", path) + } + } + if _, err := os.Stat(filepath.Join(providerDirectory, "dockersandboxes", "ledger")); !os.IsNotExist(err) { + t.Fatalf("Docker Sandboxes must use the provider-neutral pool ledger instead of a provider-local ledger") + } +} + +func TestProviderImplementationsDoNotOwnPoolControllers(t *testing.T) { + _, filename, _, ok := runtime.Caller(0) + if !ok { + t.Fatal("could not locate provider package") + } + root := filepath.Dir(filename) + err := filepath.WalkDir(root, func(path string, entry os.DirEntry, walkErr error) error { + if walkErr != nil { + return walkErr + } + if entry.IsDir() || filepath.Ext(path) != ".go" || strings.HasSuffix(path, "_test.go") { + return nil + } + parsed, err := parser.ParseFile(token.NewFileSet(), path, nil, parser.ImportsOnly) + if err != nil { + return err + } + for _, imported := range parsed.Imports { + value, err := strconv.Unquote(imported.Path.Value) + if err != nil { + return err + } + if value == "github.com/solutionforest/ephemeral-action-runner/internal/pool" || strings.HasPrefix(value, "github.com/solutionforest/ephemeral-action-runner/internal/pool/") { + t.Errorf("%s imports %q; provider packages implement host integration and must not own a pool controller", path, value) + } + } + return nil + }) + if err != nil { + t.Fatal(err) + } +} diff --git a/internal/provider/legacy_adapter.go b/internal/provider/legacy_adapter.go new file mode 100644 index 0000000..7ffc19d --- /dev/null +++ b/internal/provider/legacy_adapter.go @@ -0,0 +1,161 @@ +package provider + +import ( + "context" + "encoding/json" + "fmt" + "time" +) + +type legacyAdapter struct { + provider Provider + dryRun bool +} + +// AdaptLegacy preserves existing provider behavior behind the new lifecycle +// surface. Legacy runtime readiness is still performed by each provider's +// Start implementation, so VerifyRuntime intentionally has no additional side +// effect. The old one-second address wait is used only when a caller explicitly +// requests Address. +func AdaptLegacy(provider Provider, dryRun ...bool) Lifecycle { + adapter := &legacyAdapter{provider: provider} + if len(dryRun) != 0 { + adapter.dryRun = dryRun[0] + } + return adapter +} + +func (adapter *legacyAdapter) Create(ctx context.Context, request CreateRequest) (Instance, error) { + if adapter == nil || adapter.provider == nil { + return Instance{}, fmt.Errorf("legacy provider is nil") + } + if request.Name == "" { + return Instance{}, fmt.Errorf("instance name is required") + } + if err := adapter.provider.Clone(ctx, request.Source, request.Name); err != nil { + return Instance{}, err + } + readbackContext, cancel := context.WithTimeout(context.WithoutCancel(ctx), 60*time.Second) + defer cancel() + instances, err := adapter.provider.List(readbackContext) + if err != nil { + return Instance{}, fmt.Errorf("read exact provider identity after create: %w", err) + } + for _, instance := range instances { + if instance.Name != request.Name { + continue + } + if instance.ProviderID == "" { + return Instance{}, fmt.Errorf("provider inventory omitted immutable identity for %q", request.Name) + } + instance.ReceiptVersion = "v1" + instance.Receipt, _ = json.Marshal(map[string]string{"providerId": instance.ProviderID, "source": request.Source}) + return instance, nil + } + if adapter.dryRun { + receipt, _ := json.Marshal(map[string]string{"providerId": "dry-run:" + request.Name, "source": request.Source}) + return Instance{Name: request.Name, ProviderID: "dry-run:" + request.Name, Source: request.Source, ReceiptVersion: "v1", Receipt: receipt}, nil + } + return Instance{}, fmt.Errorf("provider inventory omitted newly created instance %q", request.Name) +} + +func (adapter *legacyAdapter) Start(ctx context.Context, instance Instance, opts StartOptions) (*RunningProcess, error) { + if err := adapter.assertExactIdentity(ctx, instance); err != nil { + return nil, err + } + return adapter.provider.Start(ctx, instance.Name, opts) +} + +func (adapter *legacyAdapter) VerifyRuntime(ctx context.Context, instance Instance) (RuntimeInfo, error) { + if err := adapter.assertExactIdentity(ctx, instance); err != nil { + return RuntimeInfo{}, err + } + if _, err := adapter.provider.Exec(ctx, instance.Name, []string{"sudo", "bash", "/opt/epar/validate-runtime.sh"}, ExecOptions{}); err != nil { + return RuntimeInfo{}, err + } + return RuntimeInfo{Ready: true}, nil +} + +func (adapter *legacyAdapter) Address(ctx context.Context, instance Instance, waitSeconds int) (string, bool, error) { + if err := adapter.assertExactIdentity(ctx, instance); err != nil { + return "", false, err + } + address, err := adapter.provider.IP(ctx, instance.Name, waitSeconds) + if err != nil { + return "", false, err + } + if address == "" { + return "", false, nil + } + return address, true, nil +} + +func (adapter *legacyAdapter) Exec(ctx context.Context, instance Instance, command []string, opts ExecOptions) (ExecResult, error) { + if err := adapter.assertExactIdentity(ctx, instance); err != nil { + return ExecResult{}, err + } + return adapter.provider.Exec(ctx, instance.Name, command, opts) +} + +func (adapter *legacyAdapter) Diagnostics(ctx context.Context, instance Instance) (Diagnostics, error) { + if err := adapter.assertExactIdentity(ctx, instance); err != nil { + return Diagnostics{}, err + } + return Diagnostics{}, nil +} + +func (adapter *legacyAdapter) Stop(ctx context.Context, instance Instance) error { + if err := adapter.assertExactIdentity(ctx, instance); err != nil { + return err + } + return adapter.provider.Stop(ctx, instance.Name) +} + +func (adapter *legacyAdapter) Delete(ctx context.Context, instance Instance) error { + if err := adapter.assertExactIdentity(ctx, instance); err != nil { + return err + } + return adapter.provider.Delete(ctx, instance.Name) +} + +func (adapter *legacyAdapter) Inventory(ctx context.Context) ([]InventoryItem, error) { + identityContext, cancel := context.WithTimeout(context.WithoutCancel(ctx), 60*time.Second) + defer cancel() + instances, err := adapter.provider.List(identityContext) + if err != nil { + return nil, err + } + items := make([]InventoryItem, 0, len(instances)) + for _, instance := range instances { + items = append(items, InventoryItem{Instance: instance, State: instance.State, Source: instance.Source}) + } + return items, nil +} + +func (adapter *legacyAdapter) assertExactIdentity(ctx context.Context, expected Instance) error { + if adapter == nil || adapter.provider == nil { + return fmt.Errorf("legacy provider is nil") + } + if adapter.dryRun && expected.ProviderID == "dry-run:"+expected.Name { + return nil + } + if expected.Name == "" || expected.ProviderID == "" { + return fmt.Errorf("exact provider name and immutable id are required") + } + identityContext, cancel := context.WithTimeout(context.WithoutCancel(ctx), 60*time.Second) + defer cancel() + instances, err := adapter.provider.List(identityContext) + if err != nil { + return err + } + for _, actual := range instances { + if actual.Name != expected.Name { + continue + } + if actual.ProviderID == "" || actual.ProviderID != expected.ProviderID { + return fmt.Errorf("provider identity mismatch for %q: expected %q, observed %q", expected.Name, expected.ProviderID, actual.ProviderID) + } + return nil + } + return fmt.Errorf("provider instance %q is missing", expected.Name) +} diff --git a/internal/provider/legacy_adapter_test.go b/internal/provider/legacy_adapter_test.go new file mode 100644 index 0000000..b5bc4b6 --- /dev/null +++ b/internal/provider/legacy_adapter_test.go @@ -0,0 +1,109 @@ +package provider + +import ( + "context" + "testing" +) + +type legacyProviderFake struct { + calls []string + ip string +} + +func (fake *legacyProviderFake) Clone(_ context.Context, source, name string) error { + fake.calls = append(fake.calls, "clone:"+source+":"+name) + return nil +} + +func (fake *legacyProviderFake) Start(_ context.Context, name string, _ StartOptions) (*RunningProcess, error) { + fake.calls = append(fake.calls, "start:"+name) + return &RunningProcess{Name: name}, nil +} + +func (fake *legacyProviderFake) Exec(_ context.Context, name string, command []string, _ ExecOptions) (ExecResult, error) { + fake.calls = append(fake.calls, "exec:"+name) + return ExecResult{Stdout: command[0]}, nil +} + +func (fake *legacyProviderFake) IP(_ context.Context, name string, waitSeconds int) (string, error) { + fake.calls = append(fake.calls, "ip:"+name) + if waitSeconds != 30 { + panic("unexpected address wait") + } + return fake.ip, nil +} + +func (fake *legacyProviderFake) Stop(_ context.Context, name string) error { + fake.calls = append(fake.calls, "stop:"+name) + return nil +} + +func (fake *legacyProviderFake) Delete(_ context.Context, name string) error { + fake.calls = append(fake.calls, "delete:"+name) + return nil +} + +func (fake *legacyProviderFake) List(context.Context) ([]Instance, error) { + fake.calls = append(fake.calls, "list") + return []Instance{{Name: "runner", ProviderID: "fake:runner-id", Source: "image", State: "running"}}, nil +} + +func TestAdaptLegacyMapsLifecycleWithoutAdditionalRuntimeProbe(t *testing.T) { + legacy := &legacyProviderFake{ip: "192.0.2.10"} + lifecycle := AdaptLegacy(legacy) + instance, err := lifecycle.Create(context.Background(), CreateRequest{Name: "runner", Source: "image"}) + if err != nil { + t.Fatal(err) + } + if _, err := lifecycle.Start(context.Background(), instance, StartOptions{}); err != nil { + t.Fatal(err) + } + runtime, err := lifecycle.VerifyRuntime(context.Background(), instance) + if err != nil || !runtime.Ready { + t.Fatalf("runtime = %#v, err = %v", runtime, err) + } + address, available, err := lifecycle.Address(context.Background(), instance, 30) + if err != nil || !available || address != "192.0.2.10" { + t.Fatalf("address = %q, available = %v, err = %v", address, available, err) + } + result, err := lifecycle.Exec(context.Background(), instance, []string{"true"}, ExecOptions{}) + if err != nil || result.Stdout != "true" { + t.Fatalf("result = %#v, err = %v", result, err) + } + if err := lifecycle.Stop(context.Background(), instance); err != nil { + t.Fatal(err) + } + if err := lifecycle.Delete(context.Background(), instance); err != nil { + t.Fatal(err) + } + items, err := lifecycle.Inventory(context.Background()) + if err != nil { + t.Fatal(err) + } + if len(items) != 1 || items[0].Instance.Name != "runner" || items[0].State != "running" { + t.Fatalf("inventory = %#v", items) + } + for _, want := range []string{"clone:image:runner", "start:runner", "ip:runner", "exec:runner", "stop:runner", "delete:runner"} { + if !containsCall(legacy.calls, want) { + t.Fatalf("calls = %#v, missing %q", legacy.calls, want) + } + } +} + +func TestAdaptLegacyAddressCanBeUnavailable(t *testing.T) { + legacy := &legacyProviderFake{} + lifecycle := AdaptLegacy(legacy) + address, available, err := lifecycle.Address(context.Background(), Instance{Name: "runner", ProviderID: "fake:runner-id"}, 30) + if err != nil || available || address != "" { + t.Fatalf("address = %q, available = %v, err = %v", address, available, err) + } +} + +func containsCall(calls []string, want string) bool { + for _, call := range calls { + if call == want { + return true + } + } + return false +} diff --git a/internal/provider/provider.go b/internal/provider/provider.go index 71b3949..e64ceb6 100644 --- a/internal/provider/provider.go +++ b/internal/provider/provider.go @@ -2,15 +2,70 @@ package provider import ( "context" + "encoding/json" + "errors" "fmt" "io" + "os" "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" ) +var ErrTemplateNotFound = errors.New("imported provider template not found") + type Instance struct { - Name string - Source string - State string + Name string + ProviderID string + Source string + State string + ReceiptVersion string + Receipt json.RawMessage +} + +// CreateRequest is the provider-neutral input for allocating one isolated +// runtime. Providers must reject fields they cannot honor instead of silently +// weakening the requested isolation or resource constraints. +type CreateRequest struct { + Name string + Source string + Template string + // TemplateDigest is the full sha256 local image/template identity recorded by + // the trusted build and load manifest. Providers must fail closed when they + // cannot bind the configured template reference to this identity. + TemplateDigest string + StagingPath string + CPUs int + Memory string + RootDisk string + DockerDisk string +} + +type RuntimeInfo struct { + Ready bool + Runtime string + Version string +} + +type Diagnostics struct { + Healthy bool + DaemonState string + ChecksPassed int + ChecksWarned int + ChecksFailed int + ChecksSkipped int + OutputLimited bool +} + +type InventoryItem struct { + Instance Instance + State string + Source string + // Workspaces are the exact host paths reported by the provider for this + // instance. They are ownership evidence for crash reconciliation; callers + // must compare canonical paths using host-platform path semantics. + Workspaces []string } type StartOptions struct { @@ -28,12 +83,14 @@ type RunningProcess struct { } type ExecOptions struct { - Stdin string - Env map[string]string - SensitiveValues []string - LogPath string - Stdout io.Writer - Stderr io.Writer + Stdin string + StdinReader io.Reader + Env map[string]string + SensitiveValues []string + LogPath string + Stdout io.Writer + Stderr io.Writer + SuppressTranscript bool } type ExecResult struct { @@ -41,6 +98,132 @@ type ExecResult struct { Stderr string } +// Lifecycle is the provider contract used by new orchestration code. Address +// discovery is explicitly optional because delegated runtimes such as Docker +// Sandboxes intentionally expose command execution without a host-routable +// guest address. +type Lifecycle interface { + Create(ctx context.Context, request CreateRequest) (Instance, error) + Start(ctx context.Context, instance Instance, opts StartOptions) (*RunningProcess, error) + VerifyRuntime(ctx context.Context, instance Instance) (RuntimeInfo, error) + Address(ctx context.Context, instance Instance, waitSeconds int) (address string, available bool, err error) + Exec(ctx context.Context, instance Instance, command []string, opts ExecOptions) (ExecResult, error) + Diagnostics(ctx context.Context, instance Instance) (Diagnostics, error) + Stop(ctx context.Context, instance Instance) error + Delete(ctx context.Context, instance Instance) error + Inventory(ctx context.Context) ([]InventoryItem, error) +} + +// ArtifactManager is an optional provider capability for runtimes whose +// reusable artifact is not prepared by the shared OCI image pipeline. +type ArtifactManager interface { + EnsureArtifacts(ctx context.Context, dryRun bool) (handled bool, err error) +} + +// TemplateArtifact is the exact immutable reusable artifact selected by the +// shared image coordinator for a template-backed provider. +type TemplateArtifact struct { + Reference string `json:"reference"` + Digest string `json:"digest"` + CacheID string `json:"cacheId"` + Platform string `json:"platform"` + RootDisk string `json:"rootDisk,omitempty"` +} + +// TemplateArtifactRuntime exposes only provider-specific template-cache +// integration. Source resolution, builds, manifests, receipts, and retention +// remain owned by the shared image and storage packages. +type TemplateArtifactRuntime interface { + ImportTemplate(ctx context.Context, archivePath string) error + VerifyImportedTemplate(ctx context.Context, artifact TemplateArtifact) error + ActivateTemplate(artifact TemplateArtifact) error +} + +// TemplateArtifactCleaner is an optional exact cleanup capability for +// template-backed providers. The shared image/storage lifecycle calls it only +// for an immutable cache identity backed by EPAR ownership evidence. +type TemplateArtifactCleaner interface { + RemoveTemplate(ctx context.Context, artifact TemplateArtifact) error +} + +// TemplateArtifactObserver performs an exact, read-only cache lookup for +// catalog reconciliation without activating or deleting the template. +type TemplateArtifactObserver interface { + ObserveTemplate(ctx context.Context, artifact TemplateArtifact) (bool, error) +} + +// StorageContribution is required for every registered provider. It describes +// the provider's measurable capacity surface and operation expansion before +// the shared pool performs provider side effects. +type StorageContribution interface { + StorageSnapshot(ctx context.Context, request StorageRequest) (StorageSnapshot, error) +} + +type StorageRequest struct { + Operation string + Now time.Time + PeakBytes uint64 + MinimumFreeBytes uint64 +} + +type StorageSnapshot struct { + Surfaces []storage.Surface + Requirements []storage.Requirement + Artifacts []storage.Artifact +} + +// AdmissionVerifier rechecks provider-wide state that can change independently +// of one sandbox. Callers use it before registration and while issuing bounded +// job-admission leases so shared host configuration cannot silently weaken an +// already-created runtime. +type AdmissionVerifier interface { + VerifyAdmission(ctx context.Context) error +} + +// InstanceAdmissionVerifier rechecks mutable provider state attached to one +// exact runtime, including kits, injected authentication, secrets, published +// ports, and management gateways. It is deliberately separate from general +// runtime health because any violation must stop job admission immediately. +type InstanceAdmissionVerifier interface { + VerifyInstanceAdmission(ctx context.Context, instance Instance) error +} + +type NetworkPolicyDecision string + +const ( + NetworkPolicyAllow NetworkPolicyDecision = "allow" + NetworkPolicyDeny NetworkPolicyDecision = "deny" +) + +// NetworkPolicyRule is the attributed effective-policy record returned by a +// policy-capable provider. Read results include every relevant resource type; +// the current mutation methods remain deliberately limited to network rules. +type NetworkPolicyRule struct { + ID string + Name string + PolicyID string + Scope string + AppliesTo string + ResourceType string + Resources []string + Decision NetworkPolicyDecision + Origin string + Status string + Editable bool + Active bool +} + +// PolicyManager is implemented only by providers that can apply and verify +// exact, instance-scoped network rules. Global policy mutation is deliberately +// absent from this contract. +type PolicyManager interface { + ApplyNetworkPolicy(ctx context.Context, instance Instance, rules []NetworkPolicyRule) error + ReadNetworkPolicy(ctx context.Context, instance Instance) ([]NetworkPolicyRule, error) + RemoveNetworkPolicy(ctx context.Context, instance Instance, rules []NetworkPolicyRule) error +} + +// Provider is the legacy EPAR provider contract. New orchestration code should +// consume Lifecycle and wrap existing providers with AdaptLegacy. type Provider interface { Clone(ctx context.Context, source, name string) error Start(ctx context.Context, name string, opts StartOptions) (*RunningProcess, error) @@ -68,6 +251,28 @@ func CopyTextAtomic(ctx context.Context, p Provider, vmName, path, mode, content return err } +// CopyFile streams one regular host file into a guest without loading the +// complete payload into memory. The provider command installs through a +// temporary file and removes it on both success and failure. +func CopyFile(ctx context.Context, p Provider, vmName, source, destination, mode string) error { + file, err := os.Open(source) + if err != nil { + return err + } + defer file.Close() + info, err := file.Stat() + if err != nil { + return err + } + if !info.Mode().IsRegular() { + return fmt.Errorf("copy source %s must be a regular file", source) + } + tmp := "/tmp/epar-copy" + cmd := []string{"bash", "-lc", fmt.Sprintf("trap 'rm -f %s' EXIT; cat > %s && if command -v sudo >/dev/null 2>&1; then sudo install -m %s %s %s; else install -m %s %s %s; fi", shellQuote(tmp), shellQuote(tmp), shellQuote(mode), shellQuote(tmp), shellQuote(destination), shellQuote(mode), shellQuote(tmp), shellQuote(destination))} + _, err = p.Exec(ctx, vmName, cmd, ExecOptions{StdinReader: file}) + return err +} + func ShellCommand(script string) []string { return []string{"bash", "-lc", script} } diff --git a/internal/provider/registry/contributions_test.go b/internal/provider/registry/contributions_test.go new file mode 100644 index 0000000..d44af06 --- /dev/null +++ b/internal/provider/registry/contributions_test.go @@ -0,0 +1,51 @@ +package registry + +import ( + "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/provider" +) + +func TestEveryProviderRegistersRequiredContributions(t *testing.T) { + seen := make(map[string]struct{}) + for _, descriptor := range Descriptors() { + if descriptor.Type == "" { + t.Fatal("provider descriptor type is empty") + } + if _, exists := seen[descriptor.Type]; exists { + t.Fatalf("duplicate provider descriptor %q", descriptor.Type) + } + seen[descriptor.Type] = struct{}{} + if !descriptor.WizardSupported { + t.Errorf("%s has no ./start wizard contribution", descriptor.Type) + } + if descriptor.WizardNumber == "" || descriptor.WizardLabel == "" || len(descriptor.WizardAliases) == 0 { + t.Errorf("%s has an incomplete ./start wizard contribution", descriptor.Type) + } + if !descriptor.ConfigurationDecoder || !descriptor.ConfigurationDefaults || !descriptor.ConfigurationValidator { + t.Errorf("%s has an incomplete configuration contribution", descriptor.Type) + } + if !descriptor.LifecycleSupported { + t.Errorf("%s has no shared lifecycle contribution", descriptor.Type) + } + if !descriptor.StorageSupported { + t.Errorf("%s has no storage contribution", descriptor.Type) + } + switch descriptor.ImageMode { + case provider.ImageModeDocker, provider.ImageModeNative, provider.ImageModeTemplate: + default: + t.Errorf("%s has unsupported image mode %q", descriptor.Type, descriptor.ImageMode) + } + if descriptor.ImageMode == provider.ImageModeTemplate { + if !descriptor.GuidedArtifacts || len(descriptor.WizardImageProfiles) == 0 { + t.Errorf("%s has no guided template provisioning contribution", descriptor.Type) + } + if descriptor.WizardImageProfiles[0].Name != "full" || descriptor.WizardImageProfiles[0].Tag != "full-latest" { + t.Errorf("%s does not register its default image profile first", descriptor.Type) + } + } + } + if len(seen) != len(SupportedTypes()) { + t.Fatalf("descriptors=%d supported types=%d", len(seen), len(SupportedTypes())) + } +} diff --git a/internal/provider/registry/registry.go b/internal/provider/registry/registry.go new file mode 100644 index 0000000..3efa6b3 --- /dev/null +++ b/internal/provider/registry/registry.go @@ -0,0 +1,228 @@ +package registry + +import ( + "fmt" + "os" + "path/filepath" + "runtime" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockercontainer" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/tart" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/wsl" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +// Runtime is the complete provider wiring needed by the pool manager. +// Legacy is populated only for providers that still use the compatibility +// adapter. +type Runtime struct { + Legacy provider.Provider + Lifecycle provider.Lifecycle + PolicyManager provider.PolicyManager + Storage provider.StorageContribution +} + +type factory func(cfg config.Config, projectRoot string, dryRun bool) Runtime + +type entry struct { + descriptor provider.Descriptor + factory factory +} + +var entries = []entry{ + { + descriptor: provider.Descriptor{Type: "docker-container", DisplayName: "Docker Container", WizardSupported: true, WizardNumber: "1", WizardLabel: "Docker Container — private daemon", WizardAliases: []string{"docker", "docker-container"}, ConfigurationDecoder: true, ConfigurationDefaults: true, ConfigurationValidator: true, LifecycleSupported: true, StorageSupported: true, ImageMode: provider.ImageModeDocker, GuidedArtifacts: true, WizardImageProfiles: catthehackerProfiles()}, + factory: func(cfg config.Config, projectRoot string, dryRun bool) Runtime { + hostGateway := config.DockerConfigNeedsHostGateway(cfg.Docker) + environment := map[string]string{ + "HTTP_PROXY": cfg.Docker.HTTPProxy, + "HTTPS_PROXY": cfg.Docker.HTTPSProxy, + "NO_PROXY": cfg.Docker.NoProxy, + } + return adaptLegacy(dockercontainer.NewWithOptions("", cfg.Provider.Platform, hostGateway, environment, dryRun), providerStorage(cfg, projectRoot), dryRun) + }, + }, + { + descriptor: provider.Descriptor{ + Type: "docker-sandboxes", + DisplayName: "Docker Sandboxes", + WizardSupported: true, + WizardNumber: "2", + WizardLabel: "Docker Sandboxes — recommended when ready", + WizardAliases: []string{"docker-sandboxes", "sandboxes"}, + ConfigurationDecoder: true, + ConfigurationDefaults: true, + ConfigurationValidator: true, + LifecycleSupported: true, + StorageSupported: true, + ImageMode: provider.ImageModeTemplate, + GuidedArtifacts: true, + WizardImageProfiles: catthehackerProfiles(), + }, + factory: func(cfg config.Config, projectRoot string, dryRun bool) Runtime { + sandboxes := dockersandboxes.NewWithDryRun("", dryRun) + return Runtime{Lifecycle: sandboxes, PolicyManager: sandboxes, Storage: providerStorage(cfg, projectRoot)} + }, + }, + { + descriptor: provider.Descriptor{Type: "wsl", DisplayName: "WSL2", WizardSupported: true, WizardNumber: "3", WizardLabel: "WSL2", WizardAliases: []string{"wsl", "wsl2"}, ConfigurationDecoder: true, ConfigurationDefaults: true, ConfigurationValidator: true, LifecycleSupported: true, StorageSupported: true, ImageMode: provider.ImageModeDocker, GuidedArtifacts: true, WizardImageProfiles: catthehackerProfiles()}, + factory: func(cfg config.Config, projectRoot string, dryRun bool) Runtime { + installRoot := config.ProjectPath(projectRoot, cfg.Provider.InstallRoot) + return adaptLegacy(wsl.New("", installRoot, projectRoot, dryRun), providerStorage(cfg, projectRoot), dryRun) + }, + }, + { + descriptor: provider.Descriptor{Type: "tart", DisplayName: "Tart (experimental)", WizardSupported: true, WizardNumber: "4", WizardLabel: "Tart (experimental)", WizardAliases: []string{"tart"}, ConfigurationDecoder: true, ConfigurationDefaults: true, ConfigurationValidator: true, LifecycleSupported: true, StorageSupported: true, ImageMode: provider.ImageModeNative}, + factory: func(cfg config.Config, projectRoot string, dryRun bool) Runtime { + return adaptLegacy(tart.New("", dryRun), providerStorage(cfg, projectRoot), dryRun) + }, + }, +} + +func catthehackerProfiles() []provider.WizardImageProfile { + return []provider.WizardImageProfile{ + {Name: "full", Tag: "full-latest"}, + {Name: "act", Tag: "act-latest"}, + {Name: "dotnet", Tag: "dotnet-latest"}, + {Name: "js", Tag: "js-latest"}, + } +} + +func Descriptors() []provider.Descriptor { + result := make([]provider.Descriptor, 0, len(entries)) + for _, registered := range entries { + descriptor := registered.descriptor + descriptor.WizardAliases = append([]string(nil), descriptor.WizardAliases...) + descriptor.WizardImageProfiles = append([]provider.WizardImageProfile(nil), descriptor.WizardImageProfiles...) + result = append(result, descriptor) + } + return result +} + +func DescriptorFor(providerType string) (provider.Descriptor, bool) { + for _, registered := range entries { + if registered.descriptor.Type == providerType { + descriptor := registered.descriptor + descriptor.WizardAliases = append([]string(nil), descriptor.WizardAliases...) + descriptor.WizardImageProfiles = append([]provider.WizardImageProfile(nil), descriptor.WizardImageProfiles...) + return descriptor, true + } + } + return provider.Descriptor{}, false +} + +func SupportedTypes() []string { + result := make([]string, 0, len(entries)) + for _, registered := range entries { + result = append(result, registered.descriptor.Type) + } + return result +} + +// New is the single construction point for concrete providers. +func New(cfg config.Config, projectRoot string, dryRun bool) (Runtime, error) { + var registered *entry + for index := range entries { + if entries[index].descriptor.Type == cfg.Provider.Type { + registered = &entries[index] + break + } + } + if registered == nil { + return Runtime{}, provider.UnsupportedTypeError(cfg.Provider.Type) + } + descriptor := registered.descriptor + if !descriptor.WizardSupported || descriptor.WizardNumber == "" || descriptor.WizardLabel == "" || len(descriptor.WizardAliases) == 0 || !descriptor.ConfigurationDecoder || !descriptor.ConfigurationDefaults || !descriptor.ConfigurationValidator || !descriptor.LifecycleSupported || !descriptor.StorageSupported || descriptor.ImageMode == "" { + return Runtime{}, fmt.Errorf("provider %q has an incomplete registry entry", cfg.Provider.Type) + } + if (descriptor.ImageMode == provider.ImageModeTemplate || descriptor.Type == "docker-container" || descriptor.Type == "wsl") && (!descriptor.GuidedArtifacts || len(descriptor.WizardImageProfiles) == 0) { + return Runtime{}, fmt.Errorf("Docker-image-capable provider %q has no guided artifact onboarding contribution", cfg.Provider.Type) + } + runtime := registered.factory(cfg, projectRoot, dryRun) + if runtime.Lifecycle == nil || runtime.Storage == nil { + return Runtime{}, fmt.Errorf("provider %q registry entry did not construct required lifecycle and storage behavior", cfg.Provider.Type) + } + if descriptor.ImageMode == provider.ImageModeTemplate { + if _, ok := runtime.Lifecycle.(provider.TemplateArtifactRuntime); !ok { + return Runtime{}, fmt.Errorf("template-backed provider %q did not construct required artifact runtime behavior", cfg.Provider.Type) + } + } + return runtime, nil +} + +func providerStorage(cfg config.Config, projectRoot string) provider.StorageContribution { + roots := []provider.StorageRoot{{ID: cfg.Provider.Type + "-project", Location: projectRoot}} + minimumExpansions := map[string]uint64{} + switch cfg.Provider.Type { + case "docker-container": + roots = append(roots, provider.StorageRoot{ID: "docker-engine-backing", Kind: storage.SurfaceDockerEngine, Location: dockerBackingRoot()}) + case "docker-sandboxes": + roots = append(roots, + provider.StorageRoot{ + ID: "docker-engine-backing", + Kind: storage.SurfaceDockerEngine, + Location: dockerBackingRoot(), + MinimumExpansions: map[string]uint64{ + "image-pull": 0, + "image-build": 0, + "source-update": 0, + "template-build": 0, + }, + }, + provider.StorageRoot{ + ID: "docker-sandboxes-backing", + Kind: storage.SurfaceSandboxCache, + Location: dockerSandboxesBackingRoot(), + MinimumExpansions: map[string]uint64{"instance-create": 0, "template-build": 0}, + }, + provider.StorageRoot{ID: "docker-sandboxes-staging", Location: config.ProjectPath(projectRoot, cfg.DockerSandboxes.StagingRoot)}, + ) + case "wsl": + roots = append(roots, + provider.StorageRoot{ID: "wsl-install-root", Location: config.ProjectPath(projectRoot, cfg.Provider.InstallRoot)}, + provider.StorageRoot{ID: "docker-engine-backing", Kind: storage.SurfaceDockerEngine, Location: dockerBackingRoot()}, + ) + case "tart": + roots = append(roots, provider.StorageRoot{ID: "tart-vm-store", Location: tartBackingRoot()}) + } + return provider.NewMultiFilesystemStorageWithMinimumExpansions(cfg.Provider.Type, roots, minimumExpansions) +} + +func dockerBackingRoot() string { + switch runtime.GOOS { + case "windows": + return filepath.Join(os.Getenv("LOCALAPPDATA"), "Docker", "wsl", "disk") + case "darwin": + home, _ := os.UserHomeDir() + return filepath.Join(home, "Library", "Containers", "com.docker.docker", "Data", "vms", "0", "data") + default: + return "/var/lib/docker" + } +} + +func dockerSandboxesBackingRoot() string { + switch runtime.GOOS { + case "windows": + return filepath.Join(os.Getenv("LOCALAPPDATA"), "DockerSandboxes", "sandboxes", "data") + case "darwin": + home, _ := os.UserHomeDir() + return filepath.Join(home, "Library", "Containers", "com.docker.docker", "Data", "docker-sandboxes") + default: + return "/var/lib/docker-sandboxes" + } +} + +func tartBackingRoot() string { + if root := os.Getenv("TART_HOME"); root != "" { + return filepath.Join(root, "vms") + } + home, _ := os.UserHomeDir() + return filepath.Join(home, ".tart", "vms") +} + +func adaptLegacy(legacy provider.Provider, storageContribution provider.StorageContribution, dryRun bool) Runtime { + return Runtime{Legacy: legacy, Lifecycle: provider.AdaptLegacy(legacy, dryRun), Storage: storageContribution} +} diff --git a/internal/provider/registry/registry_test.go b/internal/provider/registry/registry_test.go new file mode 100644 index 0000000..4a78dc1 --- /dev/null +++ b/internal/provider/registry/registry_test.go @@ -0,0 +1,218 @@ +package registry + +import ( + "context" + "go/parser" + "go/token" + "path/filepath" + "runtime" + "strconv" + "strings" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/provider" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockercontainer" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/tart" + "github.com/solutionforest/ephemeral-action-runner/internal/provider/wsl" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +func TestNewAdaptsEstablishedLegacyProviders(t *testing.T) { + tests := []struct { + providerType string + matches func(any) bool + }{ + {providerType: "tart", matches: func(value any) bool { _, ok := value.(*tart.Provider); return ok }}, + {providerType: "wsl", matches: func(value any) bool { _, ok := value.(*wsl.Provider); return ok }}, + } + for _, test := range tests { + t.Run(test.providerType, func(t *testing.T) { + cfg := config.Default() + cfg.Provider.Type = test.providerType + + providerRuntime, err := New(cfg, t.TempDir(), true) + if err != nil { + t.Fatal(err) + } + if !test.matches(providerRuntime.Legacy) { + t.Fatalf("New() legacy type = %T", providerRuntime.Legacy) + } + if providerRuntime.Lifecycle == nil { + t.Fatal("New() did not adapt the legacy provider to Lifecycle") + } + if providerRuntime.Storage == nil { + t.Fatal("New() did not register provider storage behavior") + } + if providerRuntime.PolicyManager != nil { + t.Fatalf("New() policy manager = %T, want nil", providerRuntime.PolicyManager) + } + }) + } +} + +func TestNewWiresDockerContainerOptions(t *testing.T) { + cfg := config.Default() + cfg.Provider.Type = "docker-container" + cfg.Provider.Platform = "linux/amd64" + cfg.Docker.HTTPProxy = "http://host.docker.internal:3128" + cfg.Docker.HTTPSProxy = "http://host.docker.internal:3128" + cfg.Docker.NoProxy = "localhost,127.0.0.1" + + runtime, err := New(cfg, t.TempDir(), false) + if err != nil { + t.Fatal(err) + } + dockerContainer, ok := runtime.Legacy.(*dockercontainer.Provider) + if !ok { + t.Fatalf("New() legacy type = %T, want Docker Container provider", runtime.Legacy) + } + if runtime.Lifecycle == nil { + t.Fatal("New() did not adapt the legacy provider to Lifecycle") + } + if runtime.Storage == nil { + t.Fatal("New() storage contribution is nil") + } + if !dockerContainer.HostGateway { + t.Fatal("host.docker.internal proxy did not enable host gateway") + } + for key, want := range map[string]string{ + "HTTP_PROXY": cfg.Docker.HTTPProxy, + "HTTPS_PROXY": cfg.Docker.HTTPSProxy, + "NO_PROXY": cfg.Docker.NoProxy, + } { + if got := dockerContainer.Environment[key]; got != want { + t.Errorf("provider environment %s = %q, want %q", key, got, want) + } + } +} + +func TestNewWiresDockerSandboxesCapabilitiesWithoutLegacyAdapter(t *testing.T) { + cfg := config.Default() + cfg.Provider.Type = "docker-sandboxes" + + runtime, err := New(cfg, t.TempDir(), false) + if err != nil { + t.Fatal(err) + } + if runtime.Legacy != nil { + t.Fatalf("New() legacy = %T, want nil", runtime.Legacy) + } + if runtime.Lifecycle == nil { + t.Fatal("New() lifecycle is nil") + } + if runtime.PolicyManager == nil { + t.Fatal("New() policy manager is nil") + } + if runtime.Storage == nil { + t.Fatal("New() storage contribution is nil") + } +} + +func TestDockerSandboxesStorageRoutesOperationsToTheirBackingSurfaces(t *testing.T) { + cfg := config.Default() + cfg.Provider.Type = "docker-sandboxes" + cfg.DockerSandboxes.RootDisk = "30GiB" + cfg.DockerSandboxes.DockerDisk = "100GiB" + contribution := providerStorage(cfg, t.TempDir()) + + create, err := contribution.StorageSnapshot(context.Background(), provider.StorageRequest{ + Operation: "instance-create", + Now: time.Now(), + PeakBytes: 10 << 30, + MinimumFreeBytes: 50 << 30, + }) + if err != nil { + t.Fatal(err) + } + createRequirements := map[string]uint64{} + for _, requirement := range create.Requirements { + createRequirements[requirement.SurfaceID] = requirement.PeakBytes + } + if got, want := createRequirements["docker-sandboxes-backing"], uint64(10<<30); got != want { + t.Fatalf("sandbox backing create expansion = %d, want %d", got, want) + } + if _, found := createRequirements["docker-engine-backing"]; found { + t.Fatalf("sandbox create incorrectly reserved Docker Engine storage: %v", createRequirements) + } + createSurfaces := map[string]storage.SurfaceKind{} + for _, surface := range create.Surfaces { + createSurfaces[surface.ID] = surface.Kind + } + if got, want := createSurfaces["docker-engine-backing"], storage.SurfaceDockerEngine; got != want { + t.Fatalf("Docker Engine surface kind = %q, want %q", got, want) + } + if got, want := createSurfaces["docker-sandboxes-backing"], storage.SurfaceSandboxCache; got != want { + t.Fatalf("Docker Sandboxes surface kind = %q, want %q", got, want) + } + + pull, err := contribution.StorageSnapshot(context.Background(), provider.StorageRequest{ + Operation: "image-pull", + Now: time.Now(), + PeakBytes: 20 << 30, + MinimumFreeBytes: 50 << 30, + }) + if err != nil { + t.Fatal(err) + } + pullRequirements := map[string]uint64{} + for _, requirement := range pull.Requirements { + pullRequirements[requirement.SurfaceID] = requirement.PeakBytes + } + if _, found := pullRequirements["docker-sandboxes-backing"]; found { + t.Fatalf("Docker image pull incorrectly reserved sandbox instance storage: %v", pullRequirements) + } + if got, want := pullRequirements["docker-engine-backing"], uint64(20<<30); got != want { + t.Fatalf("Docker Engine pull expansion = %d, want %d", got, want) + } +} + +func TestNewRejectsUnsupportedProvider(t *testing.T) { + cfg := config.Default() + cfg.Provider.Type = "unknown-provider" + + _, err := New(cfg, t.TempDir(), false) + if err == nil || !strings.Contains(err.Error(), `unsupported provider.type "unknown-provider"`) { + t.Fatalf("New() error = %v", err) + } +} + +func TestPoolImportsOnlyNeutralProviderContracts(t *testing.T) { + _, thisFile, _, ok := runtime.Caller(0) + if !ok { + t.Fatal("locate registry test source") + } + repositoryRoot := filepath.Clean(filepath.Join(filepath.Dir(thisFile), "..", "..", "..")) + poolFiles, err := filepath.Glob(filepath.Join(repositoryRoot, "internal", "pool", "*.go")) + if err != nil { + t.Fatalf("list pool Go files: %v", err) + } + if len(poolFiles) == 0 { + t.Fatal("no pool Go files found") + } + + concreteProviders := []string{ + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockercontainer", + "github.com/solutionforest/ephemeral-action-runner/internal/provider/dockersandboxes", + "github.com/solutionforest/ephemeral-action-runner/internal/provider/tart", + "github.com/solutionforest/ephemeral-action-runner/internal/provider/wsl", + } + for _, path := range poolFiles { + parsed, parseErr := parser.ParseFile(token.NewFileSet(), path, nil, parser.ImportsOnly) + if parseErr != nil { + t.Fatalf("parse imports in %s: %v", path, parseErr) + } + for _, imported := range parsed.Imports { + importPath, unquoteErr := strconv.Unquote(imported.Path.Value) + if unquoteErr != nil { + t.Fatalf("unquote import %s in %s: %v", imported.Path.Value, path, unquoteErr) + } + for _, concrete := range concreteProviders { + if importPath == concrete || strings.HasPrefix(importPath, concrete+"/") { + t.Errorf("%s imports concrete provider %q; pool must use internal/provider contracts", path, importPath) + } + } + } + } +} diff --git a/internal/provider/storage.go b/internal/provider/storage.go new file mode 100644 index 0000000..0ad56eb --- /dev/null +++ b/internal/provider/storage.go @@ -0,0 +1,144 @@ +package provider + +import ( + "context" + "fmt" + "os" + "path/filepath" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +type filesystemStorage struct { + providerType string + roots []StorageRoot + minimumExpansions map[string]uint64 +} + +type StorageRoot struct { + ID string + Kind storage.SurfaceKind + Location string + MinimumExpansions map[string]uint64 +} + +// NewFilesystemStorage creates the conservative common contribution used by +// providers whose EPAR-owned staging/install root is on a host filesystem. +// Provider-specific external stores are added to storage status as report-only +// surfaces until they expose an authoritative capacity API. +func NewFilesystemStorage(providerType, root string) StorageContribution { + return NewMultiFilesystemStorage(providerType, []StorageRoot{{ID: providerType + "-host-filesystem", Kind: storage.SurfaceHostFilesystem, Location: root}}) +} + +func NewMultiFilesystemStorage(providerType string, roots []StorageRoot) StorageContribution { + return NewMultiFilesystemStorageWithMinimumExpansions(providerType, roots, nil) +} + +// NewMultiFilesystemStorageWithMinimumExpansions records provider-specific +// lower bounds without leaking provider configuration into the common pool. +func NewMultiFilesystemStorageWithMinimumExpansions(providerType string, roots []StorageRoot, minimumExpansions map[string]uint64) StorageContribution { + expansions := make(map[string]uint64, len(minimumExpansions)) + for operation, bytes := range minimumExpansions { + expansions[operation] = bytes + } + return &filesystemStorage{ + providerType: providerType, + roots: append([]StorageRoot(nil), roots...), + minimumExpansions: expansions, + } +} + +func (contribution *filesystemStorage) StorageSnapshot(_ context.Context, request StorageRequest) (StorageSnapshot, error) { + if contribution == nil || contribution.providerType == "" { + return StorageSnapshot{}, fmt.Errorf("provider storage contribution is incomplete") + } + if len(contribution.roots) == 0 { + return StorageSnapshot{}, fmt.Errorf("provider %s has no required storage roots", contribution.providerType) + } + peakBytes := request.PeakBytes + if minimum := contribution.minimumExpansions[request.Operation]; minimum > peakBytes { + peakBytes = minimum + } + snapshot := StorageSnapshot{} + seen := make(map[string]struct{}, len(contribution.roots)) + for _, specification := range contribution.roots { + if specification.ID == "" || specification.Location == "" { + return StorageSnapshot{}, fmt.Errorf("provider %s has an incomplete storage root", contribution.providerType) + } + if _, duplicate := seen[specification.ID]; duplicate { + return StorageSnapshot{}, fmt.Errorf("provider %s has duplicate storage surface %q", contribution.providerType, specification.ID) + } + seen[specification.ID] = struct{}{} + surfacePeakBytes := peakBytes + requiredForOperation := true + if specification.MinimumExpansions != nil { + minimum, found := specification.MinimumExpansions[request.Operation] + requiredForOperation = found + surfacePeakBytes = request.PeakBytes + if minimum > surfacePeakBytes { + surfacePeakBytes = minimum + } + } + root, err := nearestExistingDirectory(specification.Location) + if err != nil { + return StorageSnapshot{}, fmt.Errorf("resolve %s storage root %s: %w", contribution.providerType, specification.ID, err) + } + capacity, err := storage.ProbeFilesystemCapacity(root, request.Now) + if err != nil { + return StorageSnapshot{}, err + } + kind := specification.Kind + if kind == "" { + kind = storage.SurfaceHostFilesystem + } + snapshot.Surfaces = append(snapshot.Surfaces, storage.Surface{ + ID: specification.ID, + Provider: contribution.providerType, + Kind: kind, + Location: root, + Classification: "physical", + Confidence: "authoritative-filesystem-probe", + AdmissionAuthoritative: true, + Capacity: capacity, + }) + if !requiredForOperation { + continue + } + snapshot.Requirements = append(snapshot.Requirements, storage.Requirement{ + ID: request.Operation + "-" + specification.ID, + Provider: contribution.providerType, + SurfaceID: specification.ID, + PeakBytes: surfacePeakBytes, + MinimumFreeBytes: request.MinimumFreeBytes, + }) + } + return snapshot, nil +} + +func nearestExistingDirectory(path string) (string, error) { + if path == "" { + path = "." + } + absolute, err := filepath.Abs(path) + if err != nil { + return "", err + } + for { + info, statErr := os.Stat(absolute) + if statErr == nil { + if !info.IsDir() { + absolute = filepath.Dir(absolute) + continue + } + return absolute, nil + } + if !os.IsNotExist(statErr) { + return "", statErr + } + parent := filepath.Dir(absolute) + if parent == absolute { + return "", statErr + } + absolute = parent + } +} diff --git a/internal/provider/storage_test.go b/internal/provider/storage_test.go new file mode 100644 index 0000000..1ac70f3 --- /dev/null +++ b/internal/provider/storage_test.go @@ -0,0 +1,89 @@ +package provider + +import ( + "context" + "testing" + "time" +) + +func TestFilesystemStorageAppliesProviderOperationMinimumExpansion(t *testing.T) { + contribution := NewMultiFilesystemStorageWithMinimumExpansions( + "example", + []StorageRoot{{ID: "project", Location: t.TempDir()}}, + map[string]uint64{"instance-create": 42}, + ) + + snapshot, err := contribution.StorageSnapshot(context.Background(), StorageRequest{ + Operation: "instance-create", + Now: time.Now(), + PeakBytes: 10, + MinimumFreeBytes: 20, + }) + if err != nil { + t.Fatal(err) + } + if got, want := len(snapshot.Requirements), 1; got != want { + t.Fatalf("requirement count = %d, want %d", got, want) + } + if got, want := snapshot.Requirements[0].PeakBytes, uint64(42); got != want { + t.Fatalf("peak bytes = %d, want %d", got, want) + } +} + +func TestFilesystemStorageKeepsLargerCommonExpansion(t *testing.T) { + contribution := NewMultiFilesystemStorageWithMinimumExpansions( + "example", + []StorageRoot{{ID: "project", Location: t.TempDir()}}, + map[string]uint64{"instance-create": 42}, + ) + + snapshot, err := contribution.StorageSnapshot(context.Background(), StorageRequest{ + Operation: "instance-create", + Now: time.Now(), + PeakBytes: 84, + }) + if err != nil { + t.Fatal(err) + } + if got, want := snapshot.Requirements[0].PeakBytes, uint64(84); got != want { + t.Fatalf("peak bytes = %d, want %d", got, want) + } +} + +func TestFilesystemStorageRoutesOperationRequirementsToExactSurfaces(t *testing.T) { + root := t.TempDir() + contribution := NewMultiFilesystemStorage( + "example", + []StorageRoot{ + {ID: "project", Location: root}, + {ID: "engine", Location: root, MinimumExpansions: map[string]uint64{"image-pull": 20}}, + {ID: "instance-store", Location: root, MinimumExpansions: map[string]uint64{"instance-create": 42}}, + }, + ) + + snapshot, err := contribution.StorageSnapshot(context.Background(), StorageRequest{ + Operation: "instance-create", + Now: time.Now(), + PeakBytes: 10, + MinimumFreeBytes: 20, + }) + if err != nil { + t.Fatal(err) + } + if got, want := len(snapshot.Requirements), 2; got != want { + t.Fatalf("requirement count = %d, want %d", got, want) + } + if got, want := len(snapshot.Surfaces), 3; got != want { + t.Fatalf("surface count = %d, want %d", got, want) + } + got := map[string]uint64{} + for _, requirement := range snapshot.Requirements { + got[requirement.SurfaceID] = requirement.PeakBytes + } + if got["project"] != 10 || got["instance-store"] != 42 { + t.Fatalf("instance-create requirements = %v, want project=10 and instance-store=42", got) + } + if _, found := got["engine"]; found { + t.Fatalf("instance-create incorrectly required engine capacity: %v", got) + } +} diff --git a/internal/provider/tart/tart.go b/internal/provider/tart/tart.go index f4fe72f..bc3bb6d 100644 --- a/internal/provider/tart/tart.go +++ b/internal/provider/tart/tart.go @@ -3,9 +3,13 @@ package tart import ( "bytes" "context" + "encoding/json" "fmt" "io" + "os" "os/exec" + "path/filepath" + "runtime" "strings" "github.com/solutionforest/ephemeral-action-runner/internal/provider" @@ -15,6 +19,7 @@ type Provider struct { Binary string DryRun bool runCommand runCommandFunc + identities func() (map[string]string, error) } type runCommandFunc func(ctx context.Context, stdin io.Reader, stdout, stderr io.Writer, args ...string) (provider.ExecResult, error) @@ -64,12 +69,16 @@ func (p *Provider) Start(ctx context.Context, name string, opts provider.StartOp func (p *Provider) Exec(ctx context.Context, name string, command []string, opts provider.ExecOptions) (provider.ExecResult, error) { full := []string{"exec"} - if opts.Stdin != "" { + if opts.Stdin != "" || opts.StdinReader != nil { full = append(full, "-i") } full = append(full, name) full = append(full, provider.EnvCommand(opts.Env, command)...) - return p.runWithSensitiveLog(ctx, strings.NewReader(opts.Stdin), opts.Stdout, opts.Stderr, opts.SensitiveValues, full...) + stdin := opts.StdinReader + if stdin == nil && opts.Stdin != "" { + stdin = strings.NewReader(opts.Stdin) + } + return p.runWithSensitiveLog(ctx, stdin, opts.Stdout, opts.Stderr, opts.SensitiveValues, full...) } func (p *Provider) IP(ctx context.Context, name string, waitSeconds int) (string, error) { @@ -110,9 +119,61 @@ func (p *Provider) List(ctx context.Context) ([]provider.Instance, error) { State: fields[len(fields)-1], }) } + if p.identities == nil && (runtime.GOOS != "darwin" || p.runCommand != nil) { + return out, nil + } + resolve := p.identities + if resolve == nil { + resolve = readLocalVMIdentities + } + identities, err := resolve() + if err != nil { + return nil, fmt.Errorf("read immutable Tart VM identities: %w", err) + } + for index := range out { + if identity := identities[out[index].Name]; identity != "" { + out[index].ProviderID = "tart-mac:" + strings.ToLower(identity) + } + } return out, nil } +func readLocalVMIdentities() (map[string]string, error) { + home := strings.TrimSpace(os.Getenv("TART_HOME")) + if home == "" { + userHome, err := os.UserHomeDir() + if err != nil { + return nil, err + } + home = filepath.Join(userHome, ".tart") + } + entries, err := os.ReadDir(filepath.Join(home, "vms")) + if os.IsNotExist(err) { + return map[string]string{}, nil + } + if err != nil { + return nil, err + } + result := make(map[string]string) + for _, entry := range entries { + if !entry.IsDir() || strings.ContainsAny(entry.Name(), `/\`) { + continue + } + data, readErr := os.ReadFile(filepath.Join(home, "vms", entry.Name(), "config.json")) + if readErr != nil { + continue + } + var config struct { + MACAddress string `json:"macAddress"` + } + if json.Unmarshal(data, &config) != nil || strings.TrimSpace(config.MACAddress) == "" { + continue + } + result[entry.Name()] = strings.TrimSpace(config.MACAddress) + } + return result, nil +} + func (p *Provider) run(ctx context.Context, stdin io.Reader, args ...string) (provider.ExecResult, error) { return p.runWithLog(ctx, stdin, nil, nil, args...) } @@ -136,6 +197,9 @@ func (p *Provider) runWithLogRaw(ctx context.Context, stdin io.Reader, stdoutSin return provider.ExecResult{}, nil } cmd := exec.CommandContext(ctx, p.Binary, args...) + if len(args) > 0 && args[0] == "clone" { + cmd.Env = tartCloneEnvironment(os.Environ()) + } if stdin != nil { cmd.Stdin = stdin } @@ -150,6 +214,16 @@ func (p *Provider) runWithLogRaw(ctx context.Context, stdin io.Reader, stdoutSin return result, nil } +func tartCloneEnvironment(base []string) []string { + result := make([]string, 0, len(base)+1) + for _, value := range base { + if !strings.HasPrefix(value, "TART_NO_AUTO_PRUNE=") { + result = append(result, value) + } + } + return append(result, "TART_NO_AUTO_PRUNE=1") +} + func captureWriter(capture io.Writer, sink io.Writer) io.Writer { if sink == nil { return capture diff --git a/internal/provider/tart/tart_test.go b/internal/provider/tart/tart_test.go index 8dcc363..1813eef 100644 --- a/internal/provider/tart/tart_test.go +++ b/internal/provider/tart/tart_test.go @@ -6,6 +6,7 @@ import ( "errors" "io" "os" + "path/filepath" "strings" "testing" @@ -103,3 +104,38 @@ func captureStdout(t *testing.T, fn func()) string { } return buf.String() } + +func TestReadLocalVMIdentitiesUsesStableMACAddress(t *testing.T) { + home := t.TempDir() + t.Setenv("TART_HOME", home) + vmDirectory := filepath.Join(home, "vms", "epar-tart-runner") + if err := os.MkdirAll(vmDirectory, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(vmDirectory, "config.json"), []byte(`{"macAddress":"02:00:00:12:34:56"}`), 0o600); err != nil { + t.Fatal(err) + } + identities, err := readLocalVMIdentities() + if err != nil { + t.Fatal(err) + } + if got := identities["epar-tart-runner"]; got != "02:00:00:12:34:56" { + t.Fatalf("identity = %q", got) + } +} + +func TestCloneEnvironmentDisablesTartAutomaticPruning(t *testing.T) { + environment := tartCloneEnvironment([]string{"PATH=/bin", "TART_NO_AUTO_PRUNE="}) + if got := environment[len(environment)-1]; got != "TART_NO_AUTO_PRUNE=1" { + t.Fatalf("clone environment final override = %q", got) + } + count := 0 + for _, value := range environment { + if strings.HasPrefix(value, "TART_NO_AUTO_PRUNE=") { + count++ + } + } + if count != 1 { + t.Fatalf("clone environment contains %d TART_NO_AUTO_PRUNE entries: %#v", count, environment) + } +} diff --git a/internal/provider/wsl/process_other.go b/internal/provider/wsl/process_other.go new file mode 100644 index 0000000..c73d1d6 --- /dev/null +++ b/internal/provider/wsl/process_other.go @@ -0,0 +1,7 @@ +//go:build !windows + +package wsl + +import "os/exec" + +func isolateKeepaliveProcess(*exec.Cmd) {} diff --git a/internal/provider/wsl/process_windows.go b/internal/provider/wsl/process_windows.go new file mode 100644 index 0000000..8b6c360 --- /dev/null +++ b/internal/provider/wsl/process_windows.go @@ -0,0 +1,15 @@ +//go:build windows + +package wsl + +import ( + "os/exec" + "syscall" +) + +func isolateKeepaliveProcess(command *exec.Cmd) { + command.SysProcAttr = &syscall.SysProcAttr{ + HideWindow: true, + NoInheritHandles: true, + } +} diff --git a/internal/provider/wsl/process_windows_test.go b/internal/provider/wsl/process_windows_test.go new file mode 100644 index 0000000..a6363a8 --- /dev/null +++ b/internal/provider/wsl/process_windows_test.go @@ -0,0 +1,22 @@ +//go:build windows + +package wsl + +import ( + "os/exec" + "testing" +) + +func TestIsolateKeepaliveProcessPreventsControllerLockInheritance(t *testing.T) { + command := exec.Command("wsl.exe", "--help") + isolateKeepaliveProcess(command) + if command.SysProcAttr == nil { + t.Fatal("SysProcAttr is nil") + } + if !command.SysProcAttr.HideWindow { + t.Fatal("HideWindow is false") + } + if !command.SysProcAttr.NoInheritHandles { + t.Fatal("NoInheritHandles is false") + } +} diff --git a/internal/provider/wsl/wsl.go b/internal/provider/wsl/wsl.go index f4cddd9..da7d16a 100644 --- a/internal/provider/wsl/wsl.go +++ b/internal/provider/wsl/wsl.go @@ -8,6 +8,7 @@ import ( "os" "os/exec" "path/filepath" + "runtime" "strings" "time" "unicode/utf16" @@ -21,6 +22,12 @@ type Provider struct { ProjectRoot string DryRun bool runCommand runCommandFunc + identities func(context.Context) (map[string]wslIdentity, error) +} + +type wslIdentity struct { + ID string + BasePath string } type runCommandFunc func(ctx context.Context, stdin io.Reader, logPath string, stdout, stderr io.Writer, args ...string) (provider.ExecResult, error) @@ -94,7 +101,9 @@ func (p *Provider) Start(ctx context.Context, name string, opts provider.StartOp func (p *Provider) Exec(ctx context.Context, name string, command []string, opts provider.ExecOptions) (provider.ExecResult, error) { var stdin io.Reader - if opts.Stdin != "" { + if opts.StdinReader != nil { + stdin = opts.StdinReader + } else if opts.Stdin != "" { stdin = strings.NewReader(opts.Stdin) } return p.runWithSensitiveLog(ctx, stdin, opts.LogPath, opts.Stdout, opts.Stderr, opts.SensitiveValues, p.execArgs(name, command, opts.Env)...) @@ -185,7 +194,115 @@ func (p *Provider) List(ctx context.Context) ([]provider.Instance, error) { } return nil, err } - return parseList(result.Stdout), nil + instances := parseList(result.Stdout) + if p.identities == nil && (runtime.GOOS != "windows" || p.runCommand != nil) { + return instances, nil + } + resolve := p.identities + if resolve == nil { + resolve = p.readRegistryIdentities + } + identities, err := resolve(ctx) + if err != nil { + return nil, fmt.Errorf("read immutable WSL distribution identities: %w", err) + } + for index := range instances { + identity, found := identities[strings.ToLower(instances[index].Name)] + if !found || identity.ID == "" { + continue + } + expected, pathErr := p.instanceDir(instances[index].Name) + if pathErr != nil || !sameWindowsPath(identity.BasePath, expected) { + continue + } + instances[index].ProviderID = "wsl:" + strings.ToLower(identity.ID) + } + return instances, nil +} + +func (p *Provider) readRegistryIdentities(ctx context.Context) (map[string]wslIdentity, error) { + command := exec.CommandContext(ctx, "reg.exe", "query", `HKCU\Software\Microsoft\Windows\CurrentVersion\Lxss`, "/s") + output, err := command.Output() + if err != nil { + return nil, err + } + return parseRegistryIdentities(cleanWSLOutput(output)) +} + +func parseRegistryIdentities(output string) (map[string]wslIdentity, error) { + result := make(map[string]wslIdentity) + var currentID, distributionName, basePath string + flush := func() { + if currentID != "" && distributionName != "" && basePath != "" { + result[strings.ToLower(distributionName)] = wslIdentity{ID: currentID, BasePath: expandWindowsEnvironment(basePath)} + } + currentID, distributionName, basePath = "", "", "" + } + for _, raw := range strings.Split(strings.ReplaceAll(output, "\r\n", "\n"), "\n") { + line := strings.TrimSpace(raw) + if strings.HasPrefix(strings.ToUpper(line), `HKEY_CURRENT_USER\`) { + flush() + if start := strings.LastIndex(line, `\{`); start >= 0 && strings.HasSuffix(line, "}") { + currentID = strings.Trim(line[start+1:], "{}") + } + continue + } + fields := strings.Fields(line) + if len(fields) < 3 { + continue + } + value := strings.Join(fields[2:], " ") + switch strings.ToLower(fields[0]) { + case "distributionname": + distributionName = value + case "basepath": + basePath = value + } + } + flush() + if len(result) == 0 { + return nil, fmt.Errorf("WSL registry inventory contained no complete distribution identities") + } + return result, nil +} + +func sameWindowsPath(left, right string) bool { + normalize := func(value string) string { + value = strings.TrimPrefix(strings.TrimSpace(value), `\\?\`) + value = filepath.Clean(expandWindowsEnvironment(value)) + return strings.TrimRight(strings.ToLower(value), `\/`) + } + return normalize(left) == normalize(right) +} + +func expandWindowsEnvironment(value string) string { + value = os.ExpandEnv(value) + for { + start := strings.IndexByte(value, '%') + if start < 0 { + return value + } + end := strings.IndexByte(value[start+1:], '%') + if end < 0 { + return value + } + end += start + 1 + key := value[start+1 : end] + replacement := os.Getenv(key) + if replacement == "" { + for _, entry := range os.Environ() { + parts := strings.SplitN(entry, "=", 2) + if len(parts) == 2 && strings.EqualFold(parts[0], key) { + replacement = parts[1] + break + } + } + } + if replacement == "" { + return value + } + value = value[:start] + replacement + value[end+1:] + } } func (p *Provider) Export(ctx context.Context, name, outputPath string) error { @@ -231,6 +348,7 @@ func (p *Provider) startKeepAlive(name string, stdoutSink, stderrSink io.Writer) return nil, nil } cmd := exec.Command(p.Binary, args...) + isolateKeepaliveProcess(cmd) cmd.Stdout = writerOrDiscard(stdoutSink) cmd.Stderr = writerOrDiscard(stderrSink) if err := cmd.Start(); err != nil { diff --git a/internal/provider/wsl/wsl_test.go b/internal/provider/wsl/wsl_test.go index 59dc665..5240996 100644 --- a/internal/provider/wsl/wsl_test.go +++ b/internal/provider/wsl/wsl_test.go @@ -125,6 +125,13 @@ func TestParseListParsesVerboseOutput(t *testing.T) { func TestNoInstalledDistrosReturnsEmptyList(t *testing.T) { p := New("wsl.exe", t.TempDir(), t.TempDir(), true) + p.runCommand = func(_ context.Context, _ io.Reader, _ string, _, _ io.Writer, args ...string) (provider.ExecResult, error) { + if !reflect.DeepEqual(args, []string{"--list", "--verbose"}) { + t.Fatalf("args = %#v", args) + } + message := "Windows Subsystem for Linux has no installed distributions." + return provider.ExecResult{Stderr: message}, errors.New(message) + } out, err := p.List(context.Background()) if err != nil { t.Fatalf("dry-run list failed: %v", err) @@ -212,3 +219,21 @@ func TestDeleteReturnsUnregisterFailureWhenDistroStillExists(t *testing.T) { t.Fatal("Delete() error = nil, want unregister error") } } + +func TestParseRegistryIdentities(t *testing.T) { + identities, err := parseRegistryIdentities(` +HKEY_CURRENT_USER\Software\Microsoft\Windows\CurrentVersion\Lxss\{2A6C842D-6F31-45D8-86A2-66A35D210B42} + DistributionName REG_SZ epar-wsl-runner + BasePath REG_SZ C:\repos\ephemeral-action-runner\work\wsl\epar-wsl-runner +`) + if err != nil { + t.Fatal(err) + } + identity := identities["epar-wsl-runner"] + if identity.ID != "2A6C842D-6F31-45D8-86A2-66A35D210B42" { + t.Fatalf("identity = %#v", identity) + } + if !sameWindowsPath(identity.BasePath, `c:\repos\ephemeral-action-runner\work\wsl\epar-wsl-runner`) { + t.Fatalf("base path did not compare case-insensitively: %#v", identity) + } +} diff --git a/internal/storage/capacity.go b/internal/storage/capacity.go new file mode 100644 index 0000000..d031768 --- /dev/null +++ b/internal/storage/capacity.go @@ -0,0 +1,88 @@ +package storage + +import ( + "errors" + "fmt" + "math" + "sort" + "strings" +) + +// EvaluateCapacity evaluates one requirement against one surface observation. +func EvaluateCapacity(surface Surface, requirement Requirement) (CapacityCheck, error) { + if strings.TrimSpace(surface.ID) == "" { + return CapacityCheck{}, errors.New("storage surface ID is required") + } + if strings.TrimSpace(requirement.ID) == "" { + return CapacityCheck{}, errors.New("storage requirement ID is required") + } + if requirement.SurfaceID != surface.ID { + return CapacityCheck{}, fmt.Errorf("storage requirement %q targets surface %q, not %q", requirement.ID, requirement.SurfaceID, surface.ID) + } + if requirement.MinimumFreeBytes == 0 { + requirement.MinimumFreeBytes = DefaultMinimumFreeBytes + } + if requirement.PeakBytes > math.MaxUint64-requirement.MinimumFreeBytes { + return CapacityCheck{}, fmt.Errorf("storage requirement %q overflows required available bytes", requirement.ID) + } + required := requirement.PeakBytes + requirement.MinimumFreeBytes + check := CapacityCheck{ + Requirement: requirement, + Capacity: surface.Capacity, + RequiredAvailableBytes: required, + } + if !surface.Capacity.Known { + check.Status = CapacityUnknown + check.Reason = "capacity observation is unavailable" + return check, nil + } + if surface.Capacity.TotalBytes > 0 && surface.Capacity.AvailableBytes > surface.Capacity.TotalBytes { + return CapacityCheck{}, fmt.Errorf("storage surface %q reports available bytes greater than total bytes", surface.ID) + } + if surface.Capacity.AvailableBytes < required { + check.Status = CapacityInsufficient + check.DeficitBytes = required - surface.Capacity.AvailableBytes + check.Reason = "available capacity is below peak bytes plus minimum free bytes" + return check, nil + } + check.Status = CapacityReady + check.Reason = "available capacity satisfies peak bytes plus minimum free bytes" + return check, nil +} + +func capacityPreflight(surfaces []Surface, requirements []Requirement) ([]CapacityCheck, error) { + byID := make(map[string]Surface, len(surfaces)) + for _, surface := range surfaces { + if strings.TrimSpace(surface.ID) == "" { + return nil, errors.New("storage surface ID is required") + } + switch surface.Kind { + case SurfaceHostFilesystem, SurfaceDockerEngine, SurfaceSandboxCache, SurfaceExternal: + default: + return nil, fmt.Errorf("storage surface %q has invalid kind %q", surface.ID, surface.Kind) + } + if _, exists := byID[surface.ID]; exists { + return nil, fmt.Errorf("duplicate storage surface ID %q", surface.ID) + } + byID[surface.ID] = surface + } + seenRequirements := make(map[string]struct{}, len(requirements)) + checks := make([]CapacityCheck, 0, len(requirements)) + for _, requirement := range requirements { + if _, exists := seenRequirements[requirement.ID]; exists { + return nil, fmt.Errorf("duplicate storage requirement ID %q", requirement.ID) + } + seenRequirements[requirement.ID] = struct{}{} + surface, exists := byID[requirement.SurfaceID] + if !exists { + return nil, fmt.Errorf("storage requirement %q references unknown surface %q", requirement.ID, requirement.SurfaceID) + } + check, err := EvaluateCapacity(surface, requirement) + if err != nil { + return nil, err + } + checks = append(checks, check) + } + sort.Slice(checks, func(i, j int) bool { return checks[i].Requirement.ID < checks[j].Requirement.ID }) + return checks, nil +} diff --git a/internal/storage/capacity_test.go b/internal/storage/capacity_test.go new file mode 100644 index 0000000..d3472cc --- /dev/null +++ b/internal/storage/capacity_test.go @@ -0,0 +1,68 @@ +package storage + +import ( + "math" + "testing" + "time" +) + +func TestEvaluateCapacity(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + tests := []struct { + name string + capacity Capacity + requirement Requirement + wantStatus CapacityStatus + wantRequired uint64 + wantDeficit uint64 + }{ + { + name: "unknown fails closed", + capacity: Capacity{}, + requirement: Requirement{ID: "full-build", SurfaceID: "host", PeakBytes: 30 * GiB}, + wantStatus: CapacityUnknown, + wantRequired: 31 * GiB, + }, + { + name: "insufficient includes deficit", + capacity: Capacity{Known: true, AvailableBytes: 30 * GiB, TotalBytes: 100 * GiB, ObservedAt: now}, + requirement: Requirement{ID: "full-build", SurfaceID: "host", PeakBytes: 30 * GiB}, + wantStatus: CapacityInsufficient, + wantRequired: 31 * GiB, + wantDeficit: GiB, + }, + { + name: "ready", + capacity: Capacity{Known: true, AvailableBytes: 31 * GiB, TotalBytes: 100 * GiB, ObservedAt: now}, + requirement: Requirement{ID: "full-build", SurfaceID: "host", PeakBytes: 30 * GiB}, + wantStatus: CapacityReady, + wantRequired: 31 * GiB, + }, + } + for _, test := range tests { + test := test + t.Run(test.name, func(t *testing.T) { + t.Parallel() + check, err := EvaluateCapacity(Surface{ID: "host", Kind: SurfaceHostFilesystem, Capacity: test.capacity}, test.requirement) + if err != nil { + t.Fatalf("EvaluateCapacity() error = %v", err) + } + if check.Status != test.wantStatus || check.RequiredAvailableBytes != test.wantRequired || check.DeficitBytes != test.wantDeficit { + t.Fatalf("EvaluateCapacity() = %+v, want status=%s required=%d deficit=%d", check, test.wantStatus, test.wantRequired, test.wantDeficit) + } + }) + } +} + +func TestEvaluateCapacityRejectsOverflowAndInvalidObservation(t *testing.T) { + t.Parallel() + surface := Surface{ID: "host", Capacity: Capacity{Known: true, AvailableBytes: 10, TotalBytes: 5}} + if _, err := EvaluateCapacity(surface, Requirement{ID: "build", SurfaceID: "host", MinimumFreeBytes: 1}); err == nil { + t.Fatal("EvaluateCapacity() accepted available bytes greater than total bytes") + } + surface.Capacity = Capacity{Known: true, AvailableBytes: math.MaxUint64, TotalBytes: math.MaxUint64} + if _, err := EvaluateCapacity(surface, Requirement{ID: "build", SurfaceID: "host", PeakBytes: math.MaxUint64, MinimumFreeBytes: 1}); err == nil { + t.Fatal("EvaluateCapacity() accepted overflowing requirement") + } +} diff --git a/internal/storage/catalog/catalog.go b/internal/storage/catalog/catalog.go new file mode 100644 index 0000000..3595827 --- /dev/null +++ b/internal/storage/catalog/catalog.go @@ -0,0 +1,531 @@ +// Package catalog persists exact, per-user EPAR resource custody across +// projects and configuration files. +package catalog + +import ( + "context" + "crypto/rand" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "runtime" + "sort" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/filelock" +) + +const SchemaVersion = 1 + +type Custody string + +const ( + CustodyGenerated Custody = "generated" + CustodyAcquired Custody = "acquired" +) + +type State string + +const ( + StateCurrent State = "current" + StateStaging State = "staging" + StateSuperseded State = "superseded" + StateCleanupPending State = "cleanup-pending" +) + +type Reference struct { + ConfigID string `json:"configId"` + ManifestHash string `json:"manifestHash,omitempty"` + Role string `json:"role,omitempty"` + UpdatedAt time.Time `json:"updatedAt"` +} + +type Resource struct { + Key string `json:"key"` + BackendID string `json:"backendId"` + InstallationIDs []string `json:"installationIds,omitempty"` + Kind string `json:"kind"` + Provider string `json:"provider,omitempty"` + Role string `json:"role,omitempty"` + Locator string `json:"locator"` + Identity string `json:"identity"` + Fingerprint string `json:"fingerprint,omitempty"` + Custody Custody `json:"custody"` + ManifestHash string `json:"manifestHash,omitempty"` + IntroducedTags []string `json:"introducedTags,omitempty"` + State State `json:"state"` + References []Reference `json:"references,omitempty"` + CreatedAt time.Time `json:"createdAt"` + LastSeenAt time.Time `json:"lastSeenAt"` + SupersededAt *time.Time `json:"supersededAt,omitempty"` + LeaseExpiresAt *time.Time `json:"leaseExpiresAt,omitempty"` + CleanupError string `json:"cleanupError,omitempty"` +} + +type Config struct { + ID string `json:"id"` + InstallationID string `json:"installationId"` + Path string `json:"path"` + DisplayPath string `json:"displayPath,omitempty"` + ProjectRoot string `json:"projectRoot"` + BuildCacheLimitBytes uint64 `json:"buildCacheLimitBytes,omitempty"` + ControllerLeaseUntil *time.Time `json:"controllerLeaseUntil,omitempty"` + LastSeenAt time.Time `json:"lastSeenAt"` +} + +type Journal struct { + ID string `json:"id"` + Operation string `json:"operation"` + ResourceKey string `json:"resourceKey,omitempty"` + BackendID string `json:"backendId,omitempty"` + ConfigID string `json:"configId,omitempty"` + Role string `json:"role,omitempty"` + Locator string `json:"locator,omitempty"` + PreviousIdentity string `json:"previousIdentity,omitempty"` + Phase string `json:"phase"` + StartedAt time.Time `json:"startedAt"` + UpdatedAt time.Time `json:"updatedAt"` + Error string `json:"error,omitempty"` +} + +type Catalog struct { + SchemaVersion int `json:"schemaVersion"` + InstallationID string `json:"installationId"` + UpdatedAt time.Time `json:"updatedAt"` + Configs []Config `json:"configs,omitempty"` + Resources []Resource `json:"resources,omitempty"` + Journals []Journal `json:"journals,omitempty"` +} + +type Store struct { + root string +} + +func DefaultRoot() (string, error) { + if override := strings.TrimSpace(os.Getenv("EPAR_STATE_HOME")); override != "" { + return filepath.Abs(override) + } + if runtime.GOOS == "windows" { + base := strings.TrimSpace(os.Getenv("LOCALAPPDATA")) + if base == "" { + return "", errors.New("LOCALAPPDATA is required for the EPAR host resource catalog") + } + return filepath.Join(base, "ephemeral-action-runner", "state"), nil + } + if runtime.GOOS == "linux" { + if base := strings.TrimSpace(os.Getenv("XDG_STATE_HOME")); base != "" { + return filepath.Join(base, "ephemeral-action-runner"), nil + } + home, err := os.UserHomeDir() + if err != nil { + return "", fmt.Errorf("resolve home directory for EPAR host resource catalog: %w", err) + } + return filepath.Join(home, ".local", "state", "ephemeral-action-runner"), nil + } + base, err := os.UserConfigDir() + if err != nil { + return "", fmt.Errorf("resolve user state directory for EPAR host resource catalog: %w", err) + } + return filepath.Join(base, "ephemeral-action-runner", "state"), nil +} + +func Open(root string) (*Store, error) { + if strings.TrimSpace(root) == "" { + var err error + root, err = DefaultRoot() + if err != nil { + return nil, err + } + } + absolute, err := filepath.Abs(root) + if err != nil { + return nil, fmt.Errorf("resolve EPAR host resource catalog root: %w", err) + } + if err := os.MkdirAll(absolute, 0o700); err != nil { + return nil, fmt.Errorf("create EPAR host resource catalog root: %w", err) + } + info, err := os.Lstat(absolute) + if err != nil { + return nil, err + } + if !info.IsDir() || info.Mode()&os.ModeSymlink != 0 { + return nil, fmt.Errorf("EPAR host resource catalog root must be a real directory: %s", absolute) + } + return &Store{root: absolute}, nil +} + +func (s *Store) Root() string { return s.root } + +func (s *Store) Path() string { return filepath.Join(s.root, "resources-v1.json") } + +func (s *Store) AcquireBackendLock(ctx context.Context, backendID string) (*filelock.Lock, error) { + if strings.TrimSpace(backendID) == "" { + return nil, errors.New("backend identity is required") + } + sum := sha256.Sum256([]byte(strings.TrimSpace(backendID))) + path := filepath.Join(s.root, "backend-"+hex.EncodeToString(sum[:12])+".lock") + ticker := time.NewTicker(250 * time.Millisecond) + defer ticker.Stop() + for { + lock, err := filelock.Acquire(path) + if err == nil { + return lock, nil + } + if !errors.Is(err, filelock.ErrLocked) { + return nil, fmt.Errorf("acquire EPAR backend lock for %s: %w", backendID, err) + } + select { + case <-ctx.Done(): + return nil, fmt.Errorf("wait for EPAR backend lock for %s: %w", backendID, ctx.Err()) + case <-ticker.C: + } + } +} + +func (s *Store) WithLock(now time.Time, update func(*Catalog) error) (Catalog, error) { + if update == nil { + return Catalog{}, errors.New("catalog update function is required") + } + lock, err := filelock.Acquire(filepath.Join(s.root, "resources-v1.lock")) + if err != nil { + return Catalog{}, fmt.Errorf("acquire EPAR host resource catalog lock: %w", err) + } + defer lock.Close() + value, err := s.loadUnlocked(now) + if err != nil { + return Catalog{}, err + } + if err := update(&value); err != nil { + return Catalog{}, err + } + normalize(&value) + value.UpdatedAt = now.UTC() + if err := s.writeUnlocked(value); err != nil { + return Catalog{}, err + } + return value, nil +} + +func (s *Store) Load(now time.Time) (Catalog, error) { + lock, err := filelock.Acquire(filepath.Join(s.root, "resources-v1.lock")) + if err != nil { + return Catalog{}, fmt.Errorf("acquire EPAR host resource catalog lock: %w", err) + } + defer lock.Close() + return s.loadUnlocked(now) +} + +func (s *Store) loadUnlocked(now time.Time) (Catalog, error) { + content, err := os.ReadFile(s.Path()) + if errors.Is(err, os.ErrNotExist) { + installationID, idErr := randomID() + if idErr != nil { + return Catalog{}, idErr + } + return Catalog{SchemaVersion: SchemaVersion, InstallationID: installationID, UpdatedAt: now.UTC()}, nil + } + if err != nil { + return Catalog{}, fmt.Errorf("read EPAR host resource catalog: %w", err) + } + var value Catalog + if err := json.Unmarshal(content, &value); err != nil { + return Catalog{}, fmt.Errorf("decode EPAR host resource catalog: %w", err) + } + if value.SchemaVersion != SchemaVersion || strings.TrimSpace(value.InstallationID) == "" { + return Catalog{}, fmt.Errorf("unsupported or incomplete EPAR host resource catalog schema %d", value.SchemaVersion) + } + normalize(&value) + return value, nil +} + +func (s *Store) writeUnlocked(value Catalog) error { + content, err := json.MarshalIndent(value, "", " ") + if err != nil { + return fmt.Errorf("encode EPAR host resource catalog: %w", err) + } + temp, err := os.CreateTemp(s.root, ".resources-v1-*.tmp") + if err != nil { + return err + } + tempPath := temp.Name() + defer os.Remove(tempPath) + if err := temp.Chmod(0o600); err != nil { + temp.Close() + return err + } + if _, err := temp.Write(append(content, '\n')); err != nil { + temp.Close() + return err + } + if err := temp.Sync(); err != nil { + temp.Close() + return err + } + if err := temp.Close(); err != nil { + return err + } + if err := os.Rename(tempPath, s.Path()); err != nil { + return fmt.Errorf("publish EPAR host resource catalog: %w", err) + } + return nil +} + +func ConfigID(projectRoot, configPath string) (string, error) { + root, err := CanonicalPath(projectRoot) + if err != nil { + return "", err + } + path, err := CanonicalPath(configPath) + if err != nil { + return "", err + } + sum := sha256.Sum256([]byte(root + "\x00" + path)) + return hex.EncodeToString(sum[:12]), nil +} + +func ResourceKey(backendID, kind, identity string) string { + sum := sha256.Sum256([]byte(strings.TrimSpace(backendID) + "\x00" + strings.TrimSpace(kind) + "\x00" + strings.TrimSpace(identity))) + return hex.EncodeToString(sum[:16]) +} + +func RegisterConfig(value *Catalog, projectRoot, configPath string, now time.Time) (Config, error) { + id, err := ConfigID(projectRoot, configPath) + if err != nil { + return Config{}, err + } + root, err := CanonicalPath(projectRoot) + if err != nil { + return Config{}, err + } + path, err := CanonicalPath(configPath) + if err != nil { + return Config{}, err + } + displayPath, err := filepath.Abs(configPath) + if err != nil { + return Config{}, err + } + displayPath = filepath.Clean(displayPath) + installationSum := sha256.Sum256([]byte(value.InstallationID + "\x00" + root)) + record := Config{ID: id, InstallationID: hex.EncodeToString(installationSum[:12]), Path: path, DisplayPath: displayPath, ProjectRoot: root, LastSeenAt: now.UTC()} + for index := range value.Configs { + if value.Configs[index].ID == id { + if value.Configs[index].InstallationID != "" { + record.InstallationID = value.Configs[index].InstallationID + } + record.BuildCacheLimitBytes = value.Configs[index].BuildCacheLimitBytes + record.ControllerLeaseUntil = value.Configs[index].ControllerLeaseUntil + value.Configs[index] = record + return record, nil + } + } + value.Configs = append(value.Configs, record) + return record, nil +} + +func RefreshControllerLease(value *Catalog, configID string, expiresAt time.Time) error { + if expiresAt.IsZero() { + return errors.New("controller lease expiry is required") + } + for index := range value.Configs { + if value.Configs[index].ID == configID { + expiry := expiresAt.UTC() + value.Configs[index].ControllerLeaseUntil = &expiry + return nil + } + } + return fmt.Errorf("catalog configuration %s is not registered", configID) +} + +func ReleaseControllerLease(value *Catalog, configID string) { + for index := range value.Configs { + if value.Configs[index].ID == configID { + value.Configs[index].ControllerLeaseUntil = nil + return + } + } +} + +func UpsertResource(value *Catalog, resource Resource) error { + if resource.BackendID == "" || resource.Kind == "" || resource.Identity == "" || resource.Locator == "" { + return errors.New("catalog resource backend, kind, identity, and locator are required") + } + if resource.Custody != CustodyGenerated && resource.Custody != CustodyAcquired { + return fmt.Errorf("unsupported catalog custody %q", resource.Custody) + } + if resource.Key == "" { + resource.Key = ResourceKey(resource.BackendID, resource.Kind, resource.Identity) + } + for index := range value.Resources { + if value.Resources[index].Key == resource.Key { + resource.InstallationIDs = mergeStrings(value.Resources[index].InstallationIDs, resource.InstallationIDs) + resource.CreatedAt = value.Resources[index].CreatedAt + if resource.CreatedAt.IsZero() { + resource.CreatedAt = time.Now().UTC() + } + value.Resources[index] = resource + return nil + } + } + if resource.CreatedAt.IsZero() { + resource.CreatedAt = time.Now().UTC() + } + value.Resources = append(value.Resources, resource) + return nil +} + +func mergeStrings(groups ...[]string) []string { + seen := make(map[string]bool) + var result []string + for _, group := range groups { + for _, value := range group { + value = strings.TrimSpace(value) + if value == "" || seen[value] { + continue + } + seen[value] = true + result = append(result, value) + } + } + sort.Strings(result) + return result +} + +func ReplaceConfigReferences(value *Catalog, configID string, references map[string]Reference, now time.Time) { + ReplaceConfigRoleReferences(value, configID, "", references, now) +} + +// ReplaceConfigRoleReferences atomically replaces one config's references for +// a single logical role without disturbing its other provider or bootstrap +// resources. An empty role preserves the original all-reference behavior. +func ReplaceConfigRoleReferences(value *Catalog, configID, role string, references map[string]Reference, now time.Time) { + for index := range value.Resources { + resource := &value.Resources[index] + filtered := resource.References[:0] + for _, reference := range resource.References { + if reference.ConfigID != configID || (role != "" && reference.Role != role) { + filtered = append(filtered, reference) + } + } + resource.References = filtered + if reference, found := references[resource.Key]; found { + reference.ConfigID = configID + if role != "" { + reference.Role = role + } + reference.UpdatedAt = now.UTC() + resource.References = append(resource.References, reference) + resource.State = StateCurrent + resource.SupersededAt = nil + resource.CleanupError = "" + } else if len(resource.References) == 0 && resource.State == StateCurrent { + when := now.UTC() + resource.State = StateSuperseded + resource.SupersededAt = &when + } + } +} + +// Compact removes references to missing configurations without a live lease, +// drops missing resources through the supplied exact observer, and discards +// completed journals. Observer errors preserve the resource fail-closed. +func Compact(value *Catalog, now time.Time, exists func(Resource) (bool, error)) []string { + var warnings []string + liveConfigs := make(map[string]bool) + configs := value.Configs[:0] + for _, config := range value.Configs { + configPresent := false + if info, err := os.Lstat(config.Path); err == nil && info.Mode().IsRegular() { + configPresent = true + } + leaseActive := config.ControllerLeaseUntil != nil && config.ControllerLeaseUntil.After(now) + if configPresent || leaseActive { + liveConfigs[config.ID] = true + configs = append(configs, config) + } + } + value.Configs = configs + resources := value.Resources[:0] + for _, resource := range value.Resources { + references := resource.References[:0] + for _, reference := range resource.References { + if liveConfigs[reference.ConfigID] { + references = append(references, reference) + } + } + resource.References = references + present, err := exists(resource) + if err != nil { + warnings = append(warnings, fmt.Sprintf("catalog resource %s could not be observed: %v", resource.Key, err)) + resources = append(resources, resource) + continue + } + if !present { + continue + } + resource.LastSeenAt = now.UTC() + if len(resource.References) == 0 && resource.State == StateCurrent { + when := now.UTC() + resource.State = StateSuperseded + resource.SupersededAt = &when + } + resources = append(resources, resource) + } + value.Resources = resources + resourceKeys := make(map[string]bool, len(value.Resources)) + for _, resource := range value.Resources { + resourceKeys[resource.Key] = true + } + journals := value.Journals[:0] + for _, journal := range value.Journals { + if journal.Phase != "complete" && (journal.ResourceKey == "" || resourceKeys[journal.ResourceKey]) { + journals = append(journals, journal) + } + } + value.Journals = journals + return warnings +} + +func normalize(value *Catalog) { + sort.Slice(value.Configs, func(i, j int) bool { return value.Configs[i].ID < value.Configs[j].ID }) + sort.Slice(value.Resources, func(i, j int) bool { return value.Resources[i].Key < value.Resources[j].Key }) + for index := range value.Resources { + resource := &value.Resources[index] + sort.Strings(resource.InstallationIDs) + sort.Strings(resource.IntroducedTags) + sort.Slice(resource.References, func(i, j int) bool { return resource.References[i].ConfigID < resource.References[j].ConfigID }) + } + sort.Slice(value.Journals, func(i, j int) bool { return value.Journals[i].ID < value.Journals[j].ID }) +} + +// CanonicalPath returns the stable identity used by controller locks, catalog +// records, lifecycle state, and build workspaces. Existing symlinks are +// resolved so alternate spellings of the same configuration cannot split +// ownership state. +func CanonicalPath(path string) (string, error) { + absolute, err := filepath.Abs(path) + if err != nil { + return "", err + } + if resolved, resolveErr := filepath.EvalSymlinks(absolute); resolveErr == nil { + absolute = resolved + } + absolute = filepath.Clean(absolute) + if runtime.GOOS == "windows" { + absolute = strings.ToLower(absolute) + } + return absolute, nil +} + +func randomID() (string, error) { + content := make([]byte, 16) + if _, err := rand.Read(content); err != nil { + return "", fmt.Errorf("generate EPAR installation identity: %w", err) + } + return hex.EncodeToString(content), nil +} diff --git a/internal/storage/catalog/catalog_test.go b/internal/storage/catalog/catalog_test.go new file mode 100644 index 0000000..a999967 --- /dev/null +++ b/internal/storage/catalog/catalog_test.go @@ -0,0 +1,268 @@ +package catalog + +import ( + "context" + "errors" + "os" + "path/filepath" + "testing" + "time" +) + +func TestConfigIDResolvesConfigurationSymlinks(t *testing.T) { + project := t.TempDir() + realPath := filepath.Join(project, "config.yml") + linkPath := filepath.Join(project, "config-link.yml") + if err := os.WriteFile(realPath, []byte("provider: {}\n"), 0o600); err != nil { + t.Fatal(err) + } + if err := os.Symlink(realPath, linkPath); err != nil { + t.Skipf("symlinks unavailable: %v", err) + } + realID, err := ConfigID(project, realPath) + if err != nil { + t.Fatal(err) + } + linkID, err := ConfigID(project, linkPath) + if err != nil { + t.Fatal(err) + } + if realID != linkID { + t.Fatalf("one configuration received split identities through a symlink: %q != %q", realID, linkID) + } +} + +func TestMultipleConfigsShareResourceUntilLastReferenceIsRemoved(t *testing.T) { + root := t.TempDir() + project := filepath.Join(root, "project") + if err := os.MkdirAll(project, 0o755); err != nil { + t.Fatal(err) + } + firstPath := filepath.Join(project, "first.yml") + secondPath := filepath.Join(project, "second.yml") + for _, path := range []string{firstPath, secondPath} { + if err := os.WriteFile(path, []byte("provider: test\n"), 0o600); err != nil { + t.Fatal(err) + } + } + now := time.Date(2026, 7, 29, 1, 2, 3, 0, time.UTC) + store, err := Open(filepath.Join(root, "catalog")) + if err != nil { + t.Fatal(err) + } + value, err := store.WithLock(now, func(value *Catalog) error { + first, err := RegisterConfig(value, project, firstPath, now) + if err != nil { + return err + } + second, err := RegisterConfig(value, project, secondPath, now) + if err != nil { + return err + } + resource := Resource{BackendID: "docker:test", Kind: "docker-image", Locator: "image:test", Identity: "sha256:abc", Custody: CustodyGenerated, State: StateCurrent, CreatedAt: now} + resource.Key = ResourceKey(resource.BackendID, resource.Kind, resource.Identity) + resource.References = []Reference{{ConfigID: first.ID}, {ConfigID: second.ID}} + return UpsertResource(value, resource) + }) + if err != nil { + t.Fatal(err) + } + if value.Configs[0].InstallationID == "" || value.Configs[0].InstallationID != value.Configs[1].InstallationID { + t.Fatalf("configs in one project did not share an installation identity: %#v", value.Configs) + } + key := value.Resources[0].Key + firstID, _ := ConfigID(project, firstPath) + ReplaceConfigReferences(&value, firstID, nil, now.Add(time.Minute)) + if got := len(value.Resources[0].References); got != 1 { + t.Fatalf("references after first removal = %d, want 1", got) + } + secondID, _ := ConfigID(project, secondPath) + ReplaceConfigReferences(&value, secondID, nil, now.Add(2*time.Minute)) + if value.Resources[0].State != StateSuperseded || value.Resources[0].SupersededAt == nil { + t.Fatalf("resource %s was not superseded after its final reference was removed", key) + } +} + +func TestDifferentProjectRootsHaveDifferentInstallationIdentities(t *testing.T) { + root := t.TempDir() + now := time.Now().UTC() + value := Catalog{InstallationID: "host-catalog"} + var ids []string + for _, name := range []string{"one", "two"} { + project := filepath.Join(root, name) + if err := os.MkdirAll(project, 0o755); err != nil { + t.Fatal(err) + } + configPath := filepath.Join(project, "config.yml") + if err := os.WriteFile(configPath, []byte("provider: test\n"), 0o600); err != nil { + t.Fatal(err) + } + record, err := RegisterConfig(&value, project, configPath, now) + if err != nil { + t.Fatal(err) + } + ids = append(ids, record.InstallationID) + } + if ids[0] == "" || ids[0] == ids[1] { + t.Fatalf("different project roots share installation identity %q", ids[0]) + } +} + +func TestRegisterConfigPreservesActionablePathSpelling(t *testing.T) { + root := t.TempDir() + configPath := filepath.Join(root, "Config.yml") + if err := os.WriteFile(configPath, []byte("provider: test\n"), 0o600); err != nil { + t.Fatal(err) + } + record, err := RegisterConfig(&Catalog{InstallationID: "host-catalog"}, root, configPath, time.Now().UTC()) + if err != nil { + t.Fatal(err) + } + want, err := filepath.Abs(configPath) + if err != nil { + t.Fatal(err) + } + if record.DisplayPath != filepath.Clean(want) { + t.Fatalf("display path = %q, want %q", record.DisplayPath, filepath.Clean(want)) + } +} + +func TestBackendLocksAreSeparatedAndSerializeTheSameBackend(t *testing.T) { + store, err := Open(t.TempDir()) + if err != nil { + t.Fatal(err) + } + first, err := store.AcquireBackendLock(context.Background(), "docker:first") + if err != nil { + t.Fatal(err) + } + defer first.Close() + secondBackend, err := store.AcquireBackendLock(context.Background(), "docker:second") + if err != nil { + t.Fatalf("different backend was unnecessarily blocked: %v", err) + } + if err := secondBackend.Close(); err != nil { + t.Fatal(err) + } + waitContext, cancel := context.WithTimeout(context.Background(), 25*time.Millisecond) + defer cancel() + if _, err := store.AcquireBackendLock(waitContext, "docker:first"); !errors.Is(err, context.DeadlineExceeded) { + t.Fatalf("same backend lock error = %v, want context deadline", err) + } + if err := first.Close(); err != nil { + t.Fatal(err) + } + reacquired, err := store.AcquireBackendLock(context.Background(), "docker:first") + if err != nil { + t.Fatalf("released backend lock could not be reacquired: %v", err) + } + if err := reacquired.Close(); err != nil { + t.Fatal(err) + } +} + +func TestCompactDropsMissingConfigsResourcesAndCompletedJournals(t *testing.T) { + root := t.TempDir() + now := time.Now().UTC() + value := Catalog{ + SchemaVersion: SchemaVersion, + Configs: []Config{{ID: "gone", Path: filepath.Join(root, "gone.yml")}}, + Resources: []Resource{{ + Key: "missing", BackendID: "docker:test", Kind: "docker-image", Locator: "x", Identity: "y", + Custody: CustodyGenerated, State: StateCurrent, References: []Reference{{ConfigID: "gone"}}, + }}, + Journals: []Journal{{ID: "done", Phase: "complete"}, {ID: "pending", Phase: "pull"}}, + } + warnings := Compact(&value, now, func(Resource) (bool, error) { return false, nil }) + if len(warnings) != 0 || len(value.Configs) != 0 || len(value.Resources) != 0 || len(value.Journals) != 1 || value.Journals[0].ID != "pending" { + t.Fatalf("unexpected compact result: %#v warnings=%v", value, warnings) + } +} + +func TestCompactPreservesMissingConfigWhileControllerLeaseIsActive(t *testing.T) { + root := t.TempDir() + now := time.Now().UTC() + value := Catalog{ + SchemaVersion: SchemaVersion, + Configs: []Config{{ + ID: "active", Path: filepath.Join(root, "removed.yml"), ControllerLeaseUntil: timePointer(now.Add(time.Minute)), + }}, + Resources: []Resource{{ + Key: "present", BackendID: "docker:test", Kind: "docker-image", Locator: "x", Identity: "y", + Custody: CustodyGenerated, State: StateCurrent, References: []Reference{{ConfigID: "active"}}, + }}, + } + Compact(&value, now, func(Resource) (bool, error) { return true, nil }) + if len(value.Configs) != 1 || len(value.Resources) != 1 || len(value.Resources[0].References) != 1 { + t.Fatalf("active controller lease did not preserve missing config reference: %#v", value) + } + Compact(&value, now.Add(2*time.Minute), func(Resource) (bool, error) { return true, nil }) + if len(value.Configs) != 0 || len(value.Resources[0].References) != 0 || value.Resources[0].State != StateSuperseded { + t.Fatalf("expired controller lease still protected missing config: %#v", value) + } +} + +func TestRegisterConfigPreservesControllerLease(t *testing.T) { + root := t.TempDir() + configPath := filepath.Join(root, "config.yml") + if err := os.WriteFile(configPath, []byte("provider: test\n"), 0o600); err != nil { + t.Fatal(err) + } + now := time.Now().UTC() + value := Catalog{} + record, err := RegisterConfig(&value, root, configPath, now) + if err != nil { + t.Fatal(err) + } + if err := RefreshControllerLease(&value, record.ID, now.Add(time.Minute)); err != nil { + t.Fatal(err) + } + if _, err := RegisterConfig(&value, root, configPath, now.Add(time.Second)); err != nil { + t.Fatal(err) + } + if value.Configs[0].ControllerLeaseUntil == nil || !value.Configs[0].ControllerLeaseUntil.Equal(now.Add(time.Minute)) { + t.Fatalf("registered config lost controller lease: %#v", value.Configs[0]) + } + ReleaseControllerLease(&value, record.ID) + if value.Configs[0].ControllerLeaseUntil != nil { + t.Fatalf("controller lease was not released: %#v", value.Configs[0]) + } +} + +func TestDefaultRootHonorsExplicitOverride(t *testing.T) { + override := filepath.Join(t.TempDir(), "state") + t.Setenv("EPAR_STATE_HOME", override) + got, err := DefaultRoot() + if err != nil { + t.Fatal(err) + } + want, _ := filepath.Abs(override) + if got != want { + t.Fatalf("DefaultRoot = %q, want %q", got, want) + } +} + +func TestRegisterConfigPreservesRegisteredCacheLimit(t *testing.T) { + root := t.TempDir() + configPath := filepath.Join(root, "config.yml") + if err := os.WriteFile(configPath, []byte("provider: test\n"), 0o600); err != nil { + t.Fatal(err) + } + now := time.Now().UTC() + value := Catalog{} + record, err := RegisterConfig(&value, root, configPath, now) + if err != nil { + t.Fatal(err) + } + value.Configs[0].BuildCacheLimitBytes = 20 << 30 + if _, err := RegisterConfig(&value, root, configPath, now.Add(time.Minute)); err != nil { + t.Fatal(err) + } + if value.Configs[0].ID != record.ID || value.Configs[0].BuildCacheLimitBytes != 20<<30 { + t.Fatalf("registered config lost persisted cache policy: %#v", value.Configs[0]) + } +} + +func timePointer(value time.Time) *time.Time { + return &value +} diff --git a/internal/storage/doc.go b/internal/storage/doc.go new file mode 100644 index 0000000..555900f --- /dev/null +++ b/internal/storage/doc.go @@ -0,0 +1,9 @@ +// Package storage defines provider-neutral, fail-closed storage inventory, +// capacity, retention-planning, and exact-execution contracts. +// +// The package does not delete host resources and does not implement Docker, +// Docker Sandboxes, or filesystem cleanup. Adapters inventory exact artifacts, +// Preview deterministically classifies them, and an optional ExactExecutor +// integration can apply only the exact removal entries bound into an approved +// plan hash. +package storage diff --git a/internal/storage/executor.go b/internal/storage/executor.go new file mode 100644 index 0000000..8792acb --- /dev/null +++ b/internal/storage/executor.go @@ -0,0 +1,117 @@ +package storage + +import ( + "context" + "errors" + "fmt" +) + +// Observation is the exact target state returned by an executor. +type Observation struct { + Exists bool `json:"exists"` + Target Target `json:"target"` +} + +// Removal contains exactly one immutable approved target. An implementation of +// RemoveExact must condition removal on Target.Identity and Target.Fingerprint +// and must fail closed when the locator resolves to a different object. +type Removal struct { + ArtifactID string `json:"artifactId"` + Target Target `json:"target"` + SizeBytes uint64 `json:"sizeBytes"` +} + +// ExactExecutor is deliberately unable to receive a prefix, glob, surface, or +// unbounded selector. Implementations must provide conditional exact removal +// and exact post-removal observation. +type ExactExecutor interface { + ObserveExact(context.Context, Target) (Observation, error) + RemoveExact(context.Context, Removal) error +} + +// ExecutionStatus describes an attempted exact removal. +type ExecutionStatus string + +const ( + ExecutionRemoved ExecutionStatus = "removed" + ExecutionDrifted ExecutionStatus = "drifted" + ExecutionFailed ExecutionStatus = "failed" +) + +// ExecutionEntry records one exact target outcome. +type ExecutionEntry struct { + Removal Removal `json:"removal"` + Status ExecutionStatus `json:"status"` + Error string `json:"error,omitempty"` +} + +// ExecutionReport is a partial-completion-safe journal returned in plan order. +type ExecutionReport struct { + PlanHash string `json:"planHash"` + Entries []ExecutionEntry `json:"entries"` + RemovedCount int `json:"removedCount"` + ReclaimedBytes uint64 `json:"reclaimedBytes"` +} + +// Execute applies only ActionRemove entries from a hash-approved plan. It +// re-observes identity before removal and verifies exact absence afterward. +// Execution stops on the first drift or error and returns the partial journal. +func Execute(ctx context.Context, plan Plan, approvedHash string, executor ExactExecutor) (ExecutionReport, error) { + if executor == nil { + return ExecutionReport{}, errors.New("exact storage executor is required") + } + if err := ValidatePlanHash(plan, approvedHash); err != nil { + return ExecutionReport{}, err + } + report := ExecutionReport{PlanHash: plan.Hash} + for _, decision := range plan.Decisions { + if decision.Action != ActionRemove { + continue + } + removal := Removal{ArtifactID: decision.Artifact.ID, Target: decision.Artifact.Target, SizeBytes: decision.Artifact.SizeBytes} + if err := validateExactTarget(removal.Target); err != nil { + return report, fmt.Errorf("planned removal %q is not exact: %w", removal.ArtifactID, err) + } + observation, err := executor.ObserveExact(ctx, removal.Target) + if err != nil { + entry := ExecutionEntry{Removal: removal, Status: ExecutionFailed, Error: err.Error()} + report.Entries = append(report.Entries, entry) + return report, fmt.Errorf("observe exact storage target %q: %w", removal.ArtifactID, err) + } + if !observation.Exists || observation.Target != removal.Target { + entry := ExecutionEntry{Removal: removal, Status: ExecutionDrifted, Error: "exact target identity changed or disappeared"} + report.Entries = append(report.Entries, entry) + return report, fmt.Errorf("exact storage target %q drifted", removal.ArtifactID) + } + if err := executor.RemoveExact(ctx, removal); err != nil { + entry := ExecutionEntry{Removal: removal, Status: ExecutionFailed, Error: err.Error()} + report.Entries = append(report.Entries, entry) + return report, fmt.Errorf("remove exact storage target %q: %w", removal.ArtifactID, err) + } + after, err := executor.ObserveExact(ctx, removal.Target) + if err != nil { + entry := ExecutionEntry{Removal: removal, Status: ExecutionFailed, Error: err.Error()} + report.Entries = append(report.Entries, entry) + return report, fmt.Errorf("verify exact storage target %q absence: %w", removal.ArtifactID, err) + } + if after.Exists { + entry := ExecutionEntry{Removal: removal, Status: ExecutionFailed, Error: "exact target still exists after removal"} + report.Entries = append(report.Entries, entry) + return report, fmt.Errorf("exact storage target %q still exists after removal", removal.ArtifactID) + } + report.Entries = append(report.Entries, ExecutionEntry{Removal: removal, Status: ExecutionRemoved}) + report.RemovedCount++ + report.ReclaimedBytes += removal.SizeBytes + } + return report, nil +} + +func validateExactTarget(target Target) error { + if target.Match != MatchExact { + return fmt.Errorf("target match is %q", target.Match) + } + if target.Kind == "" || target.Locator == "" || target.Identity == "" { + return errors.New("target kind, locator, and identity are required") + } + return nil +} diff --git a/internal/storage/filesystem.go b/internal/storage/filesystem.go new file mode 100644 index 0000000..29f8840 --- /dev/null +++ b/internal/storage/filesystem.go @@ -0,0 +1,138 @@ +package storage + +import ( + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "runtime" + "strings" + "time" +) + +// ProbeFilesystemCapacity returns an OS capacity observation for an exact, +// redirect-free existing filesystem path. +func ProbeFilesystemCapacity(path string, now time.Time) (Capacity, error) { + canonical, _, err := inspectFilesystemPath(path) + if err != nil { + return Capacity{}, err + } + available, total, err := platformFilesystemCapacity(canonical) + if err != nil { + return Capacity{}, fmt.Errorf("probe filesystem capacity for %q: %w", canonical, err) + } + return Capacity{Known: true, AvailableBytes: available, TotalBytes: total, ObservedAt: now.UTC()}, nil +} + +// SnapshotFilesystemTarget binds a regular file or real directory to a stable +// object identity and shallow metadata fingerprint. Symlinks, junctions, +// reparse points, special files, and redirected ancestors are rejected. +func SnapshotFilesystemTarget(path string) (Target, error) { + canonical, info, err := inspectFilesystemPath(path) + if err != nil { + return Target{}, err + } + kind := TargetFile + if info.IsDir() { + kind = TargetDirectory + } else if !info.Mode().IsRegular() { + return Target{}, fmt.Errorf("storage path %q is not a regular file or real directory", canonical) + } + identity, err := platformFilesystemIdentity(canonical, info.IsDir()) + if err != nil { + return Target{}, fmt.Errorf("read stable filesystem identity for %q: %w", canonical, err) + } + metadata := struct { + Size int64 `json:"size"` + Mode uint32 `json:"mode"` + ModTime string `json:"modTime"` + }{ + Size: info.Size(), + Mode: uint32(info.Mode()), + ModTime: info.ModTime().UTC().Format(time.RFC3339Nano), + } + encoded, err := json.Marshal(metadata) + if err != nil { + return Target{}, err + } + sum := sha256.Sum256(encoded) + return Target{ + Kind: kind, + Locator: canonical, + Identity: identity, + Fingerprint: "sha256:" + hex.EncodeToString(sum[:]), + Match: MatchExact, + }, nil +} + +func inspectFilesystemPath(path string) (string, os.FileInfo, error) { + if strings.TrimSpace(path) == "" || strings.ContainsRune(path, 0) { + return "", nil, errors.New("storage filesystem path is empty or contains NUL") + } + absolute, err := filepath.Abs(path) + if err != nil { + return "", nil, err + } + absolute = filepath.Clean(absolute) + if runtime.GOOS == "windows" { + rest := strings.TrimPrefix(absolute, filepath.VolumeName(absolute)) + if strings.Contains(rest, ":") { + return "", nil, fmt.Errorf("storage filesystem path %q contains an alternate-data-stream separator", absolute) + } + } + if err := rejectRedirectedAncestors(absolute); err != nil { + return "", nil, err + } + info, err := os.Lstat(absolute) + if err != nil { + return "", nil, err + } + if isFilesystemRedirect(info) { + return "", nil, fmt.Errorf("storage filesystem path %q is a symlink, junction, or reparse point", absolute) + } + evaluated, err := filepath.EvalSymlinks(absolute) + if err != nil { + return "", nil, err + } + evaluated, err = filepath.Abs(evaluated) + if err != nil { + return "", nil, err + } + canonicalSpelling, err := platformCanonicalFilesystemPath(absolute) + if err != nil { + return "", nil, fmt.Errorf("normalize storage filesystem path %q: %w", absolute, err) + } + canonicalSpelling = filepath.Clean(canonicalSpelling) + if !sameFilesystemPath(canonicalSpelling, filepath.Clean(evaluated)) { + return "", nil, fmt.Errorf("storage filesystem path %q contains a symlink, junction, or reparse redirection", absolute) + } + return canonicalSpelling, info, nil +} + +func rejectRedirectedAncestors(path string) error { + cursor := path + for { + info, err := os.Lstat(cursor) + if err != nil { + return err + } + if isFilesystemRedirect(info) { + return fmt.Errorf("storage filesystem path %q has redirected ancestor %q", path, cursor) + } + parent := filepath.Dir(cursor) + if parent == cursor { + return nil + } + cursor = parent + } +} + +func sameFilesystemPath(left, right string) bool { + if runtime.GOOS == "windows" { + return strings.EqualFold(left, right) + } + return left == right +} diff --git a/internal/storage/filesystem_executor.go b/internal/storage/filesystem_executor.go new file mode 100644 index 0000000..85d4f64 --- /dev/null +++ b/internal/storage/filesystem_executor.go @@ -0,0 +1,120 @@ +package storage + +import ( + "context" + "fmt" + "io/fs" + "os" + "path/filepath" + "strings" +) + +// FilesystemExecutor removes only exact files or directories strictly below +// explicitly approved roots. Every directory descendant is checked for links, +// reparse points, and special files before removal. +type FilesystemExecutor struct { + roots []string +} + +func NewFilesystemExecutor(allowedRoots ...string) (*FilesystemExecutor, error) { + executor := &FilesystemExecutor{} + for _, root := range allowedRoots { + if strings.TrimSpace(root) == "" { + continue + } + target, err := SnapshotFilesystemTarget(root) + if os.IsNotExist(err) { + continue + } + if err != nil { + return nil, fmt.Errorf("validate storage execution root %q: %w", root, err) + } + if target.Kind != TargetDirectory { + return nil, fmt.Errorf("storage execution root %q is not a directory", root) + } + executor.roots = append(executor.roots, target.Locator) + } + if len(executor.roots) == 0 { + return nil, fmt.Errorf("at least one existing exact storage execution root is required") + } + return executor, nil +} + +func (executor *FilesystemExecutor) ObserveExact(_ context.Context, target Target) (Observation, error) { + observed, err := SnapshotFilesystemTarget(target.Locator) + if os.IsNotExist(err) { + return Observation{Exists: false, Target: target}, nil + } + if err != nil { + return Observation{}, err + } + return Observation{Exists: true, Target: observed}, nil +} + +func (executor *FilesystemExecutor) RemoveExact(ctx context.Context, removal Removal) error { + if err := validateExactTarget(removal.Target); err != nil { + return err + } + if removal.Target.Kind != TargetFile && removal.Target.Kind != TargetDirectory { + return fmt.Errorf("filesystem executor does not support target kind %q", removal.Target.Kind) + } + if !executor.allowed(removal.Target.Locator) { + return fmt.Errorf("storage target %q is not strictly below an approved root", removal.Target.Locator) + } + if err := ctx.Err(); err != nil { + return err + } + observed, err := SnapshotFilesystemTarget(removal.Target.Locator) + if err != nil { + return err + } + if observed != removal.Target { + return fmt.Errorf("storage target identity changed before exact removal") + } + if observed.Kind == TargetFile { + return os.Remove(observed.Locator) + } + if err := validateRemovalTree(ctx, observed.Locator); err != nil { + return err + } + observedAgain, err := SnapshotFilesystemTarget(removal.Target.Locator) + if err != nil { + return err + } + if observedAgain != removal.Target { + return fmt.Errorf("storage directory identity changed during exact removal validation") + } + return os.RemoveAll(observed.Locator) +} + +func (executor *FilesystemExecutor) allowed(path string) bool { + for _, root := range executor.roots { + relative, err := filepath.Rel(root, path) + if err == nil && relative != "." && relative != ".." && !filepath.IsAbs(relative) && !strings.HasPrefix(relative, ".."+string(filepath.Separator)) { + return true + } + } + return false +} + +func validateRemovalTree(ctx context.Context, root string) error { + return filepath.WalkDir(root, func(path string, entry fs.DirEntry, walkErr error) error { + if walkErr != nil { + return walkErr + } + if err := ctx.Err(); err != nil { + return err + } + info, err := entry.Info() + if err != nil { + return err + } + if isFilesystemRedirect(info) { + return fmt.Errorf("refusing to remove redirected storage descendant %q", path) + } + if !info.IsDir() && !info.Mode().IsRegular() { + return fmt.Errorf("refusing to remove special storage descendant %q", path) + } + return nil + }) +} diff --git a/internal/storage/filesystem_executor_test.go b/internal/storage/filesystem_executor_test.go new file mode 100644 index 0000000..77b6227 --- /dev/null +++ b/internal/storage/filesystem_executor_test.go @@ -0,0 +1,92 @@ +package storage + +import ( + "context" + "os" + "path/filepath" + "testing" +) + +func TestFilesystemExecutorRemovesOnlyExactTargetBelowAllowedRoot(t *testing.T) { + root := t.TempDir() + targetPath := filepath.Join(root, "old", "archive.tar") + if err := os.MkdirAll(filepath.Dir(targetPath), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(targetPath, []byte("archive"), 0o600); err != nil { + t.Fatal(err) + } + target, err := SnapshotFilesystemTarget(targetPath) + if err != nil { + t.Fatal(err) + } + executor, err := NewFilesystemExecutor(root) + if err != nil { + t.Fatal(err) + } + if err := executor.RemoveExact(context.Background(), Removal{ArtifactID: "archive", Target: target}); err != nil { + t.Fatal(err) + } + if _, err := os.Stat(targetPath); !os.IsNotExist(err) { + t.Fatalf("target still exists: %v", err) + } +} + +func TestFilesystemExecutorRejectsAllowedRootAndDrift(t *testing.T) { + root := t.TempDir() + executor, err := NewFilesystemExecutor(root) + if err != nil { + t.Fatal(err) + } + rootTarget, err := SnapshotFilesystemTarget(root) + if err != nil { + t.Fatal(err) + } + if err := executor.RemoveExact(context.Background(), Removal{ArtifactID: "root", Target: rootTarget}); err == nil { + t.Fatal("RemoveExact() removed or accepted the allowed root") + } + + path := filepath.Join(root, "candidate") + if err := os.WriteFile(path, []byte("first"), 0o600); err != nil { + t.Fatal(err) + } + target, err := SnapshotFilesystemTarget(path) + if err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, []byte("replacement"), 0o600); err != nil { + t.Fatal(err) + } + if err := executor.RemoveExact(context.Background(), Removal{ArtifactID: "candidate", Target: target}); err == nil { + t.Fatal("RemoveExact() accepted a drifted target") + } + if _, err := os.Stat(path); err != nil { + t.Fatalf("drifted target was removed: %v", err) + } +} + +func TestFilesystemExecutorRejectsRedirectedDescendant(t *testing.T) { + root := t.TempDir() + targetPath := filepath.Join(root, "revision") + if err := os.Mkdir(targetPath, 0o755); err != nil { + t.Fatal(err) + } + linkPath := filepath.Join(targetPath, "redirect") + if err := os.Symlink(t.TempDir(), linkPath); err != nil { + t.Skipf("symlink unavailable: %v", err) + } + target, err := SnapshotFilesystemTarget(targetPath) + if err != nil { + t.Fatal(err) + } + executor, err := NewFilesystemExecutor(root) + if err != nil { + t.Fatal(err) + } + if err := executor.RemoveExact(context.Background(), Removal{ArtifactID: "revision", Target: target}); err == nil { + t.Fatal("RemoveExact() accepted a redirected descendant") + } + if _, err := os.Stat(targetPath); err != nil { + t.Fatalf("target directory was removed: %v", err) + } +} diff --git a/internal/storage/filesystem_test.go b/internal/storage/filesystem_test.go new file mode 100644 index 0000000..e7cc357 --- /dev/null +++ b/internal/storage/filesystem_test.go @@ -0,0 +1,85 @@ +package storage + +import ( + "os" + "path/filepath" + "runtime" + "testing" + "time" +) + +func TestSnapshotFilesystemTargetAndDrift(t *testing.T) { + t.Parallel() + path := filepath.Join(t.TempDir(), "artifact.bin") + if err := os.WriteFile(path, []byte("first"), 0o600); err != nil { + t.Fatal(err) + } + first, err := SnapshotFilesystemTarget(path) + if err != nil { + t.Fatalf("SnapshotFilesystemTarget() error = %v", err) + } + if first.Match != MatchExact || first.Kind != TargetFile || first.Identity == "" || first.Fingerprint == "" { + t.Fatalf("SnapshotFilesystemTarget() = %+v", first) + } + if err := os.WriteFile(path, []byte("different-size"), 0o600); err != nil { + t.Fatal(err) + } + second, err := SnapshotFilesystemTarget(path) + if err != nil { + t.Fatalf("second SnapshotFilesystemTarget() error = %v", err) + } + if first.Identity != second.Identity { + t.Fatalf("same filesystem object identity changed: first=%s second=%s", first.Identity, second.Identity) + } + if first.Fingerprint == second.Fingerprint { + t.Fatal("filesystem metadata drift did not change fingerprint") + } +} + +func TestSnapshotFilesystemTargetRejectsSymlinkOrReparse(t *testing.T) { + t.Parallel() + root := t.TempDir() + target := filepath.Join(root, "target") + if err := os.WriteFile(target, []byte("data"), 0o600); err != nil { + t.Fatal(err) + } + link := filepath.Join(root, "link") + if err := os.Symlink(target, link); err != nil { + t.Skipf("symlink creation unavailable on %s: %v", runtime.GOOS, err) + } + if _, err := SnapshotFilesystemTarget(link); err == nil { + t.Fatal("SnapshotFilesystemTarget() accepted symlink or reparse point") + } +} + +func TestSnapshotFilesystemTargetRejectsRedirectedAncestor(t *testing.T) { + t.Parallel() + root := t.TempDir() + realDirectory := filepath.Join(root, "real") + if err := os.Mkdir(realDirectory, 0o700); err != nil { + t.Fatal(err) + } + path := filepath.Join(realDirectory, "artifact") + if err := os.WriteFile(path, []byte("data"), 0o600); err != nil { + t.Fatal(err) + } + link := filepath.Join(root, "redirect") + if err := os.Symlink(realDirectory, link); err != nil { + t.Skipf("symlink creation unavailable on %s: %v", runtime.GOOS, err) + } + if _, err := SnapshotFilesystemTarget(filepath.Join(link, "artifact")); err == nil { + t.Fatal("SnapshotFilesystemTarget() accepted redirected ancestor") + } +} + +func TestProbeFilesystemCapacity(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + capacity, err := ProbeFilesystemCapacity(t.TempDir(), now) + if err != nil { + t.Fatalf("ProbeFilesystemCapacity() error = %v", err) + } + if !capacity.Known || capacity.TotalBytes == 0 || capacity.AvailableBytes > capacity.TotalBytes || !capacity.ObservedAt.Equal(now) { + t.Fatalf("ProbeFilesystemCapacity() = %+v", capacity) + } +} diff --git a/internal/storage/filesystem_unix.go b/internal/storage/filesystem_unix.go new file mode 100644 index 0000000..e1f980f --- /dev/null +++ b/internal/storage/filesystem_unix.go @@ -0,0 +1,40 @@ +//go:build !windows + +package storage + +import ( + "fmt" + "math" + "os" + "syscall" +) + +func platformFilesystemCapacity(path string) (uint64, uint64, error) { + var stats syscall.Statfs_t + if err := syscall.Statfs(path, &stats); err != nil { + return 0, 0, err + } + blockSize := uint64(stats.Bsize) + available := uint64(stats.Bavail) + total := uint64(stats.Blocks) + if blockSize != 0 && (available > math.MaxUint64/blockSize || total > math.MaxUint64/blockSize) { + return 0, 0, fmt.Errorf("filesystem space result overflow") + } + return available * blockSize, total * blockSize, nil +} + +func platformFilesystemIdentity(path string, _ bool) (string, error) { + info, err := os.Lstat(path) + if err != nil { + return "", err + } + stat, ok := info.Sys().(*syscall.Stat_t) + if !ok { + return "", fmt.Errorf("filesystem did not expose a stable object identity") + } + return fmt.Sprintf("unix:%x:%x", uint64(stat.Dev), uint64(stat.Ino)), nil +} + +func isFilesystemRedirect(info os.FileInfo) bool { return info.Mode()&os.ModeSymlink != 0 } + +func platformCanonicalFilesystemPath(path string) (string, error) { return path, nil } diff --git a/internal/storage/filesystem_windows.go b/internal/storage/filesystem_windows.go new file mode 100644 index 0000000..04a858a --- /dev/null +++ b/internal/storage/filesystem_windows.go @@ -0,0 +1,71 @@ +//go:build windows + +package storage + +import ( + "fmt" + "os" + "syscall" + + "golang.org/x/sys/windows" +) + +func platformFilesystemCapacity(path string) (uint64, uint64, error) { + pointer, err := windows.UTF16PtrFromString(path) + if err != nil { + return 0, 0, err + } + var available, total, free uint64 + if err := windows.GetDiskFreeSpaceEx(pointer, &available, &total, &free); err != nil { + return 0, 0, err + } + return available, total, nil +} + +func platformFilesystemIdentity(path string, directory bool) (string, error) { + pointer, err := windows.UTF16PtrFromString(path) + if err != nil { + return "", err + } + flags := uint32(windows.FILE_FLAG_OPEN_REPARSE_POINT) + if directory { + flags |= windows.FILE_FLAG_BACKUP_SEMANTICS + } + handle, err := windows.CreateFile(pointer, 0, windows.FILE_SHARE_READ|windows.FILE_SHARE_WRITE|windows.FILE_SHARE_DELETE, nil, windows.OPEN_EXISTING, flags, 0) + if err != nil { + return "", err + } + defer windows.CloseHandle(handle) + var information windows.ByHandleFileInformation + if err := windows.GetFileInformationByHandle(handle, &information); err != nil { + return "", err + } + return fmt.Sprintf("windows:%08x:%08x%08x", information.VolumeSerialNumber, information.FileIndexHigh, information.FileIndexLow), nil +} + +func isFilesystemRedirect(info os.FileInfo) bool { + if info.Mode()&os.ModeSymlink != 0 { + return true + } + data, ok := info.Sys().(*syscall.Win32FileAttributeData) + return ok && data.FileAttributes&syscall.FILE_ATTRIBUTE_REPARSE_POINT != 0 +} + +func platformCanonicalFilesystemPath(path string) (string, error) { + pointer, err := windows.UTF16PtrFromString(path) + if err != nil { + return "", err + } + size := uint32(len(path) + 1) + for { + buffer := make([]uint16, size) + length, err := windows.GetLongPathName(pointer, &buffer[0], size) + if err != nil { + return "", err + } + if length < size { + return windows.UTF16ToString(buffer[:length]), nil + } + size = length + 1 + } +} diff --git a/internal/storage/filesystem_windows_test.go b/internal/storage/filesystem_windows_test.go new file mode 100644 index 0000000..b57f144 --- /dev/null +++ b/internal/storage/filesystem_windows_test.go @@ -0,0 +1,58 @@ +//go:build windows + +package storage + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "golang.org/x/sys/windows" +) + +func TestSnapshotFilesystemTargetAcceptsWindowsShortPathAlias(t *testing.T) { + longRoot := filepath.Join(t.TempDir(), "Long Directory Name") + if err := os.Mkdir(longRoot, 0o700); err != nil { + t.Fatal(err) + } + shortRoot := windowsShortPath(t, longRoot) + if strings.EqualFold(shortRoot, longRoot) { + t.Skip("filesystem did not provide a distinct short path alias") + } + longPath := filepath.Join(longRoot, "artifact.bin") + if err := os.WriteFile(longPath, []byte("data"), 0o600); err != nil { + t.Fatal(err) + } + target, err := SnapshotFilesystemTarget(filepath.Join(shortRoot, "artifact.bin")) + if err != nil { + t.Fatalf("SnapshotFilesystemTarget() rejected a Windows short path alias: %v", err) + } + want, err := platformCanonicalFilesystemPath(longPath) + if err != nil { + t.Fatal(err) + } + if !strings.EqualFold(target.Locator, want) { + t.Fatalf("SnapshotFilesystemTarget() locator = %q, want canonical spelling %q", target.Locator, want) + } +} + +func windowsShortPath(t *testing.T, path string) string { + t.Helper() + pointer, err := windows.UTF16PtrFromString(path) + if err != nil { + t.Fatal(err) + } + size := uint32(len(path) + 1) + for { + buffer := make([]uint16, size) + length, err := windows.GetShortPathName(pointer, &buffer[0], size) + if err != nil { + t.Skipf("Windows short paths unavailable: %v", err) + } + if length < size { + return windows.UTF16ToString(buffer[:length]) + } + size = length + 1 + } +} diff --git a/internal/storage/format.go b/internal/storage/format.go new file mode 100644 index 0000000..b16f249 --- /dev/null +++ b/internal/storage/format.go @@ -0,0 +1,64 @@ +package storage + +import "fmt" + +const ( + KiB uint64 = 1 << 10 + MiB uint64 = 1 << 20 + TiB uint64 = 1 << 40 + PiB uint64 = 1 << 50 + EiB uint64 = 1 << 60 +) + +var byteUnits = []struct { + name string + bytes uint64 +}{ + {name: "EiB", bytes: EiB}, + {name: "PiB", bytes: PiB}, + {name: "TiB", bytes: TiB}, + {name: "GiB", bytes: GiB}, + {name: "MiB", bytes: MiB}, + {name: "KiB", bytes: KiB}, +} + +// FormatBytes renders a byte count in a compact binary unit suitable for +// operator-facing capacity messages. +func FormatBytes(value uint64) string { + for _, unit := range byteUnits { + if value >= unit.bytes { + return fmt.Sprintf("%.2f %s", float64(value)/float64(unit.bytes), unit.name) + } + } + return fmt.Sprintf("%d bytes", value) +} + +// CapacityAdmissionError renders a fail-closed capacity decision in terms an +// operator can act on without translating byte counts or policy arithmetic. +func CapacityAdmissionError(action string, surface Surface, requirement Requirement, check CapacityCheck, cleanupCommand string) error { + available := "unknown" + if surface.Capacity.Known { + available = FormatBytes(surface.Capacity.AvailableBytes) + } + message := fmt.Sprintf( + "not enough disk space to %s.\n\n"+ + " Storage location: %s\n"+ + " Available: %s\n"+ + " Estimated operation growth: %s\n"+ + " Free-space reserve: %s\n"+ + " Required before starting: %s", + action, + surface.Location, + available, + FormatBytes(requirement.PeakBytes), + FormatBytes(requirement.MinimumFreeBytes), + FormatBytes(check.RequiredAvailableBytes), + ) + if surface.Capacity.Known && check.DeficitBytes > 0 { + message += fmt.Sprintf("\n Additional space needed: %s", FormatBytes(check.DeficitBytes)) + } + if cleanupCommand != "" { + message += "\n\nReview reclaimable EPAR storage with:\n " + cleanupCommand + } + return fmt.Errorf("%s", message) +} diff --git a/internal/storage/format_test.go b/internal/storage/format_test.go new file mode 100644 index 0000000..c4c5f9b --- /dev/null +++ b/internal/storage/format_test.go @@ -0,0 +1,59 @@ +package storage + +import ( + "strings" + "testing" +) + +func TestFormatBytes(t *testing.T) { + tests := map[uint64]string{ + 0: "0 bytes", + 512: "512 bytes", + 1536: "1.50 KiB", + 140 * GiB: "140.00 GiB", + 9223372036854775807: "8.00 EiB", + } + for value, want := range tests { + if got := FormatBytes(value); got != want { + t.Errorf("FormatBytes(%d) = %q, want %q", value, got, want) + } + } +} + +func TestCapacityAdmissionError(t *testing.T) { + surface := Surface{ + Location: "sandbox-data", + Capacity: Capacity{Known: true, AvailableBytes: 140 * GiB}, + } + requirement := Requirement{PeakBytes: 130 * GiB, MinimumFreeBytes: 50 * GiB} + check, err := EvaluateCapacity(Surface{ + ID: "sandbox", + Location: surface.Location, + Capacity: surface.Capacity, + }, Requirement{ + ID: "create", + SurfaceID: "sandbox", + PeakBytes: requirement.PeakBytes, + MinimumFreeBytes: requirement.MinimumFreeBytes, + }) + if err != nil { + t.Fatal(err) + } + got := CapacityAdmissionError("initialize the runner", surface, requirement, check, "./start storage prune --provider docker-sandboxes").Error() + for _, want := range []string{ + "not enough disk space to initialize the runner", + "Available: 140.00 GiB", + "Estimated operation growth: 130.00 GiB", + "Free-space reserve: 50.00 GiB", + "Required before starting: 180.00 GiB", + "Additional space needed: 40.00 GiB", + "./start storage prune --provider docker-sandboxes", + } { + if !strings.Contains(got, want) { + t.Errorf("error = %q, want %q", got, want) + } + } + if strings.Contains(got, "150323855360") { + t.Fatalf("error exposes raw byte count: %q", got) + } +} diff --git a/internal/storage/hash.go b/internal/storage/hash.go new file mode 100644 index 0000000..56bedd0 --- /dev/null +++ b/internal/storage/hash.go @@ -0,0 +1,77 @@ +package storage + +import ( + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "strings" +) + +const hashPrefix = "sha256:" + +// ComputePlanHash returns the deterministic SHA-256 identity of a plan with its +// Hash field cleared. +func ComputePlanHash(plan Plan) (string, error) { + plan.Hash = "" + encoded, err := json.Marshal(plan) + if err != nil { + return "", fmt.Errorf("encode storage plan: %w", err) + } + sum := sha256.Sum256(encoded) + return hashPrefix + hex.EncodeToString(sum[:]), nil +} + +// ValidatePlanHash verifies both the embedded hash and an independently +// supplied approval hash. +func ValidatePlanHash(plan Plan, approvedHash string) error { + if strings.TrimSpace(approvedHash) == "" { + return errors.New("approved storage plan hash is required") + } + actual, err := ComputePlanHash(plan) + if err != nil { + return err + } + if plan.Hash != actual { + return fmt.Errorf("storage plan content drifted: embedded hash %q does not match %q", plan.Hash, actual) + } + if approvedHash != actual { + return fmt.Errorf("storage plan approval hash %q does not match %q", approvedHash, actual) + } + return nil +} + +// ValidatePlanArtifacts verifies that every planned removal still has the exact +// artifact snapshot supplied by the caller. Missing, added, or changed +// non-removal artifacts do not broaden the approved target set. +func ValidatePlanArtifacts(plan Plan, current []Artifact) error { + byID := make(map[string]Artifact, len(current)) + for _, artifact := range current { + if _, exists := byID[artifact.ID]; exists { + return fmt.Errorf("duplicate current storage artifact ID %q", artifact.ID) + } + byID[artifact.ID] = artifact + } + for _, decision := range plan.Decisions { + if decision.Action != ActionRemove { + continue + } + actual, exists := byID[decision.Artifact.ID] + if !exists { + return fmt.Errorf("planned storage artifact %q is missing", decision.Artifact.ID) + } + expectedJSON, err := json.Marshal(decision.Artifact) + if err != nil { + return err + } + actualJSON, err := json.Marshal(actual) + if err != nil { + return err + } + if string(expectedJSON) != string(actualJSON) { + return fmt.Errorf("planned storage artifact %q drifted", decision.Artifact.ID) + } + } + return nil +} diff --git a/internal/storage/hash_executor_test.go b/internal/storage/hash_executor_test.go new file mode 100644 index 0000000..e840b5a --- /dev/null +++ b/internal/storage/hash_executor_test.go @@ -0,0 +1,169 @@ +package storage + +import ( + "context" + "errors" + "strings" + "testing" + "time" +) + +func TestPlanHashDetectsContentAndApprovalDrift(t *testing.T) { + t.Parallel() + plan := testPlan(t) + if err := ValidatePlanHash(plan, plan.Hash); err != nil { + t.Fatalf("ValidatePlanHash() error = %v", err) + } + drifted := plan + drifted.Decisions = append([]Decision(nil), plan.Decisions...) + drifted.Decisions[0].Artifact.Provider = "changed-provider" + if err := ValidatePlanHash(drifted, plan.Hash); err == nil || !strings.Contains(err.Error(), "content drifted") { + t.Fatalf("ValidatePlanHash() drift error = %v", err) + } + if err := ValidatePlanHash(plan, "sha256:"+strings.Repeat("0", 64)); err == nil || !strings.Contains(err.Error(), "approval hash") { + t.Fatalf("ValidatePlanHash() approval error = %v", err) + } +} + +func TestValidatePlanArtifactsDetectsRemovalDrift(t *testing.T) { + t.Parallel() + plan := testPlan(t) + current := make([]Artifact, 0, len(plan.Decisions)) + for _, decision := range plan.Decisions { + current = append(current, decision.Artifact) + } + if err := ValidatePlanArtifacts(plan, current); err != nil { + t.Fatalf("ValidatePlanArtifacts() error = %v", err) + } + for index := range current { + if actionFor(plan, current[index].ID) == ActionRemove { + current[index].Target.Identity = "replacement" + break + } + } + if err := ValidatePlanArtifacts(plan, current); err == nil || !strings.Contains(err.Error(), "drifted") { + t.Fatalf("ValidatePlanArtifacts() drift error = %v", err) + } +} + +func TestExecuteOnlyPassesExactPlannedRemovals(t *testing.T) { + t.Parallel() + plan := testPlan(t) + executor := &fakeExactExecutor{existing: make(map[string]Target)} + wantRemove := make(map[string]Target) + for _, decision := range plan.Decisions { + if decision.Action == ActionRemove { + executor.existing[decision.Artifact.Target.Locator] = decision.Artifact.Target + wantRemove[decision.Artifact.ID] = decision.Artifact.Target + } + } + report, err := Execute(context.Background(), plan, plan.Hash, executor) + if err != nil { + t.Fatalf("Execute() error = %v", err) + } + if report.RemovedCount != len(wantRemove) || len(executor.removals) != len(wantRemove) { + t.Fatalf("Execute() report=%+v removals=%v want=%v", report, executor.removals, wantRemove) + } + for _, removal := range executor.removals { + want, exists := wantRemove[removal.ArtifactID] + if !exists || removal.Target != want || removal.Target.Match != MatchExact { + t.Fatalf("Execute() broadened removal: %+v", removal) + } + } +} + +func TestExecuteStopsOnIdentityDriftWithoutRemoval(t *testing.T) { + t.Parallel() + plan := testPlan(t) + executor := &fakeExactExecutor{existing: make(map[string]Target)} + for _, decision := range plan.Decisions { + if decision.Action == ActionRemove { + replacement := decision.Artifact.Target + replacement.Identity = "replacement" + executor.existing[replacement.Locator] = replacement + break + } + } + report, err := Execute(context.Background(), plan, plan.Hash, executor) + if err == nil || !strings.Contains(err.Error(), "drifted") { + t.Fatalf("Execute() error = %v, want drift", err) + } + if len(executor.removals) != 0 || len(report.Entries) != 1 || report.Entries[0].Status != ExecutionDrifted { + t.Fatalf("Execute() drift report=%+v removals=%v", report, executor.removals) + } +} + +func TestExecuteReturnsPartialJournalOnExactFailure(t *testing.T) { + t.Parallel() + plan := testPlanWithTwoRemovals(t) + executor := &fakeExactExecutor{existing: make(map[string]Target), failAt: 2} + for _, decision := range plan.Decisions { + if decision.Action == ActionRemove { + executor.existing[decision.Artifact.Target.Locator] = decision.Artifact.Target + } + } + report, err := Execute(context.Background(), plan, plan.Hash, executor) + if err == nil || !strings.Contains(err.Error(), "synthetic exact failure") { + t.Fatalf("Execute() error = %v", err) + } + if report.RemovedCount != 1 || len(report.Entries) != 2 || report.Entries[0].Status != ExecutionRemoved || report.Entries[1].Status != ExecutionFailed { + t.Fatalf("Execute() partial report = %+v", report) + } +} + +type fakeExactExecutor struct { + existing map[string]Target + removals []Removal + failAt int +} + +func (executor *fakeExactExecutor) ObserveExact(_ context.Context, target Target) (Observation, error) { + actual, exists := executor.existing[target.Locator] + return Observation{Exists: exists, Target: actual}, nil +} + +func (executor *fakeExactExecutor) RemoveExact(_ context.Context, removal Removal) error { + executor.removals = append(executor.removals, removal) + if executor.failAt > 0 && len(executor.removals) == executor.failAt { + return errors.New("synthetic exact failure") + } + delete(executor.existing, removal.Target.Locator) + return nil +} + +func testPlan(t *testing.T) Plan { + t.Helper() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + plan, err := Preview(PreviewRequest{ + Now: now, + Policy: DefaultPolicy(), + Surfaces: testSurfaces(), + Artifacts: []Artifact{ + testArtifact("remove", ArtifactTemplateArchive, now.Add(-10*24*time.Hour)), + withArtifact(testArtifact("keep", ArtifactTemplateArchive, now.Add(-time.Hour)), func(a *Artifact) {}), + withArtifact(testArtifact("protected", ArtifactTemplateArchive, now.Add(-10*24*time.Hour)), func(a *Artifact) { a.Current = true }), + }, + }) + if err != nil { + t.Fatalf("Preview() error = %v", err) + } + return plan +} + +func testPlanWithTwoRemovals(t *testing.T) Plan { + t.Helper() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + plan, err := Preview(PreviewRequest{ + Now: now, + Policy: DefaultPolicy(), + Surfaces: testSurfaces(), + Artifacts: []Artifact{ + testArtifact("remove-a", ArtifactTemplateArchive, now.Add(-11*24*time.Hour)), + testArtifact("remove-b", ArtifactTemplateArchive, now.Add(-10*24*time.Hour)), + }, + }) + if err != nil { + t.Fatalf("Preview() error = %v", err) + } + return plan +} diff --git a/internal/storage/inventory/collect.go b/internal/storage/inventory/collect.go new file mode 100644 index 0000000..eec9485 --- /dev/null +++ b/internal/storage/inventory/collect.go @@ -0,0 +1,161 @@ +package inventory + +import ( + "fmt" + "os" + "path/filepath" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +// Collect inventories only explicitly supplied or project-local EPAR roots. +// Missing optional roots are empty inventory, while unsafe roots are skipped +// with warnings. +func Collect(options Options) (Snapshot, error) { + if err := options.validate(); err != nil { + return Snapshot{}, err + } + now := options.Now.UTC() + projectTarget, err := storage.SnapshotFilesystemTarget(options.ProjectRoot) + if err != nil { + return Snapshot{}, fmt.Errorf("inspect storage inventory project root: %w", err) + } + if projectTarget.Kind != storage.TargetDirectory { + return Snapshot{}, fmt.Errorf("storage inventory project root is not a real directory") + } + projectRoot := projectTarget.Locator + capacity, capacityErr := storage.ProbeFilesystemCapacity(projectRoot, now) + snapshot := Snapshot{ + CollectedAt: now, + ProjectRoot: projectRoot, + ProviderFilter: options.Provider, + Surfaces: []storage.Surface{{ + ID: ProjectSurfaceID, + Kind: storage.SurfaceHostFilesystem, + Location: projectRoot, + Capacity: capacity, + }}, + } + if capacityErr != nil { + snapshot.Surfaces[0].Capacity = storage.Capacity{ObservedAt: now} + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("project filesystem capacity is unknown: %v", capacityErr)) + } + + logsRoot := resolveRoot(projectRoot, options.LogsRoot, filepath.Join("work", "logs")) + nativeRoot := resolveRoot(projectRoot, options.NativeRoot, filepath.Join(".local", "bin")) + templateRoot := resolveRoot(projectRoot, options.TemplateRoot, filepath.Join("work", "template-builds", "docker-sandboxes")) + + if includeProvider(options.Provider, "") { + logsSurfaceID := ensureFilesystemSurface(&snapshot, logsRoot, projectRoot, "", now) + artifacts, warnings := collectLogs(logsRoot, projectRoot) + assignSurface(artifacts, logsSurfaceID) + snapshot.Artifacts = append(snapshot.Artifacts, artifacts...) + snapshot.Warnings = append(snapshot.Warnings, warnings...) + + nativeSurfaceID := ensureFilesystemSurface(&snapshot, nativeRoot, projectRoot, "", now) + artifacts, warnings = collectNative(nativeOptions{ + Root: nativeRoot, + CurrentExecutable: options.CurrentExecutable, + CurrentRevision: options.CurrentRevision, + }) + assignSurface(artifacts, nativeSurfaceID) + snapshot.Artifacts = append(snapshot.Artifacts, artifacts...) + snapshot.Warnings = append(snapshot.Warnings, warnings...) + } + if includeProvider(options.Provider, ProviderDockerSandboxes) { + templateSurfaceID := ensureFilesystemSurface(&snapshot, templateRoot, projectRoot, ProviderDockerSandboxes, now) + artifacts, warnings, err := collectTemplates(templateOptions{ + Root: templateRoot, + Selections: options.ConfiguredTemplates, + Protections: options.TemplateProtections, + }) + if err != nil { + return Snapshot{}, err + } + assignSurface(artifacts, templateSurfaceID) + snapshot.Artifacts = append(snapshot.Artifacts, artifacts...) + snapshot.Warnings = append(snapshot.Warnings, warnings...) + } + artifacts, warnings := collectConfiguredFiles(options.ConfiguredFiles, options.Provider, projectRoot, &snapshot, now) + snapshot.Artifacts = append(snapshot.Artifacts, artifacts...) + snapshot.Warnings = append(snapshot.Warnings, warnings...) + snapshot.normalize() + return snapshot, nil +} + +func assignSurface(artifacts []storage.Artifact, surfaceID string) { + for index := range artifacts { + artifacts[index].SurfaceID = surfaceID + } +} + +func ensureFilesystemSurface(snapshot *Snapshot, root, projectRoot, provider string, now time.Time) string { + absolute, err := filepath.Abs(root) + if err != nil { + absolute = root + } + absolute = filepath.Clean(absolute) + if pathWithin(projectRoot, absolute) { + return ProjectSurfaceID + } + id := stableID("filesystem", absolute) + for index := range snapshot.Surfaces { + if snapshot.Surfaces[index].ID == id { + if snapshot.Surfaces[index].Provider != provider { + snapshot.Surfaces[index].Provider = "" + } + return id + } + } + capacity, probeErr := storage.ProbeFilesystemCapacity(absolute, now) + if probeErr != nil { + capacity = storage.Capacity{ObservedAt: now} + snapshot.Warnings = append(snapshot.Warnings, fmt.Sprintf("filesystem capacity for configured root %q is unknown: %v", absolute, probeErr)) + } + snapshot.Surfaces = append(snapshot.Surfaces, storage.Surface{ + ID: id, + Provider: provider, + Kind: storage.SurfaceHostFilesystem, + Location: absolute, + Capacity: capacity, + }) + return id +} + +func pathWithin(root, candidate string) bool { + relative, err := filepath.Rel(root, candidate) + if err != nil { + return false + } + return relative == "." || (relative != ".." && !filepath.IsAbs(relative) && !strings.HasPrefix(relative, ".."+string(filepath.Separator))) +} + +func resolveRoot(projectRoot, configured, fallback string) string { + if configured == "" { + return filepath.Join(projectRoot, fallback) + } + if filepath.IsAbs(configured) { + return filepath.Clean(configured) + } + return filepath.Join(projectRoot, configured) +} + +func inspectOptionalRoot(path string) (storage.Target, bool, error) { + _, err := os.Lstat(path) + if os.IsNotExist(err) { + return storage.Target{}, false, nil + } + if err != nil { + return storage.Target{}, false, err + } + target, err := storage.SnapshotFilesystemTarget(path) + if err != nil { + return storage.Target{}, false, err + } + if target.Kind != storage.TargetDirectory { + return storage.Target{}, false, fmt.Errorf("%q is not a real directory", path) + } + return target, true, nil +} diff --git a/internal/storage/inventory/collect_test.go b/internal/storage/inventory/collect_test.go new file mode 100644 index 0000000..6150fb0 --- /dev/null +++ b/internal/storage/inventory/collect_test.go @@ -0,0 +1,323 @@ +package inventory + +import ( + "os" + "path/filepath" + "reflect" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +func TestCollectDeterministicInventoryAndPreview(t *testing.T) { + t.Parallel() + project := t.TempDir() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + logs := filepath.Join(project, "work", "logs") + mustMkdirAll(t, logs) + mustWriteFile(t, filepath.Join(logs, "epar.log"), []byte("log")) + + nativeRoot := filepath.Join(project, ".local", "bin") + oldKey := repeatedHex("1") + currentKey := repeatedHex("2") + writeNativeRevision(t, nativeRoot, oldKey, now.Add(-16*24*time.Hour), false) + currentExecutable := writeNativeRevision(t, nativeRoot, currentKey, now.Add(-8*24*time.Hour), false) + ambiguousKey := repeatedHex("3") + mustMkdirAll(t, filepath.Join(nativeRoot, ambiguousKey)) + mustWriteFile(t, filepath.Join(nativeRoot, ambiguousKey, "ephemeral-action-runner"), []byte("unlabelled")) + + templateRoot := filepath.Join(project, "work", "template-builds", "docker-sandboxes") + oldTemplate := writeTemplateFixture(t, templateRoot, "old", "act-22.04", "linux/amd64", "old", now.Add(-16*24*time.Hour), nil) + currentTemplate := writeTemplateFixture(t, templateRoot, "current", "act-22.04", "linux/amd64", "current", now.Add(-8*24*time.Hour), nil) + + options := Options{ + ProjectRoot: project, + Now: now, + CurrentExecutable: currentExecutable, + CurrentRevision: "sha256:" + currentKey, + ConfiguredTemplates: []TemplateSelection{{ + Profile: currentTemplate.Profile, + Platform: currentTemplate.Platform, + Tag: currentTemplate.Tag, + TemplateDigest: currentTemplate.TemplateDigest, + ActivatedAt: now.Add(-8 * 24 * time.Hour), + }}, + TemplateProtections: []TemplateProtection{{ + ArchiveSHA256: oldTemplate.ArchiveSHA256, + Kind: storage.ProtectionCertification, + Detail: "release evidence", + }}, + } + first, err := Collect(options) + if err != nil { + t.Fatalf("Collect() error = %v", err) + } + second, err := Collect(options) + if err != nil { + t.Fatalf("second Collect() error = %v", err) + } + if first.CollectedAt != second.CollectedAt || first.ProjectRoot != second.ProjectRoot || first.ProviderFilter != second.ProviderFilter || !reflect.DeepEqual(first.Warnings, second.Warnings) { + t.Fatalf("Collect() stable metadata is nondeterministic:\nfirst=%+v\nsecond=%+v", first, second) + } + assertArtifactsEqual(t, first.Artifacts, second.Artifacts) + if len(first.Surfaces) != 1 || first.Surfaces[0].ID != ProjectSurfaceID || !first.Surfaces[0].Capacity.Known { + t.Fatalf("Collect() surfaces = %+v", first.Surfaces) + } + + currentNative := findArtifact(t, first.Artifacts, "native-controller:"+currentKey) + if !currentNative.Current || currentNative.Ownership.Kind != storage.OwnershipExact { + t.Fatalf("current native artifact = %+v", currentNative) + } + oldNative := findArtifact(t, first.Artifacts, "native-controller:"+oldKey) + if oldNative.SupersededAt == nil || !oldNative.SupersededAt.Equal(now.Add(-8*24*time.Hour)) { + t.Fatalf("old native supersededAt = %v", oldNative.SupersededAt) + } + unknownNative := findArtifactByLocatorSuffix(t, first.Artifacts, ambiguousKey) + if unknownNative.Ownership.Kind != storage.OwnershipUnknown { + t.Fatalf("unlabelled native ownership = %s", unknownNative.Ownership.Kind) + } + logArtifact := findArtifactByLocatorSuffix(t, first.Artifacts, "logs") + if !hasProtection(logArtifact, storage.ProtectionOperator) { + t.Fatalf("logs artifact protections = %+v", logArtifact.Protections) + } + currentArchive := findArtifactByArchiveDigest(t, first.Artifacts, currentTemplate.ArchiveSHA256) + if currentArchive.Current || hasProtection(currentArchive, storage.ProtectionConfiguration) { + t.Fatalf("imported-template selection retained its transient archive = %+v", currentArchive) + } + oldArchive := findArtifactByArchiveDigest(t, first.Artifacts, oldTemplate.ArchiveSHA256) + if !hasProtection(oldArchive, storage.ProtectionCertification) || oldArchive.SupersededAt != nil { + t.Fatalf("old archive = %+v", oldArchive) + } + + plan, err := storage.Preview(first.PreviewRequest(storage.DefaultPolicy(), nil)) + if err != nil { + t.Fatalf("Preview() inventory error = %v", err) + } + if actionForArtifact(plan, oldNative.ID) != storage.ActionRemove { + t.Fatalf("old native action = %s", actionForArtifact(plan, oldNative.ID)) + } + if actionForArtifact(plan, oldArchive.ID) != storage.ActionProtected { + t.Fatalf("certified archive action = %s", actionForArtifact(plan, oldArchive.ID)) + } +} + +func TestCollectProviderFilter(t *testing.T) { + t.Parallel() + project := t.TempDir() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + mustMkdirAll(t, filepath.Join(project, "work", "logs")) + writeNativeRevision(t, filepath.Join(project, ".local", "bin"), repeatedHex("a"), now.Add(-time.Hour), false) + writeTemplateFixture(t, filepath.Join(project, "work", "template-builds", "docker-sandboxes"), "template", "full", "linux/amd64", "current", now.Add(-time.Hour), nil) + + snapshot, err := Collect(Options{ProjectRoot: project, Provider: ProviderDockerSandboxes, Now: now}) + if err != nil { + t.Fatalf("Collect() error = %v", err) + } + if len(snapshot.Artifacts) != 3 { + t.Fatalf("provider-filtered artifacts = %+v", snapshot.Artifacts) + } + var providerSpecific int + for _, artifact := range snapshot.Artifacts { + if artifact.Provider == ProviderDockerSandboxes { + providerSpecific++ + if artifact.Kind != storage.ArtifactTemplateArchive { + t.Fatalf("provider-specific artifact = %+v", artifact) + } + } else if artifact.Provider != "" { + t.Fatalf("unrelated provider artifact = %+v", artifact) + } + } + if providerSpecific != 1 { + t.Fatalf("provider-specific artifacts = %d", providerSpecific) + } + if len(snapshot.Surfaces) != 1 { + t.Fatalf("provider-filtered surfaces = %+v", snapshot.Surfaces) + } +} + +func TestCollectAssignsExternalConfiguredRootToSeparateCapacitySurface(t *testing.T) { + t.Parallel() + project := t.TempDir() + externalLogs := t.TempDir() + mustWriteFile(t, filepath.Join(externalLogs, "epar.log"), []byte("log")) + snapshot, err := Collect(Options{ + ProjectRoot: project, + LogsRoot: externalLogs, + Now: time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC), + }) + if err != nil { + t.Fatalf("Collect() error = %v", err) + } + if len(snapshot.Surfaces) != 2 { + t.Fatalf("Collect() surfaces = %+v, want project and external", snapshot.Surfaces) + } + logArtifact := findArtifactByLocatorSuffix(t, snapshot.Artifacts, filepath.Base(externalLogs)) + if logArtifact.SurfaceID == ProjectSurfaceID || !hasProtection(logArtifact, storage.ProtectionCustomRoot) { + t.Fatalf("external logs artifact = %+v", logArtifact) + } + var found bool + for _, surface := range snapshot.Surfaces { + if surface.ID == logArtifact.SurfaceID && samePath(surface.Location, externalLogs) && surface.Capacity.Known { + found = true + } + } + if !found { + t.Fatalf("external logs surface %q not found in %+v", logArtifact.SurfaceID, snapshot.Surfaces) + } +} + +func TestCollectConfiguredProviderFilesUsesExactIdentityAndProtection(t *testing.T) { + t.Parallel() + project := t.TempDir() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + imagePath := filepath.Join(project, "work", "images", "runner.tar") + mustMkdirAll(t, filepath.Dir(imagePath)) + mustWriteFile(t, imagePath, []byte("reusable-wsl-image")) + + snapshot, err := Collect(Options{ + ProjectRoot: project, + Provider: "wsl", + Now: now, + ConfiguredFiles: []ConfiguredFile{{ + Provider: "wsl", + Role: "reusable-image", + Path: imagePath, + Kind: storage.ArtifactProviderImage, + Current: true, + ConfiguredAt: now.Add(-time.Hour), + ProtectionKind: storage.ProtectionConfiguration, + ProtectionDetail: "current reusable WSL image", + }}, + }) + if err != nil { + t.Fatalf("Collect() error = %v", err) + } + artifact := findArtifactByLocatorSuffix(t, snapshot.Artifacts, "runner.tar") + if artifact.Provider != "wsl" || artifact.Kind != storage.ArtifactProviderImage || artifact.SurfaceID != ProjectSurfaceID { + t.Fatalf("configured artifact = %+v", artifact) + } + if artifact.Ownership.Kind != storage.OwnershipExact || artifact.Target.Match != storage.MatchExact || artifact.Target.Identity == "" || artifact.Target.Fingerprint == "" { + t.Fatalf("configured artifact identity = %+v", artifact) + } + if artifact.SizeBytes != uint64(len("reusable-wsl-image")) || !artifact.Current || !hasProtection(artifact, storage.ProtectionConfiguration) || !hasProtection(artifact, storage.ProtectionCurrent) { + t.Fatalf("configured artifact protection = %+v", artifact) + } + plan, err := storage.Preview(snapshot.PreviewRequest(storage.DefaultPolicy(), nil)) + if err != nil { + t.Fatalf("Preview() error = %v", err) + } + if actionForArtifact(plan, artifact.ID) != storage.ActionProtected { + t.Fatalf("configured artifact action = %s", actionForArtifact(plan, artifact.ID)) + } +} + +func TestSnapshotPreviewRequestCopiesTopLevelSlices(t *testing.T) { + t.Parallel() + snapshot := Snapshot{ + CollectedAt: time.Now().UTC(), + Surfaces: []storage.Surface{{ID: ProjectSurfaceID}}, + Artifacts: []storage.Artifact{{ID: "artifact"}}, + } + request := snapshot.PreviewRequest(storage.DefaultPolicy(), []storage.Requirement{{ID: "requirement"}}) + request.Surfaces[0].ID = "changed" + request.Artifacts[0].ID = "changed" + if snapshot.Surfaces[0].ID != ProjectSurfaceID || snapshot.Artifacts[0].ID != "artifact" { + t.Fatal("PreviewRequest() aliased snapshot top-level slices") + } +} + +func findArtifact(t *testing.T, artifacts []storage.Artifact, id string) storage.Artifact { + t.Helper() + for _, artifact := range artifacts { + if artifact.ID == id { + return artifact + } + } + t.Fatalf("artifact %q not found in %+v", id, artifacts) + return storage.Artifact{} +} + +func findArtifactByLocatorSuffix(t *testing.T, artifacts []storage.Artifact, suffix string) storage.Artifact { + t.Helper() + for _, artifact := range artifacts { + if filepath.Base(artifact.Target.Locator) == suffix { + return artifact + } + } + t.Fatalf("artifact with locator suffix %q not found", suffix) + return storage.Artifact{} +} + +func findArtifactByKind(t *testing.T, artifacts []storage.Artifact, kind storage.ArtifactKind, provider string) storage.Artifact { + t.Helper() + for _, artifact := range artifacts { + if artifact.Kind == kind && artifact.Provider == provider { + return artifact + } + } + t.Fatalf("artifact kind=%q provider=%q not found", kind, provider) + return storage.Artifact{} +} + +func findArtifactByArchiveDigest(t *testing.T, artifacts []storage.Artifact, digest string) storage.Artifact { + t.Helper() + for _, artifact := range artifacts { + if artifact.Kind == storage.ArtifactTemplateArchive && artifact.Ownership.OwnerID == "template-archive:"+digest { + return artifact + } + } + t.Fatalf("archive artifact %q not found", digest) + return storage.Artifact{} +} + +func actionForArtifact(plan storage.Plan, id string) storage.Action { + for _, decision := range plan.Decisions { + if decision.Artifact.ID == id { + return decision.Action + } + } + return "" +} + +func hasProtection(artifact storage.Artifact, kind storage.ProtectionKind) bool { + for _, protection := range artifact.Protections { + if protection.Kind == kind { + return true + } + } + return false +} + +func repeatedHex(value string) string { + return value + value + value + value + value + value + value + value + + value + value + value + value + value + value + value + value + + value + value + value + value + value + value + value + value + + value + value + value + value + value + value + value + value + + value + value + value + value + value + value + value + value + + value + value + value + value + value + value + value + value + + value + value + value + value + value + value + value + value + + value + value + value + value + value + value + value + value +} + +func mustMkdirAll(t *testing.T, path string) { + t.Helper() + if err := os.MkdirAll(path, 0o700); err != nil { + t.Fatal(err) + } +} + +func mustWriteFile(t *testing.T, path string, data []byte) { + t.Helper() + mustMkdirAll(t, filepath.Dir(path)) + if err := os.WriteFile(path, data, 0o600); err != nil { + t.Fatal(err) + } +} + +func assertArtifactsEqual(t *testing.T, left, right []storage.Artifact) { + t.Helper() + if !reflect.DeepEqual(left, right) { + t.Fatalf("artifacts differ:\nleft=%+v\nright=%+v", left, right) + } +} diff --git a/internal/storage/inventory/configured.go b/internal/storage/inventory/configured.go new file mode 100644 index 0000000..a2b6241 --- /dev/null +++ b/internal/storage/inventory/configured.go @@ -0,0 +1,110 @@ +package inventory + +import ( + "fmt" + "os" + "path/filepath" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +func collectConfiguredFiles(files []ConfiguredFile, providerFilter, projectRoot string, snapshot *Snapshot, now time.Time) ([]storage.Artifact, []string) { + var artifacts []storage.Artifact + var warnings []string + for _, configured := range files { + if !includeProvider(providerFilter, configured.Provider) { + continue + } + path := resolveRoot(projectRoot, configured.Path, configured.Path) + info, err := os.Stat(path) + if os.IsNotExist(err) { + continue + } + if err != nil { + artifacts = append(artifacts, unknownConfiguredFile(configured, path, err)) + warnings = append(warnings, fmt.Sprintf("configured %s artifact %q was not inventoried safely: %v", configured.Role, path, err)) + continue + } + if !info.Mode().IsRegular() { + err = fmt.Errorf("configured artifact is not a regular file") + artifacts = append(artifacts, unknownConfiguredFile(configured, path, err)) + warnings = append(warnings, fmt.Sprintf("configured %s artifact %q was not inventoried safely: %v", configured.Role, path, err)) + continue + } + target, err := storage.SnapshotFilesystemTarget(path) + if err != nil { + artifacts = append(artifacts, unknownConfiguredFile(configured, path, err)) + warnings = append(warnings, fmt.Sprintf("configured %s artifact %q was not inventoried safely: %v", configured.Role, path, err)) + continue + } + after, err := os.Stat(target.Locator) + if err != nil || !after.Mode().IsRegular() || after.Size() < 0 { + if err == nil { + err = fmt.Errorf("configured artifact changed while being inventoried") + } + artifacts = append(artifacts, unknownConfiguredFile(configured, path, err)) + warnings = append(warnings, fmt.Sprintf("configured %s artifact %q was not inventoried safely: %v", configured.Role, path, err)) + continue + } + confirmed, err := storage.SnapshotFilesystemTarget(target.Locator) + if err != nil || confirmed != target { + if err == nil { + err = fmt.Errorf("configured artifact identity or metadata drifted while being inventoried") + } + artifacts = append(artifacts, unknownConfiguredFile(configured, path, err)) + warnings = append(warnings, fmt.Sprintf("configured %s artifact %q was not inventoried safely: %v", configured.Role, path, err)) + continue + } + kind := configured.Kind + if kind == "" { + kind = storage.ArtifactOther + } + protectionKind := configured.ProtectionKind + if protectionKind == "" { + protectionKind = storage.ProtectionConfiguration + } + protectionDetail := strings.TrimSpace(configured.ProtectionDetail) + if protectionDetail == "" { + protectionDetail = "explicit provider artifact configuration" + } + surfaceID := ensureFilesystemSurface(snapshot, filepath.Dir(target.Locator), projectRoot, configured.Provider, now) + artifact := storage.Artifact{ + ID: stableID("configured-file", configured.Provider+"\x00"+configured.Role+"\x00"+target.Identity), + Provider: configured.Provider, + SurfaceID: surfaceID, + Kind: kind, + RetentionGroup: configured.Provider + ":" + configured.Role, + Target: target, + Ownership: storage.Ownership{ + Kind: storage.OwnershipExact, + OwnerID: stableID("configured-owner", configured.Provider+"\x00"+configured.Role+"\x00"+target.Identity), + Evidence: "exact path persisted in EPAR configuration", + }, + SizeBytes: uint64(after.Size()), + CreatedAt: after.ModTime().UTC(), + LastUsedAt: configured.ConfiguredAt.UTC(), + Current: configured.Current, + Protections: []storage.Protection{{ + Kind: protectionKind, + Detail: protectionDetail, + }}, + } + if artifact.Current { + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionCurrent, Detail: "current configured generation"}) + } + artifacts = append(artifacts, artifact) + } + return artifacts, warnings +} + +func unknownConfiguredFile(configured ConfiguredFile, path string, cause error) storage.Artifact { + artifact := unknownEntryArtifact("configured-file", path, configured.Provider, false, configured.Kind, cause) + artifact.RetentionGroup = configured.Provider + ":" + configured.Role + artifact.Protections = append(artifact.Protections, storage.Protection{ + Kind: storage.ProtectionConfiguration, + Detail: "configured provider artifact could not be bound to an exact identity", + }) + return artifact +} diff --git a/internal/storage/inventory/doc.go b/internal/storage/inventory/doc.go new file mode 100644 index 0000000..e2ba2b4 --- /dev/null +++ b/internal/storage/inventory/doc.go @@ -0,0 +1,7 @@ +// Package inventory collects a deterministic, read-only inventory of known +// EPAR filesystem storage roots. +// +// Collection never invokes provider CLIs and never removes or rewrites host +// resources. Exact ownership is claimed only for artifacts that pass their +// complete recognition rules; ambiguous entries remain unknown and report-only. +package inventory diff --git a/internal/storage/inventory/filesystem.go b/internal/storage/inventory/filesystem.go new file mode 100644 index 0000000..e80153b --- /dev/null +++ b/internal/storage/inventory/filesystem.go @@ -0,0 +1,105 @@ +package inventory + +import ( + "crypto/sha256" + "encoding/hex" + "fmt" + "io" + "math" + "os" + "path/filepath" + "runtime" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +func hashFile(path string) (string, uint64, error) { + before, err := storage.SnapshotFilesystemTarget(path) + if err != nil { + return "", 0, err + } + if before.Kind != storage.TargetFile { + return "", 0, fmt.Errorf("%q is not a redirect-free regular file", path) + } + file, err := os.Open(path) + if err != nil { + return "", 0, err + } + defer file.Close() + hash := sha256.New() + written, err := io.Copy(hash, file) + if err != nil { + return "", 0, err + } + if written < 0 { + return "", 0, fmt.Errorf("negative byte count while hashing %q", path) + } + after, err := storage.SnapshotFilesystemTarget(path) + if err != nil { + return "", 0, err + } + if after != before { + return "", 0, fmt.Errorf("file identity or metadata drifted while hashing %q", path) + } + return "sha256:" + hex.EncodeToString(hash.Sum(nil)), uint64(written), nil +} + +func hashBytes(data []byte) string { + sum := sha256.Sum256(data) + return "sha256:" + hex.EncodeToString(sum[:]) +} + +func directoryBytes(path string) (uint64, error) { + var total uint64 + err := filepath.WalkDir(path, func(current string, entry os.DirEntry, walkErr error) error { + if walkErr != nil { + return walkErr + } + info, err := entry.Info() + if err != nil { + return err + } + if isRedirect(info) { + return fmt.Errorf("storage inventory path %q contains a symlink, junction, or reparse point", current) + } + if entry.IsDir() { + return nil + } + if !info.Mode().IsRegular() { + return fmt.Errorf("storage inventory path %q contains a non-regular file", current) + } + if info.Size() < 0 || uint64(info.Size()) > math.MaxUint64-total { + return fmt.Errorf("storage inventory byte count overflow at %q", current) + } + total += uint64(info.Size()) + return nil + }) + return total, err +} + +func stableID(prefix, value string) string { + sum := sha256.Sum256([]byte(value)) + return prefix + ":" + hex.EncodeToString(sum[:]) +} + +func samePath(left, right string) bool { + left = filepath.Clean(left) + right = filepath.Clean(right) + if runtime.GOOS == "windows" { + return strings.EqualFold(left, right) + } + return left == right +} + +func unknownTarget(path string, directory bool) storage.Target { + kind := storage.TargetFile + if directory { + kind = storage.TargetDirectory + } + absolute, err := filepath.Abs(path) + if err != nil { + absolute = path + } + return storage.Target{Kind: kind, Locator: filepath.Clean(absolute), Match: storage.MatchUnknown} +} diff --git a/internal/storage/inventory/filesystem_unix.go b/internal/storage/inventory/filesystem_unix.go new file mode 100644 index 0000000..c3ec548 --- /dev/null +++ b/internal/storage/inventory/filesystem_unix.go @@ -0,0 +1,7 @@ +//go:build !windows + +package inventory + +import "os" + +func isRedirect(info os.FileInfo) bool { return info.Mode()&os.ModeSymlink != 0 } diff --git a/internal/storage/inventory/filesystem_windows.go b/internal/storage/inventory/filesystem_windows.go new file mode 100644 index 0000000..f0618e8 --- /dev/null +++ b/internal/storage/inventory/filesystem_windows.go @@ -0,0 +1,16 @@ +//go:build windows + +package inventory + +import ( + "os" + "syscall" +) + +func isRedirect(info os.FileInfo) bool { + if info.Mode()&os.ModeSymlink != 0 { + return true + } + data, ok := info.Sys().(*syscall.Win32FileAttributeData) + return ok && data.FileAttributes&syscall.FILE_ATTRIBUTE_REPARSE_POINT != 0 +} diff --git a/internal/storage/inventory/logs.go b/internal/storage/inventory/logs.go new file mode 100644 index 0000000..587b1b8 --- /dev/null +++ b/internal/storage/inventory/logs.go @@ -0,0 +1,71 @@ +package inventory + +import ( + "fmt" + "os" + "path/filepath" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +func collectLogs(root, projectRoot string) ([]storage.Artifact, []string) { + target, exists, err := inspectOptionalRoot(root) + if !exists { + if err == nil { + return nil, nil + } + artifact := unknownRootArtifact("logs", root, "", err) + return []storage.Artifact{artifact}, []string{fmt.Sprintf("logs root was not inventoried safely: %v", err)} + } + info, err := os.Lstat(target.Locator) + if err != nil { + return nil, []string{fmt.Sprintf("logs root metadata is unavailable: %v", err)} + } + size, sizeErr := directoryBytes(target.Locator) + ownership := storage.Ownership{ + Kind: storage.OwnershipExact, + OwnerID: stableID("epar-logs", projectRoot), + Evidence: "explicit EPAR logging root", + } + protections := []storage.Protection{{Kind: storage.ProtectionOperator, Detail: "logging subsystem owns retention"}} + var warnings []string + if sizeErr != nil { + size = 0 + ownership = storage.Ownership{Kind: storage.OwnershipUnknown} + protections = append(protections, storage.Protection{Kind: storage.ProtectionUncertain, Detail: "unsafe or unreadable descendant"}) + warnings = append(warnings, fmt.Sprintf("logs root size and ownership are uncertain: %v", sizeErr)) + } + defaultRoot := filepath.Join(projectRoot, "work", "logs") + if !samePath(target.Locator, defaultRoot) { + protections = append(protections, storage.Protection{Kind: storage.ProtectionCustomRoot, Detail: "configured logging root"}) + } + return []storage.Artifact{{ + ID: stableID("logs", target.Locator), + SurfaceID: ProjectSurfaceID, + Kind: storage.ArtifactOther, + Target: target, + Ownership: ownership, + SizeBytes: size, + CreatedAt: info.ModTime().UTC(), + Protections: protections, + }}, warnings +} + +func unknownRootArtifact(kind, root, provider string, cause error) storage.Artifact { + detail := "root identity is unsafe or unreadable" + if cause != nil { + detail = cause.Error() + } + return storage.Artifact{ + ID: stableID(kind+"-unknown", root), + Provider: provider, + SurfaceID: ProjectSurfaceID, + Kind: storage.ArtifactOther, + Target: unknownTarget(root, true), + Ownership: storage.Ownership{Kind: storage.OwnershipUnknown}, + Protections: []storage.Protection{{ + Kind: storage.ProtectionUncertain, + Detail: detail, + }}, + } +} diff --git a/internal/storage/inventory/logs_test.go b/internal/storage/inventory/logs_test.go new file mode 100644 index 0000000..fa58e1a --- /dev/null +++ b/internal/storage/inventory/logs_test.go @@ -0,0 +1,40 @@ +package inventory + +import ( + "os" + "path/filepath" + "runtime" + "testing" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +func TestCollectLogsProtectsSubsystemOwnedRoot(t *testing.T) { + t.Parallel() + project := t.TempDir() + root := filepath.Join(project, "work", "logs") + mustWriteFile(t, filepath.Join(root, "instances", "runner.log"), []byte("runner")) + artifacts, warnings := collectLogs(root, project) + if len(warnings) != 0 || len(artifacts) != 1 { + t.Fatalf("collectLogs() artifacts=%+v warnings=%v", artifacts, warnings) + } + if artifacts[0].Ownership.Kind != storage.OwnershipExact || artifacts[0].SizeBytes != uint64(len("runner")) || !hasProtection(artifacts[0], storage.ProtectionOperator) { + t.Fatalf("logs artifact = %+v", artifacts[0]) + } +} + +func TestCollectLogsRejectsRedirectedDescendant(t *testing.T) { + t.Parallel() + project := t.TempDir() + root := filepath.Join(project, "work", "logs") + mustMkdirAll(t, root) + outside := t.TempDir() + link := filepath.Join(root, "redirect") + if err := os.Symlink(outside, link); err != nil { + t.Skipf("symlink creation unavailable on %s: %v", runtime.GOOS, err) + } + artifacts, warnings := collectLogs(root, project) + if len(warnings) != 1 || len(artifacts) != 1 || artifacts[0].Ownership.Kind != storage.OwnershipUnknown || !hasProtection(artifacts[0], storage.ProtectionUncertain) { + t.Fatalf("collectLogs() artifacts=%+v warnings=%v", artifacts, warnings) + } +} diff --git a/internal/storage/inventory/native.go b/internal/storage/inventory/native.go new file mode 100644 index 0000000..6834685 --- /dev/null +++ b/internal/storage/inventory/native.go @@ -0,0 +1,416 @@ +package inventory + +import ( + "bufio" + "errors" + "fmt" + "os" + "path/filepath" + "regexp" + "strconv" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +var ( + cacheKeyPattern = regexp.MustCompile(`^[0-9a-f]{64}$`) + leasePattern = regexp.MustCompile(`^lease(?:[.-]).+$`) +) + +type nativeOptions struct { + Root string + CurrentExecutable string + CurrentRevision string +} + +type nativeRevision struct { + artifact storage.Artifact + key string + completed time.Time + binary storage.Target +} + +func collectNative(options nativeOptions) ([]storage.Artifact, []string) { + rootTarget, exists, err := inspectOptionalRoot(options.Root) + if !exists { + if err == nil { + return nil, nil + } + artifact := unknownRootArtifact("native-controller", options.Root, "", err) + return []storage.Artifact{artifact}, []string{fmt.Sprintf("native-controller root was not inventoried safely: %v", err)} + } + entries, err := os.ReadDir(rootTarget.Locator) + if err != nil { + artifact := unknownRootArtifact("native-controller", rootTarget.Locator, "", err) + return []storage.Artifact{artifact}, []string{fmt.Sprintf("native-controller root is unreadable: %v", err)} + } + + currentKey, keyWarning := normalizeCurrentRevision(options.CurrentRevision) + var currentExecutable storage.Target + var warnings []string + if keyWarning != "" { + warnings = append(warnings, keyWarning) + } + if options.CurrentExecutable != "" { + currentExecutable, err = storage.SnapshotFilesystemTarget(options.CurrentExecutable) + if err != nil || currentExecutable.Kind != storage.TargetFile { + warnings = append(warnings, fmt.Sprintf("current native-controller executable is not an exact regular file: %v", err)) + currentExecutable = storage.Target{} + } + } + + var revisions []nativeRevision + var artifacts []storage.Artifact + stable, stableNames, stableCurrentMatch, stableFound, stableErr := inspectStableNativeController(rootTarget.Locator, currentExecutable) + if stableFound && stableErr == nil { + artifacts = append(artifacts, stable) + } else if stableFound { + warnings = append(warnings, fmt.Sprintf("stable native-controller files remain ownership-unknown: %v", stableErr)) + } + for _, entry := range entries { + if stableErr == nil && stableNames[entry.Name()] { + continue + } + path := filepath.Join(rootTarget.Locator, entry.Name()) + info, infoErr := entry.Info() + if infoErr != nil || isRedirect(info) { + artifact := unknownEntryArtifact("native-controller", path, "", entry.IsDir(), storage.ArtifactOther, infoErr) + artifacts = append(artifacts, artifact) + warnings = append(warnings, fmt.Sprintf("native-controller entry %q is unsafe or unreadable", entry.Name())) + continue + } + if !entry.IsDir() || !cacheKeyPattern.MatchString(entry.Name()) { + artifact := unknownEntryArtifact("native-controller", path, "", entry.IsDir(), storage.ArtifactOther, nil) + artifacts = append(artifacts, artifact) + continue + } + revision, revisionErr := inspectNativeRevision(path, entry.Name()) + if revisionErr != nil { + artifact := unknownEntryArtifact("native-controller-revision", path, "", true, storage.ArtifactNativeControllerRevision, revisionErr) + if currentKey == entry.Name() { + artifact.Current = true + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionCurrent, Detail: "configured current revision has uncertain ownership"}) + } + artifacts = append(artifacts, artifact) + warnings = append(warnings, fmt.Sprintf("native-controller revision %q remains ownership-unknown: %v", entry.Name(), revisionErr)) + continue + } + revisions = append(revisions, revision) + } + + var currentIndexes []int + for index, revision := range revisions { + matchesKey := currentKey != "" && revision.key == currentKey + matchesExecutable := currentExecutable.Identity != "" && revision.binary.Identity == currentExecutable.Identity && revision.binary.Fingerprint == currentExecutable.Fingerprint + if matchesKey || matchesExecutable { + currentIndexes = append(currentIndexes, index) + } + } + if len(currentIndexes) > 1 { + warnings = append(warnings, "native-controller current identity is ambiguous; all recognized revisions are protected") + for index := range revisions { + revisions[index].artifact.Protections = append(revisions[index].artifact.Protections, storage.Protection{Kind: storage.ProtectionUncertain, Detail: "ambiguous current revision"}) + } + } else if len(currentIndexes) == 1 { + currentIndex := currentIndexes[0] + current := revisions[currentIndex] + revisions[currentIndex].artifact.Current = true + revisions[currentIndex].artifact.Protections = append(revisions[currentIndex].artifact.Protections, storage.Protection{Kind: storage.ProtectionCurrent, Detail: "explicit current native-controller identity"}) + for index := range revisions { + if index == currentIndex || !revisions[index].completed.Before(current.completed) { + continue + } + supersededAt := current.completed + revisions[index].artifact.SupersededAt = &supersededAt + } + } else if (currentKey != "" || currentExecutable.Identity != "") && !stableCurrentMatch { + warnings = append(warnings, "explicit current native-controller identity did not match a recognized revision") + } + for _, revision := range revisions { + artifacts = append(artifacts, revision.artifact) + } + return artifacts, warnings +} + +func inspectStableNativeController(root string, currentExecutable storage.Target) (storage.Artifact, map[string]bool, bool, bool, error) { + manifestPath := filepath.Join(root, "ephemeral-action-runner.manifest") + manifestInfo, err := os.Lstat(manifestPath) + if errors.Is(err, os.ErrNotExist) { + return storage.Artifact{}, nil, false, false, nil + } + if err != nil { + return storage.Artifact{}, nil, false, true, err + } + if !manifestInfo.Mode().IsRegular() || isRedirect(manifestInfo) { + return storage.Artifact{}, nil, false, true, fmt.Errorf("manifest is not an exact regular file") + } + fields, err := parseStableManifest(manifestPath) + if err != nil { + return storage.Artifact{}, nil, false, true, err + } + executableName := fields["executable"] + if executableName != "ephemeral-action-runner" && executableName != "ephemeral-action-runner.exe" { + return storage.Artifact{}, nil, false, true, fmt.Errorf("manifest executable is invalid") + } + binaryPath := filepath.Join(root, executableName) + binaryInfo, err := os.Lstat(binaryPath) + if err != nil { + return storage.Artifact{}, nil, false, true, err + } + if !binaryInfo.Mode().IsRegular() || isRedirect(binaryInfo) { + return storage.Artifact{}, nil, false, true, fmt.Errorf("controller executable is not an exact regular file") + } + binary, err := storage.SnapshotFilesystemTarget(binaryPath) + if err != nil { + return storage.Artifact{}, nil, false, true, err + } + manifestSHA, _, err := hashFile(manifestPath) + if err != nil { + return storage.Artifact{}, nil, false, true, err + } + completed, err := time.Parse(time.RFC3339Nano, fields["completedAtUtc"]) + if err != nil { + return storage.Artifact{}, nil, false, true, fmt.Errorf("manifest completedAtUtc is invalid") + } + size := uint64(binaryInfo.Size() + manifestInfo.Size()) + artifact := storage.Artifact{ + ID: "native-controller-stable:" + fields["fingerprint"], + SurfaceID: ProjectSurfaceID, + Kind: storage.ArtifactNativeControllerRevision, + RetentionGroup: "native-controller", + Target: binary, + Ownership: storage.Ownership{ + Kind: storage.OwnershipExact, + OwnerID: "native-controller:" + fields["fingerprint"], + Evidence: "ephemeral-action-runner.manifest@" + manifestSHA, + }, + SizeBytes: size, + CreatedAt: completed.UTC(), + Current: true, + Protections: []storage.Protection{{ + Kind: storage.ProtectionCurrent, Detail: "stable native-controller manifest", + }}, + } + currentMatch := currentExecutable.Identity != "" && binary.Identity == currentExecutable.Identity && binary.Fingerprint == currentExecutable.Fingerprint + stableNames := map[string]bool{executableName: true, filepath.Base(manifestPath): true} + lockPath := filepath.Join(root, ".native-controller.lock") + if lockInfo, lockErr := os.Lstat(lockPath); lockErr == nil && lockInfo.Mode().IsRegular() && !isRedirect(lockInfo) && lockInfo.Size() == 0 { + stableNames[filepath.Base(lockPath)] = true + } + return artifact, stableNames, currentMatch, true, nil +} + +func parseStableManifest(path string) (map[string]string, error) { + file, err := os.Open(path) + if err != nil { + return nil, err + } + defer file.Close() + fields := make(map[string]string, 6) + scanner := bufio.NewScanner(file) + scanner.Buffer(make([]byte, 1024), 16*1024) + for scanner.Scan() { + key, value, ok := strings.Cut(scanner.Text(), "=") + if !ok || key == "" || value == "" { + return nil, fmt.Errorf("manifest contains an invalid field") + } + if _, exists := fields[key]; exists { + return nil, fmt.Errorf("manifest contains duplicate field %q", key) + } + fields[key] = value + } + if err := scanner.Err(); err != nil { + return nil, err + } + if len(fields) != 6 { + return nil, fmt.Errorf("manifest must contain exactly six fields") + } + for _, key := range []string{"schemaVersion", "fingerprint", "executable", "toolchainImageID", "sourceRevision", "completedAtUtc"} { + if fields[key] == "" { + return nil, fmt.Errorf("manifest is missing %q", key) + } + } + if fields["schemaVersion"] != "2" || !cacheKeyPattern.MatchString(fields["fingerprint"]) { + return nil, fmt.Errorf("manifest schema or fingerprint is invalid") + } + sourceFingerprint := strings.TrimPrefix(strings.TrimPrefix(fields["sourceRevision"], "dirty:"), "sha256:") + sourceMatches := fields["sourceRevision"] == "unknown" || sourceFingerprint == fields["fingerprint"] + if !sourceMatches || !strings.HasPrefix(fields["toolchainImageID"], "sha256:") || !cacheKeyPattern.MatchString(strings.TrimPrefix(fields["toolchainImageID"], "sha256:")) { + return nil, fmt.Errorf("manifest source or toolchain identity is invalid") + } + return fields, nil +} + +func inspectNativeRevision(path, cacheKey string) (nativeRevision, error) { + entries, err := os.ReadDir(path) + if err != nil { + return nativeRevision{}, err + } + var manifestPath, executableName string + var leaseNames []string + for _, entry := range entries { + info, err := entry.Info() + if err != nil { + return nativeRevision{}, err + } + if entry.IsDir() || !info.Mode().IsRegular() || isRedirect(info) { + return nativeRevision{}, fmt.Errorf("contains a directory, special file, or redirect at %q", entry.Name()) + } + switch { + case entry.Name() == "controller-cache.manifest": + manifestPath = filepath.Join(path, entry.Name()) + case entry.Name() == "ephemeral-action-runner" || entry.Name() == "ephemeral-action-runner.exe": + if executableName != "" { + return nativeRevision{}, fmt.Errorf("contains multiple controller executables") + } + executableName = entry.Name() + case leasePattern.MatchString(entry.Name()): + leaseNames = append(leaseNames, entry.Name()) + default: + return nativeRevision{}, fmt.Errorf("contains unexpected file %q", entry.Name()) + } + } + if manifestPath == "" || executableName == "" { + return nativeRevision{}, fmt.Errorf("requires one manifest and one controller executable") + } + fields, err := parseManifest(manifestPath) + if err != nil { + return nativeRevision{}, err + } + if fields["schemaVersion"] != "1" || fields["cacheKey"] != cacheKey || fields["executable"] != executableName { + return nativeRevision{}, fmt.Errorf("manifest identity does not match the revision directory") + } + var completed time.Time + if value, exists := fields["completedAtUnix"]; exists { + completedUnix, err := strconv.ParseInt(value, 10, 64) + if err != nil || completedUnix <= 0 { + return nativeRevision{}, fmt.Errorf("manifest completedAtUnix is invalid") + } + completed = time.Unix(completedUnix, 0).UTC() + } else { + completed, err = time.Parse(time.RFC3339Nano, fields["completedAtUtc"]) + if err != nil { + return nativeRevision{}, fmt.Errorf("manifest completedAtUtc is invalid") + } + completed = completed.UTC() + } + target, err := storage.SnapshotFilesystemTarget(path) + if err != nil { + return nativeRevision{}, err + } + binary, err := storage.SnapshotFilesystemTarget(filepath.Join(path, executableName)) + if err != nil { + return nativeRevision{}, err + } + manifestSHA, _, err := hashFile(manifestPath) + if err != nil { + return nativeRevision{}, err + } + size, err := directoryBytes(path) + if err != nil { + return nativeRevision{}, err + } + artifact := storage.Artifact{ + ID: "native-controller:" + cacheKey, + SurfaceID: ProjectSurfaceID, + Kind: storage.ArtifactNativeControllerRevision, + RetentionGroup: "native-controller", + Target: target, + Ownership: storage.Ownership{ + Kind: storage.OwnershipExact, + OwnerID: "native-controller:" + cacheKey, + Evidence: "controller-cache.manifest@" + manifestSHA, + }, + SizeBytes: size, + CreatedAt: completed, + } + if len(leaseNames) > 0 { + artifact.Protections = append(artifact.Protections, storage.Protection{Kind: storage.ProtectionLease, Detail: "revision contains one or more unexpired-status-unknown leases"}) + } + return nativeRevision{artifact: artifact, key: cacheKey, completed: completed, binary: binary}, nil +} + +func parseManifest(path string) (map[string]string, error) { + file, err := os.Open(path) + if err != nil { + return nil, err + } + defer file.Close() + fields := make(map[string]string, 4) + scanner := bufio.NewScanner(file) + scanner.Buffer(make([]byte, 1024), 16*1024) + for scanner.Scan() { + line := scanner.Text() + key, value, ok := strings.Cut(line, "=") + if !ok || key == "" || value == "" { + return nil, fmt.Errorf("manifest contains an invalid field") + } + if _, exists := fields[key]; exists { + return nil, fmt.Errorf("manifest contains duplicate field %q", key) + } + fields[key] = value + } + if err := scanner.Err(); err != nil { + return nil, err + } + if len(fields) != 4 { + return nil, fmt.Errorf("manifest must contain exactly four fields") + } + for _, key := range []string{"schemaVersion", "cacheKey", "executable"} { + if _, exists := fields[key]; !exists { + return nil, fmt.Errorf("manifest is missing %q", key) + } + } + _, hasUnix := fields["completedAtUnix"] + _, hasUTC := fields["completedAtUtc"] + if hasUnix == hasUTC { + return nil, fmt.Errorf("manifest must contain exactly one completion timestamp") + } + return fields, nil +} + +func normalizeCurrentRevision(value string) (string, string) { + if value == "" { + return "", "" + } + value = strings.TrimPrefix(value, "dirty:") + value = strings.TrimPrefix(value, "sha256:") + if !cacheKeyPattern.MatchString(value) { + return "", fmt.Sprintf("current native-controller revision %q is not an exact 64-character SHA-256 identity", value) + } + return value, "" +} + +func unknownEntryArtifact(prefix, path, provider string, directory bool, kind storage.ArtifactKind, cause error) storage.Artifact { + target, err := storage.SnapshotFilesystemTarget(path) + if err != nil { + target = unknownTarget(path, directory) + } + var size uint64 + if target.Match == storage.MatchExact { + if directory { + size, _ = directoryBytes(path) + } else if info, statErr := os.Lstat(path); statErr == nil && info.Size() >= 0 { + size = uint64(info.Size()) + } + } + detail := "entry does not satisfy an exact EPAR ownership schema" + if cause != nil { + detail = cause.Error() + } + return storage.Artifact{ + ID: stableID(prefix+"-unknown", path), + Provider: provider, + SurfaceID: ProjectSurfaceID, + Kind: kind, + Target: target, + Ownership: storage.Ownership{Kind: storage.OwnershipUnknown}, + SizeBytes: size, + Protections: []storage.Protection{{ + Kind: storage.ProtectionUncertain, + Detail: detail, + }}, + } +} diff --git a/internal/storage/inventory/native_test.go b/internal/storage/inventory/native_test.go new file mode 100644 index 0000000..d101b38 --- /dev/null +++ b/internal/storage/inventory/native_test.go @@ -0,0 +1,164 @@ +package inventory + +import ( + "fmt" + "os" + "path/filepath" + "runtime" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +func TestCollectNativeRecognitionLeaseAndMalformedManifest(t *testing.T) { + t.Parallel() + root := t.TempDir() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + validKey := repeatedHex("4") + writeNativeRevision(t, root, validKey, now.Add(-10*24*time.Hour), true) + badKey := repeatedHex("5") + badDirectory := filepath.Join(root, badKey) + mustMkdirAll(t, badDirectory) + mustWriteFile(t, filepath.Join(badDirectory, "ephemeral-action-runner"), []byte("binary")) + mustWriteFile(t, filepath.Join(badDirectory, "controller-cache.manifest"), []byte("schemaVersion=1\ncacheKey="+badKey+"\ncacheKey="+badKey+"\nexecutable=ephemeral-action-runner\ncompletedAtUnix=1\n")) + + artifacts, warnings := collectNative(nativeOptions{Root: root}) + if len(artifacts) != 2 || len(warnings) != 1 { + t.Fatalf("collectNative() artifacts=%+v warnings=%v", artifacts, warnings) + } + valid := findArtifact(t, artifacts, "native-controller:"+validKey) + if valid.Ownership.Kind != storage.OwnershipExact || !hasProtection(valid, storage.ProtectionLease) { + t.Fatalf("valid leased revision = %+v", valid) + } + bad := findArtifactByLocatorSuffix(t, artifacts, badKey) + if bad.Ownership.Kind != storage.OwnershipUnknown || !hasProtection(bad, storage.ProtectionUncertain) { + t.Fatalf("malformed revision = %+v", bad) + } +} + +func TestCollectNativeCurrentExecutableIdentity(t *testing.T) { + t.Parallel() + root := t.TempDir() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + oldKey := repeatedHex("6") + currentKey := repeatedHex("7") + writeNativeRevision(t, root, oldKey, now.Add(-20*24*time.Hour), false) + currentExecutable := writeNativeRevision(t, root, currentKey, now.Add(-10*24*time.Hour), false) + artifacts, warnings := collectNative(nativeOptions{Root: root, CurrentExecutable: currentExecutable}) + if len(warnings) != 0 { + t.Fatalf("collectNative() warnings = %v", warnings) + } + current := findArtifact(t, artifacts, "native-controller:"+currentKey) + old := findArtifact(t, artifacts, "native-controller:"+oldKey) + if !current.Current || old.SupersededAt == nil || !old.SupersededAt.Equal(now.Add(-10*24*time.Hour)) { + t.Fatalf("current=%+v old=%+v", current, old) + } +} + +func TestCollectNativeRecognizesStableControllerLayout(t *testing.T) { + t.Parallel() + root := t.TempDir() + fingerprint := repeatedHex("d") + toolchainID := repeatedHex("e") + executableName := "ephemeral-action-runner" + if runtime.GOOS == "windows" { + executableName += ".exe" + } + executable := filepath.Join(root, executableName) + mustWriteFile(t, executable, []byte("stable-binary")) + completed := time.Date(2026, 7, 29, 12, 0, 0, 0, time.UTC) + manifest := fmt.Sprintf("schemaVersion=2\nfingerprint=%s\nexecutable=%s\ntoolchainImageID=sha256:%s\nsourceRevision=dirty:sha256:%s\ncompletedAtUtc=%s\n", fingerprint, executableName, toolchainID, fingerprint, completed.Format(time.RFC3339Nano)) + mustWriteFile(t, filepath.Join(root, "ephemeral-action-runner.manifest"), []byte(manifest)) + mustWriteFile(t, filepath.Join(root, ".native-controller.lock"), nil) + artifacts, warnings := collectNative(nativeOptions{Root: root, CurrentExecutable: executable}) + if len(warnings) != 0 || len(artifacts) != 1 { + t.Fatalf("collectNative() artifacts=%+v warnings=%v", artifacts, warnings) + } + expectedTarget, err := storage.SnapshotFilesystemTarget(executable) + if err != nil { + t.Fatal(err) + } + stable := findArtifact(t, artifacts, "native-controller-stable:"+fingerprint) + if !stable.Current || stable.Ownership.Kind != storage.OwnershipExact || !hasProtection(stable, storage.ProtectionCurrent) || stable.Target.Locator != expectedTarget.Locator { + t.Fatalf("stable native controller = %+v", stable) + } +} + +func TestCollectNativeRejectsSymlinkedRevision(t *testing.T) { + t.Parallel() + root := t.TempDir() + outside := t.TempDir() + link := filepath.Join(root, repeatedHex("8")) + if err := os.Symlink(outside, link); err != nil { + t.Skipf("symlink creation unavailable on %s: %v", runtime.GOOS, err) + } + artifacts, warnings := collectNative(nativeOptions{Root: root}) + if len(artifacts) != 1 || len(warnings) != 1 || artifacts[0].Ownership.Kind != storage.OwnershipUnknown || artifacts[0].Target.Match != storage.MatchUnknown { + t.Fatalf("collectNative() symlink artifacts=%+v warnings=%v", artifacts, warnings) + } +} + +func TestParseManifestRejectsUnknownAndMissingFields(t *testing.T) { + t.Parallel() + path := filepath.Join(t.TempDir(), "manifest") + for name, content := range map[string]string{ + "unknown": "schemaVersion=1\ncacheKey=" + repeatedHex("9") + "\nexecutable=ephemeral-action-runner\ncompletedAtUnix=1\nother=value\n", + "missing": "schemaVersion=1\ncacheKey=" + repeatedHex("9") + "\nexecutable=ephemeral-action-runner\n", + } { + t.Run(name, func(t *testing.T) { + mustWriteFile(t, path, []byte(content)) + if _, err := parseManifest(path); err == nil { + t.Fatalf("parseManifest() accepted %s fields", name) + } + }) + } +} + +func TestInspectNativeRevisionAcceptsWindowsUTCManifest(t *testing.T) { + t.Parallel() + root := t.TempDir() + key := repeatedHex("b") + directory := filepath.Join(root, key) + mustMkdirAll(t, directory) + mustWriteFile(t, filepath.Join(directory, "ephemeral-action-runner.exe"), []byte("binary")) + completed := time.Date(2026, 7, 27, 12, 0, 0, 123, time.UTC) + manifest := fmt.Sprintf("schemaVersion=1\ncacheKey=%s\nexecutable=ephemeral-action-runner.exe\ncompletedAtUtc=%s\n", key, completed.Format(time.RFC3339Nano)) + mustWriteFile(t, filepath.Join(directory, "controller-cache.manifest"), []byte(manifest)) + revision, err := inspectNativeRevision(directory, key) + if err != nil { + t.Fatalf("inspectNativeRevision() error = %v", err) + } + if !revision.completed.Equal(completed) || revision.artifact.Ownership.Kind != storage.OwnershipExact { + t.Fatalf("inspectNativeRevision() = %+v", revision) + } +} + +func TestNormalizeCurrentRevisionAcceptsInjectedSourceRevisionForms(t *testing.T) { + t.Parallel() + key := repeatedHex("c") + for _, value := range []string{key, "sha256:" + key, "dirty:sha256:" + key} { + got, warning := normalizeCurrentRevision(value) + if got != key || warning != "" { + t.Fatalf("normalizeCurrentRevision(%q) = %q, %q", value, got, warning) + } + } +} + +func writeNativeRevision(t *testing.T, root, key string, completed time.Time, lease bool) string { + t.Helper() + directory := filepath.Join(root, key) + mustMkdirAll(t, directory) + executable := "ephemeral-action-runner" + if runtime.GOOS == "windows" { + executable += ".exe" + } + binary := filepath.Join(directory, executable) + mustWriteFile(t, binary, []byte("binary:"+key)) + manifest := fmt.Sprintf("schemaVersion=1\ncacheKey=%s\nexecutable=%s\ncompletedAtUnix=%d\n", key, executable, completed.Unix()) + mustWriteFile(t, filepath.Join(directory, "controller-cache.manifest"), []byte(manifest)) + if lease { + mustWriteFile(t, filepath.Join(directory, "lease.123.abcdef"), []byte("schemaVersion=1\n")) + } + return binary +} diff --git a/internal/storage/inventory/template.go b/internal/storage/inventory/template.go new file mode 100644 index 0000000..231622b --- /dev/null +++ b/internal/storage/inventory/template.go @@ -0,0 +1,340 @@ +package inventory + +import ( + "bytes" + "encoding/json" + "errors" + "fmt" + "io" + "os" + "path/filepath" + "regexp" + "sort" + "strings" + + "github.com/solutionforest/ephemeral-action-runner/internal/config" + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +const ( + maximumTemplateMetadataBytes = 4 << 20 + templateMetadataSchema = 4 +) + +var ( + templateDigestPattern = regexp.MustCompile(`^sha256:[0-9a-f]{64}$`) + templateCacheIDPattern = regexp.MustCompile(`^[0-9a-f]{12}$`) + templateTagPattern = regexp.MustCompile(`^(?:docker\.io/library/)?epar-docker-sandboxes-[a-z0-9._-]+:[a-z0-9._-]+$`) + templateProfilePattern = regexp.MustCompile(`^[a-z0-9][a-z0-9._-]*$`) + templateArchivePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9._-]*\.tar$`) +) + +type templateOptions struct { + Root string + Selections []TemplateSelection + Protections []TemplateProtection +} + +type templateMetadata struct { + SchemaVersion int `json:"schemaVersion"` + Profile string `json:"profile"` + Platform string `json:"platform"` + Template struct { + Tag string `json:"tag"` + Digest string `json:"digest"` + CacheID string `json:"cacheID"` + RootDisk string `json:"rootDisk"` + Archive string `json:"archive"` + ArchiveSHA256 string `json:"archiveSha256"` + ArchiveBytes uint64 `json:"archiveBytes"` + } `json:"template"` + Compatibility struct { + TemplateSchemaVersion int `json:"templateSchemaVersion"` + RunnerExecution string `json:"runnerExecution"` + DockerDaemonOwner string `json:"dockerDaemonOwner"` + ExpectedDockerDaemonCount int `json:"expectedDockerDaemonCount"` + } `json:"compatibility"` +} + +type templateRecord struct { + artifact storage.Artifact + metadata templateMetadata + metadataSHA256 string +} + +func collectTemplates(options templateOptions) ([]storage.Artifact, []string, error) { + if err := validateTemplateInputs(options); err != nil { + return nil, nil, err + } + rootTarget, exists, err := inspectOptionalRoot(options.Root) + if !exists { + if err == nil { + return nil, nil, nil + } + artifact := unknownRootArtifact("template-archive", options.Root, ProviderDockerSandboxes, err) + return []storage.Artifact{artifact}, []string{fmt.Sprintf("Docker Sandboxes template root was not inventoried safely: %v", err)}, nil + } + entries, err := os.ReadDir(rootTarget.Locator) + if err != nil { + artifact := unknownRootArtifact("template-archive", rootTarget.Locator, ProviderDockerSandboxes, err) + return []storage.Artifact{artifact}, []string{fmt.Sprintf("Docker Sandboxes template root is unreadable: %v", err)}, nil + } + protections := make(map[string][]storage.Protection) + for _, protection := range options.Protections { + protections[protection.ArchiveSHA256] = append(protections[protection.ArchiveSHA256], storage.Protection{Kind: protection.Kind, Detail: protection.Detail}) + } + var records []templateRecord + var artifacts []storage.Artifact + var warnings []string + for _, entry := range entries { + path := filepath.Join(rootTarget.Locator, entry.Name()) + info, infoErr := entry.Info() + if infoErr != nil || isRedirect(info) || !entry.IsDir() { + artifact := unknownEntryArtifact("template-archive", path, ProviderDockerSandboxes, entry.IsDir(), storage.ArtifactOther, infoErr) + artifacts = append(artifacts, artifact) + warnings = append(warnings, fmt.Sprintf("Docker Sandboxes template entry %q is not a safe artifact directory", entry.Name())) + continue + } + record, unknown, inspectErr := inspectTemplateDirectory(path) + if inspectErr != nil { + artifacts = append(artifacts, unknown) + warnings = append(warnings, fmt.Sprintf("Docker Sandboxes template directory %q remains ownership-unknown: %v", entry.Name(), inspectErr)) + continue + } + record.artifact.Protections = append(record.artifact.Protections, protections[record.metadata.Template.ArchiveSHA256]...) + records = append(records, record) + } + for index := range records { + sortTemplateProtections(records[index].artifact.Protections) + record := records[index] + artifacts = append(artifacts, record.artifact) + } + return artifacts, warnings, nil +} + +func inspectTemplateDirectory(path string) (templateRecord, storage.Artifact, error) { + metadataPath := filepath.Join(path, "template-metadata.json") + metadataTarget, err := storage.SnapshotFilesystemTarget(metadataPath) + if err != nil || metadataTarget.Kind != storage.TargetFile { + cause := err + if cause == nil { + cause = fmt.Errorf("template metadata is not a regular file") + } + unknown := unknownEntryArtifact("template-directory", path, ProviderDockerSandboxes, true, storage.ArtifactOther, cause) + return templateRecord{}, unknown, fmt.Errorf("template metadata is missing or unsafe: %w", cause) + } + data, err := readBoundedFile(metadataTarget.Locator, maximumTemplateMetadataBytes) + if err != nil { + unknown := unknownEntryArtifact("template-directory", path, ProviderDockerSandboxes, true, storage.ArtifactOther, err) + return templateRecord{}, unknown, err + } + if err := validateUniqueJSON(data); err != nil { + unknown := unknownEntryArtifact("template-directory", path, ProviderDockerSandboxes, true, storage.ArtifactOther, err) + return templateRecord{}, unknown, fmt.Errorf("invalid template metadata JSON: %w", err) + } + var metadata templateMetadata + if err := json.Unmarshal(data, &metadata); err != nil { + unknown := unknownEntryArtifact("template-directory", path, ProviderDockerSandboxes, true, storage.ArtifactOther, err) + return templateRecord{}, unknown, fmt.Errorf("decode template metadata: %w", err) + } + if err := validateTemplateMetadata(metadata); err != nil { + unknown := unknownEntryArtifact("template-directory", path, ProviderDockerSandboxes, true, storage.ArtifactOther, err) + return templateRecord{}, unknown, err + } + metadataSHA256, _, err := hashFile(metadataTarget.Locator) + if err != nil { + unknown := unknownEntryArtifact("template-directory", path, ProviderDockerSandboxes, true, storage.ArtifactOther, err) + return templateRecord{}, unknown, err + } + if metadataSHA256 != hashBytes(data) { + err := fmt.Errorf("template metadata drifted while being inspected") + unknown := unknownEntryArtifact("template-directory", path, ProviderDockerSandboxes, true, storage.ArtifactOther, err) + return templateRecord{}, unknown, err + } + archivePath := filepath.Join(path, metadata.Template.Archive) + archiveTarget, targetErr := storage.SnapshotFilesystemTarget(archivePath) + if targetErr != nil || archiveTarget.Kind != storage.TargetFile { + cause := targetErr + if cause == nil { + cause = fmt.Errorf("template archive is not a regular file") + } + unknown := unknownEntryArtifact("template-archive", archivePath, ProviderDockerSandboxes, false, storage.ArtifactTemplateArchive, cause) + return templateRecord{}, unknown, fmt.Errorf("template archive is missing or unsafe: %w", cause) + } + actualSHA256, actualBytes, hashErr := hashFile(archiveTarget.Locator) + info, statErr := os.Lstat(archiveTarget.Locator) + if hashErr != nil || statErr != nil || actualSHA256 != metadata.Template.ArchiveSHA256 || actualBytes != metadata.Template.ArchiveBytes { + cause := fmt.Errorf("archive integrity mismatch: expected sha256=%s bytes=%d, got sha256=%s bytes=%d", metadata.Template.ArchiveSHA256, metadata.Template.ArchiveBytes, actualSHA256, actualBytes) + if hashErr != nil { + cause = hashErr + } else if statErr != nil { + cause = statErr + } + unknown := unknownEntryArtifact("template-archive", archiveTarget.Locator, ProviderDockerSandboxes, false, storage.ArtifactTemplateArchive, cause) + unknown.SizeBytes = actualBytes + return templateRecord{}, unknown, cause + } + artifact := storage.Artifact{ + ID: stableID("template-archive", archiveTarget.Locator+"@"+actualSHA256), + Provider: ProviderDockerSandboxes, + SurfaceID: ProjectSurfaceID, + Kind: storage.ArtifactTemplateArchive, + RetentionGroup: metadata.Profile + "/" + metadata.Platform, + Target: archiveTarget, + Ownership: storage.Ownership{ + Kind: storage.OwnershipExact, + OwnerID: "template-archive:" + actualSHA256, + Evidence: "template-metadata.json@" + metadataSHA256, + }, + SizeBytes: actualBytes, + CreatedAt: info.ModTime().UTC(), + } + return templateRecord{artifact: artifact, metadata: metadata, metadataSHA256: metadataSHA256}, storage.Artifact{}, nil +} + +func validateTemplateMetadata(metadata templateMetadata) error { + if metadata.SchemaVersion != templateMetadataSchema { + return fmt.Errorf("template metadata schemaVersion must be %d", templateMetadataSchema) + } + if !templateProfilePattern.MatchString(metadata.Profile) { + return fmt.Errorf("template metadata profile is invalid") + } + if metadata.Platform != "linux/amd64" && metadata.Platform != "linux/arm64" { + return fmt.Errorf("template metadata platform is invalid") + } + if !templateTagPattern.MatchString(metadata.Template.Tag) || !templateDigestPattern.MatchString(metadata.Template.Digest) || !templateCacheIDPattern.MatchString(metadata.Template.CacheID) { + return fmt.Errorf("template metadata contains an invalid tag, digest, or cache ID") + } + if metadata.Template.CacheID != strings.TrimPrefix(metadata.Template.Digest, "sha256:")[:12] { + return fmt.Errorf("template metadata cache ID does not match image digest") + } + rootDisk, err := config.ParseByteSize(metadata.Template.RootDisk) + if err != nil || rootDisk < int64(20*storage.GiB) { + return fmt.Errorf("template metadata root disk is invalid") + } + if !templateArchivePattern.MatchString(metadata.Template.Archive) || filepath.Base(metadata.Template.Archive) != metadata.Template.Archive { + return fmt.Errorf("template metadata archive is not an exact basename") + } + if !templateDigestPattern.MatchString(metadata.Template.ArchiveSHA256) || metadata.Template.ArchiveBytes == 0 { + return fmt.Errorf("template metadata archive digest or size is invalid") + } + if metadata.Compatibility.TemplateSchemaVersion != 1 || metadata.Compatibility.RunnerExecution != "direct-actions-listener" || metadata.Compatibility.DockerDaemonOwner != "docker-sandboxes-runtime" || metadata.Compatibility.ExpectedDockerDaemonCount != 1 { + return fmt.Errorf("template metadata compatibility does not preserve the Docker Sandboxes runner runtime contract") + } + return nil +} + +func validateTemplateInputs(options templateOptions) error { + seenGroups := make(map[string]struct{}) + for _, selection := range options.Selections { + if (selection.Profile != "" && !templateProfilePattern.MatchString(selection.Profile)) || (selection.Platform != "" && selection.Platform != "linux/amd64" && selection.Platform != "linux/arm64") || !templateTagPattern.MatchString(selection.Tag) || !templateDigestPattern.MatchString(selection.TemplateDigest) { + return fmt.Errorf("configured template selection has an invalid exact identity") + } + if selection.MetadataSHA256 != "" && !templateDigestPattern.MatchString(selection.MetadataSHA256) { + return fmt.Errorf("configured template selection metadata digest is invalid") + } + group := normalizedTemplateTag(selection.Tag) + "@" + selection.TemplateDigest + if _, exists := seenGroups[group]; exists { + return fmt.Errorf("duplicate configured template selection exists for %q", group) + } + seenGroups[group] = struct{}{} + } + for _, protection := range options.Protections { + if !templateDigestPattern.MatchString(protection.ArchiveSHA256) || protection.Kind == "" { + return fmt.Errorf("template protection has an invalid archive digest or kind") + } + } + return nil +} + +func normalizedTemplateTag(value string) string { + return strings.TrimPrefix(value, "docker.io/library/") +} + +func readBoundedFile(path string, maximum int64) ([]byte, error) { + file, err := os.Open(path) + if err != nil { + return nil, err + } + defer file.Close() + data, err := io.ReadAll(io.LimitReader(file, maximum+1)) + if err != nil { + return nil, err + } + if int64(len(data)) > maximum { + return nil, fmt.Errorf("file exceeds %d bytes", maximum) + } + return data, nil +} + +func validateUniqueJSON(data []byte) error { + decoder := json.NewDecoder(bytes.NewReader(data)) + var readValue func() error + readValue = func() error { + token, err := decoder.Token() + if err != nil { + return err + } + delimiter, ok := token.(json.Delim) + if !ok { + return nil + } + switch delimiter { + case '{': + keys := make(map[string]struct{}) + for decoder.More() { + keyToken, err := decoder.Token() + if err != nil { + return err + } + key, ok := keyToken.(string) + if !ok { + return fmt.Errorf("JSON object key is not a string") + } + if _, exists := keys[key]; exists { + return fmt.Errorf("JSON object contains duplicate key %q", key) + } + keys[key] = struct{}{} + if err := readValue(); err != nil { + return err + } + } + end, err := decoder.Token() + if err != nil || end != json.Delim('}') { + return fmt.Errorf("JSON object is not terminated") + } + case '[': + for decoder.More() { + if err := readValue(); err != nil { + return err + } + } + end, err := decoder.Token() + if err != nil || end != json.Delim(']') { + return fmt.Errorf("JSON array is not terminated") + } + default: + return fmt.Errorf("unexpected JSON delimiter %q", delimiter) + } + return nil + } + if err := readValue(); err != nil { + return err + } + if _, err := decoder.Token(); !errors.Is(err, io.EOF) { + if err == nil { + return fmt.Errorf("JSON contains trailing content") + } + return err + } + return nil +} + +func sortTemplateProtections(protections []storage.Protection) { + sort.Slice(protections, func(i, j int) bool { + if protections[i].Kind == protections[j].Kind { + return protections[i].Detail < protections[j].Detail + } + return protections[i].Kind < protections[j].Kind + }) +} diff --git a/internal/storage/inventory/template_test.go b/internal/storage/inventory/template_test.go new file mode 100644 index 0000000..33691a8 --- /dev/null +++ b/internal/storage/inventory/template_test.go @@ -0,0 +1,216 @@ +package inventory + +import ( + "crypto/sha256" + "encoding/hex" + "encoding/json" + "os" + "path/filepath" + "runtime" + "strings" + "testing" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +type templateFixture struct { + Profile string + Platform string + Tag string + TemplateDigest string + ArchiveSHA256 string + MetadataSHA256 string + Directory string + Archive string +} + +func TestCollectTemplatesStrictIntegrityValidation(t *testing.T) { + t.Parallel() + root := t.TempDir() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + valid := writeTemplateFixture(t, root, "valid", "act-22.04", "linux/amd64", "valid", now.Add(-10*24*time.Hour), nil) + writeTemplateFixture(t, root, "bad-hash", "act-22.04", "linux/amd64", "bad-hash", now.Add(-10*24*time.Hour), func(metadata map[string]any) { + template := metadata["template"].(map[string]any) + template["archiveSha256"] = "sha256:" + strings.Repeat("0", 64) + }) + duplicateDirectory := filepath.Join(root, "duplicate-json") + mustMkdirAll(t, duplicateDirectory) + mustWriteFile(t, filepath.Join(duplicateDirectory, "template-metadata.json"), []byte(`{"schemaVersion":2,"schemaVersion":2}`)) + + artifacts, warnings, err := collectTemplates(templateOptions{Root: root}) + if err != nil { + t.Fatalf("collectTemplates() error = %v", err) + } + if len(artifacts) != 3 || len(warnings) != 2 { + t.Fatalf("collectTemplates() artifacts=%+v warnings=%v", artifacts, warnings) + } + exact := findArtifactByArchiveDigest(t, artifacts, valid.ArchiveSHA256) + if exact.Ownership.Kind != storage.OwnershipExact || exact.SizeBytes == 0 { + t.Fatalf("valid archive = %+v", exact) + } + unknownCount := 0 + for _, artifact := range artifacts { + if artifact.Ownership.Kind == storage.OwnershipUnknown { + unknownCount++ + } + } + if unknownCount != 2 { + t.Fatalf("ownership-unknown artifacts = %d, want 2", unknownCount) + } +} + +func TestCollectTemplatesDoesNotRetainArchiveForImportedTemplate(t *testing.T) { + t.Parallel() + root := t.TempDir() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + old := writeTemplateFixture(t, root, "old", "full", "linux/amd64", "old", now.Add(-20*24*time.Hour), nil) + current := writeTemplateFixture(t, root, "current", "full", "linux/amd64", "current", now.Add(-10*24*time.Hour), nil) + artifacts, warnings, err := collectTemplates(templateOptions{ + Root: root, + Selections: []TemplateSelection{{ + Platform: current.Platform, + Tag: current.Tag, + TemplateDigest: current.TemplateDigest, + MetadataSHA256: current.MetadataSHA256, + ActivatedAt: now.Add(-9 * 24 * time.Hour), + }}, + Protections: []TemplateProtection{{ + ArchiveSHA256: old.ArchiveSHA256, + Kind: storage.ProtectionPromotion, + Detail: "promoted bytes", + }}, + }) + if err != nil || len(warnings) != 0 { + t.Fatalf("collectTemplates() error=%v warnings=%v", err, warnings) + } + currentArtifact := findArtifactByArchiveDigest(t, artifacts, current.ArchiveSHA256) + oldArtifact := findArtifactByArchiveDigest(t, artifacts, old.ArchiveSHA256) + if currentArtifact.Current || hasProtection(currentArtifact, storage.ProtectionConfiguration) { + t.Fatalf("imported-template selection retained its transient archive = %+v", currentArtifact) + } + if oldArtifact.SupersededAt != nil || !hasProtection(oldArtifact, storage.ProtectionPromotion) { + t.Fatalf("old artifact = %+v", oldArtifact) + } +} + +func TestCollectTemplatesRejectsSymlinkArchive(t *testing.T) { + t.Parallel() + root := t.TempDir() + directory := filepath.Join(root, "symlink") + mustMkdirAll(t, directory) + outside := filepath.Join(t.TempDir(), "archive.tar") + mustWriteFile(t, outside, []byte("archive")) + link := filepath.Join(directory, "archive.tar") + if err := os.Symlink(outside, link); err != nil { + t.Skipf("symlink creation unavailable on %s: %v", runtime.GOOS, err) + } + digest := sha256.Sum256([]byte("archive")) + metadata := validTemplateMetadata("act-22.04", "linux/amd64", "symlink", "archive.tar", "sha256:"+hex.EncodeToString(digest[:]), uint64(len("archive"))) + encoded, _ := json.Marshal(metadata) + mustWriteFile(t, filepath.Join(directory, "template-metadata.json"), encoded) + + artifacts, warnings, err := collectTemplates(templateOptions{Root: root}) + if err != nil { + t.Fatalf("collectTemplates() error = %v", err) + } + if len(artifacts) != 1 || len(warnings) != 1 || artifacts[0].Ownership.Kind != storage.OwnershipUnknown { + t.Fatalf("symlink archive artifacts=%+v warnings=%v", artifacts, warnings) + } +} + +func TestValidateTemplateInputsRejectsAmbiguity(t *testing.T) { + t.Parallel() + selection := TemplateSelection{ + Profile: "full", + Platform: "linux/amd64", + Tag: "epar-docker-sandboxes-full:test-amd64", + TemplateDigest: "sha256:" + strings.Repeat("a", 64), + } + if err := validateTemplateInputs(templateOptions{Selections: []TemplateSelection{selection, selection}}); err == nil { + t.Fatal("validateTemplateInputs() accepted duplicate configured group") + } + if err := validateTemplateInputs(templateOptions{Protections: []TemplateProtection{{ArchiveSHA256: "sha256:short", Kind: storage.ProtectionCertification}}}); err == nil { + t.Fatal("validateTemplateInputs() accepted invalid protected digest") + } +} + +func TestTemplateSelectionAcceptsCanonicalDockerHubName(t *testing.T) { + selection := TemplateSelection{ + Platform: "linux/amd64", + Tag: "docker.io/library/epar-docker-sandboxes-catthehacker-full:20260723-r2-amd64", + TemplateDigest: "sha256:" + strings.Repeat("a", 64), + } + if err := validateTemplateInputs(templateOptions{Selections: []TemplateSelection{selection}}); err != nil { + t.Fatalf("canonical Docker Hub template name rejected: %v", err) + } + if got, want := normalizedTemplateTag(selection.Tag), "epar-docker-sandboxes-catthehacker-full:20260723-r2-amd64"; got != want { + t.Fatalf("normalized tag = %q, want %q", got, want) + } +} + +func writeTemplateFixture(t *testing.T, root, name, profile, platform, suffix string, created time.Time, mutate func(map[string]any)) templateFixture { + t.Helper() + directory := filepath.Join(root, name) + mustMkdirAll(t, directory) + archiveName := "epar-docker-sandboxes-" + profile + ".tar" + archive := filepath.Join(directory, archiveName) + archiveBytes := []byte("archive:" + name + ":" + suffix) + mustWriteFile(t, archive, archiveBytes) + if err := os.Chtimes(archive, created, created); err != nil { + t.Fatal(err) + } + sum := sha256.Sum256(archiveBytes) + archiveSHA := "sha256:" + hex.EncodeToString(sum[:]) + metadata := validTemplateMetadata(profile, platform, suffix, archiveName, archiveSHA, uint64(len(archiveBytes))) + if mutate != nil { + mutate(metadata) + } + encoded, err := json.Marshal(metadata) + if err != nil { + t.Fatal(err) + } + metadataPath := filepath.Join(directory, "template-metadata.json") + mustWriteFile(t, metadataPath, encoded) + metadataSum := sha256.Sum256(encoded) + template := metadata["template"].(map[string]any) + return templateFixture{ + Profile: profile, + Platform: platform, + Tag: template["tag"].(string), + TemplateDigest: template["digest"].(string), + ArchiveSHA256: archiveSHA, + MetadataSHA256: "sha256:" + hex.EncodeToString(metadataSum[:]), + Directory: directory, + Archive: archive, + } +} + +func validTemplateMetadata(profile, platform, suffix, archive string, archiveSHA string, archiveBytes uint64) map[string]any { + templateDigest := "sha256:" + strings.Repeat(digestCharacter(suffix), 64) + return map[string]any{ + "schemaVersion": templateMetadataSchema, + "profile": profile, + "platform": platform, + "template": map[string]any{ + "tag": "epar-docker-sandboxes-" + profile + ":" + suffix + "-amd64", + "digest": templateDigest, + "cacheID": strings.TrimPrefix(templateDigest, "sha256:")[:12], + "rootDisk": "30GiB", + "archive": archive, + "archiveSha256": archiveSHA, + "archiveBytes": archiveBytes, + }, + "compatibility": map[string]any{ + "templateSchemaVersion": 1, + "runnerExecution": "direct-actions-listener", + "dockerDaemonOwner": "docker-sandboxes-runtime", + "expectedDockerDaemonCount": 1, + }, + } +} + +func digestCharacter(value string) string { + sum := sha256.Sum256([]byte(value)) + return hex.EncodeToString(sum[:])[:1] +} diff --git a/internal/storage/inventory/types.go b/internal/storage/inventory/types.go new file mode 100644 index 0000000..d8e0433 --- /dev/null +++ b/internal/storage/inventory/types.go @@ -0,0 +1,111 @@ +package inventory + +import ( + "fmt" + "sort" + "strings" + "time" + + "github.com/solutionforest/ephemeral-action-runner/internal/storage" +) + +const ( + ProviderDockerSandboxes = "docker-sandboxes" + ProjectSurfaceID = "project-filesystem" +) + +// TemplateSelection is an explicit configured template identity. Profile and +// Platform may be omitted when the caller has only the configured tag and full +// template digest. ActivatedAt is optional; without it, non-current archives +// remain retention-uncertain. +type TemplateSelection struct { + Profile string `json:"profile"` + Platform string `json:"platform"` + Tag string `json:"tag"` + TemplateDigest string `json:"templateDigest"` + MetadataSHA256 string `json:"metadataSha256,omitempty"` + ActivatedAt time.Time `json:"activatedAt,omitempty"` +} + +// TemplateProtection preserves an archive by its full content digest. +type TemplateProtection struct { + ArchiveSHA256 string `json:"archiveSha256"` + Kind storage.ProtectionKind `json:"kind"` + Detail string `json:"detail,omitempty"` +} + +// ConfiguredFile identifies one provider artifact by its explicit persisted +// configuration path. Configured files are inventoried with exact filesystem +// identity, but remain protected from automatic cleanup. +type ConfiguredFile struct { + Provider string `json:"provider"` + Role string `json:"role"` + Path string `json:"path"` + Kind storage.ArtifactKind `json:"kind"` + Current bool `json:"current,omitempty"` + ConfiguredAt time.Time `json:"configuredAt,omitempty"` + ProtectionKind storage.ProtectionKind `json:"protectionKind,omitempty"` + ProtectionDetail string `json:"protectionDetail,omitempty"` +} + +// Options supplies all paths and active identities explicitly. Empty storage +// roots use their project-local defaults. +type Options struct { + ProjectRoot string `json:"projectRoot"` + Provider string `json:"provider,omitempty"` + Now time.Time `json:"now"` + LogsRoot string `json:"logsRoot,omitempty"` + NativeRoot string `json:"nativeRoot,omitempty"` + TemplateRoot string `json:"templateRoot,omitempty"` + CurrentExecutable string `json:"currentExecutable,omitempty"` + CurrentRevision string `json:"currentRevision,omitempty"` + ConfiguredTemplates []TemplateSelection `json:"configuredTemplates,omitempty"` + TemplateProtections []TemplateProtection `json:"templateProtections,omitempty"` + ConfiguredFiles []ConfiguredFile `json:"configuredFiles,omitempty"` +} + +// Snapshot is a deterministic storage-core input plus fail-closed collection +// warnings. +type Snapshot struct { + CollectedAt time.Time `json:"collectedAt"` + ProjectRoot string `json:"projectRoot"` + ProviderFilter string `json:"providerFilter,omitempty"` + Surfaces []storage.Surface `json:"surfaces"` + Artifacts []storage.Artifact `json:"artifacts,omitempty"` + Warnings []string `json:"warnings,omitempty"` +} + +// PreviewRequest converts an inventory into a storage plan request without +// changing any inventory classification. +func (snapshot Snapshot) PreviewRequest(policy storage.Policy, requirements []storage.Requirement) storage.PreviewRequest { + return storage.PreviewRequest{ + Now: snapshot.CollectedAt, + Policy: policy, + Surfaces: append([]storage.Surface(nil), snapshot.Surfaces...), + Requirements: append([]storage.Requirement(nil), requirements...), + Artifacts: append([]storage.Artifact(nil), snapshot.Artifacts...), + } +} + +func (snapshot *Snapshot) normalize() { + sort.Slice(snapshot.Surfaces, func(i, j int) bool { return snapshot.Surfaces[i].ID < snapshot.Surfaces[j].ID }) + sort.Slice(snapshot.Artifacts, func(i, j int) bool { return snapshot.Artifacts[i].ID < snapshot.Artifacts[j].ID }) + sort.Strings(snapshot.Warnings) +} + +func (options Options) validate() error { + if strings.TrimSpace(options.ProjectRoot) == "" { + return fmt.Errorf("storage inventory project root is required") + } + if options.Now.IsZero() { + return fmt.Errorf("storage inventory time is required") + } + if strings.TrimSpace(options.Provider) != options.Provider { + return fmt.Errorf("storage inventory provider filter has surrounding whitespace") + } + return nil +} + +func includeProvider(filter, provider string) bool { + return provider == "" || filter == "" || filter == provider +} diff --git a/internal/storage/planner.go b/internal/storage/planner.go new file mode 100644 index 0000000..c1f62b6 --- /dev/null +++ b/internal/storage/planner.go @@ -0,0 +1,351 @@ +package storage + +import ( + "errors" + "fmt" + "sort" + "strings" + "time" +) + +const planSchemaVersion = 1 + +var automaticKinds = map[ArtifactKind]bool{ + ArtifactNativeControllerRevision: true, + ArtifactTemplateArchive: true, + ArtifactDockerImage: true, + ArtifactSandboxTemplate: true, + ArtifactDockerVolume: true, + ArtifactProviderImage: true, + ArtifactProviderCache: true, + ArtifactOther: true, +} + +// Preview deterministically classifies a complete storage snapshot. It never +// mutates the snapshot or any host resource. +func Preview(request PreviewRequest) (Plan, error) { + now := request.Now.UTC() + if request.Now.IsZero() { + return Plan{}, errors.New("storage preview time is required") + } + policy, err := normalizePolicy(request.Policy) + if err != nil { + return Plan{}, err + } + surfaces := append([]Surface(nil), request.Surfaces...) + for index := range surfaces { + surfaces[index].Capacity.ObservedAt = surfaces[index].Capacity.ObservedAt.UTC() + } + sort.Slice(surfaces, func(i, j int) bool { return surfaces[i].ID < surfaces[j].ID }) + checks, err := capacityPreflight(surfaces, request.Requirements) + if err != nil { + return Plan{}, err + } + artifacts := append([]Artifact(nil), request.Artifacts...) + if err := normalizeAndValidateArtifacts(artifacts, surfaces); err != nil { + return Plan{}, err + } + sort.Slice(artifacts, func(i, j int) bool { return artifacts[i].ID < artifacts[j].ID }) + + decisions := make([]Decision, len(artifacts)) + candidates := make(map[string]time.Time) + for index, artifact := range artifacts { + decision, candidateAt := classifyArtifact(now, policy, artifact) + decisions[index] = decision + if candidateAt != nil { + candidates[artifact.ID] = *candidateAt + } + } + applyGenerationCount(policy, decisions, candidates) + applyCacheBudgets(policy, decisions, candidates) + + plan := Plan{ + SchemaVersion: planSchemaVersion, + CreatedAt: now, + Policy: policy, + Surfaces: surfaces, + CapacityChecks: checks, + Decisions: decisions, + } + for _, check := range checks { + if check.Status != CapacityReady { + plan.Warnings = append(plan.Warnings, fmt.Sprintf("capacity %s: requirement=%s surface=%s reason=%s", check.Status, check.Requirement.ID, check.Requirement.SurfaceID, check.Reason)) + } + } + for _, decision := range decisions { + if decision.Action == ActionRemove { + plan.RemovalCount++ + plan.ReclaimableBytes += decision.Artifact.SizeBytes + } + } + plan.Hash, err = ComputePlanHash(plan) + if err != nil { + return Plan{}, err + } + return plan, nil +} + +func normalizePolicy(policy Policy) (Policy, error) { + if policy.GracePeriod == 0 && policy.KeepPrevious == 0 && len(policy.Budgets) == 0 { + policy = DefaultPolicy() + } + if policy.GracePeriod <= 0 { + return Policy{}, errors.New("storage retention grace period must be positive") + } + if policy.KeepPrevious < 0 { + return Policy{}, errors.New("storage keepPrevious must not be negative") + } + policy.Budgets = append([]Budget(nil), policy.Budgets...) + sort.Slice(policy.Budgets, func(i, j int) bool { return policy.Budgets[i].Kind < policy.Budgets[j].Kind }) + seen := make(map[ArtifactKind]struct{}, len(policy.Budgets)) + for _, budget := range policy.Budgets { + if budget.Kind != ArtifactBuildKitCache && budget.Kind != ArtifactGoCache { + return Policy{}, fmt.Errorf("storage budget for unsupported automatic artifact kind %q", budget.Kind) + } + if budget.MaxBytes == 0 { + return Policy{}, fmt.Errorf("storage budget for %q must be positive", budget.Kind) + } + if _, exists := seen[budget.Kind]; exists { + return Policy{}, fmt.Errorf("duplicate storage budget for %q", budget.Kind) + } + seen[budget.Kind] = struct{}{} + } + return policy, nil +} + +func normalizeAndValidateArtifacts(artifacts []Artifact, surfaces []Surface) error { + surfaceIDs := make(map[string]struct{}, len(surfaces)) + for _, surface := range surfaces { + surfaceIDs[surface.ID] = struct{}{} + } + ids := make(map[string]struct{}, len(artifacts)) + for index := range artifacts { + artifact := &artifacts[index] + artifact.CreatedAt = artifact.CreatedAt.UTC() + artifact.LastUsedAt = artifact.LastUsedAt.UTC() + if artifact.SupersededAt != nil { + supersededAt := artifact.SupersededAt.UTC() + artifact.SupersededAt = &supersededAt + } + if artifact.Lease != nil { + lease := *artifact.Lease + lease.ExpiresAt = lease.ExpiresAt.UTC() + artifact.Lease = &lease + } + artifact.Protections = append([]Protection(nil), artifact.Protections...) + if strings.TrimSpace(artifact.ID) == "" { + return errors.New("storage artifact ID is required") + } + if _, exists := ids[artifact.ID]; exists { + return fmt.Errorf("duplicate storage artifact ID %q", artifact.ID) + } + ids[artifact.ID] = struct{}{} + if _, exists := surfaceIDs[artifact.SurfaceID]; !exists { + return fmt.Errorf("storage artifact %q references unknown surface %q", artifact.ID, artifact.SurfaceID) + } + if artifact.Kind == "" { + return fmt.Errorf("storage artifact %q kind is required", artifact.ID) + } + if artifact.Target.Kind == "" || strings.TrimSpace(artifact.Target.Locator) == "" { + return fmt.Errorf("storage artifact %q target kind and locator are required", artifact.ID) + } + if artifact.Target.Match == MatchExact && strings.TrimSpace(artifact.Target.Identity) == "" { + return fmt.Errorf("storage artifact %q exact target identity is required", artifact.ID) + } + if artifact.Target.Match != MatchExact && artifact.Target.Match != MatchPrefix && artifact.Target.Match != MatchUnknown { + return fmt.Errorf("storage artifact %q has invalid target match %q", artifact.ID, artifact.Target.Match) + } + if artifact.Ownership.Kind != OwnershipExact && artifact.Ownership.Kind != OwnershipShared && artifact.Ownership.Kind != OwnershipUnknown { + return fmt.Errorf("storage artifact %q has invalid ownership %q", artifact.ID, artifact.Ownership.Kind) + } + if artifact.Ownership.Kind == OwnershipExact && (strings.TrimSpace(artifact.Ownership.OwnerID) == "" || strings.TrimSpace(artifact.Ownership.Evidence) == "") { + return fmt.Errorf("storage artifact %q exact ownership requires owner ID and evidence", artifact.ID) + } + sort.Slice(artifact.Protections, func(i, j int) bool { + if artifact.Protections[i].Kind == artifact.Protections[j].Kind { + return artifact.Protections[i].Detail < artifact.Protections[j].Detail + } + return artifact.Protections[i].Kind < artifact.Protections[j].Kind + }) + } + return nil +} + +func classifyArtifact(now time.Time, policy Policy, artifact Artifact) (Decision, *time.Time) { + decision := Decision{Artifact: artifact} + add := func(action Action, reasons ...string) (Decision, *time.Time) { + decision.Action = action + decision.Reasons = append(decision.Reasons, reasons...) + return decision, nil + } + if artifact.Current { + return add(ActionProtected, "current") + } + if artifact.Active { + return add(ActionProtected, "active") + } + if len(artifact.Protections) > 0 { + for _, protection := range artifact.Protections { + decision.Reasons = append(decision.Reasons, "protected:"+string(protection.Kind)) + } + decision.Action = ActionProtected + return decision, nil + } + if artifact.Lease != nil { + if strings.TrimSpace(artifact.Lease.ID) == "" || artifact.Lease.ExpiresAt.IsZero() { + return add(ActionProtected, "lease-uncertain") + } + if !artifact.Lease.ExpiresAt.UTC().Before(now) { + return add(ActionProtected, "lease-active") + } + } + if artifact.Ownership.Kind != OwnershipExact { + return add(ActionReportOnly, "ownership-"+string(artifact.Ownership.Kind)) + } + if artifact.Target.Match != MatchExact { + return add(ActionReportOnly, "target-"+string(artifact.Target.Match)) + } + if !automaticKinds[artifact.Kind] { + return add(ActionReportOnly, "explicit-cleanup-only") + } + if !automaticTargetKind(artifact.Kind, artifact.Target.Kind) { + return add(ActionReportOnly, "target-kind-not-automatic") + } + + var candidateAt time.Time + if artifact.LifecycleState == "superseded" || artifact.LifecycleState == "cleanup-pending" { + if artifact.SupersededAt == nil || artifact.SupersededAt.IsZero() { + return add(ActionReportOnly, "superseded-state-uncertain") + } + candidateAt = artifact.SupersededAt.UTC() + if candidateAt.After(now) { + return add(ActionReportOnly, "retention-clock-skew") + } + decision.Action = ActionKeep + decision.Reasons = []string{"eligible-immediately"} + return decision, &candidateAt + } + switch artifact.Kind { + case ArtifactNativeControllerRevision, ArtifactTemplateArchive: + if artifact.RetentionGroup == "" || artifact.SupersededAt == nil || artifact.SupersededAt.IsZero() { + return add(ActionReportOnly, "superseded-state-uncertain") + } + candidateAt = artifact.SupersededAt.UTC() + default: + return add(ActionReportOnly, "explicit-cleanup-only") + } + if candidateAt.After(now) { + return add(ActionReportOnly, "retention-clock-skew") + } + if now.Sub(candidateAt) < policy.GracePeriod { + return add(ActionKeep, "within-grace-period") + } + decision.Action = ActionKeep + decision.Reasons = []string{"eligible"} + return decision, &candidateAt +} + +func automaticTargetKind(artifactKind ArtifactKind, targetKind TargetKind) bool { + switch artifactKind { + case ArtifactNativeControllerRevision: + return targetKind == TargetDirectory || targetKind == TargetFile + case ArtifactTemplateArchive: + return targetKind == TargetFile + case ArtifactDockerImage: + return targetKind == TargetDockerImageTag + case ArtifactSandboxTemplate: + return targetKind == TargetSandboxTemplate + case ArtifactDockerVolume: + return targetKind == TargetDockerVolume + case ArtifactProviderImage, ArtifactProviderCache, ArtifactOther: + return targetKind == TargetFile || targetKind == TargetDirectory || targetKind == TargetExternal || targetKind == TargetBuildKitRecord + default: + return false + } +} + +func applyGenerationCount(policy Policy, decisions []Decision, candidates map[string]time.Time) { + groups := make(map[string][]int) + for index, decision := range decisions { + if _, candidate := candidates[decision.Artifact.ID]; !candidate { + continue + } + switch decision.Artifact.Kind { + case ArtifactNativeControllerRevision, ArtifactTemplateArchive, ArtifactDockerImage, ArtifactSandboxTemplate, ArtifactDockerVolume, ArtifactProviderImage, ArtifactProviderCache, ArtifactOther: + key := string(decision.Artifact.Kind) + "\x00" + decision.Artifact.RetentionGroup + groups[key] = append(groups[key], index) + } + } + for _, indexes := range groups { + sort.Slice(indexes, func(i, j int) bool { + left, right := decisions[indexes[i]].Artifact, decisions[indexes[j]].Artifact + leftAt, rightAt := candidates[left.ID], candidates[right.ID] + if leftAt.Equal(rightAt) { + return left.ID < right.ID + } + return leftAt.After(rightAt) + }) + for position, index := range indexes { + if position < policy.KeepPrevious { + decisions[index].Action = ActionKeep + decisions[index].Reasons = []string{"keep-previous"} + continue + } + decisions[index].Action = ActionRemove + if decisions[index].Artifact.LifecycleState == "superseded" || decisions[index].Artifact.LifecycleState == "cleanup-pending" { + decisions[index].Reasons = []string{"superseded", "immediate-retirement"} + } else { + decisions[index].Reasons = []string{"superseded", "grace-period-expired"} + } + } + } +} + +func applyCacheBudgets(policy Policy, decisions []Decision, candidates map[string]time.Time) { + budgets := make(map[ArtifactKind]uint64, len(policy.Budgets)) + for _, budget := range policy.Budgets { + budgets[budget.Kind] = budget.MaxBytes + } + for kind, maxBytes := range budgets { + var total uint64 + var eligible []int + overflow := false + for index, decision := range decisions { + if decision.Artifact.Kind != kind { + continue + } + next := total + decision.Artifact.SizeBytes + if next < total { + overflow = true + break + } + total = next + if _, candidate := candidates[decision.Artifact.ID]; candidate { + eligible = append(eligible, index) + } + } + if overflow || total <= maxBytes { + continue + } + sort.Slice(eligible, func(i, j int) bool { + left, right := decisions[eligible[i]].Artifact, decisions[eligible[j]].Artifact + leftAt, rightAt := candidates[left.ID], candidates[right.ID] + if leftAt.Equal(rightAt) { + return left.ID < right.ID + } + return leftAt.Before(rightAt) + }) + for _, index := range eligible { + if total <= maxBytes { + break + } + decisions[index].Action = ActionRemove + decisions[index].Reasons = []string{"aggregate-budget-exceeded", "grace-period-expired"} + if decisions[index].Artifact.SizeBytes > total { + total = 0 + } else { + total -= decisions[index].Artifact.SizeBytes + } + } + } +} diff --git a/internal/storage/planner_test.go b/internal/storage/planner_test.go new file mode 100644 index 0000000..162b809 --- /dev/null +++ b/internal/storage/planner_test.go @@ -0,0 +1,235 @@ +package storage + +import ( + "encoding/json" + "math/rand" + "strings" + "testing" + "time" +) + +func TestPreviewConservativeClassification(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + expired := now.Add(-DefaultGracePeriod - time.Hour) + withinGrace := now.Add(-DefaultGracePeriod + time.Hour) + activeLease := &Lease{ID: "lease-active", ExpiresAt: now.Add(time.Hour)} + expiredLease := &Lease{ID: "lease-expired", ExpiresAt: now.Add(-time.Hour)} + artifacts := []Artifact{ + testArtifact("expired", ArtifactNativeControllerRevision, expired), + testArtifact("archive-expired", ArtifactTemplateArchive, expired), + withArtifact(testArtifact("within-grace", ArtifactNativeControllerRevision, withinGrace), func(a *Artifact) {}), + withArtifact(testArtifact("current", ArtifactNativeControllerRevision, expired), func(a *Artifact) { a.Current = true }), + withArtifact(testArtifact("active", ArtifactNativeControllerRevision, expired), func(a *Artifact) { a.Active = true }), + withArtifact(testArtifact("active-lease", ArtifactNativeControllerRevision, expired), func(a *Artifact) { a.Lease = activeLease }), + withArtifact(testArtifact("expired-lease", ArtifactNativeControllerRevision, expired), func(a *Artifact) { a.Lease = expiredLease }), + withArtifact(testArtifact("uncertain-lease", ArtifactNativeControllerRevision, expired), func(a *Artifact) { a.Lease = &Lease{} }), + withArtifact(testArtifact("shared", ArtifactNativeControllerRevision, expired), func(a *Artifact) { a.Ownership = Ownership{Kind: OwnershipShared} }), + withArtifact(testArtifact("unknown-owner", ArtifactNativeControllerRevision, expired), func(a *Artifact) { a.Ownership = Ownership{Kind: OwnershipUnknown} }), + withArtifact(testArtifact("prefix", ArtifactNativeControllerRevision, expired), func(a *Artifact) { a.Target.Match = MatchPrefix }), + withArtifact(testArtifact("protected", ArtifactTemplateArchive, expired), func(a *Artifact) { + a.Protections = []Protection{{Kind: ProtectionCertification, Detail: "release evidence"}} + }), + withArtifact(testArtifact("docker-image", ArtifactDockerImage, expired), func(a *Artifact) {}), + withArtifact(testArtifact("archive-wrong-target", ArtifactTemplateArchive, expired), func(a *Artifact) { a.Target.Kind = TargetDockerVolume }), + withArtifact(testArtifact("no-superseded-state", ArtifactTemplateArchive, expired), func(a *Artifact) { a.SupersededAt = nil }), + withArtifact(testArtifact("clock-skew", ArtifactTemplateArchive, now.Add(time.Hour)), func(a *Artifact) {}), + } + plan, err := Preview(PreviewRequest{ + Now: now, + Policy: DefaultPolicy(), + Surfaces: []Surface{{ID: "host", Kind: SurfaceHostFilesystem, Capacity: Capacity{Known: true, AvailableBytes: 100 * GiB, TotalBytes: 200 * GiB}}}, + Artifacts: artifacts, + }) + if err != nil { + t.Fatalf("Preview() error = %v", err) + } + want := map[string]Action{ + "expired": ActionRemove, + "archive-expired": ActionRemove, + "within-grace": ActionKeep, + "current": ActionProtected, + "active": ActionProtected, + "active-lease": ActionProtected, + "expired-lease": ActionRemove, + "uncertain-lease": ActionProtected, + "shared": ActionReportOnly, + "unknown-owner": ActionReportOnly, + "prefix": ActionReportOnly, + "protected": ActionProtected, + "docker-image": ActionReportOnly, + "archive-wrong-target": ActionReportOnly, + "no-superseded-state": ActionReportOnly, + "clock-skew": ActionReportOnly, + } + for _, decision := range plan.Decisions { + if decision.Action != want[decision.Artifact.ID] { + t.Errorf("artifact %q action = %q reasons=%v, want %q", decision.Artifact.ID, decision.Action, decision.Reasons, want[decision.Artifact.ID]) + } + delete(want, decision.Artifact.ID) + } + if len(want) != 0 { + t.Fatalf("Preview() omitted decisions: %v", want) + } + if plan.RemovalCount != 3 { + t.Fatalf("Preview() removal count = %d, want 3", plan.RemovalCount) + } +} + +func TestPreviewKeepPrevious(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + policy := DefaultPolicy() + policy.KeepPrevious = 1 + artifacts := []Artifact{ + testArtifact("oldest", ArtifactNativeControllerRevision, now.Add(-10*24*time.Hour)), + testArtifact("newest", ArtifactNativeControllerRevision, now.Add(-8*24*time.Hour)), + } + plan, err := Preview(PreviewRequest{Now: now, Policy: policy, Surfaces: testSurfaces(), Artifacts: artifacts}) + if err != nil { + t.Fatalf("Preview() error = %v", err) + } + if actionFor(plan, "newest") != ActionKeep || actionFor(plan, "oldest") != ActionRemove { + t.Fatalf("keepPrevious classifications = newest:%s oldest:%s", actionFor(plan, "newest"), actionFor(plan, "oldest")) + } +} + +func TestPreviewDedicatedCachesRemainSelfManaged(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + policy := DefaultPolicy() + policy.Budgets = []Budget{{Kind: ArtifactGoCache, MaxBytes: 10 * GiB}, {Kind: ArtifactBuildKitCache, MaxBytes: 64 * GiB}} + makeCache := func(id string, kind ArtifactKind, size uint64, age time.Duration) Artifact { + artifact := testArtifact(id, kind, now.Add(-age)) + artifact.SupersededAt = nil + artifact.LastUsedAt = now.Add(-age) + artifact.SizeBytes = size + if kind == ArtifactBuildKitCache { + artifact.Target.Kind = TargetBuildKitRecord + } else { + artifact.Target.Kind = TargetDirectory + } + return artifact + } + artifacts := []Artifact{ + makeCache("go-old", ArtifactGoCache, 4*GiB, 10*24*time.Hour), + makeCache("go-new", ArtifactGoCache, 4*GiB, 8*24*time.Hour), + makeCache("go-within-grace", ArtifactGoCache, 4*GiB, 2*24*time.Hour), + makeCache("buildkit-under-budget", ArtifactBuildKitCache, 60*GiB, 10*24*time.Hour), + } + plan, err := Preview(PreviewRequest{Now: now, Policy: policy, Surfaces: testSurfaces(), Artifacts: artifacts}) + if err != nil { + t.Fatalf("Preview() error = %v", err) + } + for _, id := range []string{"go-old", "go-new", "go-within-grace", "buildkit-under-budget"} { + if actionFor(plan, id) != ActionReportOnly { + t.Fatalf("%s action = %s, want report-only", id, actionFor(plan, id)) + } + } +} + +func TestPreviewIsDeterministicAcrossInputOrder(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.FixedZone("offset", 8*60*60)) + artifacts := []Artifact{ + testArtifact("c", ArtifactTemplateArchive, now.Add(-10*24*time.Hour)), + testArtifact("a", ArtifactNativeControllerRevision, now.Add(-10*24*time.Hour)), + testArtifact("b", ArtifactDockerImage, now.Add(-10*24*time.Hour)), + } + request := PreviewRequest{ + Now: now, + Policy: DefaultPolicy(), + Surfaces: []Surface{{ID: "z", Provider: "docker-sandboxes", Kind: SurfaceDockerEngine}, {ID: "host", Kind: SurfaceHostFilesystem}}, + Requirements: []Requirement{ + {ID: "z-check", Provider: "docker-sandboxes", SurfaceID: "z", PeakBytes: GiB}, + {ID: "a-check", SurfaceID: "host", PeakBytes: GiB}, + }, + Artifacts: artifacts, + } + first, err := Preview(request) + if err != nil { + t.Fatalf("first Preview() error = %v", err) + } + rand.New(rand.NewSource(42)).Shuffle(len(request.Artifacts), func(i, j int) { + request.Artifacts[i], request.Artifacts[j] = request.Artifacts[j], request.Artifacts[i] + }) + request.Surfaces[0], request.Surfaces[1] = request.Surfaces[1], request.Surfaces[0] + request.Requirements[0], request.Requirements[1] = request.Requirements[1], request.Requirements[0] + second, err := Preview(request) + if err != nil { + t.Fatalf("second Preview() error = %v", err) + } + firstJSON, _ := json.Marshal(first) + secondJSON, _ := json.Marshal(second) + if string(firstJSON) != string(secondJSON) { + t.Fatalf("Preview() is nondeterministic:\n%s\n%s", firstJSON, secondJSON) + } +} + +func TestPreviewDoesNotMutateArtifactProtectionOrder(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + artifact := testArtifact("protected", ArtifactTemplateArchive, now.Add(-10*24*time.Hour)) + artifact.Protections = []Protection{{Kind: ProtectionOperator, Detail: "z"}, {Kind: ProtectionCertification, Detail: "a"}} + original := append([]Protection(nil), artifact.Protections...) + if _, err := Preview(PreviewRequest{Now: now, Policy: DefaultPolicy(), Surfaces: testSurfaces(), Artifacts: []Artifact{artifact}}); err != nil { + t.Fatalf("Preview() error = %v", err) + } + if artifact.Protections[0] != original[0] || artifact.Protections[1] != original[1] { + t.Fatalf("Preview() mutated caller protections: got=%v want=%v", artifact.Protections, original) + } +} + +func TestPreviewCapacityWarningsAndValidation(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 27, 12, 0, 0, 0, time.UTC) + plan, err := Preview(PreviewRequest{ + Now: now, + Policy: DefaultPolicy(), + Surfaces: []Surface{{ID: "host", Kind: SurfaceHostFilesystem}}, + Requirements: []Requirement{{ID: "build", SurfaceID: "host", PeakBytes: 30 * GiB}}, + }) + if err != nil { + t.Fatalf("Preview() error = %v", err) + } + if len(plan.Warnings) != 1 || !strings.Contains(plan.Warnings[0], "capacity unknown") { + t.Fatalf("Preview() warnings = %v, want unknown capacity", plan.Warnings) + } + if _, err := Preview(PreviewRequest{Now: now, Policy: DefaultPolicy(), Surfaces: testSurfaces(), Artifacts: []Artifact{testArtifact("duplicate", ArtifactTemplateArchive, now.Add(-10*24*time.Hour)), testArtifact("duplicate", ArtifactTemplateArchive, now.Add(-9*24*time.Hour))}}); err == nil { + t.Fatal("Preview() accepted duplicate artifact IDs") + } +} + +func testArtifact(id string, kind ArtifactKind, lifecycleAt time.Time) Artifact { + at := lifecycleAt.UTC() + return Artifact{ + ID: id, + Provider: "test-provider", + SurfaceID: "host", + Kind: kind, + RetentionGroup: "default", + Target: Target{Kind: TargetFile, Locator: "/exact/" + id, Identity: "identity:" + id, Fingerprint: "fingerprint:" + id, Match: MatchExact}, + Ownership: Ownership{Kind: OwnershipExact, OwnerID: "installation:test", Evidence: "signed-manifest"}, + SizeBytes: GiB, + CreatedAt: at.Add(-time.Hour), + SupersededAt: &at, + } +} + +func withArtifact(artifact Artifact, mutate func(*Artifact)) Artifact { + mutate(&artifact) + return artifact +} + +func testSurfaces() []Surface { + return []Surface{{ID: "host", Kind: SurfaceHostFilesystem, Capacity: Capacity{Known: true, AvailableBytes: 100 * GiB, TotalBytes: 200 * GiB}}} +} + +func actionFor(plan Plan, id string) Action { + for _, decision := range plan.Decisions { + if decision.Artifact.ID == id { + return decision.Action + } + } + return "" +} diff --git a/internal/storage/types.go b/internal/storage/types.go new file mode 100644 index 0000000..a19c57e --- /dev/null +++ b/internal/storage/types.go @@ -0,0 +1,264 @@ +package storage + +import "time" + +const ( + GiB uint64 = 1 << 30 + + DefaultMinimumFreeBytes = 1 * GiB + DefaultGracePeriod = 168 * time.Hour + DefaultKeepPrevious = 0 + DefaultBuildKitMaxBytes = 20 * GiB + DefaultGoCacheMaxBytes = 10 * GiB +) + +// SurfaceKind identifies a capacity and reclaim domain. Reclaim on one surface +// must not be reported as free space on another surface. +type SurfaceKind string + +const ( + SurfaceHostFilesystem SurfaceKind = "host-filesystem" + SurfaceDockerEngine SurfaceKind = "docker-engine" + SurfaceSandboxCache SurfaceKind = "sandbox-template-cache" + SurfaceExternal SurfaceKind = "external" +) + +// Capacity is an observation for one storage surface. Unknown capacity is a +// first-class fail-closed state, not zero available bytes. +type Capacity struct { + Known bool `json:"known"` + AvailableBytes uint64 `json:"availableBytes,omitempty"` + TotalBytes uint64 `json:"totalBytes,omitempty"` + ObservedAt time.Time `json:"observedAt,omitempty"` +} + +// Surface identifies one capacity and reclaim domain. +type Surface struct { + ID string `json:"id"` + Provider string `json:"provider,omitempty"` + Kind SurfaceKind `json:"kind"` + Location string `json:"location,omitempty"` + Classification string `json:"classification,omitempty"` + Sparse bool `json:"sparse,omitempty"` + VirtualMaximumBytes uint64 `json:"virtualMaximumBytes,omitempty"` + AllocatedBytes uint64 `json:"allocatedBytes,omitempty"` + Confidence string `json:"confidence,omitempty"` + AdmissionAuthoritative bool `json:"admissionAuthoritative"` + Advisory bool `json:"advisory,omitempty"` + Capacity Capacity `json:"capacity"` +} + +// Requirement is the peak additional capacity needed by an operation on one +// surface. MinimumFreeBytes defaults to DefaultMinimumFreeBytes when zero. +type Requirement struct { + ID string `json:"id"` + Provider string `json:"provider,omitempty"` + SurfaceID string `json:"surfaceId"` + PeakBytes uint64 `json:"peakBytes"` + MinimumFreeBytes uint64 `json:"minimumFreeBytes"` +} + +// CapacityStatus is the result of a capacity preflight. +type CapacityStatus string + +const ( + CapacityReady CapacityStatus = "ready" + CapacityInsufficient CapacityStatus = "insufficient" + CapacityUnknown CapacityStatus = "unknown" +) + +// CapacityCheck binds a requirement to the exact capacity observation used to +// evaluate it. +type CapacityCheck struct { + Requirement Requirement `json:"requirement"` + Capacity Capacity `json:"capacity"` + Status CapacityStatus `json:"status"` + RequiredAvailableBytes uint64 `json:"requiredAvailableBytes"` + DeficitBytes uint64 `json:"deficitBytes,omitempty"` + Reason string `json:"reason"` +} + +// ArtifactKind selects the retention policy applicable to an artifact. +type ArtifactKind string + +const ( + ArtifactNativeControllerRevision ArtifactKind = "native-controller-revision" + ArtifactTemplateArchive ArtifactKind = "template-archive" + ArtifactDockerImage ArtifactKind = "docker-image" + ArtifactSandboxTemplate ArtifactKind = "sandbox-template" + ArtifactDockerVolume ArtifactKind = "docker-volume" + ArtifactBuildKitCache ArtifactKind = "buildkit-cache" + ArtifactGoCache ArtifactKind = "go-cache" + ArtifactProviderImage ArtifactKind = "provider-image" + ArtifactProviderCache ArtifactKind = "provider-cache" + ArtifactOther ArtifactKind = "other" +) + +// TargetKind describes the one exact object an executor may receive. +type TargetKind string + +const ( + TargetFile TargetKind = "file" + TargetDirectory TargetKind = "directory" + TargetDockerImageTag TargetKind = "docker-image-tag" + TargetDockerVolume TargetKind = "docker-volume" + TargetBuildKitRecord TargetKind = "buildkit-record" + TargetSandboxTemplate TargetKind = "sandbox-template" + TargetExternal TargetKind = "external" +) + +// MatchKind records whether a locator identifies one object or a broad set. +type MatchKind string + +const ( + MatchExact MatchKind = "exact" + MatchPrefix MatchKind = "prefix" + MatchUnknown MatchKind = "unknown" +) + +// Target binds a locator to an immutable object identity. Fingerprint may bind +// additional mutable metadata, such as a filesystem object's size and mtime. +type Target struct { + Kind TargetKind `json:"kind"` + Locator string `json:"locator"` + Identity string `json:"identity,omitempty"` + Fingerprint string `json:"fingerprint,omitempty"` + Match MatchKind `json:"match"` +} + +// OwnershipKind expresses how confidently an artifact belongs exclusively to +// this EPAR installation. +type OwnershipKind string + +const ( + OwnershipExact OwnershipKind = "exact" + OwnershipShared OwnershipKind = "shared" + OwnershipUnknown OwnershipKind = "unknown" +) + +// Ownership carries the exact owner identity and its evidence. Names or +// prefixes alone are not exact ownership evidence. +type Ownership struct { + Kind OwnershipKind `json:"kind"` + OwnerID string `json:"ownerId,omitempty"` + Evidence string `json:"evidence,omitempty"` +} + +// ProtectionKind identifies a reason an artifact must not be selected. +type ProtectionKind string + +const ( + ProtectionActive ProtectionKind = "active" + ProtectionCurrent ProtectionKind = "current" + ProtectionLease ProtectionKind = "lease" + ProtectionConfiguration ProtectionKind = "configuration" + ProtectionLock ProtectionKind = "source-lock" + ProtectionPromotion ProtectionKind = "promotion" + ProtectionCertification ProtectionKind = "certification" + ProtectionCustomRoot ProtectionKind = "custom-root" + ProtectionOperator ProtectionKind = "operator" + ProtectionUncertain ProtectionKind = "uncertain" +) + +// Protection is adapter-supplied evidence that an artifact must be retained. +type Protection struct { + Kind ProtectionKind `json:"kind"` + Detail string `json:"detail,omitempty"` +} + +// Lease protects an artifact through ExpiresAt. A missing ID or expiry is +// treated as uncertain and remains protected. +type Lease struct { + ID string `json:"id"` + ExpiresAt time.Time `json:"expiresAt"` +} + +// Artifact is an evidence-bearing snapshot of one storage object. +type Artifact struct { + ID string `json:"id"` + Provider string `json:"provider,omitempty"` + SurfaceID string `json:"surfaceId"` + Kind ArtifactKind `json:"kind"` + RetentionGroup string `json:"retentionGroup,omitempty"` + Target Target `json:"target"` + Ownership Ownership `json:"ownership"` + SizeBytes uint64 `json:"sizeBytes"` + CreatedAt time.Time `json:"createdAt,omitempty"` + LastUsedAt time.Time `json:"lastUsedAt,omitempty"` + SupersededAt *time.Time `json:"supersededAt,omitempty"` + Current bool `json:"current,omitempty"` + Active bool `json:"active,omitempty"` + Lease *Lease `json:"lease,omitempty"` + Protections []Protection `json:"protections,omitempty"` + BackendID string `json:"backendId,omitempty"` + Custody string `json:"custody,omitempty"` + LifecycleState string `json:"lifecycleState,omitempty"` + ConfigRefs []string `json:"configReferences,omitempty"` + CleanupError string `json:"cleanupError,omitempty"` +} + +// Budget bounds one automatically managed artifact kind. +type Budget struct { + Kind ArtifactKind `json:"kind"` + MaxBytes uint64 `json:"maxBytes"` +} + +// Policy is the complete deterministic retention policy embedded into a plan. +type Policy struct { + GracePeriod time.Duration `json:"gracePeriod"` + KeepPrevious int `json:"keepPrevious"` + Budgets []Budget `json:"budgets"` +} + +// DefaultPolicy returns the approved conservative automatic defaults. +func DefaultPolicy() Policy { + return Policy{ + GracePeriod: DefaultGracePeriod, + KeepPrevious: DefaultKeepPrevious, + Budgets: []Budget{ + {Kind: ArtifactBuildKitCache, MaxBytes: DefaultBuildKitMaxBytes}, + {Kind: ArtifactGoCache, MaxBytes: DefaultGoCacheMaxBytes}, + }, + } +} + +// Action is the deterministic classification for one artifact. +type Action string + +const ( + ActionKeep Action = "keep" + ActionProtected Action = "protected" + ActionReportOnly Action = "report-only" + ActionRemove Action = "remove" +) + +// Decision is one artifact's retention outcome and evidence. +type Decision struct { + Artifact Artifact `json:"artifact"` + Action Action `json:"action"` + Reasons []string `json:"reasons"` +} + +// PreviewRequest supplies a complete storage snapshot and an explicit clock. +type PreviewRequest struct { + Now time.Time `json:"now"` + Policy Policy `json:"policy"` + Surfaces []Surface `json:"surfaces"` + Requirements []Requirement `json:"requirements,omitempty"` + Artifacts []Artifact `json:"artifacts,omitempty"` +} + +// Plan is an immutable, deterministic preview. Hash is the SHA-256 of the plan +// with the Hash field empty. +type Plan struct { + SchemaVersion int `json:"schemaVersion"` + CreatedAt time.Time `json:"createdAt"` + Policy Policy `json:"policy"` + Surfaces []Surface `json:"surfaces"` + CapacityChecks []CapacityCheck `json:"capacityChecks,omitempty"` + Decisions []Decision `json:"decisions,omitempty"` + RemovalCount int `json:"removalCount"` + ReclaimableBytes uint64 `json:"reclaimableBytes"` + Warnings []string `json:"warnings,omitempty"` + Hash string `json:"hash"` +} diff --git a/scripts/bootstrap-trust/main.go b/scripts/bootstrap-trust/main.go new file mode 100644 index 0000000..8e08a4b --- /dev/null +++ b/scripts/bootstrap-trust/main.go @@ -0,0 +1,296 @@ +package main + +import ( + "bytes" + "crypto/sha256" + "crypto/x509" + "encoding/hex" + "encoding/json" + "encoding/pem" + "errors" + "flag" + "fmt" + "io" + "os" + "path/filepath" + "sort" + "strings" + "time" +) + +const ( + feedSchemaVersion = 1 + maxFeedBytes = 32 << 20 + maxFeedAge = 30 * time.Second + maxCertificates = 4096 +) + +type feedDocument struct { + SchemaVersion int `json:"schemaVersion"` + HostOS string `json:"hostOS"` + Scopes []string `json:"scopes"` + GeneratedAt time.Time `json:"generatedAt"` + ExpiresAt time.Time `json:"expiresAt"` + Certificates []feedCertificate `json:"certificates"` + DistrustSHA256 []string `json:"distrustSHA256"` +} + +type feedCertificate struct { + SHA256 string `json:"sha256"` + PEM string `json:"pem"` +} + +type bundleSummary struct { + HostOS string + Scopes []string + Certificates int + FeedSHA256 string + BundleSHA256 string +} + +func main() { + feedPath := flag.String("feed", "", "host trust feed path") + outputPath := flag.String("output", "", "validated PEM bundle output path") + expectedHostOS := flag.String("expected-host-os", "", "expected feed host OS") + flag.Parse() + if flag.NArg() != 0 || strings.TrimSpace(*feedPath) == "" || strings.TrimSpace(*outputPath) == "" || strings.TrimSpace(*expectedHostOS) == "" { + fmt.Fprintln(os.Stderr, "usage: bootstrap-trust --feed --output --expected-host-os ") + os.Exit(2) + } + summary, err := materializeBundle(*feedPath, *outputPath, *expectedHostOS, time.Now().UTC()) + if err != nil { + fmt.Fprintf(os.Stderr, "bootstrap build trust rejected: %v\n", err) + os.Exit(1) + } + fmt.Printf("hostOS=%s scopes=%s certificates=%d feedSHA256=%s bundleSHA256=%s\n", summary.HostOS, strings.Join(summary.Scopes, ","), summary.Certificates, summary.FeedSHA256, summary.BundleSHA256) +} + +func materializeBundle(feedPath, outputPath, expectedHostOS string, now time.Time) (bundleSummary, error) { + feedPath = filepath.Clean(feedPath) + info, err := os.Lstat(feedPath) + if err != nil { + return bundleSummary{}, fmt.Errorf("inspect feed: %w", err) + } + if !info.Mode().IsRegular() || info.Mode()&os.ModeSymlink != 0 { + return bundleSummary{}, fmt.Errorf("feed must be a regular non-symlink file") + } + file, err := os.Open(feedPath) + if err != nil { + return bundleSummary{}, fmt.Errorf("open feed: %w", err) + } + content, readErr := io.ReadAll(io.LimitReader(file, maxFeedBytes+1)) + closeErr := file.Close() + if readErr != nil { + return bundleSummary{}, fmt.Errorf("read feed: %w", readErr) + } + if closeErr != nil { + return bundleSummary{}, fmt.Errorf("close feed: %w", closeErr) + } + if len(content) > maxFeedBytes { + return bundleSummary{}, fmt.Errorf("feed exceeds %d bytes", maxFeedBytes) + } + feedHash := sha256.Sum256(content) + + decoder := json.NewDecoder(bytes.NewReader(content)) + decoder.DisallowUnknownFields() + var document feedDocument + if err := decoder.Decode(&document); err != nil { + return bundleSummary{}, fmt.Errorf("parse feed: %w", err) + } + if err := requireJSONEOF(decoder); err != nil { + return bundleSummary{}, err + } + hostOS := strings.ToLower(strings.TrimSpace(document.HostOS)) + expected := strings.ToLower(strings.TrimSpace(expectedHostOS)) + if document.SchemaVersion != feedSchemaVersion { + return bundleSummary{}, fmt.Errorf("unsupported schemaVersion %d", document.SchemaVersion) + } + if expected != "windows" && expected != "darwin" && expected != "linux" { + return bundleSummary{}, fmt.Errorf("unsupported expected host OS %q", expectedHostOS) + } + if hostOS != expected { + return bundleSummary{}, fmt.Errorf("feed hostOS %q does not match expected %q", hostOS, expected) + } + scopes, err := validateScopes(document.Scopes, hostOS) + if err != nil { + return bundleSummary{}, err + } + now = now.UTC() + generatedAt := document.GeneratedAt.UTC() + expiresAt := document.ExpiresAt.UTC() + if generatedAt.IsZero() || expiresAt.IsZero() || !expiresAt.After(generatedAt) { + return bundleSummary{}, fmt.Errorf("feed has invalid generatedAt/expiresAt") + } + if generatedAt.After(now.Add(5 * time.Second)) { + return bundleSummary{}, fmt.Errorf("feed generatedAt is in the future") + } + if now.Sub(generatedAt) > maxFeedAge { + return bundleSummary{}, fmt.Errorf("feed is older than %s", maxFeedAge) + } + if now.After(expiresAt) { + return bundleSummary{}, fmt.Errorf("feed expired at %s", expiresAt.Format(time.RFC3339)) + } + if len(document.Certificates) == 0 || len(document.Certificates) > maxCertificates { + return bundleSummary{}, fmt.Errorf("feed certificate count %d is outside 1..%d", len(document.Certificates), maxCertificates) + } + + distrusted := make(map[string]struct{}, len(document.DistrustSHA256)) + for index, value := range document.DistrustSHA256 { + hash, err := normalizedSHA256(value) + if err != nil { + return bundleSummary{}, fmt.Errorf("distrustSHA256[%d]: %w", index, err) + } + if _, exists := distrusted[hash]; exists { + return bundleSummary{}, fmt.Errorf("distrustSHA256[%d] duplicates %s", index, hash) + } + distrusted[hash] = struct{}{} + } + + certificates := make(map[string][]byte, len(document.Certificates)) + for index, encoded := range document.Certificates { + declaredHash, err := normalizedSHA256(encoded.SHA256) + if err != nil { + return bundleSummary{}, fmt.Errorf("certificate %d SHA-256: %w", index, err) + } + if _, exists := certificates[declaredHash]; exists { + return bundleSummary{}, fmt.Errorf("certificate %d duplicates %s", index, declaredHash) + } + block, rest := pem.Decode([]byte(encoded.PEM)) + if block == nil || block.Type != "CERTIFICATE" || len(bytes.TrimSpace(rest)) != 0 { + return bundleSummary{}, fmt.Errorf("certificate %d must contain exactly one CERTIFICATE PEM block", index) + } + certificate, err := x509.ParseCertificate(block.Bytes) + if err != nil { + return bundleSummary{}, fmt.Errorf("certificate %d parse: %w", index, err) + } + if !certificate.IsCA || !certificate.BasicConstraintsValid { + return bundleSummary{}, fmt.Errorf("certificate %d is not a valid CA certificate", index) + } + actualHash := sha256.Sum256(block.Bytes) + actualHex := hex.EncodeToString(actualHash[:]) + if actualHex != declaredHash { + return bundleSummary{}, fmt.Errorf("certificate %d SHA-256 mismatch: declared %s, got %s", index, declaredHash, actualHex) + } + if _, denied := distrusted[actualHex]; denied { + return bundleSummary{}, fmt.Errorf("certificate %d is also present in distrustSHA256", index) + } + certificates[actualHex] = pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: block.Bytes}) + } + + hashes := make([]string, 0, len(certificates)) + for hash := range certificates { + hashes = append(hashes, hash) + } + sort.Strings(hashes) + var bundle bytes.Buffer + for _, hash := range hashes { + bundle.Write(certificates[hash]) + } + bundleHash := sha256.Sum256(bundle.Bytes()) + if err := writeAtomically(outputPath, bundle.Bytes()); err != nil { + return bundleSummary{}, err + } + return bundleSummary{ + HostOS: hostOS, + Scopes: scopes, + Certificates: len(hashes), + FeedSHA256: hex.EncodeToString(feedHash[:]), + BundleSHA256: hex.EncodeToString(bundleHash[:]), + }, nil +} + +func requireJSONEOF(decoder *json.Decoder) error { + var extra any + err := decoder.Decode(&extra) + if errors.Is(err, io.EOF) { + return nil + } + if err == nil { + return fmt.Errorf("feed contains multiple JSON values") + } + return fmt.Errorf("parse trailing feed content: %w", err) +} + +func validateScopes(values []string, hostOS string) ([]string, error) { + if len(values) == 0 { + return nil, fmt.Errorf("feed scopes are empty") + } + seen := make(map[string]struct{}, len(values)) + scopes := make([]string, 0, len(values)) + for index, value := range values { + scope := strings.ToLower(strings.TrimSpace(value)) + if scope != "system" && scope != "user" { + return nil, fmt.Errorf("feed scope %d is unsupported: %q", index, value) + } + if hostOS == "linux" && scope == "user" { + return nil, fmt.Errorf("Linux build trust cannot declare user scope") + } + if _, exists := seen[scope]; exists { + return nil, fmt.Errorf("feed scope %d duplicates %q", index, scope) + } + seen[scope] = struct{}{} + scopes = append(scopes, scope) + } + if _, exists := seen["system"]; !exists { + return nil, fmt.Errorf("build trust feed must include system scope") + } + sort.Strings(scopes) + return scopes, nil +} + +func normalizedSHA256(value string) (string, error) { + hash := strings.ToLower(strings.TrimSpace(value)) + if len(hash) != sha256.Size*2 { + return "", fmt.Errorf("must be %d hexadecimal characters", sha256.Size*2) + } + if _, err := hex.DecodeString(hash); err != nil { + return "", fmt.Errorf("invalid hexadecimal value: %w", err) + } + return hash, nil +} + +func writeAtomically(path string, content []byte) (err error) { + path = filepath.Clean(path) + parent := filepath.Dir(path) + if err := os.MkdirAll(parent, 0o700); err != nil { + return fmt.Errorf("create bundle directory: %w", err) + } + if info, statErr := os.Lstat(path); statErr == nil { + if !info.Mode().IsRegular() || info.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("bundle output must be a regular non-symlink file") + } + } else if !errors.Is(statErr, os.ErrNotExist) { + return fmt.Errorf("inspect bundle output: %w", statErr) + } + temporary, err := os.CreateTemp(parent, ".bootstrap-ca-*.tmp") + if err != nil { + return fmt.Errorf("create temporary bundle: %w", err) + } + temporaryPath := temporary.Name() + defer func() { + if temporary != nil { + _ = temporary.Close() + } + if err != nil { + _ = os.Remove(temporaryPath) + } + }() + if err = temporary.Chmod(0o600); err != nil { + return fmt.Errorf("protect temporary bundle: %w", err) + } + if _, err = temporary.Write(content); err != nil { + return fmt.Errorf("write temporary bundle: %w", err) + } + if err = temporary.Sync(); err != nil { + return fmt.Errorf("sync temporary bundle: %w", err) + } + if err = temporary.Close(); err != nil { + temporary = nil + return fmt.Errorf("close temporary bundle: %w", err) + } + temporary = nil + if err = os.Rename(temporaryPath, path); err != nil { + return fmt.Errorf("activate bundle: %w", err) + } + return nil +} diff --git a/scripts/bootstrap-trust/main_test.go b/scripts/bootstrap-trust/main_test.go new file mode 100644 index 0000000..74e2797 --- /dev/null +++ b/scripts/bootstrap-trust/main_test.go @@ -0,0 +1,192 @@ +package main + +import ( + "crypto/ed25519" + "crypto/rand" + "crypto/sha256" + "crypto/x509" + "encoding/hex" + "encoding/json" + "encoding/pem" + "math/big" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +func TestMaterializeBundleValidatesAndWritesCanonicalCA(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 29, 12, 0, 0, 0, time.UTC) + certificatePEM, certificateHash := testCA(t, now) + document := testFeed(now, certificatePEM, certificateHash) + root := t.TempDir() + feedPath := writeTestFeed(t, root, document) + outputPath := filepath.Join(root, "out", "ca.pem") + + summary, err := materializeBundle(feedPath, outputPath, "windows", now) + if err != nil { + t.Fatalf("materializeBundle: %v", err) + } + if summary.HostOS != "windows" || strings.Join(summary.Scopes, ",") != "system,user" || summary.Certificates != 1 { + t.Fatalf("unexpected summary: %+v", summary) + } + content, err := os.ReadFile(outputPath) + if err != nil { + t.Fatalf("read output: %v", err) + } + block, rest := pem.Decode(content) + if block == nil || block.Type != "CERTIFICATE" || len(strings.TrimSpace(string(rest))) != 0 { + t.Fatalf("output is not one canonical certificate") + } + actual := sha256.Sum256(block.Bytes) + if hex.EncodeToString(actual[:]) != certificateHash { + t.Fatalf("output hash mismatch") + } +} + +func TestMaterializeBundleFailsClosed(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 29, 12, 0, 0, 0, time.UTC) + certificatePEM, certificateHash := testCA(t, now) + tests := []struct { + name string + mutate func(*feedDocument) + expectedOS string + want string + }{ + { + name: "expired", + mutate: func(document *feedDocument) { + document.GeneratedAt = now.Add(-time.Minute) + document.ExpiresAt = now.Add(-30 * time.Second) + }, + expectedOS: "windows", + want: "older than", + }, + { + name: "wrong host", + mutate: func(document *feedDocument) { + document.HostOS = "darwin" + }, + expectedOS: "windows", + want: "does not match", + }, + { + name: "hash mismatch", + mutate: func(document *feedDocument) { + document.Certificates[0].SHA256 = strings.Repeat("0", 64) + }, + expectedOS: "windows", + want: "SHA-256 mismatch", + }, + { + name: "distrusted certificate", + mutate: func(document *feedDocument) { + document.DistrustSHA256 = []string{certificateHash} + }, + expectedOS: "windows", + want: "also present in distrustSHA256", + }, + { + name: "missing system scope", + mutate: func(document *feedDocument) { + document.Scopes = []string{"user"} + }, + expectedOS: "windows", + want: "must include system scope", + }, + } + for _, test := range tests { + test := test + t.Run(test.name, func(t *testing.T) { + t.Parallel() + document := testFeed(now, certificatePEM, certificateHash) + test.mutate(&document) + root := t.TempDir() + feedPath := writeTestFeed(t, root, document) + _, err := materializeBundle(feedPath, filepath.Join(root, "ca.pem"), test.expectedOS, now) + if err == nil || !strings.Contains(err.Error(), test.want) { + t.Fatalf("error = %v, want substring %q", err, test.want) + } + }) + } +} + +func TestMaterializeBundleRejectsSymlinkPaths(t *testing.T) { + t.Parallel() + now := time.Date(2026, 7, 29, 12, 0, 0, 0, time.UTC) + certificatePEM, certificateHash := testCA(t, now) + root := t.TempDir() + feedPath := writeTestFeed(t, root, testFeed(now, certificatePEM, certificateHash)) + feedLink := filepath.Join(root, "feed-link.json") + if err := os.Symlink(feedPath, feedLink); err != nil { + t.Skipf("symlink creation is unavailable: %v", err) + } + if _, err := materializeBundle(feedLink, filepath.Join(root, "ca.pem"), "windows", now); err == nil || !strings.Contains(err.Error(), "non-symlink") { + t.Fatalf("symlink feed error = %v", err) + } + + outputTarget := filepath.Join(root, "target.pem") + if err := os.WriteFile(outputTarget, []byte("target"), 0o600); err != nil { + t.Fatalf("write output target: %v", err) + } + outputLink := filepath.Join(root, "output-link.pem") + if err := os.Symlink(outputTarget, outputLink); err != nil { + t.Fatalf("create output symlink: %v", err) + } + if _, err := materializeBundle(feedPath, outputLink, "windows", now); err == nil || !strings.Contains(err.Error(), "non-symlink") { + t.Fatalf("symlink output error = %v", err) + } +} + +func testFeed(now time.Time, certificatePEM []byte, certificateHash string) feedDocument { + return feedDocument{ + SchemaVersion: 1, + HostOS: "windows", + Scopes: []string{"system", "user"}, + GeneratedAt: now.Add(-time.Second), + ExpiresAt: now.Add(29 * time.Second), + Certificates: []feedCertificate{{ + SHA256: certificateHash, + PEM: string(certificatePEM), + }}, + DistrustSHA256: []string{}, + } +} + +func writeTestFeed(t *testing.T, root string, document feedDocument) string { + t.Helper() + content, err := json.Marshal(document) + if err != nil { + t.Fatalf("marshal feed: %v", err) + } + path := filepath.Join(root, "feed.json") + if err := os.WriteFile(path, content, 0o600); err != nil { + t.Fatalf("write feed: %v", err) + } + return path +} + +func testCA(t *testing.T, now time.Time) ([]byte, string) { + t.Helper() + publicKey, privateKey, err := ed25519.GenerateKey(rand.Reader) + if err != nil { + t.Fatalf("generate key: %v", err) + } + template := &x509.Certificate{ + SerialNumber: big.NewInt(1), + NotBefore: now.Add(-time.Hour), + NotAfter: now.Add(24 * time.Hour), + IsCA: true, + BasicConstraintsValid: true, + KeyUsage: x509.KeyUsageCertSign, + } + der, err := x509.CreateCertificate(rand.Reader, template, template, publicKey, privateKey) + if err != nil { + t.Fatalf("create certificate: %v", err) + } + hash := sha256.Sum256(der) + return pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: der}), hex.EncodeToString(hash[:]) +} diff --git a/scripts/build-native-controller.ps1 b/scripts/build-native-controller.ps1 new file mode 100644 index 0000000..1deaeaa --- /dev/null +++ b/scripts/build-native-controller.ps1 @@ -0,0 +1,873 @@ +[CmdletBinding()] +param( + [Parameter(ValueFromRemainingArguments = $true)] + [string[]] $EparArgs +) + +$ErrorActionPreference = 'Stop' +$RepoRoot = Split-Path -Parent (Split-Path -Parent $MyInvocation.MyCommand.Path) +. (Join-Path $RepoRoot 'scripts\host-trust\wrapper-lib.ps1') +$GoImage = if ($env:GO_DOCKER_IMAGE) { $env:GO_DOCKER_IMAGE } else { 'golang:latest' } +$DevImage = if ($env:EPAR_DEV_IMAGE) { $env:EPAR_DEV_IMAGE } else { 'epar-dev-toolchain' } +$repoRootPath = [System.IO.Path]::GetFullPath($RepoRoot) +$projectHasher = [System.Security.Cryptography.SHA256]::Create() +try { + $projectMaterial = [System.Text.Encoding]::UTF8.GetBytes($repoRootPath.ToLowerInvariant()) + $ProjectID = ([BitConverter]::ToString($projectHasher.ComputeHash($projectMaterial))).Replace('-', '').ToLowerInvariant().Substring(0, 12) +} finally { + $projectHasher.Dispose() +} +$GomodVolume = if ($env:EPAR_GOMOD_VOLUME) { $env:EPAR_GOMOD_VOLUME } else { "epar-$ProjectID-gomod" } +$GocacheVolume = if ($env:EPAR_GOCACHE_VOLUME) { $env:EPAR_GOCACHE_VOLUME } else { "epar-$ProjectID-gocache" } +$ManageGoCache = -not $env:EPAR_GOMOD_VOLUME -and -not $env:EPAR_GOCACHE_VOLUME +$GoCacheLimitBytes = [uint64](10GB) +if ($env:EPAR_GO_CACHE_LIMIT_BYTES) { + $parsedGoCacheLimit = [uint64] 0 + if (-not [uint64]::TryParse($env:EPAR_GO_CACHE_LIMIT_BYTES, [ref] $parsedGoCacheLimit) -or $parsedGoCacheLimit -eq 0) { + Write-Error 'EPAR_GO_CACHE_LIMIT_BYTES must be a positive integer byte count.' + exit 1 + } + $GoCacheLimitBytes = $parsedGoCacheLimit +} +$BootstrapMinimumFreeBytes = [uint64](1GB) +if ($env:EPAR_BOOTSTRAP_MIN_FREE_BYTES) { + $parsedBootstrapMinimum = [uint64] 0 + if (-not [uint64]::TryParse($env:EPAR_BOOTSTRAP_MIN_FREE_BYTES, [ref] $parsedBootstrapMinimum) -or $parsedBootstrapMinimum -eq 0) { + Write-Error 'EPAR_BOOTSTRAP_MIN_FREE_BYTES must be a positive integer byte count.' + exit 1 + } + $BootstrapMinimumFreeBytes = $parsedBootstrapMinimum +} + +if (-not (Get-Command docker -ErrorAction SilentlyContinue)) { + Write-Error 'docker command not found. Install Docker and make sure it is available on PATH.' + exit 1 +} + +try { + $repoVolumeRoot = [System.IO.Path]::GetPathRoot($repoRootPath) + $repoDrive = [System.IO.DriveInfo]::new($repoVolumeRoot) + $bootstrapAvailableBytes = [uint64] $repoDrive.AvailableFreeSpace +} catch { + Write-Error "cannot measure bootstrap storage for ${RepoRoot}: $($_.Exception.Message)" + exit 1 +} +if ($bootstrapAvailableBytes -lt $BootstrapMinimumFreeBytes) { + if ($EparArgs -contains '--allow-insufficient-storage') { + Write-Warning ("bootstrap storage on {0} is below the {1}-byte reserve; continuing because --allow-insufficient-storage was explicitly supplied." -f $repoVolumeRoot, $BootstrapMinimumFreeBytes) + } else { + Write-Error ("insufficient bootstrap storage on {0}: available={1} required-reserve={2}. Free space, inspect storage, or retry this invocation with --allow-insufficient-storage." -f $repoVolumeRoot, $bootstrapAvailableBytes, $BootstrapMinimumFreeBytes) + exit 1 + } +} + +function Get-EparGoCacheVolumeIdentity { + param([Parameter(Mandatory = $true)][string] $Name) + $previousErrorActionPreference = $ErrorActionPreference + try { + $ErrorActionPreference = 'SilentlyContinue' + $inspectionJSON = @((docker volume inspect $Name 2>$null)) + $inspectionExitCode = $LASTEXITCODE + } finally { + $ErrorActionPreference = $previousErrorActionPreference + } + if ($inspectionExitCode -ne 0) { + return [pscustomobject]@{ Exists = $false; Identity = '' } + } + try { + $records = @(($inspectionJSON -join [Environment]::NewLine) | ConvertFrom-Json) + } catch { + throw "Docker returned invalid inspection JSON for Go cache volume ${Name}: $($_.Exception.Message)" + } + if ($records.Count -ne 1 -or $null -eq $records[0].Labels) { + throw "Docker returned an incomplete inspection record for Go cache volume $Name" + } + $labels = $records[0].Labels + $identity = '{0}|{1}|{2}|{3}' -f $labels.'io.solutionforest.epar.project', $labels.'io.solutionforest.epar.cache', $labels.'io.solutionforest.epar.schema', $labels.'io.solutionforest.epar.root' + return [pscustomobject]@{ Exists = $true; Identity = $identity } +} + +function Initialize-EparGoCacheVolume { + param( + [Parameter(Mandatory = $true)][string] $Name, + [Parameter(Mandatory = $true)][string] $Role + ) + $expected = "$ProjectID|$Role|1|$repoRootPath" + $inspection = Get-EparGoCacheVolumeIdentity -Name $Name + if ($inspection.Exists) { + if ($inspection.Identity -cne $expected) { + throw "refusing Go cache volume ${Name}: EPAR ownership labels do not match this project" + } + return + } + docker volume create ` + --label "io.solutionforest.epar.project=$ProjectID" ` + --label "io.solutionforest.epar.cache=$Role" ` + --label 'io.solutionforest.epar.schema=1' ` + --label "io.solutionforest.epar.root=$repoRootPath" ` + $Name | Out-Null + if ($LASTEXITCODE -ne 0) { throw "failed to create exact EPAR Go cache volume $Name" } + $inspection = Get-EparGoCacheVolumeIdentity -Name $Name + if (-not $inspection.Exists -or $inspection.Identity -cne $expected) { + throw "refusing Go cache volume ${Name}: post-create ownership labels do not match this project" + } +} + +function Invoke-EparGoCacheLimit { + $active = @(@( + docker ps -q --filter "volume=$GomodVolume" + docker ps -q --filter "volume=$GocacheVolume" + ) | Where-Object { $_ } | Sort-Object -Unique) + if (@($active).Count -gt 0) { + Write-Warning 'EPAR Go cache limit check skipped because an exact cache volume is active.' + return + } + $gcName = "epar-$ProjectID-go-cache-gc" + $existingGC = @((docker ps -aq --filter "name=^/$gcName$") | Where-Object { $_ }) + if (@($existingGC).Count -gt 0) { + Write-Warning "EPAR Go cache limit check skipped because $gcName already exists." + return + } + $usageLines = @(docker run --rm ` + --name $gcName ` + -v "${GomodVolume}:/go/pkg/mod" ` + -v "${GocacheVolume}:/root/.cache/go-build" ` + $DevImage ` + du -sk /go/pkg/mod /root/.cache/go-build) + if ($LASTEXITCODE -ne 0) { throw 'failed to measure the exact EPAR Go cache volumes' } + [uint64] $usedKiB = 0 + foreach ($line in $usageLines) { + if ($line -notmatch '^\s*([0-9]+)\s+') { + throw "Docker returned an invalid Go cache usage line: $line" + } + [uint64] $entryKiB = 0 + if (-not [uint64]::TryParse($Matches[1], [ref] $entryKiB) -or $usedKiB -gt ([uint64]::MaxValue - $entryKiB)) { + throw 'Go cache usage exceeds the supported range' + } + $usedKiB += $entryKiB + } + if ($usageLines.Count -ne 2 -or $usedKiB -gt ([uint64]::MaxValue / 1024)) { + throw 'Docker returned an incomplete or overflowing Go cache measurement' + } + [uint64] $usedBytes = $usedKiB * 1024 + if ($usedBytes -gt $GoCacheLimitBytes) { + docker run --rm ` + --name $gcName ` + -v "${GomodVolume}:/go/pkg/mod" ` + -v "${GocacheVolume}:/root/.cache/go-build" ` + $DevImage ` + go clean -cache -modcache + if ($LASTEXITCODE -ne 0) { throw 'failed to clear the exact EPAR Go cache volumes' } + } +} + +function Get-EparNativeSourceHash { + param( + [Parameter(Mandatory = $true)][string] $DevImageID, + [Parameter(Mandatory = $true)][string] $GitCommit, + [Parameter(Mandatory = $true)][string] $SourceState + ) + $sourceFiles = @( + Get-ChildItem -LiteralPath (Join-Path $RepoRoot 'cmd'), (Join-Path $RepoRoot 'internal') -Filter '*.go' -File -Recurse + Get-ChildItem -LiteralPath (Join-Path $RepoRoot 'scripts\docker') -File -Recurse + Get-Item -LiteralPath (Join-Path $RepoRoot 'go.mod'), (Join-Path $RepoRoot 'go.sum') + Get-Item -LiteralPath $MyInvocation.ScriptName + ) | Sort-Object FullName + $material = [System.Text.StringBuilder]::new() + [void] $material.AppendLine('windows/amd64') + [void] $material.AppendLine($DevImageID) + [void] $material.AppendLine($GitCommit) + [void] $material.AppendLine($SourceState) + foreach ($file in $sourceFiles) { + $relative = $file.FullName.Substring($RepoRoot.Length).TrimStart([char[]]@('\', '/')).Replace('\', '/') + $digest = (Get-FileHash -LiteralPath $file.FullName -Algorithm SHA256).Hash.ToLowerInvariant() + [void] $material.AppendLine($relative) + [void] $material.AppendLine($digest) + } + $sha = [System.Security.Cryptography.SHA256]::Create() + try { + $bytes = [System.Text.Encoding]::UTF8.GetBytes($material.ToString()) + return ([BitConverter]::ToString($sha.ComputeHash($bytes))).Replace('-', '').ToLowerInvariant() + } finally { + $sha.Dispose() + } +} + +function Write-EparBootstrapAcquisitionJournal { + param( + [Parameter(Mandatory = $true)][string] $Phase, + [AllowEmptyString()][string] $PreviousGoImageID = '', + [AllowEmptyString()][string] $ResolvedGoImageID = '', + [AllowEmptyString()][string] $ResolvedDevImageID = '', + [AllowEmptyString()][string] $PreviousDevImageID = '' + ) + # The native wrapper runs before the controller can update the shared host + # catalog. Keep a deliberately narrow, atomic hand-off record for it. + $journalDirectory = Join-Path $RepoRoot '.local\storage\bootstrap' + New-Item -ItemType Directory -Force -Path $journalDirectory | Out-Null + $journalPath = Join-Path $journalDirectory 'native-controller-acquisition.json' + $temporaryPath = Join-Path $journalDirectory ('.native-controller-acquisition-' + [guid]::NewGuid().ToString('N') + '.tmp') + $record = [ordered]@{ + schemaVersion = 1 + projectID = $ProjectID + projectRoot = $repoRootPath + phase = $Phase + goImage = $GoImage + devImage = $DevImage + previousGoImageID = $PreviousGoImageID + previousDevImageID = $PreviousDevImageID + resolvedGoImageID = $ResolvedGoImageID + resolvedDevImageID = $ResolvedDevImageID + updatedAtUtc = [DateTime]::UtcNow.ToString('o') + } + [System.IO.File]::WriteAllText($temporaryPath, ($record | ConvertTo-Json -Compress), [System.Text.UTF8Encoding]::new($false)) + Move-Item -LiteralPath $temporaryPath -Destination $journalPath -Force +} + +function Get-EparDockerImageID { + param([Parameter(Mandatory = $true)][string] $Reference) + $previousErrorActionPreference = $ErrorActionPreference + try { + $ErrorActionPreference = 'Continue' + $imageID = ((docker image inspect --format '{{.Id}}' $Reference 2>$null) -join '').Trim() + $inspectExitCode = $LASTEXITCODE + } finally { + $ErrorActionPreference = $previousErrorActionPreference + } + if ($inspectExitCode -ne 0) { return '' } + if ($imageID -notmatch '^sha256:[0-9a-f]{64}$') { + throw "Docker returned an invalid immutable image ID for $Reference" + } + return $imageID +} + +function Resolve-EparGoToolchainImage { + param([AllowEmptyString()][string] $PreviousDevImageID = '') + $previousImageID = Get-EparDockerImageID -Reference $GoImage + Write-EparBootstrapAcquisitionJournal -Phase 'pulling-go-toolchain' -PreviousGoImageID $previousImageID -PreviousDevImageID $PreviousDevImageID + docker pull $GoImage | Out-Host + if ($LASTEXITCODE -ne 0) { throw "failed to resolve the current Go toolchain image $GoImage" } + $resolvedImageID = Get-EparDockerImageID -Reference $GoImage + if (-not $resolvedImageID) { throw "could not resolve the immutable Docker image ID for $GoImage after pull" } + Write-EparBootstrapAcquisitionJournal -Phase 'go-toolchain-resolved' -PreviousGoImageID $previousImageID -ResolvedGoImageID $resolvedImageID -PreviousDevImageID $PreviousDevImageID + return [pscustomobject]@{ PreviousImageID = $previousImageID; ResolvedImageID = $resolvedImageID } +} + +function Read-EparStableNativeControllerManifest { + param([Parameter(Mandatory = $true)][string] $Path) + if (-not (Test-Path -LiteralPath $Path -PathType Leaf)) { return $null } + $fields = @{} + foreach ($line in @(Get-Content -LiteralPath $Path -ErrorAction Stop)) { + $separator = $line.IndexOf('=') + if ($separator -le 0) { return $null } + $key = $line.Substring(0, $separator) + if ($fields.ContainsKey($key)) { return $null } + $fields[$key] = $line.Substring($separator + 1) + } + if ($fields.schemaVersion -ne '2' -or $fields.executable -ne 'ephemeral-action-runner.exe' -or $fields.fingerprint -notmatch '^[0-9a-f]{64}$' -or $fields.toolchainImageID -notmatch '^sha256:[0-9a-f]{64}$') { return $null } + return $fields +} + +function Enter-EparStableNativeControllerBuildLock { + param([Parameter(Mandatory = $true)][string] $Path) + $deadline = [DateTime]::UtcNow.AddMinutes(2) + while ($true) { + try { + return [System.IO.File]::Open($Path, [System.IO.FileMode]::OpenOrCreate, [System.IO.FileAccess]::ReadWrite, [System.IO.FileShare]::None) + } catch [System.IO.IOException] { + if ([DateTime]::UtcNow -ge $deadline) { throw 'another EPAR native-controller build is still in progress; wait for it to finish and retry.' } + Start-Sleep -Milliseconds 200 + } + } +} + +function Get-EparDirectoryBytes { + param([Parameter(Mandatory = $true)][string] $Path) + $measurement = Get-ChildItem -LiteralPath $Path -File -Force -Recurse -ErrorAction Stop | Measure-Object -Property Length -Sum + if ($null -eq $measurement.Sum) { return [int64] 0 } + return [int64] $measurement.Sum +} + +function Test-EparNativeControllerLeaseActive { + param([Parameter(Mandatory = $true)][string] $Directory) + $now = [DateTime]::UtcNow + foreach ($lease in @(Get-ChildItem -LiteralPath $Directory -File -Force -ErrorAction SilentlyContinue | Where-Object { $_.Name -match '^lease(?:-|\.)' })) { + $fields = @{} + try { + foreach ($line in @(Get-Content -LiteralPath $lease.FullName -ErrorAction Stop)) { + $separator = $line.IndexOf('=') + if ($separator -gt 0) { + $fields[$line.Substring(0, $separator)] = $line.Substring($separator + 1) + } + } + } catch { + return $true + } + $startedAt = [DateTime]::MinValue + if (-not [DateTime]::TryParse($fields.startedAtUtc, [ref] $startedAt)) { + return $true + } + $startedAt = $startedAt.ToUniversalTime() + if ($fields.host -ne [Environment]::MachineName) { + if (($now - $startedAt) -lt [TimeSpan]::FromDays(30)) { return $true } + continue + } + $leasePID = 0 + if (-not [int]::TryParse($fields.pid, [ref] $leasePID) -or $leasePID -le 0) { + return $true + } + try { + $process = Get-Process -Id $leasePID -ErrorAction Stop + if ($fields.processStartUtc) { + $expectedStart = [DateTime]::MinValue + if (-not [DateTime]::TryParse($fields.processStartUtc, [ref] $expectedStart)) { + return $true + } + if ($process.StartTime.ToUniversalTime() -ne $expectedStart.ToUniversalTime()) { + continue + } + } + return $true + } catch { + continue + } + } + return $false +} + +function Test-EparNativeControllerBuildLeaseValid { + param([Parameter(Mandatory = $true)][System.IO.FileInfo] $Lease) + if (($Lease.Attributes -band [System.IO.FileAttributes]::ReparsePoint) -ne 0 -or $Lease.Name -notmatch '^lease-build-([1-9][0-9]*)-[0-9a-f]{32}\.txt$') { + return $false + } + $namePID = $Matches[1] + $fields = @{} + try { + foreach ($line in @(Get-Content -LiteralPath $Lease.FullName -ErrorAction Stop)) { + $separator = $line.IndexOf('=') + if ($separator -le 0) { return $false } + $key = $line.Substring(0, $separator) + if ($fields.ContainsKey($key)) { return $false } + $fields[$key] = $line.Substring($separator + 1) + } + } catch { + return $false + } + if ($fields.Count -ne 5 -or $fields.schemaVersion -ne '1' -or -not $fields.host) { return $false } + $leasePID = 0 + if (-not [int]::TryParse($fields.pid, [ref] $leasePID) -or $leasePID -le 0 -or $leasePID.ToString() -ne $namePID) { return $false } + $processStartUtc = [DateTime]::MinValue + if (-not [DateTime]::TryParse($fields.processStartUtc, [ref] $processStartUtc)) { return $false } + $startedAtUtc = [DateTime]::MinValue + return [DateTime]::TryParse($fields.startedAtUtc, [ref] $startedAtUtc) +} + +function Invoke-EparNativeControllerCacheRetention { + param( + [Parameter(Mandatory = $true)][string] $CacheRoot, + [Parameter(Mandatory = $true)][ValidatePattern('^[0-9a-f]{64}$')][string] $CurrentCacheKey, + [ValidateRange(0, 50)][int] $KeepPrevious = 5, + [ValidateRange(1, [long]::MaxValue)][long] $MaxBytes = 268435456, + [TimeSpan] $GracePeriod = ([TimeSpan]::FromDays(7)), + [TimeSpan] $AbandonedBuildGracePeriod = ([TimeSpan]::FromHours(24)), + [switch] $RemoveCurrent + ) + if (-not (Test-Path -LiteralPath $CacheRoot -PathType Container)) { return } + $resolvedRoot = [System.IO.Path]::GetFullPath($CacheRoot).TrimEnd([System.IO.Path]::DirectorySeparatorChar, [System.IO.Path]::AltDirectorySeparatorChar) + $expectedCurrent = [System.IO.Path]::GetFullPath((Join-Path $resolvedRoot $CurrentCacheKey)) + $now = [DateTime]::UtcNow + + foreach ($directory in @(Get-ChildItem -LiteralPath $resolvedRoot -Directory -Force | Where-Object { $_.Name -match '^\.build[-.][0-9A-Za-z]+$' })) { + if (($directory.Attributes -band [System.IO.FileAttributes]::ReparsePoint) -ne 0) { continue } + $buildLeases = @(Get-ChildItem -LiteralPath $directory.FullName -File -Force -ErrorAction SilentlyContinue | Where-Object { Test-EparNativeControllerBuildLeaseValid -Lease $_ }) + if ($buildLeases.Count -ne 1) { continue } + if (Test-EparNativeControllerLeaseActive -Directory $directory.FullName) { continue } + if (($now - $directory.LastWriteTimeUtc) -lt $AbandonedBuildGracePeriod) { continue } + $candidate = [System.IO.Path]::GetFullPath($directory.FullName) + if ([System.IO.Path]::GetDirectoryName($candidate) -ne $resolvedRoot) { continue } + Remove-Item -LiteralPath $candidate -Recurse -Force -ErrorAction Stop + } + + $entries = @() + foreach ($directory in @(Get-ChildItem -LiteralPath $resolvedRoot -Directory -Force | Where-Object { $_.Name -match '^[0-9a-f]{64}$' })) { + if (($directory.Attributes -band [System.IO.FileAttributes]::ReparsePoint) -ne 0) { continue } + $candidate = [System.IO.Path]::GetFullPath($directory.FullName) + if ($candidate -eq $expectedCurrent -and -not $RemoveCurrent) { continue } + if ([System.IO.Path]::GetDirectoryName($candidate) -ne $resolvedRoot) { continue } + if (Test-EparNativeControllerLeaseActive -Directory $candidate) { continue } + + $files = @(Get-ChildItem -LiteralPath $candidate -File -Force -ErrorAction Stop) + $directories = @(Get-ChildItem -LiteralPath $candidate -Directory -Force -ErrorAction Stop) + if ($directories.Count -ne 0) { continue } + if (@($files | Where-Object { ($_.Attributes -band [System.IO.FileAttributes]::ReparsePoint) -ne 0 }).Count -ne 0) { continue } + $manifest = $files | Where-Object { $_.Name -eq 'controller-cache.manifest' } | Select-Object -First 1 + if ($null -eq $manifest) { continue } + $manifestLines = @(Get-Content -LiteralPath $manifest.FullName -ErrorAction Stop) + if (-not $manifestLines.Contains('schemaVersion=1') -or -not $manifestLines.Contains("cacheKey=$($directory.Name)")) { continue } + $executableLines = @($manifestLines | Where-Object { $_ -match '^executable=' }) + if ($executableLines.Count -ne 1) { continue } + $executable = $executableLines[0].Substring('executable='.Length) + if ($executable -notin @('ephemeral-action-runner', 'ephemeral-action-runner.exe')) { continue } + if (-not (Test-Path -LiteralPath (Join-Path $candidate $executable) -PathType Leaf)) { continue } + $unexpected = @($files | Where-Object { $_.Name -notin @($executable, 'controller-cache.manifest') -and $_.Name -notmatch '^lease(?:-|\.)' }) + if ($unexpected.Count -ne 0) { continue } + $entries += [pscustomobject]@{ + Path = $candidate + Name = $directory.Name + LastWriteTimeUtc = $directory.LastWriteTimeUtc + Bytes = Get-EparDirectoryBytes -Path $candidate + } + } + + $retainedCount = 0 + $retainedBytes = if (Test-Path -LiteralPath $expectedCurrent -PathType Container) { Get-EparDirectoryBytes -Path $expectedCurrent } else { [int64] 0 } + foreach ($entry in @($entries | Sort-Object -Property @{ Expression = 'LastWriteTimeUtc'; Descending = $true }, @{ Expression = 'Name'; Descending = $false })) { + $withinGrace = ($now - $entry.LastWriteTimeUtc) -lt $GracePeriod + $withinCount = $retainedCount -lt $KeepPrevious + $withinBudget = $retainedBytes -le ($MaxBytes - $entry.Bytes) + if ($withinGrace -or ($withinCount -and $withinBudget)) { + $retainedCount++ + $retainedBytes += $entry.Bytes + continue + } + $candidate = [System.IO.Path]::GetFullPath($entry.Path) + if ([System.IO.Path]::GetDirectoryName($candidate) -ne $resolvedRoot -or [System.IO.Path]::GetFileName($candidate) -notmatch '^[0-9a-f]{64}$') { + throw "refusing native-controller cache retention outside the exact cache root: $candidate" + } + Remove-Item -LiteralPath $candidate -Recurse -Force -ErrorAction Stop + } +} + +function Test-EparBenignDockerDesktopPrefaceDiagnostic { + param([Parameter(Mandatory = $true)][string] $Transcript) + $normalized = ($Transcript -replace '\s+', ' ').Trim() + return $normalized -match '^(?:docker\s*:\s*)?\d{4}/\d{2}/\d{2} \d{2}:\d{2}:\d{2} http2: server: error reading preface from client //\./pipe/(?:dockerDesktopLinuxEngine|docker_engine): file has already been closed(?: At .* FullyQualifiedErrorId\s*:\s*NativeCommandError)?$' +} + +function Test-EparRetryableDockerContextMetadataDiagnostic { + param([Parameter(Mandatory = $true)][string] $Transcript) + $normalized = ($Transcript -replace '\s+', ' ').Trim() + return $normalized -match '^(?:docker(?:\.exe|\.cmd)?\s*:\s*)?ERROR: failed to build: failed to read metadata: open [^:*?"<>|\r\n]+:\\[^:*?"<>|\r\n]*\\\.docker\\contexts\\meta\\[0-9a-f]{64}\\meta\.json: The process cannot access the file because it is being used by another process\.(?: At .* FullyQualifiedErrorId\s*:\s*NativeCommandError)?$' +} + +function Get-EparTLSFailureHost { + param([Parameter(Mandatory = $true)][string] $Transcript) + if ($Transcript -notmatch '(?i)(?:x509:\s*certificate signed by unknown authority|certificate verify failed|unable to (?:get local issuer certificate|verify the first certificate))') { + return '' + } + foreach ($match in [regex]::Matches($Transcript, 'https://(?[A-Za-z0-9.-]+)(?=[:/"])', [System.Text.RegularExpressions.RegexOptions]::IgnoreCase)) { + $hostName = $match.Groups['host'].Value.Trim().ToLowerInvariant() + if ($hostName -match '^[a-z0-9](?:[a-z0-9.-]*[a-z0-9])?$') { + return $hostName + } + } + return '' +} + +function Get-EparCertificateSHA256 { + param([Parameter(Mandatory = $true)][System.Security.Cryptography.X509Certificates.X509Certificate2] $Certificate) + $algorithm = [System.Security.Cryptography.SHA256]::Create() + try { + return ([BitConverter]::ToString($algorithm.ComputeHash($Certificate.RawData))).Replace('-', '') + } finally { + $algorithm.Dispose() + } +} + +function Find-EparWindowsIssuerRoots { + param([Parameter(Mandatory = $true)][string] $IssuerCommonName) + $matches = [System.Collections.Generic.List[object]]::new() + foreach ($store in @( + [pscustomobject]@{ Name = 'LocalMachine\Root'; Path = 'Cert:\LocalMachine\Root' }, + [pscustomobject]@{ Name = 'CurrentUser\Root'; Path = 'Cert:\CurrentUser\Root' } + )) { + if (-not (Test-Path -LiteralPath $store.Path)) { continue } + foreach ($certificate in Get-ChildItem -LiteralPath $store.Path -ErrorAction SilentlyContinue) { + if ($certificate.GetNameInfo([System.Security.Cryptography.X509Certificates.X509NameType]::SimpleName, $false) -cne $IssuerCommonName) { continue } + $matches.Add([pscustomobject]@{ + Store = $store.Name + Subject = $certificate.Subject + SHA1 = $certificate.Thumbprint + SHA256 = Get-EparCertificateSHA256 -Certificate $certificate + NotAfter = $certificate.NotAfter.ToString('o') + }) + } + } + return @($matches) +} + +function Invoke-EparTLSFailureDiagnostic { + param( + [Parameter(Mandatory = $true)][string] $Transcript, + [Parameter(Mandatory = $true)][string] $LogPath + ) + $hostName = Get-EparTLSFailureHost -Transcript $Transcript + if (-not $hostName) { return } + + $diagnosticScript = @' +set -u +raw="$(mktemp)" +leaf="$(mktemp)" +cleanup() { rm -f -- "$raw" "$leaf"; } +trap cleanup EXIT +openssl s_client -connect "${EPAR_TLS_DIAGNOSTIC_HOST}:443" -servername "${EPAR_TLS_DIAGNOSTIC_HOST}" -showcerts "$raw" 2>&1 || true +awk '/-----BEGIN CERTIFICATE-----/{capture=1} capture{print} /-----END CERTIFICATE-----/{exit}' "$raw" >"$leaf" +grep -E 'verify error|Verify return code' "$raw" || true +if [ -s "$leaf" ]; then + openssl x509 -in "$leaf" -noout -subject -issuer -fingerprint -sha256 -dates +fi +'@ + $previousErrorActionPreference = $ErrorActionPreference + try { + $ErrorActionPreference = 'Continue' + $diagnosticOutput = @(& docker run --rm -e "EPAR_TLS_DIAGNOSTIC_HOST=$hostName" $DevImage sh -c $diagnosticScript 2>&1 | ForEach-Object { "$_" }) + $diagnosticExitCode = $LASTEXITCODE + } finally { + $ErrorActionPreference = $previousErrorActionPreference + } + + $subject = (($diagnosticOutput | Where-Object { $_ -match '^subject=' } | Select-Object -First 1) -replace '^subject=', '').Trim() + $issuer = (($diagnosticOutput | Where-Object { $_ -match '^issuer=' } | Select-Object -First 1) -replace '^issuer=', '').Trim() + $fingerprint = (($diagnosticOutput | Where-Object { $_ -match '^sha256 Fingerprint=' } | Select-Object -First 1) -replace '^sha256 Fingerprint=', '').Replace(':', '').Trim() + $notBefore = (($diagnosticOutput | Where-Object { $_ -match '^notBefore=' } | Select-Object -First 1) -replace '^notBefore=', '').Trim() + $notAfter = (($diagnosticOutput | Where-Object { $_ -match '^notAfter=' } | Select-Object -First 1) -replace '^notAfter=', '').Trim() + $verifyErrors = @($diagnosticOutput | Where-Object { $_ -match '(?i)verify error|Verify return code' }) + + $report = [System.Collections.Generic.List[string]]::new() + $report.Add('') + $report.Add('EPAR TLS certificate diagnostic') + $report.Add(" Requested host: $hostName`:443") + $report.Add(" Toolchain image: $DevImage") + if ($diagnosticExitCode -ne 0 -or -not $subject) { + $report.Add(' Certificate inspection: unavailable; see the raw build error above.') + } else { + $report.Add(' Certificate presented to the build container:') + $report.Add(" Subject: $subject") + $report.Add(" Issuer: $issuer") + if ($fingerprint) { $report.Add(" SHA-256: $fingerprint") } + if ($notBefore) { $report.Add(" Valid from: $notBefore") } + if ($notAfter) { $report.Add(" Valid until: $notAfter") } + foreach ($verifyError in $verifyErrors) { $report.Add(" OpenSSL: $($verifyError.Trim())") } + + $issuerCommonName = '' + if ($issuer -match '(?:^|,)\s*CN\s*=\s*(?[^,]+)') { + $issuerCommonName = $Matches['cn'].Trim() + } + $roots = if ($issuerCommonName) { @(Find-EparWindowsIssuerRoots -IssuerCommonName $issuerCommonName) } else { @() } + if ($roots.Count -eq 0) { + $report.Add(' Candidate Windows root: none with the same issuer name was found in LocalMachine\Root or CurrentUser\Root.') + } else { + $report.Add(' Candidate Windows root certificate(s) with the same issuer name:') + foreach ($root in $roots) { + $report.Add(" Store: $($root.Store)") + $report.Add(" Subject: $($root.Subject)") + $report.Add(" SHA-1: $($root.SHA1)") + $report.Add(" SHA-256: $($root.SHA256)") + $report.Add(" Valid until: $($root.NotAfter)") + } + $report.Add(' Interpretation: Windows has candidate issuer trust, but the Linux bootstrap container does not currently trust that issuer.') + } + } + $report.Add(' TLS verification was not disabled, and EPAR did not retry the download insecurely.') + $report.Add(" Full native-controller build log: $LogPath") + + foreach ($line in $report) { [Console]::Error.WriteLine($line) } + [System.IO.File]::AppendAllLines($LogPath, $report, [System.Text.UTF8Encoding]::new($false)) +} + +function Invoke-EparDockerBuild { + $stderrPath = [System.IO.Path]::GetTempFileName() + $previousErrorActionPreference = $ErrorActionPreference + try { + # Windows PowerShell otherwise promotes native stderr to terminating + # ErrorRecord objects under Stop, before the Docker exit code and the + # complete transcript can be classified below. + $ErrorActionPreference = 'Continue' + $maximumAttempts = 5 + for ($attempt = 1; $attempt -le $maximumAttempts; $attempt++) { + docker build --quiet --provenance=false --build-arg "GO_IMAGE=$GoImage" -t $DevImage -f (Join-Path $RepoRoot 'scripts\docker\dev.Dockerfile') (Join-Path $RepoRoot 'scripts\docker') 2> $stderrPath | Out-Null + $exitCode = $LASTEXITCODE + $stderrTranscript = if (Test-Path -LiteralPath $stderrPath) { Get-Content -Raw -LiteralPath $stderrPath } else { '' } + if ($exitCode -ne 0 -and $attempt -lt $maximumAttempts -and (Test-EparRetryableDockerContextMetadataDiagnostic -Transcript $stderrTranscript)) { + $delayMilliseconds = 250 * [math]::Pow(2, $attempt - 1) + Write-Warning "Docker context metadata is temporarily busy; retrying toolchain build in $delayMilliseconds ms (attempt $($attempt + 1) of $maximumAttempts)." + Start-Sleep -Milliseconds $delayMilliseconds + continue + } + if ($stderrTranscript -and -not ($exitCode -eq 0 -and (Test-EparBenignDockerDesktopPrefaceDiagnostic -Transcript $stderrTranscript))) { + [Console]::Error.Write($stderrTranscript) + } + return $exitCode + } + throw 'unreachable Docker build retry state' + } finally { + $ErrorActionPreference = $previousErrorActionPreference + Remove-Item -LiteralPath $stderrPath -Force -ErrorAction SilentlyContinue + } +} + +function Initialize-EparBootstrapBuildTrust { + param( + [Parameter(Mandatory = $true)][string] $ProjectRoot, + [string[]] $Arguments + ) + + $configPath = Get-EparHostTrustConfigPath -ProjectRoot $ProjectRoot -Arguments $Arguments + $helper = Join-Path $ProjectRoot 'scripts\host-trust\host-trust-feed.ps1' + $previousErrorActionPreference = $ErrorActionPreference + try { + $ErrorActionPreference = 'Continue' + $feedOutput = @(& $helper sync -ProjectRoot $ProjectRoot -Config $configPath -Purpose build 2>&1 | ForEach-Object { "$_" }) + $feedExitCode = $LASTEXITCODE + } finally { + $ErrorActionPreference = $previousErrorActionPreference + } + if ($feedExitCode -ne 0) { + throw "could not publish host trust for the native-controller build: $($feedOutput -join [Environment]::NewLine)" + } + $feedPath = ($feedOutput | Where-Object { -not [string]::IsNullOrWhiteSpace($_) } | Select-Object -Last 1) + if (-not $feedPath) { + throw 'the host trust publisher did not return a build feed' + } + $feedPath = [System.IO.Path]::GetFullPath($feedPath.Trim()) + $feedItem = Get-Item -LiteralPath $feedPath -Force -ErrorAction Stop + if (-not (Test-Path -LiteralPath $feedPath -PathType Leaf)) { + throw "bootstrap build trust feed is not a regular file: $feedPath" + } + if ($feedItem.Attributes -band [System.IO.FileAttributes]::ReparsePoint) { + throw "bootstrap build trust feed must not be a reparse point: $feedPath" + } + + $configHasher = [System.Security.Cryptography.SHA256]::Create() + try { + $configMaterial = [System.Text.Encoding]::UTF8.GetBytes($configPath.ToLowerInvariant()) + $configID = ([BitConverter]::ToString($configHasher.ComputeHash($configMaterial))).Replace('-', '').ToLowerInvariant().Substring(0, 32) + } finally { + $configHasher.Dispose() + } + $bundleDirectory = Join-Path $ProjectRoot ".local\storage\bootstrap-trust\$configID" + New-Item -ItemType Directory -Force -Path $bundleDirectory | Out-Null + $bundleDirectoryItem = Get-Item -LiteralPath $bundleDirectory -Force + if ($bundleDirectoryItem.Attributes -band [System.IO.FileAttributes]::ReparsePoint) { + throw "bootstrap build trust directory must not be a reparse point: $bundleDirectory" + } + $bundlePath = Join-Path $bundleDirectory 'ca.pem' + if (Test-Path -LiteralPath $bundlePath) { + $existingBundle = Get-Item -LiteralPath $bundlePath -Force + if (-not (Test-Path -LiteralPath $bundlePath -PathType Leaf) -or ($existingBundle.Attributes -band [System.IO.FileAttributes]::ReparsePoint)) { + throw "bootstrap build trust output must be a regular non-reparse file: $bundlePath" + } + } + + $validatorDirectory = Join-Path $ProjectRoot 'scripts\bootstrap-trust' + try { + $ErrorActionPreference = 'Continue' + $validatorOutput = @(& docker run --rm ` + --network none ` + -e GO111MODULE=off ` + -e GOTOOLCHAIN=local ` + -v "${validatorDirectory}:/bootstrap:ro" ` + -v "${feedPath}:/feed/current.json:ro" ` + -v "${bundleDirectory}:/out" ` + $DevImage ` + /usr/local/go/bin/go run /bootstrap/main.go --feed /feed/current.json --output /out/ca.pem --expected-host-os windows 2>&1 | ForEach-Object { "$_" }) + $validatorExitCode = $LASTEXITCODE + } finally { + $ErrorActionPreference = $previousErrorActionPreference + } + if ($validatorExitCode -ne 0) { + throw "bootstrap build trust validation failed: $($validatorOutput -join [Environment]::NewLine)" + } + $bundleItem = Get-Item -LiteralPath $bundlePath -Force -ErrorAction Stop + if (-not (Test-Path -LiteralPath $bundlePath -PathType Leaf) -or $bundleItem.Length -eq 0 -or ($bundleItem.Attributes -band [System.IO.FileAttributes]::ReparsePoint)) { + throw "bootstrap build trust validator did not produce a regular nonempty bundle: $bundlePath" + } + $summary = ($validatorOutput | Where-Object { -not [string]::IsNullOrWhiteSpace($_) } | Select-Object -Last 1) + return [pscustomobject]@{ BundlePath = $bundlePath; Summary = $summary; ConfigPath = $configPath } +} + +$previousDevImageID = Get-EparDockerImageID -Reference $DevImage +$goToolchain = Resolve-EparGoToolchainImage -PreviousDevImageID $previousDevImageID +$buildExit = Invoke-EparDockerBuild +if ($buildExit -ne 0) { exit $buildExit } +$DevImageID = Get-EparDockerImageID -Reference $DevImage +if (-not $DevImageID) { Write-Error "could not resolve the immutable Docker toolchain image ID for $DevImage"; exit 1 } +Write-EparBootstrapAcquisitionJournal -Phase 'toolchain-built' -PreviousGoImageID $goToolchain.PreviousImageID -ResolvedGoImageID $goToolchain.ResolvedImageID -ResolvedDevImageID $DevImageID -PreviousDevImageID $previousDevImageID +if ($ManageGoCache) { + Initialize-EparGoCacheVolume -Name $GomodVolume -Role 'gomod' + Initialize-EparGoCacheVolume -Name $GocacheVolume -Role 'gobuild' + Invoke-EparGoCacheLimit +} + +$gitCommit = 'unknown' +$sourceState = 'unknown' +if (Get-Command git -ErrorAction SilentlyContinue) { + $gitCommitOutput = ((& git -C $RepoRoot rev-parse --verify HEAD 2>$null) -join '').Trim() + if ($LASTEXITCODE -eq 0 -and $gitCommitOutput -match '^[0-9a-f]{40}$') { + $gitCommit = $gitCommitOutput + $gitStatus = @(& git -C $RepoRoot status --porcelain=v1 --untracked-files=all 2>$null) + if ($LASTEXITCODE -eq 0) { + $sourceState = if ($gitStatus.Count -eq 0) { 'clean' } else { 'dirty' } + } + } +} + +$fingerprint = Get-EparNativeSourceHash -DevImageID $DevImageID -GitCommit $gitCommit -SourceState $sourceState +$controllerSourceRevision = if ($sourceState -eq 'clean') { "sha256:$fingerprint" } elseif ($sourceState -eq 'dirty') { "dirty:sha256:$fingerprint" } else { 'unknown' } +$cacheRoot = Join-Path $RepoRoot '.local\bin' +$binary = Join-Path $cacheRoot 'ephemeral-action-runner.exe' +$manifestPath = Join-Path $cacheRoot 'ephemeral-action-runner.manifest' +New-Item -ItemType Directory -Force -Path $cacheRoot | Out-Null + +# Historical hash directories were created by older no-Go wrappers. Delete only +# complete, exactly shaped, inactive revisions; unknown paths are intentionally +# left for storage's legacy inventory. +try { + Invoke-EparNativeControllerCacheRetention -CacheRoot $cacheRoot -CurrentCacheKey ([string]::new([char]'0', 64)) -KeepPrevious 0 -MaxBytes 1 -GracePeriod ([TimeSpan]::Zero) -RemoveCurrent +} catch { + Write-Warning "Native-controller legacy revision cleanup skipped after an error: $($_.Exception.Message)" +} + +$existingManifest = Read-EparStableNativeControllerManifest -Path $manifestPath +$needsBuild = -not (Test-Path -LiteralPath $binary -PathType Leaf) -or $null -eq $existingManifest -or $existingManifest.fingerprint -cne $fingerprint -or $existingManifest.toolchainImageID -cne $DevImageID +if ($needsBuild -and $null -ne $existingManifest -and (Test-Path -LiteralPath $binary -PathType Leaf) -and (Test-EparNativeControllerLeaseActive -Directory $cacheRoot)) { + # A mutable bootstrap dependency can produce a new dev-toolchain image + # while an already-built controller is serving another invocation. Reuse + # that stable binary only when the current source still fingerprints + # exactly against the toolchain recorded beside it. Real source changes + # continue to fail with the stop-running-process instruction below. + $activeControllerFingerprint = Get-EparNativeSourceHash -DevImageID $existingManifest.toolchainImageID -GitCommit $gitCommit -SourceState $sourceState + if ($existingManifest.fingerprint -ceq $activeControllerFingerprint) { + $needsBuild = $false + } +} +if ($needsBuild) { + $buildLock = Enter-EparStableNativeControllerBuildLock -Path (Join-Path $cacheRoot '.native-controller.lock') + try { + $existingManifest = Read-EparStableNativeControllerManifest -Path $manifestPath + $needsBuild = -not (Test-Path -LiteralPath $binary -PathType Leaf) -or $null -eq $existingManifest -or $existingManifest.fingerprint -cne $fingerprint -or $existingManifest.toolchainImageID -cne $DevImageID + if ($needsBuild) { + if (Test-EparNativeControllerLeaseActive -Directory $cacheRoot) { + throw 'EPAR source or its Go toolchain changed while a native EPAR controller is running. Stop the running EPAR process, then run ./start again; EPAR keeps one stable native controller binary and will not create another versioned copy.' + } + $buildLogDirectory = Join-Path $RepoRoot 'work\logs' + New-Item -ItemType Directory -Force -Path $buildLogDirectory | Out-Null + $buildLogPath = Join-Path $buildLogDirectory 'epar-native-controller-build.log' + $bootstrapBuildTrust = Initialize-EparBootstrapBuildTrust -ProjectRoot $RepoRoot -Arguments $EparArgs + [System.IO.File]::WriteAllLines($buildLogPath, @( + "EPAR native-controller build started at $([DateTime]::UtcNow.ToString('o'))", + "Toolchain image: $DevImage", + "Target: windows/amd64", + "Bootstrap build trust: $($bootstrapBuildTrust.Summary)", + '' + ), [System.Text.UTF8Encoding]::new($false)) + Write-Output "Native controller build log: $buildLogPath" + $temporaryDirectory = Join-Path $cacheRoot ('.build-' + [guid]::NewGuid().ToString('N')) + New-Item -ItemType Directory -Path $temporaryDirectory | Out-Null + $buildLeasePath = Join-Path $temporaryDirectory ("lease-build-{0}-{1}.txt" -f $PID, [guid]::NewGuid().ToString('N')) + $buildProcessStartUtc = (Get-Process -Id $PID).StartTime.ToUniversalTime().ToString('o') + [System.IO.File]::WriteAllLines($buildLeasePath, @('schemaVersion=1', "host=$([Environment]::MachineName)", "pid=$PID", "processStartUtc=$buildProcessStartUtc", "startedAtUtc=$([DateTime]::UtcNow.ToString('o'))"), [System.Text.UTF8Encoding]::new($false)) + try { + $previousErrorActionPreference = $ErrorActionPreference + try { + $ErrorActionPreference = 'Continue' + $buildOutput = @(& docker run --rm ` + -e CGO_ENABLED=0 ` + -e GOOS=windows ` + -e GOARCH=amd64 ` + -e GOTOOLCHAIN=local ` + -e SSL_CERT_FILE=/run/epar-bootstrap-ca.pem ` + -v "${RepoRoot}:/src:ro" ` + -v "${temporaryDirectory}:/out" ` + -v "${GomodVolume}:/go/pkg/mod" ` + -v "${GocacheVolume}:/root/.cache/go-build" ` + -v "$($bootstrapBuildTrust.BundlePath):/run/epar-bootstrap-ca.pem:ro" ` + -w /src ` + $DevImage ` + go build -trimpath -ldflags "-X main.sourceRevision=$controllerSourceRevision" -o /out/ephemeral-action-runner.exe ./cmd/ephemeral-action-runner 2>&1 | ForEach-Object { "$_" }) + $nativeBuildExitCode = $LASTEXITCODE + } finally { + $ErrorActionPreference = $previousErrorActionPreference + } + $buildTranscript = if ($buildOutput.Count -ne 0) { ($buildOutput -join [Environment]::NewLine) + [Environment]::NewLine } else { '' } + if ($buildTranscript) { + [System.IO.File]::AppendAllText($buildLogPath, $buildTranscript, [System.Text.UTF8Encoding]::new($false)) + } + if ($nativeBuildExitCode -ne 0) { + $tlsFailureHost = Get-EparTLSFailureHost -Transcript $buildTranscript + if ($tlsFailureHost) { + [Console]::Error.WriteLine("Native controller build failed while downloading dependencies from https://$tlsFailureHost.") + [Console]::Error.WriteLine(' The build container rejected the presented TLS certificate as an unknown issuer.') + [Console]::Error.WriteLine(" Full compiler output: $buildLogPath") + } elseif ($buildTranscript) { + [Console]::Error.Write($buildTranscript) + } + Invoke-EparTLSFailureDiagnostic -Transcript $buildTranscript -LogPath $buildLogPath + exit $nativeBuildExitCode + } + if ($buildTranscript) { [Console]::Error.Write($buildTranscript) } + $temporaryBinary = Join-Path $temporaryDirectory 'ephemeral-action-runner.exe' + if (-not (Test-Path -LiteralPath $temporaryBinary -PathType Leaf)) { throw 'native EPAR build completed without producing the expected Windows binary' } + try { + Move-Item -LiteralPath $temporaryBinary -Destination $binary -Force + } catch { + throw "could not atomically replace $binary. Stop every EPAR process using the native controller and retry: $($_.Exception.Message)" + } + $manifestTemporaryPath = Join-Path $cacheRoot ('.native-controller-manifest-' + [guid]::NewGuid().ToString('N') + '.tmp') + [System.IO.File]::WriteAllLines($manifestTemporaryPath, @('schemaVersion=2', "fingerprint=$fingerprint", 'executable=ephemeral-action-runner.exe', "toolchainImageID=$DevImageID", "sourceRevision=$controllerSourceRevision", "completedAtUtc=$([DateTime]::UtcNow.ToString('o'))"), [System.Text.UTF8Encoding]::new($false)) + Move-Item -LiteralPath $manifestTemporaryPath -Destination $manifestPath -Force + } finally { + if ($temporaryDirectory -and (Test-Path -LiteralPath $temporaryDirectory)) { Remove-Item -LiteralPath $temporaryDirectory -Recurse -Force -ErrorAction SilentlyContinue } + } + } + } finally { + $buildLock.Dispose() + } +} +if ($ManageGoCache) { + $configuredGoCacheLimit = ((& $binary storage effective-go-cache-limit --project-root $RepoRoot) -join '').Trim() + $parsedConfiguredGoCacheLimit = [uint64] 0 + if ($LASTEXITCODE -ne 0 -or -not [uint64]::TryParse($configuredGoCacheLimit, [ref] $parsedConfiguredGoCacheLimit) -or $parsedConfiguredGoCacheLimit -eq 0) { + throw 'EPAR returned an invalid configured Go cache limit' + } + $GoCacheLimitBytes = $parsedConfiguredGoCacheLimit + Invoke-EparGoCacheLimit +} + +$leasePath = Join-Path $cacheRoot ("lease-native-{0}-{1}.txt" -f $PID, [guid]::NewGuid().ToString('N')) +$processStartUtc = (Get-Process -Id $PID).StartTime.ToUniversalTime().ToString('o') +[System.IO.File]::WriteAllLines($leasePath, @( + 'schemaVersion=1', + "host=$([Environment]::MachineName)", + "pid=$PID", + "processStartUtc=$processStartUtc", + "startedAtUtc=$([DateTime]::UtcNow.ToString('o'))" +), [System.Text.UTF8Encoding]::new($false)) + +$previousNative = $env:EPAR_NATIVE_CONTROLLER +$previousControllerOS = $env:EPAR_CONTROLLER_HOST_OS +$previousHostName = $env:EPAR_HOST_NAME +$previousHints = $env:DOCKER_CLI_HINTS +$controllerCommand = if ($EparArgs -and $EparArgs.Count -gt 0) { [string]$EparArgs[0] } else { 'start' } +$bridge = if ($controllerCommand -eq 'init') { Start-EparHostTrustBridge -ProjectRoot $RepoRoot -Command $controllerCommand -Arguments $EparArgs } else { $null } +try { + $env:EPAR_NATIVE_CONTROLLER = '1' + $env:EPAR_CONTROLLER_HOST_OS = 'windows' + if (-not $env:EPAR_HOST_NAME) { + $env:EPAR_HOST_NAME = if ($env:COMPUTERNAME) { $env:COMPUTERNAME } else { [System.Net.Dns]::GetHostName() } + } + if (-not $env:DOCKER_CLI_HINTS) { $env:DOCKER_CLI_HINTS = 'false' } + & $binary @EparArgs + $nativeExitCode = $LASTEXITCODE + if ($nativeExitCode -eq 0 -and $controllerCommand -eq 'init') { + Complete-EparHostTrustInit -ProjectRoot $RepoRoot -Bridge $bridge + } + exit $nativeExitCode +} finally { + Stop-EparHostTrustBridge -Bridge $bridge + Remove-Item -LiteralPath $leasePath -Force -ErrorAction SilentlyContinue + if ($null -eq $previousNative) { Remove-Item Env:EPAR_NATIVE_CONTROLLER -ErrorAction SilentlyContinue } else { $env:EPAR_NATIVE_CONTROLLER = $previousNative } + if ($null -eq $previousControllerOS) { Remove-Item Env:EPAR_CONTROLLER_HOST_OS -ErrorAction SilentlyContinue } else { $env:EPAR_CONTROLLER_HOST_OS = $previousControllerOS } + if ($null -eq $previousHostName) { Remove-Item Env:EPAR_HOST_NAME -ErrorAction SilentlyContinue } else { $env:EPAR_HOST_NAME = $previousHostName } + if ($null -eq $previousHints) { Remove-Item Env:DOCKER_CLI_HINTS -ErrorAction SilentlyContinue } else { $env:DOCKER_CLI_HINTS = $previousHints } +} diff --git a/scripts/build-native-controller.sh b/scripts/build-native-controller.sh new file mode 100644 index 0000000..9504847 --- /dev/null +++ b/scripts/build-native-controller.sh @@ -0,0 +1,542 @@ +#!/usr/bin/env bash +set -euo pipefail + +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)" +EPAR_HOST_TRUST_HELPER="${repo_root}/scripts/host-trust/host-trust-feed.sh" +source "${repo_root}/scripts/host-trust/wrapper-lib.sh" +go_image="${GO_DOCKER_IMAGE:-golang:latest}" +dev_image="${EPAR_DEV_IMAGE:-epar-dev-toolchain}" +native_cache_keep_previous=5 +native_cache_max_bytes=$((256 * 1024 * 1024)) +native_cache_grace_seconds=$((7 * 24 * 60 * 60)) +abandoned_build_grace_seconds=$((24 * 60 * 60)) +bootstrap_minimum_free_bytes="${EPAR_BOOTSTRAP_MIN_FREE_BYTES:-$((1 * 1024 * 1024 * 1024))}" +go_cache_limit_bytes="${EPAR_GO_CACHE_LIMIT_BYTES:-$((10 * 1024 * 1024 * 1024))}" + +command -v docker >/dev/null 2>&1 || { echo "docker command not found. Install Docker and make sure it is available on PATH." >&2; exit 1; } +command -v shasum >/dev/null 2>&1 || { echo "shasum is required to build the native EPAR cache key." >&2; exit 1; } +[[ "$bootstrap_minimum_free_bytes" =~ ^[1-9][0-9]*$ ]] || { echo "EPAR_BOOTSTRAP_MIN_FREE_BYTES must be a positive integer byte count." >&2; exit 1; } +[[ "$go_cache_limit_bytes" =~ ^[1-9][0-9]*$ ]] || { echo "EPAR_GO_CACHE_LIMIT_BYTES must be a positive integer byte count." >&2; exit 1; } +project_id="$(printf '%s' "$repo_root" | shasum -a 256 | awk '{print substr($1,1,12)}')" +gomod_volume="${EPAR_GOMOD_VOLUME:-epar-${project_id}-gomod}" +gocache_volume="${EPAR_GOCACHE_VOLUME:-epar-${project_id}-gocache}" +manage_go_cache=0 +if [[ -z "${EPAR_GOMOD_VOLUME:-}" && -z "${EPAR_GOCACHE_VOLUME:-}" ]]; then manage_go_cache=1; fi +bootstrap_available_kib="$(df -Pk "$repo_root" | awk 'NR == 2 { print $4 }')" +[[ "$bootstrap_available_kib" =~ ^[0-9]+$ ]] || { echo "cannot measure bootstrap storage for ${repo_root}" >&2; exit 1; } +bootstrap_available_bytes=$((bootstrap_available_kib * 1024)) +if ((bootstrap_available_bytes < bootstrap_minimum_free_bytes)); then + if [[ " $* " == *" --allow-insufficient-storage "* ]]; then + echo "WARNING: bootstrap storage is below the ${bootstrap_minimum_free_bytes}-byte reserve; continuing because --allow-insufficient-storage was explicitly supplied." >&2 + else + echo "insufficient bootstrap storage for ${repo_root}: available=${bootstrap_available_bytes} required-reserve=${bootstrap_minimum_free_bytes}. Free space, inspect storage, or retry this invocation with --allow-insufficient-storage." >&2 + exit 1 + fi +fi + +epar_ensure_go_cache_volume() { + local volume="$1" + local role="$2" + local expected="${project_id}|${role}|1|${repo_root}" + local actual="" + if actual="$(docker volume inspect --format '{{ index .Labels "io.solutionforest.epar.project" }}|{{ index .Labels "io.solutionforest.epar.cache" }}|{{ index .Labels "io.solutionforest.epar.schema" }}|{{ index .Labels "io.solutionforest.epar.root" }}' "$volume" 2>/dev/null)"; then + [[ "$actual" == "$expected" ]] || { echo "refusing Go cache volume ${volume}: EPAR ownership labels do not match this project" >&2; return 1; } + return 0 + fi + docker volume create \ + --label "io.solutionforest.epar.project=${project_id}" \ + --label "io.solutionforest.epar.cache=${role}" \ + --label 'io.solutionforest.epar.schema=1' \ + --label "io.solutionforest.epar.root=${repo_root}" \ + "$volume" >/dev/null + actual="$(docker volume inspect --format '{{ index .Labels "io.solutionforest.epar.project" }}|{{ index .Labels "io.solutionforest.epar.cache" }}|{{ index .Labels "io.solutionforest.epar.schema" }}|{{ index .Labels "io.solutionforest.epar.root" }}' "$volume")" + [[ "$actual" == "$expected" ]] || { echo "refusing Go cache volume ${volume}: post-create ownership labels do not match this project" >&2; return 1; } +} + +epar_enforce_go_cache_limit() { + local active gc_name + active="$( + { + docker ps -q --filter "volume=${gomod_volume}" + docker ps -q --filter "volume=${gocache_volume}" + } | sort -u | sed '/^$/d' + )" + if [[ -n "$active" ]]; then + echo "warning: EPAR Go cache limit check skipped because an exact cache volume is active" >&2 + return 0 + fi + gc_name="epar-${project_id}-go-cache-gc" + if [[ -n "$(docker ps -aq --filter "name=^/${gc_name}$")" ]]; then + echo "warning: EPAR Go cache limit check skipped because ${gc_name} already exists" >&2 + return 0 + fi + docker run --rm \ + --name "$gc_name" \ + -e "EPAR_GO_CACHE_LIMIT_BYTES=${go_cache_limit_bytes}" \ + -v "${gomod_volume}:/go/pkg/mod" \ + -v "${gocache_volume}:/root/.cache/go-build" \ + "$dev_image" \ + sh -ceu 'mod_kib="$(du -sk /go/pkg/mod | awk "{print \$1}")"; build_kib="$(du -sk /root/.cache/go-build | awk "{print \$1}")"; used_bytes="$(((mod_kib + build_kib) * 1024))"; if [ "$used_bytes" -gt "$EPAR_GO_CACHE_LIMIT_BYTES" ]; then go clean -cache -modcache; fi' +} + +epar_write_bootstrap_acquisition_journal() { + local phase="$1" + local previous_id="${2:-}" + local resolved_go_id="${3:-}" + local resolved_dev_id="${4:-}" + local previous_dev_id="${5:-}" + local journal_directory="${repo_root}/.local/storage/bootstrap" + local journal_path="${journal_directory}/native-controller-acquisition.json" + local temporary_path + epar_bootstrap_json_escape() { + local value="$1" + value="${value//\\/\\\\}" + value="${value//\"/\\\"}" + value="${value//$'\n'/\\n}" + value="${value//$'\r'/\\r}" + value="${value//$'\t'/\\t}" + printf '%s' "$value" + } + mkdir -p "$journal_directory" + temporary_path="$(mktemp "${journal_directory}/.native-controller-acquisition.XXXXXX")" + printf '{"schemaVersion":1,"projectID":"%s","projectRoot":"%s","phase":"%s","goImage":"%s","devImage":"%s","previousGoImageID":"%s","previousDevImageID":"%s","resolvedGoImageID":"%s","resolvedDevImageID":"%s","updatedAtUnix":%s}\n' \ + "$(epar_bootstrap_json_escape "$project_id")" "$(epar_bootstrap_json_escape "$repo_root")" "$(epar_bootstrap_json_escape "$phase")" "$(epar_bootstrap_json_escape "$go_image")" "$(epar_bootstrap_json_escape "$dev_image")" "$(epar_bootstrap_json_escape "$previous_id")" "$(epar_bootstrap_json_escape "$previous_dev_id")" "$(epar_bootstrap_json_escape "$resolved_go_id")" "$(epar_bootstrap_json_escape "$resolved_dev_id")" "$(date +%s)" >"$temporary_path" + mv -f -- "$temporary_path" "$journal_path" +} + +epar_docker_image_id() { + local reference="$1" + local image_id + image_id="$(docker image inspect --format '{{.Id}}' "$reference" 2>/dev/null || true)" + if [[ -z "$image_id" ]]; then return 0; fi + [[ "$image_id" =~ ^sha256:[0-9a-f]{64}$ ]] || { echo "Docker returned an invalid immutable image ID for ${reference}" >&2; return 1; } + printf '%s\n' "$image_id" +} + +epar_resolve_go_toolchain_image() { + local previous_id resolved_id + previous_id="$(epar_docker_image_id "$go_image")" + epar_write_bootstrap_acquisition_journal pulling-go-toolchain "$previous_id" '' '' "$previous_dev_image_id" + docker pull "$go_image" >&2 + resolved_id="$(epar_docker_image_id "$go_image")" + [[ -n "$resolved_id" ]] || { echo "could not resolve the immutable Docker image ID for ${go_image} after pull" >&2; return 1; } + epar_write_bootstrap_acquisition_journal go-toolchain-resolved "$previous_id" "$resolved_id" '' "$previous_dev_image_id" + EPAR_GO_PREVIOUS_IMAGE_ID="$previous_id" + EPAR_GO_RESOLVED_IMAGE_ID="$resolved_id" +} + +epar_prepare_bootstrap_build_trust() { + local config_path feed_path config_id bundle_directory validator_output host_os + config_path="$(epar_host_trust_config_path "$repo_root" "$@")" + feed_path="$("$EPAR_HOST_TRUST_HELPER" sync --project-root "$repo_root" --config "$config_path" --purpose build)" + [[ -n "$feed_path" && -f "$feed_path" && ! -L "$feed_path" ]] || { echo "the host trust publisher did not return a regular build feed" >&2; return 1; } + config_id="$(printf '%s' "$config_path" | shasum -a 256 | awk '{print substr($1,1,32)}')" + bundle_directory="${repo_root}/.local/storage/bootstrap-trust/${config_id}" + mkdir -p "$bundle_directory" + [[ -d "$bundle_directory" && ! -L "$bundle_directory" ]] || { echo "bootstrap build trust directory must be a regular directory: ${bundle_directory}" >&2; return 1; } + bootstrap_trust_bundle="${bundle_directory}/ca.pem" + if [[ -e "$bootstrap_trust_bundle" && (! -f "$bootstrap_trust_bundle" || -L "$bootstrap_trust_bundle") ]]; then + echo "bootstrap build trust output must be a regular non-symlink file: ${bootstrap_trust_bundle}" >&2 + return 1 + fi + host_os="$(epar_host_trust_host_os)" + validator_output="$( + docker run --rm \ + --network none \ + -e GO111MODULE=off \ + -e GOTOOLCHAIN=local \ + -v "${repo_root}/scripts/bootstrap-trust:/bootstrap:ro" \ + -v "${feed_path}:/feed/current.json:ro" \ + -v "${bundle_directory}:/out" \ + "$dev_image" \ + /usr/local/go/bin/go run /bootstrap/main.go --feed /feed/current.json --output /out/ca.pem --expected-host-os "$host_os" + )" + [[ -s "$bootstrap_trust_bundle" && ! -L "$bootstrap_trust_bundle" ]] || { echo "bootstrap build trust validator did not produce a regular nonempty bundle" >&2; return 1; } + bootstrap_trust_summary="$(printf '%s\n' "$validator_output" | sed -n '$p')" +} + +epar_tls_failure_host() { + local transcript="$1" + grep -Eqi 'x509: certificate signed by unknown authority|certificate verify failed|unable to (get local issuer certificate|verify the first certificate)' "$transcript" || return 0 + sed -nE 's#.*https://([A-Za-z0-9.-]+)([:/"].*)?#\1#p' "$transcript" | sed -n '1p' | tr '[:upper:]' '[:lower:]' +} + +epar_report_tls_failure() { + local transcript="$1" + local log_path="$2" + local host_name diagnostic_output subject issuer fingerprint not_before not_after + host_name="$(epar_tls_failure_host "$transcript")" + [[ -n "$host_name" ]] || return 0 + diagnostic_output="$( + docker run --rm \ + -e "EPAR_TLS_DIAGNOSTIC_HOST=${host_name}" \ + "$dev_image" \ + sh -c ' + set -u + raw="$(mktemp)" + leaf="$(mktemp)" + cleanup() { rm -f -- "$raw" "$leaf"; } + trap cleanup EXIT + openssl s_client -connect "${EPAR_TLS_DIAGNOSTIC_HOST}:443" -servername "${EPAR_TLS_DIAGNOSTIC_HOST}" -showcerts "$raw" 2>&1 || true + awk '"'"'/-----BEGIN CERTIFICATE-----/{capture=1} capture{print} /-----END CERTIFICATE-----/{exit}'"'"' "$raw" >"$leaf" + grep -E "verify error|Verify return code" "$raw" || true + if [ -s "$leaf" ]; then + openssl x509 -in "$leaf" -noout -subject -issuer -fingerprint -sha256 -dates + fi + ' 2>&1 + )" || true + subject="$(printf '%s\n' "$diagnostic_output" | sed -n 's/^subject=//p' | sed -n '1p')" + issuer="$(printf '%s\n' "$diagnostic_output" | sed -n 's/^issuer=//p' | sed -n '1p')" + fingerprint="$(printf '%s\n' "$diagnostic_output" | sed -n 's/^sha256 Fingerprint=//p' | sed -n '1p' | tr -d ':')" + not_before="$(printf '%s\n' "$diagnostic_output" | sed -n 's/^notBefore=//p' | sed -n '1p')" + not_after="$(printf '%s\n' "$diagnostic_output" | sed -n 's/^notAfter=//p' | sed -n '1p')" + { + printf '\n%s\n' 'EPAR TLS certificate diagnostic' + printf ' Requested host: %s:443\n' "$host_name" + printf ' Toolchain image: %s\n' "$dev_image" + if [[ -z "$subject" ]]; then + printf '%s\n' ' Certificate inspection: unavailable; see the raw build error above.' + else + printf '%s\n' ' Certificate presented to the build container:' + printf ' Subject: %s\n' "$subject" + printf ' Issuer: %s\n' "$issuer" + [[ -z "$fingerprint" ]] || printf ' SHA-256: %s\n' "$fingerprint" + [[ -z "$not_before" ]] || printf ' Valid from: %s\n' "$not_before" + [[ -z "$not_after" ]] || printf ' Valid until: %s\n' "$not_after" + printf '%s\n' "$diagnostic_output" | grep -E 'verify error|Verify return code' | sed 's/^/ OpenSSL: /' || true + printf '%s\n' ' Interpretation: the host network presented a certificate whose issuer is unavailable to the Linux bootstrap container.' + printf '%s\n' ' Check the host system or user trust store for a root matching the issuer above.' + fi + printf '%s\n' ' TLS verification was not disabled, and EPAR did not retry the download insecurely.' + printf ' Full native-controller build log: %s\n' "$log_path" + } | tee -a "$log_path" >&2 +} + +epar_directory_mtime() { + stat -c %Y "$1" 2>/dev/null || stat -f %m "$1" +} + +epar_directory_bytes() { + du -sk "$1" | awk '{print $1 * 1024}' +} + +epar_native_controller_lease_active() { + local directory="$1" + local lease lease_host lease_pid lease_started now host + now="$(date +%s)" + host="$(hostname 2>/dev/null || true)" + for lease in "${directory}"/lease-* "${directory}"/lease.*; do + [[ -f "$lease" ]] || continue + lease_host="$(sed -n 's/^host=//p' "$lease" | head -n 1)" + lease_pid="$(sed -n 's/^pid=//p' "$lease" | head -n 1)" + lease_started="$(sed -n 's/^startedAtUnix=//p' "$lease" | head -n 1)" + [[ "$lease_started" =~ ^[0-9]+$ ]] || return 0 + if [[ "$lease_host" != "$host" ]]; then + ((now - lease_started >= 30 * 24 * 60 * 60)) || return 0 + continue + fi + [[ "$lease_pid" =~ ^[1-9][0-9]*$ ]] || return 0 + kill -0 "$lease_pid" 2>/dev/null && return 0 + done + return 1 +} + +epar_native_controller_build_lease_valid() { + local lease="$1" + local name="${lease##*/}" + [[ -f "$lease" && ! -L "$lease" ]] || return 1 + [[ "$name" =~ ^lease-build-([1-9][0-9]*)\.[0-9A-Za-z]{6}$ ]] || return 1 + local name_pid="${BASH_REMATCH[1]}" + [[ "$(wc -l <"$lease" | tr -d ' ')" == "4" ]] || return 1 + [[ "$(grep -c '^schemaVersion=' "$lease")" == "1" ]] || return 1 + grep -Fqx 'schemaVersion=1' "$lease" || return 1 + [[ "$(grep -c '^host=.' "$lease")" == "1" ]] || return 1 + [[ "$(grep -c '^pid=[1-9][0-9]*$' "$lease")" == "1" ]] || return 1 + [[ "$(grep -c '^startedAtUnix=[0-9][0-9]*$' "$lease")" == "1" ]] || return 1 + [[ "$(sed -n 's/^pid=//p' "$lease")" == "$name_pid" ]] || return 1 +} + +epar_prune_native_controller_cache() { + local cache_root="$1" + local current_cache_key="$2" + local remove_current="${3:-0}" + local now path name mtime bytes manifest unexpected lease valid_build_leases executable + local retained_count=0 + local retained_bytes=0 + local within_grace=0 + [[ "$current_cache_key" =~ ^[0-9a-f]{64}$ ]] || return 1 + [[ -d "$cache_root" ]] || return 0 + now="$(date +%s)" + + for path in "${cache_root}"/.build-* "${cache_root}"/.build.*; do + [[ -d "$path" ]] || continue + [[ ! -L "$path" ]] || continue + name="${path##*/}" + [[ "$name" =~ ^\.build[-.][0-9A-Za-z]+$ ]] || continue + valid_build_leases=0 + for lease in "${path}"/lease-build-*; do + epar_native_controller_build_lease_valid "$lease" || continue + valid_build_leases=$((valid_build_leases + 1)) + done + ((valid_build_leases == 1)) || continue + epar_native_controller_lease_active "$path" && continue + mtime="$(epar_directory_mtime "$path")" || continue + if ((now - mtime >= abandoned_build_grace_seconds)); then + rm -rf -- "${cache_root:?}/${name}" + fi + done + + retention_inventory_file="$(mktemp "${cache_root}/.retention.XXXXXX")" + for path in "${cache_root}"/*; do + [[ -d "$path" ]] || continue + [[ ! -L "$path" ]] || continue + name="${path##*/}" + [[ "$name" =~ ^[0-9a-f]{64}$ ]] || continue + [[ "$name" != "$current_cache_key" || "$remove_current" == 1 ]] || continue + epar_native_controller_lease_active "$path" && continue + manifest="${path}/controller-cache.manifest" + [[ -f "$manifest" && ! -L "$manifest" ]] || continue + grep -Fqx 'schemaVersion=1' "$manifest" || continue + grep -Fqx "cacheKey=${name}" "$manifest" || continue + [[ "$(grep -c '^executable=' "$manifest")" == "1" ]] || continue + executable="$(sed -n 's/^executable=//p' "$manifest")" + [[ "$executable" == ephemeral-action-runner || "$executable" == ephemeral-action-runner.exe ]] || continue + [[ -f "${path}/${executable}" && ! -L "${path}/${executable}" ]] || continue + unexpected="$(find "$path" -mindepth 1 -maxdepth 1 ! \( -type f \( -name "$executable" -o -name controller-cache.manifest -o -name 'lease-*' -o -name 'lease.*' \) \) -print -quit)" + [[ -z "$unexpected" ]] || continue + mtime="$(epar_directory_mtime "$path")" || continue + bytes="$(epar_directory_bytes "$path")" || continue + printf '%s:%s:%s\n' "$mtime" "$bytes" "$name" >>"$retention_inventory_file" + done + + if [[ -d "${cache_root}/${current_cache_key}" ]]; then + retained_bytes="$(epar_directory_bytes "${cache_root}/${current_cache_key}")" + fi + while IFS=: read -r mtime bytes name; do + [[ "$mtime" =~ ^[0-9]+$ && "$bytes" =~ ^[0-9]+$ && "$name" =~ ^[0-9a-f]{64}$ ]] || continue + within_grace=0 + ((now - mtime < native_cache_grace_seconds)) && within_grace=1 + if ((within_grace == 1 || (retained_count < native_cache_keep_previous && retained_bytes + bytes <= native_cache_max_bytes))); then + retained_count=$((retained_count + 1)) + retained_bytes=$((retained_bytes + bytes)) + continue + fi + path="${cache_root}/${name}" + [[ "${path%/*}" == "$cache_root" && "${path##*/}" =~ ^[0-9a-f]{64}$ ]] || return 1 + rm -rf -- "$path" + done < <(sort -t: -k1,1nr -k3,3 "$retention_inventory_file") + rm -f -- "$retention_inventory_file" + retention_inventory_file="" +} + +case "$(uname -s)/$(uname -m)" in + Darwin/arm64) goos=darwin; goarch=arm64 ;; + Linux/x86_64|Linux/amd64) goos=linux; goarch=amd64 ;; + Linux/aarch64|Linux/arm64) goos=linux; goarch=arm64 ;; + *) echo "unsupported native EPAR controller platform: $(uname -s)/$(uname -m)" >&2; exit 1 ;; +esac + +previous_dev_image_id="$(epar_docker_image_id "$dev_image")" +epar_resolve_go_toolchain_image +go_toolchain_previous_id="$EPAR_GO_PREVIOUS_IMAGE_ID" +go_toolchain_resolved_id="$EPAR_GO_RESOLVED_IMAGE_ID" +docker build --quiet \ + --provenance=false \ + --build-arg "GO_IMAGE=${go_image}" \ + -t "$dev_image" \ + -f "${repo_root}/scripts/docker/dev.Dockerfile" \ + "${repo_root}/scripts/docker" >/dev/null +dev_image_id="$(epar_docker_image_id "$dev_image")" +[[ "$dev_image_id" =~ ^sha256:[0-9a-f]{64}$ ]] || { echo "could not resolve the immutable Docker toolchain image ID for ${dev_image}" >&2; exit 1; } +epar_write_bootstrap_acquisition_journal toolchain-built "$go_toolchain_previous_id" "$go_toolchain_resolved_id" "$dev_image_id" "$previous_dev_image_id" +if ((manage_go_cache == 1)); then + epar_ensure_go_cache_volume "$gomod_volume" gomod + epar_ensure_go_cache_volume "$gocache_volume" gobuild + epar_enforce_go_cache_limit +fi + +git_commit=unknown +source_state=unknown +if command -v git >/dev/null 2>&1; then + git_commit_candidate="$(git -C "$repo_root" rev-parse --verify HEAD 2>/dev/null || true)" + if [[ "$git_commit_candidate" =~ ^[0-9a-f]{40}$ ]]; then + git_commit="$git_commit_candidate" + if git -C "$repo_root" status --porcelain=v1 --untracked-files=all >/dev/null 2>&1; then + if [[ -z "$(git -C "$repo_root" status --porcelain=v1 --untracked-files=all)" ]]; then source_state=clean; else source_state=dirty; fi + fi + fi +fi + +source_manifest="$( + printf '%s\n%s\n%s\n%s\n' "${goos}/${goarch}" "$dev_image_id" "$git_commit" "$source_state" + { + find "${repo_root}/cmd" "${repo_root}/internal" -type f -name '*.go' -print + find "${repo_root}/scripts/docker" -type f -print + printf '%s\n' "${repo_root}/go.mod" "${repo_root}/go.sum" "${repo_root}/scripts/build-native-controller.sh" + } | LC_ALL=C sort | while IFS= read -r file; do + printf '%s\n' "${file#"${repo_root}/"}" + shasum -a 256 "$file" | awk '{print $1}' + done +)" +fingerprint="$(printf '%s' "$source_manifest" | shasum -a 256 | awk '{print $1}')" +case "$source_state" in + clean) controller_source_revision="sha256:${fingerprint}" ;; + dirty) controller_source_revision="dirty:sha256:${fingerprint}" ;; + *) controller_source_revision=unknown ;; +esac +cache_root="${repo_root}/.local/bin" +binary="${cache_root}/ephemeral-action-runner" +manifest_path="${cache_root}/ephemeral-action-runner.manifest" +lock_directory="${cache_root}/.native-controller.lock" + +temporary_directory="" +lease_file="" +build_lease_file="" +retention_inventory_file="" +manifest_temporary="" +lock_lease_file="" +cleanup_build() { + if [[ -n "$temporary_directory" && -d "$temporary_directory" ]]; then rm -rf -- "$temporary_directory"; fi + if [[ -n "$lease_file" && -f "$lease_file" ]]; then rm -f -- "$lease_file"; fi + if [[ -n "$build_lease_file" && -f "$build_lease_file" ]]; then rm -f -- "$build_lease_file"; fi + if [[ -n "$retention_inventory_file" && -f "$retention_inventory_file" ]]; then rm -f -- "$retention_inventory_file"; fi + if [[ -n "$manifest_temporary" && -f "$manifest_temporary" ]]; then rm -f -- "$manifest_temporary"; fi + if [[ -n "$lock_lease_file" && -f "$lock_lease_file" ]]; then rm -f -- "$lock_lease_file"; fi + if [[ -n "$lock_directory" && -d "$lock_directory" ]]; then rmdir -- "$lock_directory" 2>/dev/null || true; fi +} +trap cleanup_build EXIT INT TERM + +mkdir -p "$cache_root" +# Retire only old hash-directory revisions whose manifest and inactive lease +# prove ownership. Unknown paths remain available to storage's legacy preview. +native_cache_keep_previous=0 +native_cache_max_bytes=1 +native_cache_grace_seconds=0 +if ! epar_prune_native_controller_cache "$cache_root" "$(printf '%064d' 0)" 1; then + echo "warning: native-controller legacy revision cleanup skipped after an error" >&2 +fi + +epar_stable_manifest_matches() { + [[ -f "$manifest_path" && ! -L "$manifest_path" && -x "$binary" && ! -L "$binary" ]] || return 1 + grep -Fqx 'schemaVersion=2' "$manifest_path" && grep -Fqx "fingerprint=${fingerprint}" "$manifest_path" && grep -Fqx 'executable=ephemeral-action-runner' "$manifest_path" && grep -Fqx "toolchainImageID=${dev_image_id}" "$manifest_path" +} + +epar_acquire_stable_build_lock() { + local deadline=$(( $(date +%s) + 120 )) + while ! mkdir "$lock_directory" 2>/dev/null; do + if [[ -d "$lock_directory" && ! -L "$lock_directory" ]]; then + local valid=0 candidate unexpected + for candidate in "$lock_directory"/lease-build-*; do + epar_native_controller_build_lease_valid "$candidate" && valid=$((valid + 1)) + done + unexpected="$(find "$lock_directory" -mindepth 1 -maxdepth 1 ! \( -type f -name 'lease-build-*' \) -print -quit)" + if ((valid == 1)) && [[ -z "$unexpected" ]] && ! epar_native_controller_lease_active "$lock_directory"; then + rm -rf -- "$lock_directory" + continue + fi + fi + (( $(date +%s) < deadline )) || { echo 'another EPAR native-controller build is still in progress; wait for it to finish and retry.' >&2; return 1; } + sleep 0.2 + done + lock_lease_file="$(mktemp "${lock_directory}/lease-build-$$.XXXXXX")" + printf '%s\n' 'schemaVersion=1' "host=$(hostname 2>/dev/null || true)" "pid=$$" "startedAtUnix=$(date +%s)" >"$lock_lease_file" +} + +if ! epar_stable_manifest_matches; then + epar_acquire_stable_build_lock + if ! epar_stable_manifest_matches; then + if epar_native_controller_lease_active "$cache_root"; then + echo 'EPAR source or its Go toolchain changed while a native EPAR controller is running. Stop the running EPAR process, then run ./start again; EPAR keeps one stable native controller binary and will not create another versioned copy.' >&2 + exit 1 + fi + epar_prepare_bootstrap_build_trust "$@" + build_log_directory="${repo_root}/work/logs" + build_log_path="${build_log_directory}/epar-native-controller-build.log" + mkdir -p "$build_log_directory" + printf '%s\n' \ + "EPAR native-controller build started at $(date -u +%Y-%m-%dT%H:%M:%SZ)" \ + "Toolchain image: ${dev_image}" \ + "Target: ${goos}/${goarch}" \ + "Bootstrap build trust: ${bootstrap_trust_summary}" \ + '' >"$build_log_path" + printf 'Native controller build log: %s\n' "$build_log_path" + temporary_directory="$(mktemp -d "${cache_root}/.build.XXXXXX")" + build_stderr="${temporary_directory}/native-controller-build.stderr" + build_lease_file="$(mktemp "${temporary_directory}/lease-build-$$.XXXXXX")" + printf '%s\n' 'schemaVersion=1' "host=$(hostname 2>/dev/null || true)" "pid=$$" "startedAtUnix=$(date +%s)" >"$build_lease_file" + set +e + docker run --rm \ + -e CGO_ENABLED=0 \ + -e "GOOS=${goos}" \ + -e "GOARCH=${goarch}" \ + -e GOTOOLCHAIN=local \ + -e SSL_CERT_FILE=/run/epar-bootstrap-ca.pem \ + -v "${repo_root}:/src:ro" \ + -v "${temporary_directory}:/out" \ + -v "${gomod_volume}:/go/pkg/mod" \ + -v "${gocache_volume}:/root/.cache/go-build" \ + -v "${bootstrap_trust_bundle}:/run/epar-bootstrap-ca.pem:ro" \ + -w /src \ + "$dev_image" \ + go build -trimpath -ldflags "-X main.sourceRevision=${controller_source_revision}" -o /out/ephemeral-action-runner ./cmd/ephemeral-action-runner 2>"$build_stderr" + native_build_exit_code=$? + set -e + if [[ -s "$build_stderr" ]]; then cat "$build_stderr" >>"$build_log_path"; fi + if ((native_build_exit_code != 0)); then + tls_failure_host="$(epar_tls_failure_host "$build_stderr")" + if [[ -n "$tls_failure_host" ]]; then + printf 'Native controller build failed while downloading dependencies from https://%s.\n' "$tls_failure_host" >&2 + printf '%s\n' ' The build container rejected the presented TLS certificate as an unknown issuer.' >&2 + printf ' Full compiler output: %s\n' "$build_log_path" >&2 + elif [[ -s "$build_stderr" ]]; then + cat "$build_stderr" >&2 + fi + epar_report_tls_failure "$build_stderr" "$build_log_path" + exit "$native_build_exit_code" + fi + if [[ -s "$build_stderr" ]]; then cat "$build_stderr" >&2; fi + [[ -f "${temporary_directory}/ephemeral-action-runner" ]] || { echo "native EPAR build did not produce the expected binary" >&2; exit 1; } + chmod 0755 "${temporary_directory}/ephemeral-action-runner" + mv -f -- "${temporary_directory}/ephemeral-action-runner" "$binary" + manifest_temporary="$(mktemp "${cache_root}/.native-controller-manifest.XXXXXX")" + printf '%s\n' 'schemaVersion=2' "fingerprint=${fingerprint}" 'executable=ephemeral-action-runner' "toolchainImageID=${dev_image_id}" "sourceRevision=${controller_source_revision}" "completedAtUnix=$(date +%s)" >"$manifest_temporary" + mv -f -- "$manifest_temporary" "$manifest_path" + manifest_temporary="" + fi +fi +if ((manage_go_cache == 1)); then + go_cache_limit_bytes="$("$binary" storage effective-go-cache-limit --project-root "$repo_root")" + [[ "$go_cache_limit_bytes" =~ ^[1-9][0-9]*$ ]] || { echo "EPAR returned an invalid configured Go cache limit" >&2; exit 1; } + epar_enforce_go_cache_limit +fi + +lease_file="$(mktemp "${cache_root}/lease-native-$$.XXXXXX")" +printf '%s\n' \ + 'schemaVersion=1' \ + "host=$(hostname 2>/dev/null || true)" \ + "pid=$$" \ + "startedAtUnix=$(date +%s)" >"$lease_file" +export EPAR_NATIVE_CONTROLLER=1 +export DOCKER_CLI_HINTS="${DOCKER_CLI_HINTS:-false}" +export EPAR_HOST_NAME="${EPAR_HOST_NAME:-$(hostname 2>/dev/null || true)}" +controller_command="${1:-start}" + +# Bootstrap trust is mounted only into the compiler container above. The cached +# native controller reads the host stores directly, so legacy bridge state must +# not survive into first-run continuation or suppress its native preflight. +unset EPAR_BUILD_TRUST_FEED EPAR_HOST_TRUST_FEED EPAR_CONTROLLER_HOST_OS EPAR_HOST_TRUST_INIT_DEFERRED + +# Keep explicit-init's post-write verification. Ordinary starts must not create +# a pre-wizard feed because the native controller resolves the new config itself. +if [[ "$controller_command" == "init" ]]; then + epar_host_trust_prepare "$repo_root" "$controller_command" "$@" +fi +status=0 +"$binary" "$@" || status=$? +if [[ "$status" == "0" && "$controller_command" == "init" ]]; then + epar_host_trust_post_init "$repo_root" || status=$? +fi +epar_host_trust_cleanup +cleanup_build +trap - EXIT INT TERM +exit "$status" diff --git a/scripts/ci/core-runner-controller.sh b/scripts/ci/core-runner-controller.sh index 5aae410..75b0365 100644 --- a/scripts/ci/core-runner-controller.sh +++ b/scripts/ci/core-runner-controller.sh @@ -248,6 +248,8 @@ image: upstreamDir: third_party/runner-images upstreamLock: third_party/runner-images.lock runnerVersion: latest + updateFrequency: weekly + updateTime: "07:00" customInstallScripts: EOF if [[ -n "${EPAR_TRUSTED_CA_CERTIFICATE_PATH:-}" ]]; then @@ -287,8 +289,18 @@ runner: includeHostLabel: false ephemeral: true +security: + runnerGroup: + enforcement: enforce + requireExplicitGroup: true + requireNonDefaultGroup: true + requiredRepositoryAccess: selected + # This public project uses a protected trusted-workflow canary. Public + # repository access is the only runner-group safety requirement relaxed. + requirePublicRepositoriesDisabled: false + provider: - type: docker-dind + type: docker-container sourceImage: ${EPAR_OUTPUT_IMAGE} platform: linux/amd64 network: default diff --git a/scripts/ci/core-runner-controller_test.sh b/scripts/ci/core-runner-controller_test.sh index 000f496..cb2d807 100644 --- a/scripts/ci/core-runner-controller_test.sh +++ b/scripts/ci/core-runner-controller_test.sh @@ -175,9 +175,9 @@ if [[ " $* " == *" pool up "* ]]; then printf 'Authorization: token %s\n' "${MOCK_GUEST_AUTH_TOKEN}" } >"${log_dir}/mock.guest.log" { - echo 'Docker-DinD safe diagnostic' + echo 'Docker Container safe diagnostic' printf 'EPAR_APP_PRIVATE_KEY=%s\n' "${MOCK_PRIVATE_KEY_ENV}" - } >"${log_dir}/mock.docker-dind.log" + } >"${log_dir}/mock.docker-container.log" echo 'pool safe diagnostic' printf 'RUNNER_TOKEN=%s\n' "${MOCK_POOL_TOKEN}" printf -- '--token %s\n' "${MOCK_CLI_TOKEN}" @@ -333,10 +333,10 @@ fi grep -q 'EPAR pool supervisor exited unexpectedly with status 7' "${failure_output}" grep -q 'EPAR pool supervisor log (last 200 lines, sanitized)' "${failure_output}" grep -q 'EPAR runner log: mock.guest.log (last 200 lines, sanitized)' "${failure_output}" -grep -q 'EPAR runner log: mock.docker-dind.log (last 200 lines, sanitized)' "${failure_output}" +grep -q 'EPAR runner log: mock.docker-container.log (last 200 lines, sanitized)' "${failure_output}" grep -q '| pool safe diagnostic' "${failure_output}" grep -q '| guest safe diagnostic' "${failure_output}" -grep -q '| Docker-DinD safe diagnostic' "${failure_output}" +grep -q '| Docker Container safe diagnostic' "${failure_output}" grep -q '| RUNNER_TOKEN=\*\*\*' "${failure_output}" grep -q '| Authorization: Bearer \*\*\*' "${failure_output}" grep -q '| Authorization: Basic \*\*\*' "${failure_output}" diff --git a/scripts/container/ubuntu/entrypoint.sh b/scripts/container/ubuntu/entrypoint.sh index 1d01ac3..d77933a 100644 --- a/scripts/container/ubuntu/entrypoint.sh +++ b/scripts/container/ubuntu/entrypoint.sh @@ -8,9 +8,9 @@ dockerd_args=(--host=unix:///var/run/docker.sock) storage_driver="${EPAR_DOCKERD_STORAGE_DRIVER-vfs}" if [[ -n "${storage_driver}" && "${storage_driver}" != "auto" ]]; then dockerd_args+=(--storage-driver="${storage_driver}") - echo "EPAR Docker-DinD: starting inner Docker daemon with ${storage_driver} storage driver" + echo "EPAR Docker Container: starting inner Docker daemon with ${storage_driver} storage driver" else - echo "EPAR Docker-DinD: starting inner Docker daemon with Docker's default storage driver" + echo "EPAR Docker Container: starting inner Docker daemon with Docker's default storage driver" fi dockerd "${dockerd_args[@]}" >/var/log/epar-dockerd.log 2>&1 & diff --git a/scripts/docker-sandboxes/build-template.ps1 b/scripts/docker-sandboxes/build-template.ps1 new file mode 100644 index 0000000..2d8591e --- /dev/null +++ b/scripts/docker-sandboxes/build-template.ps1 @@ -0,0 +1,25 @@ +[CmdletBinding()] +param( + [string] $Config = '.local/config.yml', + [string] $ProjectRoot, + [switch] $Execute, + [Parameter(ValueFromRemainingArguments = $true)] + [string[]] $RemainingArguments +) + +$ErrorActionPreference = 'Stop' +$repositoryRoot = Split-Path -Parent (Split-Path -Parent $PSScriptRoot) +if ([string]::IsNullOrWhiteSpace($ProjectRoot)) { + $ProjectRoot = $repositoryRoot +} +$startScript = Join-Path $repositoryRoot 'start.ps1' +$arguments = @('image', 'build', '--config', $Config, '--project-root', $ProjectRoot) +if (-not $Execute) { + $arguments += '--dry-run' +} +if ($RemainingArguments) { + $arguments += $RemainingArguments +} +Write-Warning 'This compatibility wrapper delegates to the common EPAR image build path. Prefer ./start image build.' +& $startScript @arguments +exit $LASTEXITCODE diff --git a/scripts/docker-sandboxes/load-template.ps1 b/scripts/docker-sandboxes/load-template.ps1 new file mode 100644 index 0000000..0a395f2 --- /dev/null +++ b/scripts/docker-sandboxes/load-template.ps1 @@ -0,0 +1,25 @@ +[CmdletBinding()] +param( + [string] $Config = '.local/config.yml', + [string] $ProjectRoot, + [switch] $Execute, + [Parameter(ValueFromRemainingArguments = $true)] + [string[]] $RemainingArguments +) + +$ErrorActionPreference = 'Stop' +$repositoryRoot = Split-Path -Parent (Split-Path -Parent $PSScriptRoot) +if ([string]::IsNullOrWhiteSpace($ProjectRoot)) { + $ProjectRoot = $repositoryRoot +} +$startScript = Join-Path $repositoryRoot 'start.ps1' +$arguments = @('image', 'build', '--config', $Config, '--project-root', $ProjectRoot) +if (-not $Execute) { + $arguments += '--dry-run' +} +if ($RemainingArguments) { + $arguments += $RemainingArguments +} +Write-Warning 'Template import is now part of the common EPAR image build. This compatibility wrapper delegates to ./start image build.' +& $startScript @arguments +exit $LASTEXITCODE diff --git a/scripts/docker-sandboxes/validate-assets.ps1 b/scripts/docker-sandboxes/validate-assets.ps1 new file mode 100644 index 0000000..f6088fa --- /dev/null +++ b/scripts/docker-sandboxes/validate-assets.ps1 @@ -0,0 +1,310 @@ +[CmdletBinding()] +param( + [ValidateSet('linux/amd64', 'linux/arm64')] + [string]$Platform = 'linux/amd64', + [switch]$VerifyRemote, + [switch]$DockerfileCheck, + [string]$Builder +) + +$ErrorActionPreference = 'Stop' +$scriptDirectory = Split-Path -Parent $MyInvocation.MyCommand.Path +$repositoryRoot = [System.IO.Path]::GetFullPath((Join-Path $scriptDirectory '..\..')) +$templateDirectory = Join-Path $repositoryRoot 'templates\docker-sandboxes' +$lockPath = Join-Path $templateDirectory 'sources.lock.json' +$lock = Get-Content -Raw -LiteralPath $lockPath | ConvertFrom-Json + +function Assert-Equal { + param([string]$Name, $Actual, $Expected) + if ($Actual -ne $Expected) { + throw "$Name mismatch: expected $Expected, got $Actual" + } +} + +function Get-Sha256Text { + param([string]$Text) + $sha = [System.Security.Cryptography.SHA256]::Create() + try { + $bytes = [System.Text.UTF8Encoding]::new($false).GetBytes($Text) + return 'sha256:' + ([System.BitConverter]::ToString($sha.ComputeHash($bytes)).Replace('-', '').ToLowerInvariant()) + } + finally { + $sha.Dispose() + } +} + +function Test-RemoteIndex { + param([string]$Name, [string]$Reference, [string]$ExpectedIndexDigest, [string]$ExpectedManifestDigest, [string]$Platform) + $rawLines = @(& docker buildx imagetools inspect --raw $Reference) + if ($LASTEXITCODE -ne 0) { + throw "Remote inspection failed for $Reference" + } + # OCI manifests are UTF-8 JSON with LF separators. Reconstructing them with + # the host newline would turn LF into CRLF on Windows and change the digest. + $raw = [string]::Join("`n", $rawLines) + Assert-Equal "$Name index" (Get-Sha256Text $raw) $ExpectedIndexDigest + $index = $raw | ConvertFrom-Json + $platformParts = $Platform -split '/', 2 + $matching = @($index.manifests | Where-Object { $_.platform.os -eq $platformParts[0] -and $_.platform.architecture -eq $platformParts[1] }) + Assert-Equal "$Name $Platform manifest count" $matching.Count 1 + Assert-Equal "$Name $Platform manifest" $matching[0].digest $ExpectedManifestDigest +} + +Write-Host '[1/6] Checking pinned constants and exact plural naming.' +Assert-Equal 'source lock schema' $lock.schemaVersion 2 +Assert-Equal 'default platform' $lock.defaultPlatform 'linux/amd64' +Assert-Equal 'supported platform count' @($lock.supportedPlatforms).Count 2 +Assert-Equal 'first supported platform' $lock.supportedPlatforms[0] 'linux/amd64' +Assert-Equal 'second supported platform' $lock.supportedPlatforms[1] 'linux/arm64' +Assert-Equal 'Dockerfile frontend index' $lock.dockerfileFrontend.indexDigest 'sha256:a57df69d0ea827fb7266491f2813635de6f17269be881f696fbfdf2d83dda33e' +Assert-Equal 'SBOM generator index' $lock.sbomGenerator.indexDigest 'sha256:79e7b013cbec16bbb436f312819a49a4a57752b2270c1a9332ae1a10fcc82a68' +Assert-Equal 'Go builder version' $lock.goBuilder.version '1.25.12' +Assert-Equal 'Go builder index' $lock.goBuilder.indexDigest 'sha256:9006890ecba0a168034d99516084099ae3114d9f2b7d6572c77f2dde57ebc980' +Assert-Equal 'hook launcher source checksum' $lock.hookLauncher.sha256 '7fe07f10f484fa6888481a4165e81570187c0aeff422738d3ea5add6b95dd9b7' +Assert-Equal 'Tini version' $lock.tini.version '0.19.0' +$expectedPlatforms = [ordered]@{ + 'linux/amd64' = [ordered]@{ + architecture = 'amd64' + frontendManifest = 'sha256:b5f3b260a9678e1d83d2fce86eeddf79420b79147eaba2a25986f47133d73720' + goBuilderManifest = 'sha256:12e171e33ce7ade87ac8ab2bbe65cea9371527285bdab43ca02780a9e6ac60e5' + sbomManifest = 'sha256:13864237fb990943433f89d698590aad1de38d4a7e13d38e7b12f2488c1952e7' + tiniUrl = 'https://github.com/krallin/tini/releases/download/v0.19.0/tini-amd64' + tiniSha256 = '93dcc18adc78c65a028a84799ecf8ad40c936fdfc5f2a57b1acda5a8117fa82c' + } + 'linux/arm64' = [ordered]@{ + architecture = 'arm64' + frontendManifest = 'sha256:c8678869a83fab70232869ba24acc1c0be661f4d65135c0eeacb6a8e78420fdd' + goBuilderManifest = 'sha256:afe53a4752b49f57ddebc97501a99394e2f7715236b4241efa830d54efb44434' + sbomManifest = 'sha256:860305b3d1667c35142f11f6e9485e322c1c6173702a0831dc68739a34847f2d' + tiniUrl = 'https://github.com/krallin/tini/releases/download/v0.19.0/tini-arm64' + tiniSha256 = '07952557df20bfd2a95f9bef198b445e006171969499a1d361bd9e6f8e5e0e81' + } +} +foreach ($platformName in $expectedPlatforms.Keys) { + $platformRecord = $lock.platforms.PSObject.Properties[$platformName].Value + $expectedPlatform = $expectedPlatforms[$platformName] + Assert-Equal "$platformName architecture" $platformRecord.architecture $expectedPlatform.architecture + Assert-Equal "$platformName Dockerfile frontend manifest" $platformRecord.dockerfileFrontendManifestDigest $expectedPlatform.frontendManifest + Assert-Equal "$platformName Go builder manifest" $platformRecord.goBuilderManifestDigest $expectedPlatform.goBuilderManifest + Assert-Equal "$platformName Go builder reference" $platformRecord.goBuilderReference ("docker.io/library/golang@{0}" -f $expectedPlatform.goBuilderManifest) + Assert-Equal "$platformName SBOM generator manifest" $platformRecord.sbomGeneratorManifestDigest $expectedPlatform.sbomManifest + Assert-Equal "$platformName SBOM generator reference" $platformRecord.sbomGeneratorReference ("docker.io/docker/buildkit-syft-scanner@{0}" -f $expectedPlatform.sbomManifest) + Assert-Equal "$platformName Tini URL" $platformRecord.tini.url $expectedPlatform.tiniUrl + Assert-Equal "$platformName Tini checksum" $platformRecord.tini.sha256 $expectedPlatform.tiniSha256 +} +$expectedProfiles = [ordered]@{ + 'act-22.04' = [ordered]@{ + observedTag = 'ghcr.io/catthehacker/ubuntu:act-22.04' + index = 'sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac' + legacyManifest = 'sha256:f3d493b10df1582ce631e0213bd90aa5f8196287c8a9f8ef546ecb44ca256655' + legacyTag = 'epar-docker-sandboxes-catthehacker-act-22.04:20260723-r3-amd64' + platforms = [ordered]@{ + 'linux/amd64' = [ordered]@{ manifest = 'sha256:f3d493b10df1582ce631e0213bd90aa5f8196287c8a9f8ef546ecb44ca256655'; status = 'planned'; tag = 'epar-docker-sandboxes-catthehacker-act-22.04:20260723-r4-amd64'; compatibilityFile = 'act-22.04.amd64.compatibility.json' } + 'linux/arm64' = [ordered]@{ manifest = 'sha256:72b9ec71ee5972e02df5053f0000d34dbd2a3d0165b912bf25bbeabd72fba160'; status = 'unvalidated'; tag = 'epar-docker-sandboxes-catthehacker-act-22.04:20260723-r4-arm64'; compatibilityFile = 'act-22.04.arm64.compatibility.json' } + } + } + 'full' = [ordered]@{ + observedTag = 'ghcr.io/catthehacker/ubuntu:full-latest' + index = 'sha256:76581ac3f31aa1ad7cb558b47c3e836b9cbcd82dc08fc69349f77e3967bea50c' + legacyManifest = 'sha256:58314fa8cbf0f0e5384a37b3444811033320038816ef7c16f30b3e841ed65e51' + legacyTag = 'epar-docker-sandboxes-catthehacker-full:20260723-r1-amd64' + platforms = [ordered]@{ + 'linux/amd64' = [ordered]@{ manifest = 'sha256:58314fa8cbf0f0e5384a37b3444811033320038816ef7c16f30b3e841ed65e51'; status = 'planned'; tag = 'epar-docker-sandboxes-catthehacker-full:20260723-r2-amd64'; compatibilityFile = 'full.amd64.compatibility.json' } + 'linux/arm64' = [ordered]@{ manifest = 'sha256:245c8981fbf4ac268db015463c6c446b9411481f7e0001537128dc384d46dd0c'; status = 'unvalidated'; tag = 'epar-docker-sandboxes-catthehacker-full:20260723-r2-arm64'; compatibilityFile = 'full.arm64.compatibility.json' } + } + } +} +foreach ($profileName in $expectedProfiles.Keys) { + $profile = $lock.profiles.PSObject.Properties[$profileName].Value + $expectedProfile = $expectedProfiles[$profileName] + Assert-Equal "$profileName observed source channel" $profile.observedTagReference $expectedProfile.observedTag + Assert-Equal "$profileName index" $profile.indexDigest $expectedProfile.index + foreach ($forbiddenActiveProperty in @('amd64ManifestDigest', 'status', 'templateTag')) { + if ($profile.PSObject.Properties.Name -contains $forbiddenActiveProperty) { + throw "$profileName must not expose historical $forbiddenActiveProperty as an active profile property" + } + } + $supersededRecord = $lock.supersededRecords.'linux/amd64'.PSObject.Properties[$profileName].Value + Assert-Equal "$profileName superseded record authority" $supersededRecord.authoritative $false + Assert-Equal "$profileName superseded amd64 manifest" $supersededRecord.manifestDigest $expectedProfile.legacyManifest + Assert-Equal "$profileName superseded template tag" $supersededRecord.templateTag $expectedProfile.legacyTag + Assert-Equal "$profileName superseded reason" $supersededRecord.reason 'Predates the current runner-template helper and architecture changes' + if ($supersededRecord.PSObject.Properties.Name -contains 'validationStatus') { + throw "$profileName superseded record must not carry a current validation status" + } + $expectedReferenceSuffix = '@' + $expectedProfile.index + if (-not $profile.immutableReference.EndsWith($expectedReferenceSuffix, [System.StringComparison]::Ordinal)) { + throw "$profileName immutable reference is not pinned to its index digest" + } + foreach ($platformName in $expectedPlatforms.Keys) { + $profilePlatform = $profile.platforms.PSObject.Properties[$platformName].Value + $expectedProfilePlatform = $expectedProfile.platforms[$platformName] + Assert-Equal "$profileName $platformName manifest" $profilePlatform.manifestDigest $expectedProfilePlatform.manifest + Assert-Equal "$profileName $platformName status" $profilePlatform.validationStatus $expectedProfilePlatform.status + Assert-Equal "$profileName $platformName template tag" $profilePlatform.templateTag $expectedProfilePlatform.tag + Assert-Equal "$profileName $platformName compatibility file" $profilePlatform.compatibilityFile $expectedProfilePlatform.compatibilityFile + if ($profilePlatform.templateTag -notmatch '^epar-docker-sandboxes-') { + throw "$profileName $platformName template tag violates the plural naming contract" + } + } +} + +Write-Host '[2/6] Verifying deterministic guest-helper hashes.' +$launcherPath = Join-Path (Join-Path $templateDirectory 'hook-launcher') 'main.go' +$launcherHash = (Get-FileHash -Algorithm SHA256 -LiteralPath $launcherPath).Hash.ToLowerInvariant() +Assert-Equal 'hook launcher source' $launcherHash $lock.hookLauncher.sha256 +$hashManifestPath = Join-Path $templateDirectory 'helpers.sha256' +$manifestEntries = Get-Content -LiteralPath $hashManifestPath +$guestDirectory = Join-Path $templateDirectory 'guest' +$guestAssetFiles = @(Get-ChildItem -LiteralPath $guestDirectory -File | Sort-Object Name) +$guestScripts = @($guestAssetFiles | Where-Object Extension -EQ '.sh') +Assert-Equal 'helper manifest entry count' $manifestEntries.Count $guestAssetFiles.Count +$manifestFileNames = @() +foreach ($line in $manifestEntries) { + if ($line -notmatch '^([0-9a-f]{64}) \./((?:[a-z0-9.-]+\.sh)|docker-daemon\.json)$') { + throw "Invalid helper hash entry: $line" + } + $manifestFileNames += $Matches[2] + $helperPath = Join-Path $guestDirectory $Matches[2] + if (-not (Test-Path -LiteralPath $helperPath -PathType Leaf)) { + throw "Helper hash references missing file: $helperPath" + } + $actualHash = (Get-FileHash -Algorithm SHA256 -LiteralPath $helperPath).Hash.ToLowerInvariant() + Assert-Equal "helper $($Matches[2])" $actualHash $Matches[1] +} +Assert-Equal 'unique helper manifest entry count' @($manifestFileNames | Sort-Object -Unique).Count $guestAssetFiles.Count + +Write-Host '[3/6] Checking Dockerfile and entrypoint invariants.' +$dockerfilePath = Join-Path $templateDirectory 'Dockerfile' +$dockerfile = Get-Content -Raw -LiteralPath $dockerfilePath +$dockerignore = Get-Content -Raw -LiteralPath (Join-Path $templateDirectory '.dockerignore') +foreach ($required in @( + '# syntax=docker/dockerfile:1.7.1@sha256:a57df69d0ea827fb7266491f2813635de6f17269be881f696fbfdf2d83dda33e', + 'FROM --platform=$BUILDPLATFORM ${GO_BUILDER_IMAGE} AS hook-builder', + 'FROM --platform=${TEMPLATE_PLATFORM} ${SOURCE_IMAGE}', + 'COPY --from=hook-builder --chmod=0555 /out/epar-hook-bash /opt/epar/hook-bin/bash', + 'com.docker.sandboxes.start-docker=true', + 'USER agent', + 'ENTRYPOINT ["/usr/local/bin/tini", "-g", "--", "/opt/epar/template-entrypoint.sh"]', + 'ARG ACTIONS_RUNNER_VERSION', + 'ARG ACTIONS_RUNNER_SHA256', + 'COPY inputs/actions-runner.tar.gz /tmp/actions-runner.tar.gz', + 'echo "${ACTIONS_RUNNER_SHA256#sha256:} /tmp/actions-runner.tar.gz" | sha256sum --check -', + 'sha256:93dcc18adc78c65a028a84799ecf8ad40c936fdfc5f2a57b1acda5a8117fa82c' +)) { + if (-not $dockerfile.Contains($required)) { + throw "Dockerfile is missing required invariant: $required" + } +} +if ($dockerfile -match '(?im)apt-get\s+update|(?im)\blatest\b|(?im)COPY\s+.*var/lib/docker|(?im)--privileged|(?im)--secret') { + throw 'Dockerfile contains an unpinned, privileged, secret, or /var/lib/docker preload pattern' +} +foreach ($requiredContextEntry in @('!Dockerfile', '!helpers.sha256', '!guest/*.sh', '!guest/docker-daemon.json', '!hook-launcher/*.go', '!custom-install/run.sh', '!profiles/*.compatibility.json')) { + if (-not ($dockerignore -split "`r?`n").Contains($requiredContextEntry)) { + throw ".dockerignore is missing deterministic context entry: $requiredContextEntry" + } +} +$guestText = ($guestScripts | ForEach-Object { Get-Content -Raw -LiteralPath $_.FullName }) -join "`n" +if ($guestText -match '(?im)apt-get\s+update|(?im)(^|[;&|]\s*)dockerd(?:\s|$)|(?im)-----BEGIN .*PRIVATE KEY-----|(?im)AKIA[0-9A-Z]{16}') { + throw 'Guest helpers contain a boot-time package update, dockerd start, or credential pattern' +} +$configureRunner = Get-Content -Raw -LiteralPath (Join-Path (Join-Path $templateDirectory 'guest') 'configure-runner.sh') +if ($configureRunner -match '(?m)(^|\s)--replace(\s|$)') { + throw 'configure-runner.sh must not allow runner replacement' +} +$runnerDiagnostics = Get-Content -Raw -LiteralPath (Join-Path (Join-Path $templateDirectory 'guest') 'collect-runner-diagnostics.sh') +if ($runnerDiagnostics -match '(?im)\btail\b|_diag|(?:^|,)cmd=') { + throw 'collect-runner-diagnostics.sh must not emit command lines or runner/job log content' +} +if ($guestText -match '(?im)\btail\s+(?:-[^\s]+\s+)*["'']?\$\{?(?:log_file|runner_log|job_log)') { + throw 'Guest helpers must not copy runner or job log content into controller-visible output' +} + +Write-Host '[4/6] Parsing compatibility metadata.' +foreach ($profileName in $expectedProfiles.Keys) { + $profile = $lock.profiles.PSObject.Properties[$profileName].Value + foreach ($platformName in $expectedPlatforms.Keys) { + $profilePlatform = $profile.platforms.PSObject.Properties[$platformName].Value + $compatibilityPath = Join-Path (Join-Path $templateDirectory 'profiles') $profilePlatform.compatibilityFile + $compatibility = Get-Content -Raw -LiteralPath $compatibilityPath | ConvertFrom-Json + Assert-Equal "$profileName $platformName compatibility schema" $compatibility.schemaVersion 2 + Assert-Equal "$profileName $platformName template schema" $compatibility.templateSchemaVersion 1 + Assert-Equal "$profileName $platformName compatibility profile" $compatibility.profile $profileName + Assert-Equal "$profileName $platformName compatibility status" $compatibility.validationStatus $profilePlatform.validationStatus + Assert-Equal "$profileName $platformName compatibility platform" $compatibility.platform $platformName + Assert-Equal "$profileName $platformName source reference" $compatibility.source.reference $profile.immutableReference + Assert-Equal "$profileName $platformName source index" $compatibility.source.indexDigest $profile.indexDigest + Assert-Equal "$profileName $platformName source manifest" $compatibility.source.manifestDigest $profilePlatform.manifestDigest + Assert-Equal "$profileName $platformName daemon count" $compatibility.docker.expectedDaemonCount 1 + Assert-Equal "$profileName $platformName daemon owner" $compatibility.docker.daemonOwner 'docker-sandboxes-runtime' + Assert-Equal "$profileName $platformName /var/lib/docker preload" $compatibility.docker.imagePreloadsVarLibDocker $false + } +} + +Write-Host '[5/6] Parsing PowerShell and Bash sources.' +$parseErrors = @() +Get-ChildItem -LiteralPath $scriptDirectory -Filter '*.ps1' -File | ForEach-Object { + $tokens = $null + $errors = $null + [void][System.Management.Automation.Language.Parser]::ParseFile($_.FullName, [ref]$tokens, [ref]$errors) + $parseErrors += @($errors) +} +if ($parseErrors.Count -gt 0) { + throw "PowerShell parse errors: $($parseErrors | ForEach-Object Message | Sort-Object -Unique -join '; ')" +} +$gitBashPath = 'C:\Program Files\Git\bin\bash.exe' +if (Test-Path -LiteralPath $gitBashPath -PathType Leaf) { + $bashPath = $gitBashPath +} +else { + $bashCommand = Get-Command bash -ErrorAction SilentlyContinue + $bashPath = if ($null -eq $bashCommand) { $null } else { $bashCommand.Source } +} +if ($null -eq $bashPath) { + throw 'bash is required to syntax-check guest helpers' +} +foreach ($guestFile in $guestScripts) { + & $bashPath -n $guestFile.FullName + if ($LASTEXITCODE -ne 0) { + throw "bash -n failed for $($guestFile.FullName)" + } +} + +Write-Host '[6/6] Running optional remote and Dockerfile frontend checks.' +if ($VerifyRemote) { + $platformLock = $lock.platforms.PSObject.Properties[$Platform].Value + Test-RemoteIndex 'Dockerfile frontend' $lock.dockerfileFrontend.inspectionReference $lock.dockerfileFrontend.indexDigest $platformLock.dockerfileFrontendManifestDigest $Platform + Test-RemoteIndex 'SBOM generator' $lock.sbomGenerator.inspectionReference $lock.sbomGenerator.indexDigest $platformLock.sbomGeneratorManifestDigest $Platform + Test-RemoteIndex 'Go hook-launcher builder' $lock.goBuilder.inspectionReference $lock.goBuilder.indexDigest $platformLock.goBuilderManifestDigest $Platform + foreach ($profileName in $expectedProfiles.Keys) { + $profile = $lock.profiles.PSObject.Properties[$profileName].Value + $profilePlatform = $profile.platforms.PSObject.Properties[$Platform].Value + Test-RemoteIndex "Catthehacker $profileName" $profile.inspectionReference $profile.indexDigest $profilePlatform.manifestDigest $Platform + } +} +if ($DockerfileCheck) { + if ([string]::IsNullOrWhiteSpace($Builder)) { + $buildxMetadataPath = Join-Path $repositoryRoot '.local\storage\buildx.json' + if (Test-Path -LiteralPath $buildxMetadataPath -PathType Leaf) { + $buildxMetadata = Get-Content -Raw -LiteralPath $buildxMetadataPath | ConvertFrom-Json + $Builder = [string]$buildxMetadata.builder + } + } + if ([string]::IsNullOrWhiteSpace($Builder)) { + throw 'DockerfileCheck requires the exact EPAR-owned Buildx builder. Run ./start image build first or pass -Builder with that owned builder identity; the validation script will not use Docker''s current/default builder.' + } + & docker buildx inspect $Builder *> $null + if ($LASTEXITCODE -ne 0) { + throw "EPAR-owned Buildx builder '$Builder' is unavailable; the validation script will not fall back to Docker's current/default builder." + } + $platformLock = $lock.platforms.PSObject.Properties[$Platform].Value + foreach ($profileName in $expectedProfiles.Keys) { + $profile = $lock.profiles.PSObject.Properties[$profileName].Value + $profilePlatform = $profile.platforms.PSObject.Properties[$Platform].Value + & docker buildx build --builder $Builder --call check --platform $Platform --build-arg ("TEMPLATE_PLATFORM={0}" -f $Platform) --build-arg ("SOURCE_IMAGE={0}" -f $profile.immutableReference) --build-arg ("GO_BUILDER_IMAGE={0}" -f $platformLock.goBuilderReference) --build-arg ("HOOK_LAUNCHER_SHA256={0}" -f $lock.hookLauncher.sha256) --build-arg ("SOURCE_PROFILE={0}" -f $profileName) --build-arg ("SOURCE_INDEX_DIGEST={0}" -f $profile.indexDigest) --build-arg ("SOURCE_MANIFEST_DIGEST={0}" -f $profilePlatform.manifestDigest) --build-arg ("SOURCE_REVISION={0}" -f $profile.sourceRevision) --build-arg ("TEMPLATE_VERSION={0}" -f (($profilePlatform.templateTag -split ':', 2)[1])) --build-arg ("COMPATIBILITY_FILE={0}" -f $profilePlatform.compatibilityFile) --build-arg 'ACTIONS_RUNNER_VERSION=0.0.0' --build-arg ('ACTIONS_RUNNER_SHA256=sha256:' + ('0' * 64)) --build-arg ("TINI_SHA256=sha256:{0}" -f $platformLock.tini.sha256) --file $dockerfilePath $templateDirectory + if ($LASTEXITCODE -ne 0) { + throw "Dockerfile frontend check failed for $profileName" + } + } +} +Write-Host 'Docker Sandboxes runner-template assets passed validation.' diff --git a/scripts/docker/dev.Dockerfile b/scripts/docker/dev.Dockerfile index 2d2e38f..126f1e7 100644 --- a/scripts/docker/dev.Dockerfile +++ b/scripts/docker/dev.Dockerfile @@ -1,7 +1,7 @@ # Go toolchain + Docker CLI, for running EPAR from source with no local Go # install (scripts/run-with-docker.sh). The Docker CLI lets EPAR's own # runtime docker calls reach the host daemon over the mounted socket. -ARG GO_IMAGE=golang:1.25 +ARG GO_IMAGE=golang:latest FROM ${GO_IMAGE} COPY --from=docker:27-cli /usr/local/bin/docker /usr/local/bin/docker COPY --from=docker:27-cli /usr/local/libexec/docker/cli-plugins /usr/local/libexec/docker/cli-plugins diff --git a/scripts/guest/ubuntu/check-host-trust-generation.sh b/scripts/guest/ubuntu/check-host-trust-generation.sh index 6e822b8..5262b9b 100644 --- a/scripts/guest/ubuntu/check-host-trust-generation.sh +++ b/scripts/guest/ubuntu/check-host-trust-generation.sh @@ -1,8 +1,16 @@ #!/usr/bin/env bash set -euo pipefail -marker="${EPAR_HOST_TRUST_MARKER:-/opt/epar/host-trust-generation.json}" -lease="${EPAR_HOST_TRUST_LEASE:-/run/epar/host-trust-lease.json}" +if [[ "$#" -eq 0 ]]; then + marker="/opt/epar/host-trust-generation.json" + lease="/run/epar/host-trust-lease.json" +elif [[ "$#" -eq 2 ]]; then + marker="$1" + lease="$2" +else + echo "EPAR host-trust gate: invalid invocation" >&2 + exit 1 +fi if [[ ! -s "${marker}" ]]; then echo "EPAR host-trust gate: image generation marker is missing" >&2 @@ -12,12 +20,12 @@ if [[ ! -s "${lease}" ]]; then echo "EPAR host-trust gate: controller lease is missing" >&2 exit 1 fi -if ! command -v python3 >/dev/null 2>&1; then +if [[ ! -x /usr/bin/python3 ]]; then echo "EPAR host-trust gate: python3 is required" >&2 exit 1 fi -python3 - "${marker}" "${lease}" <<'PY' +/usr/bin/env -i PATH=/usr/bin:/bin LANG=C.UTF-8 /usr/bin/python3 -I -S - "${marker}" "${lease}" <<'PY' import datetime import json import sys @@ -44,8 +52,10 @@ for key in ("generation", "hostOS", "mode", "scopes"): f"(image={marker.get(key)!r}, lease={lease.get(key)!r})" ) -if marker.get("mode") != "overlay" or not marker.get("generation"): +if marker.get("mode") not in ("overlay", "disabled") or not marker.get("generation"): raise SystemExit("EPAR host-trust gate: invalid image trust policy") +if marker.get("mode") == "disabled" and marker.get("scopes") != []: + raise SystemExit("EPAR host-trust gate: disabled trust mode must not carry scopes") expires = lease.get("expiresAt") if not isinstance(expires, str) or not expires: diff --git a/scripts/guest/ubuntu/install-runner.sh b/scripts/guest/ubuntu/install-runner.sh index de40c66..c78929d 100755 --- a/scripts/guest/ubuntu/install-runner.sh +++ b/scripts/guest/ubuntu/install-runner.sh @@ -1,7 +1,13 @@ #!/usr/bin/env bash set -euo pipefail -RUNNER_VERSION="${1:-latest}" +RUNNER_VERSION="${1:-}" +RUNNER_PACKAGE="${2:-}" +RUNNER_SHA256="${3:-}" +if [[ -z "${RUNNER_VERSION}" || -z "${RUNNER_PACKAGE}" || ! -f "${RUNNER_PACKAGE}" || ! "${RUNNER_SHA256}" =~ ^sha256:[0-9a-f]{64}$ ]]; then + echo "Usage: install-runner.sh " >&2 + exit 1 +fi ARCH="$(uname -m)" case "${ARCH}" in aarch64|arm64) RUNNER_ARCH="arm64" ;; @@ -14,22 +20,24 @@ export NEEDRESTART_MODE=l export NEEDRESTART_SUSPEND=1 bash /opt/epar/wait-apt-ready.sh apt-get update -apt-get install -y --no-install-recommends ca-certificates curl jq sudo tar - -if [[ "${RUNNER_VERSION}" == "latest" ]]; then - RUNNER_VERSION="$(curl -fsSL https://api.github.com/repos/actions/runner/releases/latest | jq -r '.tag_name' | sed 's/^v//')" -fi +apt-get install -y --no-install-recommends ca-certificates sudo tar id -u runner >/dev/null 2>&1 || useradd --create-home --shell /bin/bash runner usermod -aG docker runner 2>/dev/null || true install -d -o runner -g runner /opt/actions-runner cd /opt/actions-runner -RUNNER_TGZ="actions-runner-linux-${RUNNER_ARCH}-${RUNNER_VERSION}.tar.gz" -curl -fL "https://github.com/actions/runner/releases/download/v${RUNNER_VERSION}/${RUNNER_TGZ}" -o "/tmp/${RUNNER_TGZ}" -tar xzf "/tmp/${RUNNER_TGZ}" +echo "${RUNNER_SHA256#sha256:} ${RUNNER_PACKAGE}" | sha256sum --check - +tar xzf "${RUNNER_PACKAGE}" +rm -f "${RUNNER_PACKAGE}" chown -R runner:runner /opt/actions-runner +INSTALLED_RUNNER_VERSION="$(sudo -u runner -H ./bin/Runner.Listener --version | tr -d '\r' | tail -n 1)" +if [[ "${INSTALLED_RUNNER_VERSION}" != "${RUNNER_VERSION}" ]]; then + echo "Actions runner package version ${INSTALLED_RUNNER_VERSION:-} does not match expected version ${RUNNER_VERSION}" >&2 + exit 1 +fi + ./bin/installdependencies.sh install -d /var/log/actions-runner diff --git a/scripts/host-trust/host-trust-feed.ps1 b/scripts/host-trust/host-trust-feed.ps1 index ad494a4..b31d311 100644 --- a/scripts/host-trust/host-trust-feed.ps1 +++ b/scripts/host-trust/host-trust-feed.ps1 @@ -7,6 +7,8 @@ param( [string] $ProjectRoot, [Parameter(Mandatory = $true)] [string] $Config, + [ValidateSet('runner', 'build')] + [string] $Purpose = 'runner', [ValidateRange(1, 3600)] [int] $Interval = 10 ) @@ -14,8 +16,11 @@ param( $ErrorActionPreference = 'Stop' function Get-OverlayConfiguration { - param([string] $Path) - if (-not (Test-Path -LiteralPath $Path -PathType Leaf)) { return $null } + param([string] $Path, [string] $FeedPurpose) + if (-not (Test-Path -LiteralPath $Path -PathType Leaf)) { + if ($FeedPurpose -eq 'build') { return [pscustomobject]@{ Scopes = [string[]]@('system') } } + return $null + } $inImage = $false $listMode = $false $mode = '' @@ -52,6 +57,12 @@ function Get-OverlayConfiguration { } $listMode = $false } + if ($FeedPurpose -eq 'build') { + $buildScopes = [System.Collections.Generic.List[string]]::new() + [void]$buildScopes.Add('system') + if ($mode -eq 'overlay' -and $scopes -contains 'user') { [void]$buildScopes.Add('user') } + return [pscustomobject]@{ Scopes = $buildScopes } + } if ($mode -ne 'overlay') { return $null } if ($scopes.Count -eq 0) { [void]$scopes.Add('system') } foreach ($scope in $scopes) { @@ -219,11 +230,11 @@ if (Test-Path -LiteralPath $Config -PathType Leaf) { } } $Config = $Config.ToLowerInvariant() -$settings = Get-OverlayConfiguration $Config +$settings = Get-OverlayConfiguration $Config $Purpose if ($null -eq $settings) { exit 0 } $cacheBase = if ($env:LOCALAPPDATA) { Join-Path $env:LOCALAPPDATA 'ephemeral-action-runner\host-trust' } else { Join-Path $env:TEMP 'ephemeral-action-runner\host-trust' } -$configId = Get-ConfigId $Config +$configId = Get-ConfigId ($Purpose + [char]0 + $Config) $feedRoot = Join-Path $cacheBase $configId $lockDir = $feedRoot + '.lock' function Acquire-EparSharedLock { diff --git a/scripts/host-trust/host-trust-feed.sh b/scripts/host-trust/host-trust-feed.sh index 10ccc4d..cc22c63 100755 --- a/scripts/host-trust/host-trust-feed.sh +++ b/scripts/host-trust/host-trust-feed.sh @@ -7,7 +7,7 @@ set -euo pipefail usage() { cat >&2 <<'EOF' -Usage: host-trust-feed.sh sync|watch --project-root --config [--interval ] +Usage: host-trust-feed.sh sync|watch --project-root --config [--purpose runner|build] [--interval ] The config must opt in with: image: @@ -20,6 +20,7 @@ shift || true project_root="" config_path="" interval=10 +purpose="runner" script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" while (($#)); do @@ -27,6 +28,7 @@ while (($#)); do --project-root) project_root="${2:?missing value for --project-root}"; shift 2 ;; --config) config_path="${2:?missing value for --config}"; shift 2 ;; --interval) interval="${2:?missing value for --interval}"; shift 2 ;; + --purpose) purpose="${2:?missing value for --purpose}"; shift 2 ;; *) usage; exit 2 ;; esac done @@ -35,6 +37,10 @@ if [[ "$command_name" != "sync" && "$command_name" != "watch" ]] || [[ -z "$proj usage exit 2 fi +if [[ "$purpose" != "runner" && "$purpose" != "build" ]]; then + echo "trust feed purpose must be runner or build" >&2 + exit 2 +fi if [[ ! "$interval" =~ ^[1-9][0-9]*$ ]]; then echo "host trust interval must be a positive integer" >&2 exit 2 @@ -45,13 +51,15 @@ if [[ "$config_path" != /* ]]; then config_path="$project_root/$config_path" fi if [[ ! -f "$config_path" ]]; then - # The first `start` can create config interactively. Treat missing config as - # disabled: native controller code will re-evaluate after init. - exit 0 -fi -config_path="$(cd "$(dirname "$config_path")" && pwd -P)/$(basename "$config_path")" -if command -v realpath >/dev/null 2>&1; then - config_path="$(realpath "$config_path")" + # A first `start` still needs operational system trust to compile the native + # controller before the wizard can create a config. Runner inheritance + # remains disabled until an existing config explicitly enables it. + if [[ "$purpose" != "build" ]]; then exit 0; fi +else + config_path="$(cd "$(dirname "$config_path")" && pwd -P)/$(basename "$config_path")" + if command -v realpath >/dev/null 2>&1; then + config_path="$(realpath "$config_path")" + fi fi sha256_text() { @@ -63,6 +71,7 @@ sha256_text() { } config_values() { + [[ -f "$config_path" ]] || return 0 # Supported deliberately-small YAML subset: EPAR's own parser is flat in # each section and supports inline or block lists. Emit mode followed by # one scope per line. @@ -98,7 +107,13 @@ while IFS= read -r value; do scope=*) scopes+=("$(printf '%s' "${value#scope=}" | tr '[:upper:]' '[:lower:]')") ;; esac done < <(config_values) -if [[ "$mode" != "overlay" ]]; then +if [[ "$purpose" == "build" ]]; then + build_scopes=(system) + if [[ "$mode" == "overlay" ]] && printf '%s\n' "${scopes[@]}" | grep -Fxq user; then + build_scopes+=(user) + fi + scopes=("${build_scopes[@]}") +elif [[ "$mode" != "overlay" ]]; then exit 0 fi @@ -123,7 +138,7 @@ for scope in "${scopes[@]}"; do fi done -config_id="$(printf '%s' "$config_path" | sha256_text | cut -c1-32)" +config_id="$(printf '%s\0%s' "$purpose" "$config_path" | sha256_text | cut -c1-32)" feed_root="$cache_root/$config_id" lock_dir="$feed_root.lock" diff --git a/scripts/host-trust/wrapper-lib.ps1 b/scripts/host-trust/wrapper-lib.ps1 index 8707e5d..eaeb254 100644 --- a/scripts/host-trust/wrapper-lib.ps1 +++ b/scripts/host-trust/wrapper-lib.ps1 @@ -94,42 +94,57 @@ function Start-EparHostTrustBridge { $config = Get-EparHostTrustConfigPath -ProjectRoot $ProjectRoot -Arguments $Arguments if ($Command -eq "init") { - return [pscustomobject]@{ FeedDir = $null; WatchProcess = $null; Config = $config; PostInit = $true } + return [pscustomobject]@{ FeedDir = $null; BuildFeedDir = $null; RunnerFeedDir = $null; WatchProcess = $null; WatchProcesses = @(); Config = $config; PostInit = $true } } $subcommand = if ($Arguments -and $Arguments.Count -gt 1) { [string]$Arguments[1] } else { "" } $needsBridge = $Command -eq "start" -or - ($Command -eq "image" -and $subcommand -eq "build") -or + ($Command -eq "image" -and $subcommand -in @("build", "update")) -or ($Command -eq "pool" -and $subcommand -in @("up", "verify")) if (-not $needsBridge) { - return [pscustomobject]@{ FeedDir = $null; WatchProcess = $null; Config = $config; PostInit = $false } + return [pscustomobject]@{ FeedDir = $null; BuildFeedDir = $null; RunnerFeedDir = $null; WatchProcess = $null; WatchProcesses = @(); Config = $config; PostInit = $false } } $helper = Join-Path $ProjectRoot "scripts\host-trust\host-trust-feed.ps1" - $feedLines = @(& $helper sync -ProjectRoot $ProjectRoot -Config $config 2>&1) - if ($LASTEXITCODE -ne 0) { - throw "Host-trust preflight failed: $($feedLines -join [Environment]::NewLine)" - } - $feedDir = ($feedLines | Where-Object { $_ -is [string] -and $_.Trim() } | Select-Object -Last 1) - if (-not $feedDir) { - return [pscustomobject]@{ FeedDir = $null; WatchProcess = $null; Config = $config; PostInit = $false } - } - $feedDir = Split-Path -Parent $feedDir.Trim() - $watchOut = Join-Path $feedDir "watcher.log" - $watchErr = Join-Path $feedDir "watcher-error.log" $powershell = (Get-Process -Id $PID).Path - $watchCommand = '& ' + (ConvertTo-EparPowerShellLiteral $helper) + - ' watch -ProjectRoot ' + (ConvertTo-EparPowerShellLiteral $ProjectRoot) + - ' -Config ' + (ConvertTo-EparPowerShellLiteral $config) + - ' -Interval 10' - $encodedWatchCommand = [Convert]::ToBase64String([System.Text.Encoding]::Unicode.GetBytes($watchCommand)) - $watch = Start-Process -FilePath $powershell -WindowStyle Hidden -PassThru ` - -ArgumentList @("-NoLogo", "-NoProfile", "-ExecutionPolicy", "Bypass", "-EncodedCommand", $encodedWatchCommand) ` - -RedirectStandardOutput $watchOut -RedirectStandardError $watchErr - Start-Sleep -Milliseconds 150 - if ($watch.HasExited) { - throw "Host-trust watcher exited during startup. See $watchErr" + $watchers = [System.Collections.Generic.List[object]]::new() + $feedDirectories = @{} + foreach ($purpose in @('build', 'runner')) { + $feedLines = @(& $helper sync -ProjectRoot $ProjectRoot -Config $config -Purpose $purpose 2>&1) + if ($LASTEXITCODE -ne 0) { + throw "$purpose trust preflight failed: $($feedLines -join [Environment]::NewLine)" + } + $feedPath = ($feedLines | Where-Object { $_ -is [string] -and $_.Trim() } | Select-Object -Last 1) + if (-not $feedPath) { + $feedDirectories[$purpose] = $null + continue + } + $feedDir = Split-Path -Parent $feedPath.Trim() + $feedDirectories[$purpose] = $feedDir + $watchOut = Join-Path $feedDir "watcher.log" + $watchErr = Join-Path $feedDir "watcher-error.log" + $watchCommand = '& ' + (ConvertTo-EparPowerShellLiteral $helper) + + ' watch -ProjectRoot ' + (ConvertTo-EparPowerShellLiteral $ProjectRoot) + + ' -Config ' + (ConvertTo-EparPowerShellLiteral $config) + + ' -Purpose ' + (ConvertTo-EparPowerShellLiteral $purpose) + + ' -Interval 10 >> ' + (ConvertTo-EparPowerShellLiteral $watchOut) + + ' 2>> ' + (ConvertTo-EparPowerShellLiteral $watchErr) + $encodedWatchCommand = [Convert]::ToBase64String([System.Text.Encoding]::Unicode.GetBytes($watchCommand)) + $startInfo = [System.Diagnostics.ProcessStartInfo]::new() + $startInfo.FileName = $powershell + $startInfo.Arguments = "-NoLogo -NoProfile -ExecutionPolicy Bypass -EncodedCommand $encodedWatchCommand" + $startInfo.UseShellExecute = $false + $startInfo.CreateNoWindow = $true + $watch = [System.Diagnostics.Process]::Start($startInfo) + Start-Sleep -Milliseconds 150 + if ($watch.HasExited) { + throw "$purpose trust watcher exited during startup. See $watchErr" + } + [void]$watchers.Add([pscustomobject]@{ Process = $watch; FeedDir = $feedDir }) } - return [pscustomobject]@{ FeedDir = $feedDir; WatchProcess = $watch; Config = $config; PostInit = $false } + $runnerFeedDir = $feedDirectories['runner'] + $buildFeedDir = $feedDirectories['build'] + $firstWatcher = if ($watchers.Count -gt 0) { $watchers[0].Process } else { $null } + return [pscustomobject]@{ FeedDir = $runnerFeedDir; BuildFeedDir = $buildFeedDir; RunnerFeedDir = $runnerFeedDir; WatchProcess = $firstWatcher; WatchProcesses = @($watchers); Config = $config; PostInit = $false } } function Complete-EparHostTrustInit { @@ -150,21 +165,26 @@ function Complete-EparHostTrustInit { function Stop-EparHostTrustBridge { param($Bridge) - if ($null -eq $Bridge -or $null -eq $Bridge.WatchProcess) { return } - try { - if (-not $Bridge.WatchProcess.HasExited) { - Stop-Process -Id $Bridge.WatchProcess.Id -ErrorAction SilentlyContinue - $Bridge.WatchProcess.WaitForExit(3000) | Out-Null + if ($null -eq $Bridge) { return } + $entries = @($Bridge.WatchProcesses) + if ($entries.Count -eq 0 -and $null -ne $Bridge.WatchProcess) { + $entries = @([pscustomobject]@{ Process = $Bridge.WatchProcess; FeedDir = $Bridge.FeedDir }) + } + foreach ($entry in $entries) { + try { + if (-not $entry.Process.HasExited) { + Stop-Process -Id $entry.Process.Id -ErrorAction SilentlyContinue + $entry.Process.WaitForExit(3000) | Out-Null + } + } catch { + Write-Warning "Could not stop trust-feed watcher: $($_.Exception.Message)" } - } catch { - Write-Warning "Could not stop host-trust watcher: $($_.Exception.Message)" - } - if ($Bridge.FeedDir -and $Bridge.WatchProcess.HasExited) { - $lockDir = $Bridge.FeedDir + '.lock' + if (-not $entry.FeedDir -or -not $entry.Process.HasExited) { continue } + $lockDir = $entry.FeedDir + '.lock' $ownerPath = Join-Path $lockDir 'pid' $owner = 0 [void][int]::TryParse((Get-Content -LiteralPath $ownerPath -ErrorAction SilentlyContinue | Select-Object -First 1), [ref]$owner) - if ($owner -eq $Bridge.WatchProcess.Id) { + if ($owner -eq $entry.Process.Id) { Remove-Item -LiteralPath $ownerPath -Force -ErrorAction SilentlyContinue Remove-Item -LiteralPath $lockDir -Force -ErrorAction SilentlyContinue } diff --git a/scripts/host-trust/wrapper-lib.sh b/scripts/host-trust/wrapper-lib.sh index d689aa5..1585b14 100644 --- a/scripts/host-trust/wrapper-lib.sh +++ b/scripts/host-trust/wrapper-lib.sh @@ -4,7 +4,10 @@ # set EPAR_HOST_TRUST_HELPER to the real-host helper script before sourcing. EPAR_HOST_TRUST_FEED_DIR="" +EPAR_BUILD_TRUST_FEED_DIR="" +EPAR_RUNNER_TRUST_FEED_DIR="" EPAR_HOST_TRUST_WATCH_PID="" +EPAR_TRUST_WATCH_PIDS=() EPAR_HOST_TRUST_POST_INIT_CONFIG="" epar_host_trust_config_path() { @@ -72,9 +75,12 @@ epar_host_trust_prepare() { local project_root="$1" command="$2" shift 2 EPAR_HOST_TRUST_FEED_DIR="" + EPAR_BUILD_TRUST_FEED_DIR="" + EPAR_RUNNER_TRUST_FEED_DIR="" EPAR_HOST_TRUST_WATCH_PID="" + EPAR_TRUST_WATCH_PIDS=() EPAR_HOST_TRUST_POST_INIT_CONFIG="" - local config_path feed_path watcher_log subcommand="" + local config_path feed_path feed_dir watcher_log subcommand="" purpose watcher_pid if (($# >= 2)); then subcommand="$2"; fi config_path="$(epar_host_trust_config_path "$project_root" "$@")" case "$command" in @@ -85,25 +91,28 @@ epar_host_trust_prepare() { return 0 ;; start) ;; - image) [[ "$subcommand" == build ]] || return 0 ;; + image) [[ "$subcommand" == build || "$subcommand" == update ]] || return 0 ;; pool) [[ "$subcommand" == up || "$subcommand" == verify ]] || return 0 ;; *) return 0 ;; esac - feed_path="$("$EPAR_HOST_TRUST_HELPER" sync --project-root "$project_root" --config "$config_path")" || return $? - [[ -n "$feed_path" ]] || return 0 - EPAR_HOST_TRUST_FEED_DIR="$(dirname "$feed_path")" - watcher_log="${EPAR_HOST_TRUST_FEED_DIR}/watcher.log" - "$EPAR_HOST_TRUST_HELPER" watch --project-root "$project_root" --config "$config_path" --interval 10 >>"$watcher_log" 2>&1 & - EPAR_HOST_TRUST_WATCH_PID="$!" - # Fail closed when the singleton watcher rejects the lock or exits before - # the controller receives its first feed generation. - sleep 0.1 - if ! kill -0 "$EPAR_HOST_TRUST_WATCH_PID" 2>/dev/null; then - wait "$EPAR_HOST_TRUST_WATCH_PID" || true - EPAR_HOST_TRUST_WATCH_PID="" - echo "host trust watcher failed to start; see $watcher_log" >&2 - return 1 - fi + for purpose in build runner; do + feed_path="$("$EPAR_HOST_TRUST_HELPER" sync --project-root "$project_root" --config "$config_path" --purpose "$purpose")" || return $? + [[ -n "$feed_path" ]] || continue + feed_dir="$(dirname "$feed_path")" + if [[ "$purpose" == build ]]; then EPAR_BUILD_TRUST_FEED_DIR="$feed_dir"; else EPAR_RUNNER_TRUST_FEED_DIR="$feed_dir"; fi + watcher_log="${feed_dir}/watcher.log" + "$EPAR_HOST_TRUST_HELPER" watch --project-root "$project_root" --config "$config_path" --purpose "$purpose" --interval 10 >>"$watcher_log" 2>&1 & + watcher_pid="$!" + EPAR_TRUST_WATCH_PIDS+=("$watcher_pid") + sleep 0.1 + if ! kill -0 "$watcher_pid" 2>/dev/null; then + wait "$watcher_pid" || true + echo "$purpose trust watcher failed to start; see $watcher_log" >&2 + return 1 + fi + done + EPAR_HOST_TRUST_FEED_DIR="$EPAR_RUNNER_TRUST_FEED_DIR" + if ((${#EPAR_TRUST_WATCH_PIDS[@]} > 0)); then EPAR_HOST_TRUST_WATCH_PID="${EPAR_TRUST_WATCH_PIDS[0]}"; fi } epar_host_trust_post_init() { @@ -125,9 +134,12 @@ epar_host_trust_post_init() { } epar_host_trust_cleanup() { - if [[ -n "${EPAR_HOST_TRUST_WATCH_PID:-}" ]]; then - kill "$EPAR_HOST_TRUST_WATCH_PID" 2>/dev/null || true - wait "$EPAR_HOST_TRUST_WATCH_PID" 2>/dev/null || true - EPAR_HOST_TRUST_WATCH_PID="" - fi + local watcher_pid + for watcher_pid in "${EPAR_TRUST_WATCH_PIDS[@]:-}"; do + [[ -n "$watcher_pid" ]] || continue + kill "$watcher_pid" 2>/dev/null || true + wait "$watcher_pid" 2>/dev/null || true + done + EPAR_TRUST_WATCH_PIDS=() + EPAR_HOST_TRUST_WATCH_PID="" } diff --git a/scripts/run-with-docker.ps1 b/scripts/run-with-docker.ps1 index 7310c4c..72eb382 100644 --- a/scripts/run-with-docker.ps1 +++ b/scripts/run-with-docker.ps1 @@ -3,22 +3,36 @@ param( [string[]] $EparArgs ) -# Runs EPAR from source with no local Go install: a containerized Go -# toolchain compiles and executes the source with `go run`, the same as the -# documented source-first path (docs/usage.md) - just inside a container -# instead of on the host. No binary is built or left on disk. +# Runs EPAR from source with no local Go install. By default, a containerized +# Go toolchain builds a CGO-disabled native controller under .local/bin and the +# wrapper executes that binary on the host. Set +# EPAR_LEGACY_CONTROLLER_IN_DOCKER=1 only for compatible legacy providers. # -# Docker is still required (both for this wrapper and for EPAR's own -# Docker-DinD provider, reached here via the mounted host socket). +# Docker is required for the build toolchain and for the Docker Container +# provider. Docker Sandboxes also requires its separately installed sbx CLI. # # Usage: scripts\run-with-docker.ps1 [epar-args...] $ErrorActionPreference = "Stop" +$OriginalInvocationExists = Test-Path Env:EPAR_INVOCATION +$OriginalInvocation = $env:EPAR_INVOCATION +$OwnInvocationMarker = [string]::IsNullOrWhiteSpace($env:EPAR_INVOCATION) +if (-not $env:EPAR_INVOCATION) { + $env:EPAR_INVOCATION = "run-with-docker-powershell" +} + +if ($env:EPAR_LEGACY_CONTROLLER_IN_DOCKER -ne '1') { + try { + & (Join-Path $PSScriptRoot 'build-native-controller.ps1') @EparArgs + $nativeExitCode = $LASTEXITCODE + } finally { + if ($OwnInvocationMarker -and $OriginalInvocationExists) { $env:EPAR_INVOCATION = $OriginalInvocation } elseif ($OwnInvocationMarker) { Remove-Item Env:EPAR_INVOCATION -ErrorAction SilentlyContinue } + } + exit $nativeExitCode +} -$Image = if ($env:GO_DOCKER_IMAGE) { $env:GO_DOCKER_IMAGE } else { "golang:1.25" } +$Image = if ($env:GO_DOCKER_IMAGE) { $env:GO_DOCKER_IMAGE } else { "golang:latest" } $DevImage = if ($env:EPAR_DEV_IMAGE) { $env:EPAR_DEV_IMAGE } else { "epar-dev-toolchain" } -$GomodVolume = if ($env:EPAR_GOMOD_VOLUME) { $env:EPAR_GOMOD_VOLUME } else { "epar-gomod" } -$GocacheVolume = if ($env:EPAR_GOCACHE_VOLUME) { $env:EPAR_GOCACHE_VOLUME } else { "epar-gocache" } $DockerSock = if ($env:EPAR_DOCKER_SOCK) { $env:EPAR_DOCKER_SOCK } else { "/var/run/docker.sock" } $OriginalDockerCliHintsExists = Test-Path Env:DOCKER_CLI_HINTS $OriginalDockerCliHints = $env:DOCKER_CLI_HINTS @@ -37,6 +51,8 @@ if (-not $HostName) { } $DockerEnvFlags = @() $DockerEnvFlags += @("-e", "DOCKER_CLI_HINTS=$DockerCliHints") +$DockerEnvFlags += @("-e", "EPAR_CONTROLLER_IN_DOCKER=1") +$DockerEnvFlags += @("-e", "EPAR_INVOCATION=$($env:EPAR_INVOCATION)") if ($env:EPAR_CONFIG) { $DockerEnvFlags += @("-e", "EPAR_CONFIG=$($env:EPAR_CONFIG)") } @@ -45,10 +61,44 @@ if ($HostName) { } if (-not (Get-Command docker -ErrorAction SilentlyContinue)) { - Write-Error "docker command not found. Install Docker Desktop or another working Docker host." + Write-Error "docker command not found. Install Docker and make sure it is available on PATH." exit 1 } +function Test-EparBenignDockerDesktopPrefaceDiagnostic { + param([Parameter(Mandatory = $true)][string] $Transcript) + $normalized = ($Transcript -replace '\s+', ' ').Trim() + return $normalized -match '^(?:docker\s*:\s*)?\d{4}/\d{2}/\d{2} \d{2}:\d{2}:\d{2} http2: server: error reading preface from client //\./pipe/(?:dockerDesktopLinuxEngine|docker_engine): file has already been closed(?: At .* FullyQualifiedErrorId\s*:\s*NativeCommandError)?$' +} + +function Invoke-EparBootstrapDockerBuild { + $stderrPath = [System.IO.Path]::GetTempFileName() + $previousErrorActionPreference = $ErrorActionPreference + try { + # Windows PowerShell converts native stderr into ErrorRecord objects + # when the preference is Stop. Keep the native transcript as bytes so + # the one known successful Docker Desktop diagnostic can be classified + # without terminating this wrapper or losing real failure output. + $ErrorActionPreference = 'Continue' + docker build --quiet ` + --build-arg "GO_IMAGE=$Image" ` + -t $DevImage ` + -f (Join-Path $RepoRoot "scripts\docker\dev.Dockerfile") ` + (Join-Path $RepoRoot "scripts\docker") 2> $stderrPath | Out-Null + $buildExitCode = $LASTEXITCODE + if (Test-Path -LiteralPath $stderrPath) { + $stderrTranscript = Get-Content -Raw -LiteralPath $stderrPath + if ($stderrTranscript -and -not ($buildExitCode -eq 0 -and (Test-EparBenignDockerDesktopPrefaceDiagnostic -Transcript $stderrTranscript))) { + [Console]::Error.Write($stderrTranscript) + } + } + return $buildExitCode + } finally { + $ErrorActionPreference = $previousErrorActionPreference + Remove-Item -LiteralPath $stderrPath -Force -ErrorAction SilentlyContinue + } +} + $RepoRoot = Split-Path -Parent (Split-Path -Parent $MyInvocation.MyCommand.Path) . (Join-Path $RepoRoot "scripts\host-trust\wrapper-lib.ps1") $EparCommand = if ($EparArgs -and $EparArgs.Count -gt 0) { [string] $EparArgs[0] } else { "start" } @@ -59,6 +109,9 @@ if ($EparCommand -eq "init" -or $ImplicitInit) { $DockerEnvFlags += @("-e", "EPAR_CONTROLLER_HOST_OS=$(Get-EparHostTrustHostOS)") } $DockerRunFlags = @("--rm", "-i") +$GoCacheFlags = @() +if ($env:EPAR_GOMOD_VOLUME) { $GoCacheFlags += @("-v", "$($env:EPAR_GOMOD_VOLUME):/go/pkg/mod") } +if ($env:EPAR_GOCACHE_VOLUME) { $GoCacheFlags += @("-v", "$($env:EPAR_GOCACHE_VOLUME):/root/.cache/go-build") } try { if (-not [Console]::IsInputRedirected) { $DockerRunFlags += "-t" @@ -70,45 +123,44 @@ try { $ExitCode = 0 $bridge = $null try { - docker build --quiet ` - --build-arg "GO_IMAGE=$Image" ` - -t $DevImage ` - -f (Join-Path $RepoRoot "scripts\docker\dev.Dockerfile") ` - (Join-Path $RepoRoot "scripts\docker") | Out-Null - - if ($LASTEXITCODE -ne 0) { - $ExitCode = $LASTEXITCODE + $ExitCode = Invoke-EparBootstrapDockerBuild + if ($ExitCode -ne 0) { + # Invoke-EparBootstrapDockerBuild already preserved the complete Docker stderr on failure. } else { if ($ImplicitInit) { $InitArgs = @(Get-EparHostTrustInitArguments -Arguments $EparArgs) docker run @DockerRunFlags ` @DockerEnvFlags ` + @GoCacheFlags ` -v "${RepoRoot}:/app" -w /app ` - -v "${GomodVolume}:/go/pkg/mod" ` - -v "${GocacheVolume}:/root/.cache/go-build" ` -v "${DockerSock}:/var/run/docker.sock" ` $DevImage ` go run ./cmd/ephemeral-action-runner @InitArgs $ExitCode = $LASTEXITCODE if ($ExitCode -eq 0) { - $initBridge = [pscustomobject]@{ FeedDir = $null; WatchProcess = $null; Config = $ConfigPath; PostInit = $true } + $initBridge = [pscustomobject]@{ FeedDir = $null; BuildFeedDir = $null; RunnerFeedDir = $null; WatchProcess = $null; WatchProcesses = @(); Config = $ConfigPath; PostInit = $true } Complete-EparHostTrustInit -ProjectRoot $RepoRoot -Bridge $initBridge } } if ($ExitCode -eq 0) { $bridge = Start-EparHostTrustBridge -ProjectRoot $RepoRoot -Command $EparCommand -Arguments $EparArgs $HostTrustFlags = @() - if ($bridge.FeedDir) { + if ($bridge.BuildFeedDir -or $bridge.RunnerFeedDir) { $HostTrustFlags += @("-e", "EPAR_CONTROLLER_HOST_OS=$(Get-EparHostTrustHostOS)") + } + if ($bridge.BuildFeedDir) { + $HostTrustFlags += @("-e", "EPAR_BUILD_TRUST_FEED=/run/epar-build-trust/current.json") + $HostTrustFlags += @("-v", "$($bridge.BuildFeedDir):/run/epar-build-trust:ro") + } + if ($bridge.RunnerFeedDir) { $HostTrustFlags += @("-e", "EPAR_HOST_TRUST_FEED=/run/epar-host-trust/current.json") - $HostTrustFlags += @("-v", "$($bridge.FeedDir):/run/epar-host-trust:ro") + $HostTrustFlags += @("-v", "$($bridge.RunnerFeedDir):/run/epar-host-trust:ro") } docker run @DockerRunFlags ` @DockerEnvFlags ` @HostTrustFlags ` + @GoCacheFlags ` -v "${RepoRoot}:/app" -w /app ` - -v "${GomodVolume}:/go/pkg/mod" ` - -v "${GocacheVolume}:/root/.cache/go-build" ` -v "${DockerSock}:/var/run/docker.sock" ` $DevImage ` go run ./cmd/ephemeral-action-runner @EparArgs @@ -126,6 +178,7 @@ try { } else { Remove-Item Env:DOCKER_CLI_HINTS -ErrorAction SilentlyContinue } + if ($OwnInvocationMarker -and $OriginalInvocationExists) { $env:EPAR_INVOCATION = $OriginalInvocation } elseif ($OwnInvocationMarker) { Remove-Item Env:EPAR_INVOCATION -ErrorAction SilentlyContinue } } exit $ExitCode diff --git a/scripts/run-with-docker.sh b/scripts/run-with-docker.sh index 5c25ea1..f5f8155 100755 --- a/scripts/run-with-docker.sh +++ b/scripts/run-with-docker.sh @@ -1,6 +1,8 @@ #!/usr/bin/env bash set -euo pipefail +export EPAR_INVOCATION="${EPAR_INVOCATION:-run-with-docker}" + case "$(uname -s)" in MINGW*|MSYS*|CYGWIN*) script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" @@ -10,20 +12,24 @@ case "$(uname -s)" in ;; esac -# Runs EPAR from source with no local Go install: a containerized Go -# toolchain compiles and executes the source with `go run`, the same as the -# documented source-first path (docs/usage.md) — just inside a container -# instead of on the host. No binary is built or left on disk. +if [[ "${EPAR_LEGACY_CONTROLLER_IN_DOCKER:-0}" != "1" ]]; then + exec bash "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)/build-native-controller.sh" "$@" +fi + +# Runs EPAR from source with no local Go install. By default, a containerized +# Go toolchain builds a CGO-disabled native controller under .local/bin and the +# wrapper executes that binary on the host. Set +# EPAR_LEGACY_CONTROLLER_IN_DOCKER=1 only for compatible legacy providers. # -# Docker is still required (both for this wrapper and for EPAR's own -# Docker-DinD provider, reached here via the mounted host socket). +# Docker is required for the build toolchain and for the Docker Container +# provider. Docker Sandboxes also requires its separately installed sbx CLI. # # Usage: scripts/run-with-docker.sh [epar-args...] -image="${GO_DOCKER_IMAGE:-golang:1.25}" +image="${GO_DOCKER_IMAGE:-golang:latest}" dev_image="${EPAR_DEV_IMAGE:-epar-dev-toolchain}" -gomod_volume="${EPAR_GOMOD_VOLUME:-epar-gomod}" -gocache_volume="${EPAR_GOCACHE_VOLUME:-epar-gocache}" +gomod_volume="${EPAR_GOMOD_VOLUME:-}" +gocache_volume="${EPAR_GOCACHE_VOLUME:-}" docker_sock="${EPAR_DOCKER_SOCK:-/var/run/docker.sock}" export DOCKER_CLI_HINTS="${DOCKER_CLI_HINTS:-false}" host_name="${EPAR_HOST_NAME:-}" @@ -31,13 +37,15 @@ if [[ -z "${host_name}" ]]; then host_name="$(hostname 2>/dev/null || true)" fi docker_env_flags=(-e "DOCKER_CLI_HINTS=${DOCKER_CLI_HINTS}") +docker_env_flags+=(-e "EPAR_CONTROLLER_IN_DOCKER=1") +docker_env_flags+=(-e "EPAR_INVOCATION=${EPAR_INVOCATION}") if [[ -n "${EPAR_CONFIG:-}" ]]; then docker_env_flags+=(-e "EPAR_CONFIG=${EPAR_CONFIG}"); fi if [[ -n "${host_name}" ]]; then docker_env_flags+=(-e "EPAR_HOST_NAME=${host_name}") fi if ! command -v docker >/dev/null 2>&1; then - echo "docker command not found. Install Docker Desktop, Docker Engine, or a compatible Docker host." >&2 + echo "docker command not found. Install Docker and make sure it is available on PATH." >&2 exit 1 fi @@ -68,6 +76,9 @@ if [[ -t 0 ]]; then fi host_trust_docker_flags=() +go_cache_docker_flags=() +if [[ -n "$gomod_volume" ]]; then go_cache_docker_flags+=(-v "${gomod_volume}:/go/pkg/mod"); fi +if [[ -n "$gocache_volume" ]]; then go_cache_docker_flags+=(-v "${gocache_volume}:/root/.cache/go-build"); fi run_controller() { local -a docker_args=(run) docker_args+=("${tty_flags[@]}" "${docker_env_flags[@]}") @@ -76,12 +87,12 @@ run_controller() { fi docker_args+=( -v "${repo_root}:/app" -w /app - -v "${gomod_volume}:/go/pkg/mod" - -v "${gocache_volume}:/root/.cache/go-build" -v "${docker_sock}:/var/run/docker.sock" - "$dev_image" - go run ./cmd/ephemeral-action-runner "$@" ) + if ((${#go_cache_docker_flags[@]})); then + docker_args+=("${go_cache_docker_flags[@]}") + fi + docker_args+=("$dev_image" go run ./cmd/ephemeral-action-runner "$@") docker "${docker_args[@]}" } @@ -100,11 +111,19 @@ fi if [[ "$status" == 0 ]]; then epar_host_trust_prepare "${repo_root}" "${controller_command:-start}" "$@" || status=$? fi -if [[ "$status" == 0 && -n "${EPAR_HOST_TRUST_FEED_DIR}" ]]; then +if [[ "$status" == 0 && ( -n "${EPAR_BUILD_TRUST_FEED_DIR}" || -n "${EPAR_RUNNER_TRUST_FEED_DIR}" ) ]]; then + host_trust_docker_flags+=(-e "EPAR_CONTROLLER_HOST_OS=$(epar_host_trust_host_os)") +fi +if [[ "$status" == 0 && -n "${EPAR_BUILD_TRUST_FEED_DIR}" ]]; then + host_trust_docker_flags+=( + -e "EPAR_BUILD_TRUST_FEED=/run/epar-build-trust/current.json" + -v "${EPAR_BUILD_TRUST_FEED_DIR}:/run/epar-build-trust:ro" + ) +fi +if [[ "$status" == 0 && -n "${EPAR_RUNNER_TRUST_FEED_DIR}" ]]; then host_trust_docker_flags+=( - -e "EPAR_CONTROLLER_HOST_OS=$(epar_host_trust_host_os)" -e "EPAR_HOST_TRUST_FEED=/run/epar-host-trust/current.json" - -v "${EPAR_HOST_TRUST_FEED_DIR}:/run/epar-host-trust:ro" + -v "${EPAR_RUNNER_TRUST_FEED_DIR}:/run/epar-host-trust:ro" ) fi diff --git a/scripts/sync-wiki.sh b/scripts/sync-wiki.sh index 43c9ed2..7fabba8 100755 --- a/scripts/sync-wiki.sh +++ b/scripts/sync-wiki.sh @@ -26,6 +26,14 @@ copy_assets() { mkdir -p "$out_dir/assets" cp -R "$repo_root/docs/assets/." "$out_dir/assets/" fi + if [ -d "$repo_root/configs" ]; then + mkdir -p "$out_dir/configs" + cp -R "$repo_root/configs/." "$out_dir/configs/" + fi + if [ -f "$repo_root/templates/docker-sandboxes/sources.lock.json" ]; then + mkdir -p "$out_dir/templates/docker-sandboxes" + cp "$repo_root/templates/docker-sandboxes/sources.lock.json" "$out_dir/templates/docker-sandboxes/" + fi } copy_page() { @@ -44,38 +52,82 @@ rewrite_links() { perl -0pi -e ' my %map = ( "README.md" => "Home", + "docs/README.md" => "Documentation", "docs/usage.md" => "Usage", "usage.md" => "Usage", + "docs/configuration.md" => "Configuration", + "configuration.md" => "Configuration", "docs/github-app.md" => "GitHub-App-Setup", "github-app.md" => "GitHub-App-Setup", + "docs/runner-groups.md" => "Runner-Groups", + "runner-groups.md" => "Runner-Groups", "docs/image-build.md" => "Image-Build", "image-build.md" => "Image-Build", - "docs/design.md" => "Design", + "docs/development/design.md" => "Design", + "development/design.md" => "Design", "design.md" => "Design", + "docs/development/principles.md" => "Development-Principles", + "development/principles.md" => "Development-Principles", + "principles.md" => "Development-Principles", "docs/operations.md" => "Operations", "operations.md" => "Operations", + "docs/logging.md" => "Logging", + "logging.md" => "Logging", + "docs/storage.md" => "Storage", + "storage.md" => "Storage", + "docs/troubleshooting.md" => "Troubleshooting", + "troubleshooting.md" => "Troubleshooting", "docs/security.md" => "Security", "security.md" => "Security", - "docs/background.md" => "Background", - "background.md" => "Background", "docs/providers/tart.md" => "Tart-Provider", "providers/tart.md" => "Tart-Provider", "tart.md" => "Tart-Provider", "docs/providers/wsl.md" => "WSL-Provider", "providers/wsl.md" => "WSL-Provider", "wsl.md" => "WSL-Provider", - "docs/providers/docker-dind.md" => "Docker-DinD-Provider", - "providers/docker-dind.md" => "Docker-DinD-Provider", - "docker-dind.md" => "Docker-DinD-Provider", - "docs/providers/adding-provider.md" => "Adding-A-Provider", - "providers/adding-provider.md" => "Adding-A-Provider", + "docs/providers/docker-container.md" => "Docker-Container-Provider", + "providers/docker-container.md" => "Docker-Container-Provider", + "docker-container.md" => "Docker-Container-Provider", + "docs/providers/docker-sandboxes.md" => "Docker-Sandboxes-Provider", + "providers/docker-sandboxes.md" => "Docker-Sandboxes-Provider", + "docker-sandboxes.md" => "Docker-Sandboxes-Provider", + "docs/development/adding-provider.md" => "Adding-A-Provider", + "development/adding-provider.md" => "Adding-A-Provider", "adding-provider.md" => "Adding-A-Provider", "docs/advanced/docker-registry-mirrors.md" => "Docker-Registry-Mirrors", "advanced/docker-registry-mirrors.md" => "Docker-Registry-Mirrors", "docker-registry-mirrors.md" => "Docker-Registry-Mirrors", + "docs/advanced/docker-sandboxes-template.md" => "Docker-Sandboxes-Template", + "advanced/docker-sandboxes-template.md" => "Docker-Sandboxes-Template", + "docker-sandboxes-template.md" => "Docker-Sandboxes-Template", + "docs/advanced/cross-architecture-containers.md" => "Cross-Architecture-Containers", + "advanced/cross-architecture-containers.md" => "Cross-Architecture-Containers", + "cross-architecture-containers.md" => "Cross-Architecture-Containers", + "docs/advanced/no-go-install.md" => "No-Go-Install", + "advanced/no-go-install.md" => "No-Go-Install", + "no-go-install.md" => "No-Go-Install", + "docs/advanced/windows-startup.md" => "Windows-Startup", + "advanced/windows-startup.md" => "Windows-Startup", + "windows-startup.md" => "Windows-Startup", "docs/advanced/macos-startup.md" => "macOS-Startup", "advanced/macos-startup.md" => "macOS-Startup", "macos-startup.md" => "macOS-Startup", + "docs/development/README.md" => "Development", + "development/README.md" => "Development", + "docs/development/" => "Development", + "development/" => "Development", + "" => "Documentation", + "docs/development/core-runner-verification.md" => "Core-Runner-Verification", + "development/core-runner-verification.md" => "Core-Runner-Verification", + "core-runner-verification.md" => "Core-Runner-Verification", + "docs/development/releases.md" => "Releases", + "development/releases.md" => "Releases", + "releases.md" => "Releases", + "examples/observability/README.md" => "Observability-Examples", + "observability/README.md" => "Observability-Examples", + "SUPPORT.md" => "Support", + "CONTRIBUTING.md" => "Contributing", + "CODE_OF_CONDUCT.md" => "Code-of-Conduct", ); s{\]\(([^)]+)\)}{ @@ -89,6 +141,8 @@ rewrite_links() { while ($path =~ s{^\.\./}{}) {} if ($path =~ s{^docs/assets/}{assets/}) { "]($path$anchor)"; + } elsif ($path =~ m{^(?:configs|templates)/}) { + "]($path$anchor)"; } elsif (exists $map{$path}) { "]($map{$path}$anchor)"; } else { @@ -102,19 +156,37 @@ rewrite_links() { copy_assets copy_page "README.md" "Home" +copy_page "docs/README.md" "Documentation" copy_page "docs/usage.md" "Usage" +copy_page "docs/configuration.md" "Configuration" copy_page "docs/github-app.md" "GitHub-App-Setup" +copy_page "docs/runner-groups.md" "Runner-Groups" copy_page "docs/image-build.md" "Image-Build" -copy_page "docs/design.md" "Design" +copy_page "docs/development/design.md" "Design" +copy_page "docs/development/principles.md" "Development-Principles" copy_page "docs/operations.md" "Operations" +copy_page "docs/logging.md" "Logging" +copy_page "docs/storage.md" "Storage" +copy_page "docs/troubleshooting.md" "Troubleshooting" copy_page "docs/security.md" "Security" -copy_page "docs/background.md" "Background" copy_page "docs/providers/tart.md" "Tart-Provider" copy_page "docs/providers/wsl.md" "WSL-Provider" -copy_page "docs/providers/docker-dind.md" "Docker-DinD-Provider" -copy_page "docs/providers/adding-provider.md" "Adding-A-Provider" +copy_page "docs/providers/docker-container.md" "Docker-Container-Provider" +copy_page "docs/providers/docker-sandboxes.md" "Docker-Sandboxes-Provider" +copy_page "docs/development/adding-provider.md" "Adding-A-Provider" copy_page "docs/advanced/docker-registry-mirrors.md" "Docker-Registry-Mirrors" +copy_page "docs/advanced/docker-sandboxes-template.md" "Docker-Sandboxes-Template" +copy_page "docs/advanced/cross-architecture-containers.md" "Cross-Architecture-Containers" +copy_page "docs/advanced/no-go-install.md" "No-Go-Install" +copy_page "docs/advanced/windows-startup.md" "Windows-Startup" copy_page "docs/advanced/macos-startup.md" "macOS-Startup" +copy_page "docs/development/README.md" "Development" +copy_page "docs/development/core-runner-verification.md" "Core-Runner-Verification" +copy_page "docs/development/releases.md" "Releases" +copy_page "examples/observability/README.md" "Observability-Examples" +copy_page "SUPPORT.md" "Support" +copy_page "CONTRIBUTING.md" "Contributing" +copy_page "CODE_OF_CONDUCT.md" "Code-of-Conduct" for file in "$out_dir"/*.md; do rewrite_links "$file" @@ -124,13 +196,16 @@ cat > "$out_dir/_Sidebar.md" <<'SIDEBAR' # EPAR Docs - [Home](Home) +- [Documentation](Documentation) - [Usage](Usage) +- [Configuration](Configuration) - [GitHub App Setup](GitHub-App-Setup) - [Image Build](Image-Build) ## Providers -- [Docker-DinD](Docker-DinD-Provider) +- [Docker Container](Docker-Container-Provider) +- [Docker Sandboxes](Docker-Sandboxes-Provider) - [Tart](Tart-Provider) - [WSL](WSL-Provider) - [Adding A Provider](Adding-A-Provider) @@ -138,11 +213,30 @@ cat > "$out_dir/_Sidebar.md" <<'SIDEBAR' ## Operations - [Operations](Operations) +- [Logging](Logging) +- [Observability Examples](Observability-Examples) +- [Storage](Storage) +- [Troubleshooting](Troubleshooting) - [Security](Security) -- [Design](Design) -- [Background](Background) + +## Advanced + +- [Docker Sandboxes Template](Docker-Sandboxes-Template) +- [Cross-Architecture Containers](Cross-Architecture-Containers) - [Docker Registry Mirrors](Docker-Registry-Mirrors) +- [No-Go Installation](No-Go-Install) +- [Windows Startup](Windows-Startup) - [macOS Startup](macOS-Startup) + +## Development + +- [Development Guide](Development) +- [Design](Design) +- [Development Principles](Development-Principles) +- [Core Runner Verification](Core-Runner-Verification) +- [Releases](Releases) +- [Contributing](Contributing) +- [Support](Support) SIDEBAR source_ref="${GITHUB_SHA:-}" diff --git a/scripts/test/docker-sandboxes-plan-smoke.ps1 b/scripts/test/docker-sandboxes-plan-smoke.ps1 new file mode 100644 index 0000000..f890a6f --- /dev/null +++ b/scripts/test/docker-sandboxes-plan-smoke.ps1 @@ -0,0 +1,29 @@ +[CmdletBinding()] +param() + +$ErrorActionPreference = 'Stop' +$repositoryRoot = Split-Path -Parent (Split-Path -Parent $PSScriptRoot) +$wrappers = @( + (Join-Path $repositoryRoot 'scripts\docker-sandboxes\build-template.ps1'), + (Join-Path $repositoryRoot 'scripts\docker-sandboxes\load-template.ps1') +) + +foreach ($wrapper in $wrappers) { + $tokens = $null + $parseErrors = $null + [void][System.Management.Automation.Language.Parser]::ParseFile($wrapper, [ref] $tokens, [ref] $parseErrors) + if (@($parseErrors).Count -ne 0) { + throw "$wrapper has PowerShell parse errors: $(@($parseErrors).Message -join '; ')" + } + $source = Get-Content -Raw -LiteralPath $wrapper + foreach ($required in @("'start.ps1'", "@('image', 'build'", "'--dry-run'")) { + if (-not $source.Contains($required)) { + throw "$wrapper does not delegate to the common image build path: missing $required" + } + } + if ($source -match '(?i)\bsbx\b|\bdocker\s+(?:build|image|template)\b') { + throw "$wrapper contains provider build/import operations instead of delegating" + } +} + +Write-Host 'Docker Sandboxes compatibility wrappers passed delegation and syntax checks.' diff --git a/scripts/test/host-trust-docker-e2e.sh b/scripts/test/host-trust-docker-e2e.sh index 604b51d..422c4fb 100644 --- a/scripts/test/host-trust-docker-e2e.sh +++ b/scripts/test/host-trust-docker-e2e.sh @@ -6,7 +6,7 @@ set -euo pipefail # MSYS leaves untouched; bind arguments still receive normal drive mapping. export MSYS2_ARG_CONV_EXCL='/CN=' -# Disposable Linux-container fixture for the Docker-DinD host-trust contract. +# Disposable Linux-container fixture for the Docker Container host-trust contract. # It creates test-only CAs under a temporary directory and never reads or # modifies the host's real trust store. diff --git a/scripts/test/host-trust-wrapper-smoke.ps1 b/scripts/test/host-trust-wrapper-smoke.ps1 index 71d6672..fc2cdbd 100644 --- a/scripts/test/host-trust-wrapper-smoke.ps1 +++ b/scripts/test/host-trust-wrapper-smoke.ps1 @@ -70,6 +70,8 @@ image: $bridge = Start-EparHostTrustBridge -ProjectRoot $ProjectRoot -Command pool -Arguments @('pool', 'up', '--config', $config) if (-not $bridge.WatchProcess -or $bridge.WatchProcess.HasExited) { throw 'Windows host-trust watcher did not start' } + $publishedFeed = Join-Path $bridge.RunnerFeedDir 'current.json' + $firstPublishedAt = (Get-Content -LiteralPath $publishedFeed -Raw | ConvertFrom-Json).generatedAt $liveLock = $bridge.FeedDir + '.lock' $deadline = [DateTime]::UtcNow.AddSeconds(5) while (-not (Test-Path -LiteralPath $liveLock -PathType Container) -and [DateTime]::UtcNow -lt $deadline) { @@ -79,6 +81,13 @@ image: $lockRejected = $false try { & $helper sync -ProjectRoot $ProjectRoot -Config $config *> $null } catch { $lockRejected = $true } if (-not $lockRejected) { throw 'second controller unexpectedly acquired the live Windows wrapper lock' } + $refreshDeadline = [DateTime]::UtcNow.AddSeconds(15) + $refreshedPublishedAt = $firstPublishedAt + while ($refreshedPublishedAt -eq $firstPublishedAt -and [DateTime]::UtcNow -lt $refreshDeadline) { + Start-Sleep -Milliseconds 250 + $refreshedPublishedAt = (Get-Content -LiteralPath $publishedFeed -Raw | ConvertFrom-Json).generatedAt + } + if ($refreshedPublishedAt -eq $firstPublishedAt) { throw 'Windows host-trust watcher did not refresh its published feed' } Stop-EparHostTrustBridge -Bridge $bridge $bridge = $null if (Test-Path -LiteralPath $liveLock) { throw 'Windows wrapper shutdown left its singleton lock behind' } @@ -109,6 +118,16 @@ image: throw 'Windows wrapper quoted mode/block-scope parsing failed' } + $disabledConfig = Join-Path $temporary 'disabled.yml' + [System.IO.File]::WriteAllText($disabledConfig, "image:`n hostTrustMode: disabled`n hostTrustScopes: [system, user]`n", [System.Text.UTF8Encoding]::new($false)) + $disabledRunnerFeed = [string](& $helper sync -ProjectRoot $ProjectRoot -Config $disabledConfig -Purpose runner) + if ($LASTEXITCODE -ne 0 -or $disabledRunnerFeed) { throw 'disabled runner trust unexpectedly published a feed' } + $disabledBuildCurrent = [string](& $helper sync -ProjectRoot $ProjectRoot -Config $disabledConfig -Purpose build) + $disabledBuildFeed = Get-Content -LiteralPath $disabledBuildCurrent -Raw | ConvertFrom-Json + if ($LASTEXITCODE -ne 0 -or @($disabledBuildFeed.scopes).Count -ne 1 -or $disabledBuildFeed.scopes[0] -ne 'system') { + throw 'disabled runner trust did not retain automatic system-only build trust' + } + Write-Output 'Windows host-trust wrapper lifecycle smoke passed' } finally { diff --git a/scripts/test/host-trust-wrapper-smoke.sh b/scripts/test/host-trust-wrapper-smoke.sh index ef5ecea..a5fcf92 100644 --- a/scripts/test/host-trust-wrapper-smoke.sh +++ b/scripts/test/host-trust-wrapper-smoke.sh @@ -204,13 +204,15 @@ EPAR_HOST_TRUST_HELPER="$helper" missing_config="$temporary/missing-project/.local/config.yml" missing_output="$("$helper" sync --project-root "$project_root" --config "$missing_config")" [[ -z "$missing_output" ]] || { echo "missing config unexpectedly produced a host-trust feed" >&2; exit 1; } +missing_build_output="$("$helper" sync --project-root "$project_root" --config "$missing_config" --purpose build)" +[[ -s "$missing_build_output" ]] || { echo "missing config did not produce automatic system build trust" >&2; exit 1; } native_project="$temporary/native-go-project" fake_go="$temporary/fake-go" fake_go_log="$temporary/fake-go.log" mkdir -p "$native_project/scripts/host-trust" cp "$project_root/start" "$native_project/start" -cp "$project_root/scripts/host-trust/wrapper-lib.sh" "$project_root/scripts/host-trust/host-trust-feed.sh" "$native_project/scripts/host-trust/" +cp "$project_root/scripts/host-trust/wrapper-lib.sh" "$project_root/scripts/host-trust/host-trust-feed.sh" "$project_root/scripts/host-trust/macos-trust-settings.js" "$native_project/scripts/host-trust/" cat >"$fake_go" <<'SH' #!/usr/bin/env bash set -euo pipefail @@ -219,10 +221,12 @@ if [[ "${1:-}" == version ]]; then exit 0 fi printf '%s\n' "$*" >>"$FAKE_GO_LOG" +printf 'trust build=<%s> runner=<%s> os=<%s> deferred=<%s>\n' "${EPAR_BUILD_TRUST_FEED:-}" "${EPAR_HOST_TRUST_FEED:-}" "${EPAR_CONTROLLER_HOST_OS:-}" "${EPAR_HOST_TRUST_INIT_DEFERRED:-}" >>"$FAKE_GO_LOG" SH chmod +x "$fake_go" -(cd "$native_project" && EPAR_GO_BIN="$fake_go" FAKE_GO_LOG="$fake_go_log" ./start) +(cd "$native_project" && EPAR_GO_BIN="$fake_go" FAKE_GO_LOG="$fake_go_log" EPAR_BUILD_TRUST_FEED=stale-build EPAR_HOST_TRUST_FEED=stale-runner EPAR_CONTROLLER_HOST_OS=darwin EPAR_HOST_TRUST_INIT_DEFERRED=1 ./start) grep -Fxq 'run ./cmd/ephemeral-action-runner start' "$fake_go_log" +grep -Fxq 'trust build=<> runner=<> os=<> deferred=<>' "$fake_go_log" nested_root="$temporary/nested-project" mkdir -p "$nested_root/.local" diff --git a/scripts/test/native-controller-cache-retention.sh b/scripts/test/native-controller-cache-retention.sh new file mode 100644 index 0000000..7358b6c --- /dev/null +++ b/scripts/test/native-controller-cache-retention.sh @@ -0,0 +1,264 @@ +#!/usr/bin/env bash +set -euo pipefail + +source_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd -P)" +builder="${source_root}/scripts/build-native-controller.sh" +grep -q -- '--provenance=false' "$builder" || { echo 'native controller toolchain build must disable nondeterministic default provenance' >&2; exit 1; } +eval "$(sed -n '/^epar_tls_failure_host()/,/^case "$(uname -s)\/$(uname -m)"/p' "$builder" | sed '$d')" + +native_cache_keep_previous=2 +native_cache_max_bytes=$((1024 * 1024)) +native_cache_grace_seconds=0 +abandoned_build_grace_seconds=0 +retention_inventory_file="" + +temporary="$(mktemp -d)" +cleanup() { + rm -rf -- "$temporary" +} +trap cleanup EXIT INT TERM + +tls_transcript="${temporary}/tls-error.log" +printf '%s\n' 'module: Get "https://proxy.golang.org/example/@v/v1.0.0.zip": tls: failed to verify certificate: x509: certificate signed by unknown authority' >"$tls_transcript" +[[ "$(epar_tls_failure_host "$tls_transcript")" == 'proxy.golang.org' ]] || { echo 'native-controller TLS failure host was not extracted' >&2; exit 1; } +printf '%s\n' 'ordinary build failure' >"$tls_transcript" +[[ -z "$(epar_tls_failure_host "$tls_transcript")" ]] || { echo 'ordinary native-controller failure was misclassified as TLS' >&2; exit 1; } + +repeat_character() { + printf '%64s' '' | tr ' ' "$1" +} + +write_manifested_revision() { + local root="$1" + local key="$2" + local directory="${root}/${key}" + mkdir -p "$directory" + printf '%s' "$key" >"${directory}/ephemeral-action-runner" + printf '%s\n' \ + 'schemaVersion=1' \ + "cacheKey=${key}" \ + 'executable=ephemeral-action-runner' >"${directory}/controller-cache.manifest" +} + +cache_root="${temporary}/bin" +mkdir -p "$cache_root" +keys=() +for character in 0 1 2 3 4 5 6; do + keys+=("$(repeat_character "$character")") +done +for index in 0 1 2 3 4; do + write_manifested_revision "$cache_root" "${keys[$index]}" + touch -t "20200101010${index}" "${cache_root}/${keys[$index]}" +done + +printf '%s\n' \ + 'schemaVersion=1' \ + "host=$(hostname 2>/dev/null || true)" \ + "pid=$$" \ + 'startedAtUnix=1' >"${cache_root}/${keys[4]}/lease-active" + +mkdir -p "${cache_root}/${keys[5]}" +printf '%s' "${keys[5]}" >"${cache_root}/${keys[5]}/ephemeral-action-runner" +mkdir -p "${cache_root}/${keys[6]}" +printf '%s' "${keys[6]}" >"${cache_root}/${keys[6]}/ephemeral-action-runner" +printf '%s\n' \ + 'schemaVersion=1' \ + "cacheKey=${keys[5]}" \ + 'executable=ephemeral-action-runner' >"${cache_root}/${keys[6]}/controller-cache.manifest" + +stale_build="${cache_root}/.build-stale" +mkdir -p "$stale_build" +printf '%s\n' \ + 'schemaVersion=1' \ + "host=$(hostname 2>/dev/null || true)" \ + 'pid=2147483646' \ + 'startedAtUnix=1' >"${stale_build}/lease-build-2147483646.ABC123" +touch -t 202001010100 "$stale_build" + +active_build="${cache_root}/.build-active" +mkdir -p "$active_build" +printf '%s\n' \ + 'schemaVersion=1' \ + "host=$(hostname 2>/dev/null || true)" \ + "pid=$$" \ + 'startedAtUnix=1' >"${active_build}/lease-build-$$.DEF456" +touch -t 202001010100 "$active_build" + +unmarked_build="${cache_root}/.build-unmarked" +mkdir -p "$unmarked_build" +touch -t 202001010100 "$unmarked_build" + +malformed_build="${cache_root}/.build-malformed" +mkdir -p "$malformed_build" +printf '%s\n' \ + "host=$(hostname 2>/dev/null || true)" \ + 'pid=2147483646' \ + 'startedAtUnix=1' >"${malformed_build}/lease-build-2147483646.GHI789" +touch -t 202001010100 "$malformed_build" + +foreign_build="${cache_root}/.build-foreign" +mkdir -p "$foreign_build" +printf '%s\n' \ + 'schemaVersion=1' \ + "host=foreign-$(hostname 2>/dev/null || true)" \ + 'pid=2147483646' \ + "startedAtUnix=$(date +%s)" >"${foreign_build}/lease-build-2147483646.JKL012" +touch -t 202001010100 "$foreign_build" + +symlink_key="$(repeat_character 8)" +symlink_created=0 +if ln -s "${cache_root}/${keys[0]}" "${cache_root}/${symlink_key}" 2>/dev/null && [[ -L "${cache_root}/${symlink_key}" ]]; then + symlink_created=1 +fi + +epar_prune_native_controller_cache "$cache_root" "${keys[0]}" + +expected=("${keys[0]}" "${keys[2]}" "${keys[3]}" "${keys[4]}" "${keys[5]}" "${keys[6]}") +for key in "${expected[@]}"; do + [[ -d "${cache_root}/${key}" ]] || { echo "retention removed protected revision ${key}" >&2; exit 1; } +done +[[ ! -e "${cache_root}/${keys[1]}" ]] || { echo "retention kept the expired excess revision" >&2; exit 1; } +[[ ! -e "$stale_build" ]] || { echo "retention kept an inactive leased build" >&2; exit 1; } +[[ -d "$active_build" ]] || { echo "retention removed an active build" >&2; exit 1; } +[[ -d "$unmarked_build" ]] || { echo "retention removed a markerless build" >&2; exit 1; } +[[ -d "$malformed_build" ]] || { echo "retention treated a malformed build lease as ownership evidence" >&2; exit 1; } +[[ -d "$foreign_build" ]] || { echo "retention removed a recent foreign-host build lease" >&2; exit 1; } +if ((symlink_created == 1)); then + [[ -L "${cache_root}/${symlink_key}" ]] || { echo "retention followed or removed a cache symlink" >&2; exit 1; } +fi + +policy_root="${temporary}/policy" +mkdir -p "$policy_root" +policy_current="$(repeat_character a)" +policy_expired="$(repeat_character b)" +policy_grace="$(repeat_character c)" +write_manifested_revision "$policy_root" "$policy_current" +write_manifested_revision "$policy_root" "$policy_expired" +write_manifested_revision "$policy_root" "$policy_grace" +printf '%1024s' '' >"${policy_root}/${policy_expired}/ephemeral-action-runner" +printf '%1024s' '' >"${policy_root}/${policy_grace}/ephemeral-action-runner" +touch -t 202001010100 "${policy_root}/${policy_expired}" +native_cache_keep_previous=5 +native_cache_max_bytes=512 +native_cache_grace_seconds=$((7 * 24 * 60 * 60)) +epar_prune_native_controller_cache "$policy_root" "$policy_current" +[[ ! -e "${policy_root}/${policy_expired}" ]] || { echo "retention kept an expired revision beyond the byte budget" >&2; exit 1; } +[[ -d "${policy_root}/${policy_grace}" ]] || { echo "retention removed a grace-protected revision beyond the byte budget" >&2; exit 1; } + +builder_source="$(cat "$builder")" +for required in 'golang:latest' 'ephemeral-action-runner.manifest' 'schemaVersion=2' 'lease-native-' 'epar_write_bootstrap_acquisition_journal' 'epar_resolve_go_toolchain_image' 'previousDevImageID' 'previous_dev_image_id' 'epar-native-controller-build.log' 'epar_report_tls_failure' 'TLS verification was not disabled' 'epar_prepare_bootstrap_build_trust' '--network none' 'GO111MODULE=off' 'GOTOOLCHAIN=local' 'SSL_CERT_FILE=/run/epar-bootstrap-ca.pem' 'scripts/bootstrap-trust' ':/run/epar-bootstrap-ca.pem:ro'; do + [[ "$builder_source" == *"$required"* ]] || { echo "stable native-controller wrapper contract is missing: ${required}" >&2; exit 1; } +done + +native_smoke_root="${temporary}/native-runtime-smoke" +native_smoke_project="${native_smoke_root}/project" +native_smoke_bin="${native_smoke_root}/bin" +mkdir -p "${native_smoke_project}/scripts/host-trust" "${native_smoke_project}/scripts/docker" "${native_smoke_project}/scripts/bootstrap-trust" "${native_smoke_project}/cmd" "${native_smoke_project}/internal" "$native_smoke_bin" +cp "$builder" "${native_smoke_project}/scripts/build-native-controller.sh" +: >"${native_smoke_project}/scripts/docker/dev.Dockerfile" +: >"${native_smoke_project}/go.mod" +: >"${native_smoke_project}/go.sum" +cat >"${native_smoke_project}/scripts/host-trust/wrapper-lib.sh" <<'SH' +#!/usr/bin/env bash +EPAR_HOST_TRUST_POST_INIT_CONFIG="" +EPAR_BUILD_TRUST_FEED_DIR="" +EPAR_RUNNER_TRUST_FEED_DIR="" +epar_host_trust_config_path() { printf '%s/.local/config.yml\n' "$1"; } +epar_host_trust_prepare() { EPAR_HOST_TRUST_POST_INIT_CONFIG="$(epar_host_trust_config_path "$1")"; } +epar_host_trust_post_init() { [[ -n "$EPAR_HOST_TRUST_POST_INIT_CONFIG" ]] && "$EPAR_HOST_TRUST_HELPER" sync --project-root "$1" --config "$EPAR_HOST_TRUST_POST_INIT_CONFIG" >/dev/null; } +epar_host_trust_cleanup() { :; } +epar_host_trust_host_os() { printf '%s\n' linux; } +SH +cat >"${native_smoke_project}/scripts/host-trust/host-trust-feed.sh" <<'SH' +#!/usr/bin/env bash +set -euo pipefail +case "${1:-}" in + sync) + if [[ " $* " == *" --purpose build "* ]]; then + printf 'bootstrap\n' >>"$FAKE_HELPER_LOG" + printf '{}\n' >"$FAKE_BOOTSTRAP_FEED" + printf '%s\n' "$FAKE_BOOTSTRAP_FEED" + else + printf 'post-init\n' >>"$FAKE_HELPER_LOG" + fi + ;; + watch) + trap 'exit 0' INT TERM + while :; do sleep 1; done + ;; + *) + echo "unexpected host-trust helper command: $*" >&2 + exit 1 + ;; +esac +SH +cat >"${native_smoke_bin}/docker" <<'SH' +#!/usr/bin/env bash +set -euo pipefail +printf 'CALL' >>"$FAKE_DOCKER_LOG" +printf ' <%s>' "$@" >>"$FAKE_DOCKER_LOG" +printf '\n' >>"$FAKE_DOCKER_LOG" +case "${1:-}" in + image) + printf 'sha256:%064d\n' 0 + ;; + pull|build) + ;; + run) + output_directory="" + for argument in "$@"; do + case "$argument" in + *:/out) output_directory="${argument%:/out}" ;; + esac + done + if [[ " $* " == *' /bootstrap/main.go '* ]]; then + printf 'bootstrap trust\n' + printf 'fake bootstrap trust\n' >"${output_directory}/ca.pem" + elif [[ " $* " == *' go build '* ]]; then + cat >"${output_directory}/ephemeral-action-runner" <<'NATIVE' +#!/usr/bin/env bash +set -euo pipefail +printf 'runtime build=<%s> runner=<%s> os=<%s> deferred=<%s> args=<%s>\n' "${EPAR_BUILD_TRUST_FEED:-}" "${EPAR_HOST_TRUST_FEED:-}" "${EPAR_CONTROLLER_HOST_OS:-}" "${EPAR_HOST_TRUST_INIT_DEFERRED:-}" "$*" >>"${FAKE_NATIVE_LOG:?}" +if [[ "${1:-}" == init ]]; then + mkdir -p .local + printf '%s\n' 'image:' ' hostTrustMode: overlay' ' hostTrustScopes: [system, user]' >.local/config.yml +fi +NATIVE + chmod +x "${output_directory}/ephemeral-action-runner" + fi + ;; + *) + echo "unexpected fake Docker command: $*" >&2 + exit 1 + ;; +esac +SH +chmod +x "${native_smoke_project}/scripts/build-native-controller.sh" "${native_smoke_project}/scripts/host-trust/host-trust-feed.sh" "${native_smoke_bin}/docker" + +native_smoke_env=( + "PATH=${native_smoke_bin}:$PATH" + "EPAR_GOMOD_VOLUME=native-smoke-gomod" + "EPAR_GOCACHE_VOLUME=native-smoke-gocache" + 'EPAR_BOOTSTRAP_MIN_FREE_BYTES=1' + "FAKE_HELPER_LOG=${native_smoke_root}/helper.log" + "FAKE_BOOTSTRAP_FEED=${native_smoke_root}/bootstrap-feed.json" + "FAKE_DOCKER_LOG=${native_smoke_root}/docker.log" + "FAKE_NATIVE_LOG=${native_smoke_root}/native.log" + 'EPAR_BUILD_TRUST_FEED=stale-build' + 'EPAR_HOST_TRUST_FEED=stale-runner' + 'EPAR_CONTROLLER_HOST_OS=darwin' + 'EPAR_HOST_TRUST_INIT_DEFERRED=1' +) +(cd "$native_smoke_project" && env "${native_smoke_env[@]}" scripts/build-native-controller.sh start) +grep -Fxq 'runtime build=<> runner=<> os=<> deferred=<> args=' "${native_smoke_root}/native.log" +grep -Fxq 'bootstrap' "${native_smoke_root}/helper.log" +[[ "$(wc -l <"${native_smoke_root}/helper.log" | tr -d ' ')" == 1 ]] || { echo 'ordinary cached-native start unexpectedly used a runtime trust bridge' >&2; exit 1; } +grep -Fq ':/feed/current.json:ro>' "${native_smoke_root}/docker.log" +grep -Fq ' ' "${native_smoke_root}/docker.log" + +: >"${native_smoke_root}/helper.log" +(cd "$native_smoke_project" && env "${native_smoke_env[@]}" scripts/build-native-controller.sh init) +grep -Fxq 'runtime build=<> runner=<> os=<> deferred=<> args=' "${native_smoke_root}/native.log" +grep -Fxq 'post-init' "${native_smoke_root}/helper.log" + +echo "Unix native-controller cache retention contract passed" diff --git a/scripts/test/no-go-first-run-smoke.ps1 b/scripts/test/no-go-first-run-smoke.ps1 deleted file mode 100644 index a7c4be2..0000000 --- a/scripts/test/no-go-first-run-smoke.ps1 +++ /dev/null @@ -1,77 +0,0 @@ -[CmdletBinding()] -param( - [string] $ProjectRoot = (Split-Path -Parent (Split-Path -Parent $PSScriptRoot)) -) - -$ErrorActionPreference = 'Stop' -$temporary = Join-Path ([System.IO.Path]::GetTempPath()) ('epar-no-go-first-run-' + [guid]::NewGuid().ToString('N')) -$oldPath = $env:PATH -$oldLocalAppData = $env:LOCALAPPDATA -$oldFakeProject = $env:FAKE_PROJECT -$oldFakeDockerLog = $env:FAKE_DOCKER_LOG -$oldFakeInitFail = $env:FAKE_INIT_FAIL -try { - $hostPowerShell = (Get-Process -Id $PID).Path - $project = Join-Path $temporary 'project' - $hostTrust = Join-Path $project 'scripts\host-trust' - $dockerScripts = Join-Path $project 'scripts\docker' - $fakeBin = Join-Path $temporary 'bin' - New-Item -ItemType Directory -Force -Path $hostTrust, $dockerScripts, $fakeBin | Out-Null - Copy-Item (Join-Path $ProjectRoot 'scripts\run-with-docker.ps1') (Join-Path $project 'scripts\run-with-docker.ps1') - Copy-Item (Join-Path $ProjectRoot 'scripts\host-trust\wrapper-lib.ps1') (Join-Path $hostTrust 'wrapper-lib.ps1') - Copy-Item (Join-Path $ProjectRoot 'scripts\host-trust\host-trust-feed.ps1') (Join-Path $hostTrust 'host-trust-feed.ps1') - [System.IO.File]::WriteAllText((Join-Path $dockerScripts 'dev.Dockerfile'), '', [System.Text.UTF8Encoding]::new($false)) - - $fakeDocker = @' -$ErrorActionPreference = 'Stop' -$line = 'CALL' + (($args | ForEach-Object { ' <' + $_ + '>' }) -join '') -Add-Content -LiteralPath $env:FAKE_DOCKER_LOG -Value $line -Encoding utf8 -if ($args.Count -gt 0 -and $args[0] -eq 'build') { exit 0 } -if ($args -contains 'init') { - if ($env:FAKE_INIT_FAIL -eq '1') { exit 23 } - $local = Join-Path $env:FAKE_PROJECT '.local' - New-Item -ItemType Directory -Force -Path $local | Out-Null - $config = "provider:`n type: docker-dind`nrunner:`n ephemeral: true`nimage:`n hostTrustMode: overlay`n hostTrustScopes: [system, user]`n" - [System.IO.File]::WriteAllText((Join-Path $local 'config.yml'), $config, [System.Text.UTF8Encoding]::new($false)) -} -exit 0 -'@ - [System.IO.File]::WriteAllText((Join-Path $fakeBin 'fake-docker.ps1'), $fakeDocker, [System.Text.UTF8Encoding]::new($false)) - $cmd = "@echo off`r`npwsh.exe -NoLogo -NoProfile -File `"%~dp0fake-docker.ps1`" %*`r`nexit /b %ERRORLEVEL%`r`n" - [System.IO.File]::WriteAllText((Join-Path $fakeBin 'docker.cmd'), $cmd, [System.Text.ASCIIEncoding]::new()) - - $env:PATH = $fakeBin + [System.IO.Path]::PathSeparator + $oldPath - $env:LOCALAPPDATA = Join-Path $temporary 'cache' - $env:FAKE_PROJECT = $project - $env:FAKE_DOCKER_LOG = Join-Path $temporary 'docker.log' - Remove-Item Env:FAKE_INIT_FAIL -ErrorAction SilentlyContinue - - & $hostPowerShell -NoLogo -NoProfile -ExecutionPolicy Bypass -File (Join-Path $project 'scripts\run-with-docker.ps1') start - if ($LASTEXITCODE -ne 0) { throw "first-run wrapper exited $LASTEXITCODE" } - $calls = @(Get-Content -LiteralPath $env:FAKE_DOCKER_LOG | Where-Object { $_ -like '* *' }) - if ($calls.Count -ne 2) { throw "expected two controller runs, got $($calls.Count)" } - if ($calls[0] -notlike '* *' -or $calls[0] -notlike '* *' -or $calls[0] -notlike '* *') { - throw "implicit init did not receive the real Windows host/deferred-init contract: $($calls[0])" - } - if ($calls[1] -notlike '* *' -or $calls[1] -notlike '*:/run/epar-host-trust:ro>*' -or $calls[1] -notlike '* *') { - throw "second start did not receive the read-only host-trust feed: $($calls[1])" - } - - Remove-Item -LiteralPath (Join-Path $project '.local') -Recurse -Force - Remove-Item -LiteralPath $env:LOCALAPPDATA -Recurse -Force -ErrorAction SilentlyContinue - [System.IO.File]::WriteAllText($env:FAKE_DOCKER_LOG, '', [System.Text.UTF8Encoding]::new($false)) - $env:FAKE_INIT_FAIL = '1' - & $hostPowerShell -NoLogo -NoProfile -ExecutionPolicy Bypass -File (Join-Path $project 'scripts\run-with-docker.ps1') start - if ($LASTEXITCODE -ne 23) { throw "failing implicit init exited $LASTEXITCODE, want 23" } - - Write-Output 'Windows no-Go first-run start lifecycle smoke passed' - exit 0 -} -finally { - $env:PATH = $oldPath - if ($null -eq $oldLocalAppData) { Remove-Item Env:LOCALAPPDATA -ErrorAction SilentlyContinue } else { $env:LOCALAPPDATA = $oldLocalAppData } - if ($null -eq $oldFakeProject) { Remove-Item Env:FAKE_PROJECT -ErrorAction SilentlyContinue } else { $env:FAKE_PROJECT = $oldFakeProject } - if ($null -eq $oldFakeDockerLog) { Remove-Item Env:FAKE_DOCKER_LOG -ErrorAction SilentlyContinue } else { $env:FAKE_DOCKER_LOG = $oldFakeDockerLog } - if ($null -eq $oldFakeInitFail) { Remove-Item Env:FAKE_INIT_FAIL -ErrorAction SilentlyContinue } else { $env:FAKE_INIT_FAIL = $oldFakeInitFail } - Remove-Item -LiteralPath $temporary -Recurse -Force -ErrorAction SilentlyContinue -} diff --git a/scripts/test/no-go-first-run-smoke.sh b/scripts/test/no-go-first-run-smoke.sh index 95a1901..4e820db 100644 --- a/scripts/test/no-go-first-run-smoke.sh +++ b/scripts/test/no-go-first-run-smoke.sh @@ -24,7 +24,7 @@ if [[ " $* " == *" go run ./cmd/ephemeral-action-runner init "* ]]; then mkdir -p "$FAKE_PROJECT/.local" cat >"$FAKE_PROJECT/.local/config.yml" <<'YAML' provider: - type: docker-dind + type: docker-container runner: ephemeral: true image: @@ -63,6 +63,7 @@ if [[ "$host_os" == Linux ]]; then fi export FAKE_PROJECT="$project" export FAKE_DOCKER_LOG="$temporary/docker.log" +export EPAR_LEGACY_CONTROLLER_IN_DOCKER=1 (cd "$project" && scripts/run-with-docker.sh start) [[ "$(grep -c ' ' "$FAKE_DOCKER_LOG")" == 2 ]] diff --git a/scripts/test/start-command-forwarding.sh b/scripts/test/start-command-forwarding.sh new file mode 100644 index 0000000..e620d0b --- /dev/null +++ b/scripts/test/start-command-forwarding.sh @@ -0,0 +1,73 @@ +#!/usr/bin/env bash +set -euo pipefail + +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd -P)" +test_root="$(mktemp -d)" +trap 'rm -rf "$test_root"' EXIT + +test_uname="$(uname -s)" +case "$test_uname" in + MINGW*|MSYS*|CYGWIN*) test_uname=Linux ;; +esac +cat >"$test_root/uname" <<'SCRIPT' +#!/usr/bin/env bash +printf '%s\n' "$EPAR_TEST_UNAME" +SCRIPT +cat >"$test_root/go" <<'SCRIPT' +#!/usr/bin/env bash +if [[ "${1:-}" == "version" ]]; then + exit 0 +fi +printf '%s\n' "$@" >"$EPAR_START_FORWARD_LOG" +SCRIPT +chmod +x "$test_root/uname" "$test_root/go" + +export PATH="$test_root:$PATH" +export EPAR_TEST_UNAME="$test_uname" +export EPAR_GO_BIN=go +export EPAR_USE_DOCKER_RUN=0 +export EPAR_START_FORWARD_LOG="$test_root/arguments" +export EPAR_CONFIG="$test_root/config.yml" +printf 'image:\n hostTrustMode: disabled\n' >"$EPAR_CONFIG" + +"$repo_root/start" +actual="$(tr '\n' ' ' <"$EPAR_START_FORWARD_LOG")" +expected="run ./cmd/ephemeral-action-runner start " +if [[ "$actual" != "$expected" ]]; then + echo "default start forwarding mismatch: got '$actual', want '$expected'" >&2 + exit 1 +fi + +"$repo_root/start" --config .local/custom-config.yml --instances 2 +actual="$(tr '\n' ' ' <"$EPAR_START_FORWARD_LOG")" +expected="run ./cmd/ephemeral-action-runner start --config .local/custom-config.yml --instances 2 " +if [[ "$actual" != "$expected" ]]; then + echo "start flag forwarding mismatch: got '$actual', want '$expected'" >&2 + exit 1 +fi + +"$repo_root/start" storage prune --provider docker-sandboxes +actual="$(tr '\n' ' ' <"$EPAR_START_FORWARD_LOG")" +expected="run ./cmd/ephemeral-action-runner storage prune --provider docker-sandboxes " +if [[ "$actual" != "$expected" ]]; then + echo "explicit command forwarding mismatch: got '$actual', want '$expected'" >&2 + exit 1 +fi + +"$repo_root/start" image update --config .local/custom-config.yml +actual="$(tr '\n' ' ' <"$EPAR_START_FORWARD_LOG")" +expected="run ./cmd/ephemeral-action-runner image update --config .local/custom-config.yml " +if [[ "$actual" != "$expected" ]]; then + echo "image update forwarding mismatch: got '$actual', want '$expected'" >&2 + exit 1 +fi + +"$repo_root/start" version +actual="$(tr '\n' ' ' <"$EPAR_START_FORWARD_LOG")" +expected="run ./cmd/ephemeral-action-runner version " +if [[ "$actual" != "$expected" ]]; then + echo "version forwarding mismatch: got '$actual', want '$expected'" >&2 + exit 1 +fi + +printf 'start command-forwarding smoke passed\n' diff --git a/scripts/test/windows-native-controller-contract.ps1 b/scripts/test/windows-native-controller-contract.ps1 new file mode 100644 index 0000000..e83224b --- /dev/null +++ b/scripts/test/windows-native-controller-contract.ps1 @@ -0,0 +1,407 @@ +[CmdletBinding()] +param( + [string] $ProjectRoot +) + +$ErrorActionPreference = 'Stop' +if ([string]::IsNullOrWhiteSpace($ProjectRoot)) { + $scriptDirectory = Split-Path -Parent $MyInvocation.MyCommand.Path + $ProjectRoot = Split-Path -Parent (Split-Path -Parent $scriptDirectory) +} +$ProjectRoot = [System.IO.Path]::GetFullPath($ProjectRoot) +$wrapperPath = Join-Path $ProjectRoot 'scripts\run-with-docker.ps1' +$builderPath = Join-Path $ProjectRoot 'scripts\build-native-controller.ps1' +$startPath = Join-Path $ProjectRoot 'start.ps1' +$wrapperSource = Get-Content -Raw -LiteralPath $wrapperPath +$builderSource = Get-Content -Raw -LiteralPath $builderPath +$startSource = Get-Content -Raw -LiteralPath $startPath + +$tokens = $null +$parseErrors = $null +$wrapperAst = [System.Management.Automation.Language.Parser]::ParseFile($wrapperPath, [ref]$tokens, [ref]$parseErrors) +if (@($parseErrors).Count -ne 0) { + throw "run-with-docker.ps1 failed to parse: $(@($parseErrors).Message -join '; ')" +} +$startTokens = $null +$startParseErrors = $null +[System.Management.Automation.Language.Parser]::ParseFile($startPath, [ref]$startTokens, [ref]$startParseErrors) | Out-Null +if (@($startParseErrors).Count -ne 0) { + throw "start.ps1 failed to parse: $(@($startParseErrors).Message -join '; ')" +} + +$classifier = $wrapperAst.Find({ + param($node) + $node -is [System.Management.Automation.Language.FunctionDefinitionAst] -and $node.Name -eq 'Test-EparBenignDockerDesktopPrefaceDiagnostic' +}, $true) +if ($null -eq $classifier) { + throw 'Docker Desktop transcript classifier is missing' +} +Invoke-Expression $classifier.Extent.Text +$benign = '2026/07/22 02:36:46 http2: server: error reading preface from client //./pipe/dockerDesktopLinuxEngine: file has already been closed' +if (-not (Test-EparBenignDockerDesktopPrefaceDiagnostic -Transcript $benign)) { + throw 'known successful Docker Desktop preface diagnostic was not classified as benign' +} +if (Test-EparBenignDockerDesktopPrefaceDiagnostic -Transcript ($benign + [Environment]::NewLine + 'real build failure')) { + throw 'mixed Docker stderr was incorrectly discarded as benign' +} + +$builderTokens = $null +$builderParseErrors = $null +$builderAst = [System.Management.Automation.Language.Parser]::ParseFile($builderPath, [ref]$builderTokens, [ref]$builderParseErrors) +if (@($builderParseErrors).Count -ne 0) { + throw "build-native-controller.ps1 failed to parse: $(@($builderParseErrors).Message -join '; ')" +} +if ($builderSource -notmatch 'docker build --quiet --provenance=false') { + throw 'native controller toolchain build must disable nondeterministic default provenance' +} +foreach ($functionName in @('Get-EparDirectoryBytes', 'Test-EparNativeControllerLeaseActive', 'Test-EparNativeControllerBuildLeaseValid', 'Invoke-EparNativeControllerCacheRetention', 'Get-EparGoCacheVolumeIdentity', 'Invoke-EparGoCacheLimit', 'Read-EparStableNativeControllerManifest', 'Get-EparDockerImageID', 'Get-EparTLSFailureHost', 'Get-EparCertificateSHA256', 'Find-EparWindowsIssuerRoots', 'Invoke-EparTLSFailureDiagnostic', 'Initialize-EparBootstrapBuildTrust')) { + $function = $builderAst.Find({ + param($node) + $node -is [System.Management.Automation.Language.FunctionDefinitionAst] -and $node.Name -eq $functionName + }, $true) + if ($null -eq $function) { + throw "native controller cache retention function is missing: $functionName" + } + Invoke-Expression $function.Extent.Text +} +$previousErrorActionPreferenceForImageProbe = $ErrorActionPreference +try { + function docker { + Write-Error 'No such image: contract-missing' + $global:LASTEXITCODE = 1 + } + if (Get-EparDockerImageID -Reference 'contract-missing') { + throw 'missing Docker image probe returned an identity' + } + if ($ErrorActionPreference -ne $previousErrorActionPreferenceForImageProbe) { + throw 'missing Docker image probe did not restore ErrorActionPreference' + } +} finally { + Remove-Item Function:\docker -ErrorAction SilentlyContinue +} +$tlsFailureTranscript = 'module: Get "https://proxy.golang.org/example/@v/v1.0.0.zip": tls: failed to verify certificate: x509: certificate signed by unknown authority' +if ((Get-EparTLSFailureHost -Transcript $tlsFailureTranscript) -cne 'proxy.golang.org') { + throw 'native-controller TLS failure host was not extracted' +} +if (Get-EparTLSFailureHost -Transcript 'ordinary build failure without a certificate error') { + throw 'ordinary native-controller failure was misclassified as a TLS certificate failure' +} +$tlsDiagnosticRoot = Join-Path ([System.IO.Path]::GetTempPath()) ('epar-native-tls-diagnostic-' + [guid]::NewGuid().ToString('N')) +New-Item -ItemType Directory -Path $tlsDiagnosticRoot | Out-Null +$previousDevImage = $DevImage +try { + $tlsDiagnosticLog = Join-Path $tlsDiagnosticRoot 'build.log' + [System.IO.File]::WriteAllText($tlsDiagnosticLog, $tlsFailureTranscript) + $DevImage = 'contract-dev-image' + function docker { + Write-Output 'depth=0 CN=proxy.golang.org' + Write-Output 'verify error:num=20:unable to get local issuer certificate' + Write-Output 'subject=CN=proxy.golang.org' + Write-Output 'issuer=CN=EPAR contract root' + Write-Output 'sha256 Fingerprint=AA:BB' + Write-Output 'notBefore=Jul 29 00:00:00 2026 GMT' + Write-Output 'notAfter=Jul 29 00:00:00 2027 GMT' + $global:LASTEXITCODE = 0 + } + Invoke-EparTLSFailureDiagnostic -Transcript $tlsFailureTranscript -LogPath $tlsDiagnosticLog + $tlsDiagnosticContent = Get-Content -Raw -LiteralPath $tlsDiagnosticLog + foreach ($required in @('EPAR TLS certificate diagnostic', 'Requested host: proxy.golang.org:443', 'Issuer: CN=EPAR contract root', 'SHA-256: AABB', 'TLS verification was not disabled')) { + if (-not $tlsDiagnosticContent.Contains($required)) { + throw "native-controller TLS diagnostic log is missing: $required" + } + } +} finally { + $DevImage = $previousDevImage + Remove-Item Function:\docker -ErrorAction SilentlyContinue + Remove-Item -LiteralPath $tlsDiagnosticRoot -Recurse -Force -ErrorAction SilentlyContinue +} +$stableManifestRoot = Join-Path ([System.IO.Path]::GetTempPath()) ('epar-native-stable-manifest-' + [guid]::NewGuid().ToString('N')) +New-Item -ItemType Directory -Path $stableManifestRoot | Out-Null +try { + $stableManifestPath = Join-Path $stableManifestRoot 'ephemeral-action-runner.manifest' + $stableFingerprint = [string]::new([char]'a', 64) + [System.IO.File]::WriteAllLines($stableManifestPath, @('schemaVersion=2', "fingerprint=$stableFingerprint", 'executable=ephemeral-action-runner.exe', ('toolchainImageID=sha256:' + [string]::new([char]'b', 64)), 'sourceRevision=sha256:test', 'completedAtUtc=2026-07-29T00:00:00Z')) + $stableManifest = Read-EparStableNativeControllerManifest -Path $stableManifestPath + if ($null -eq $stableManifest -or $stableManifest.fingerprint -ne $stableFingerprint) { throw 'stable native-controller manifest was not parsed' } + Add-Content -LiteralPath $stableManifestPath -Value 'fingerprint=duplicate' + if ($null -ne (Read-EparStableNativeControllerManifest -Path $stableManifestPath)) { throw 'stable native-controller manifest accepted duplicate keys' } +} finally { + Remove-Item -LiteralPath $stableManifestRoot -Recurse -Force -ErrorAction SilentlyContinue +} +$GomodVolume = 'contract-gomod' +$GocacheVolume = 'contract-gocache' +$ProjectID = 'contract' +$GoCacheLimitBytes = [uint64](10GB) +$DevImage = 'contract-dev-image' +$dockerCalls = [System.Collections.Generic.List[string]]::new() +function docker { + $dockerCalls.Add(($args -join ' ')) + if ($args -contains 'du') { + Write-Output "1`t/go/pkg/mod" + Write-Output "1`t/root/.cache/go-build" + } + $global:LASTEXITCODE = 0 +} +try { + Invoke-EparGoCacheLimit + if (@($dockerCalls | Where-Object { $_ -like 'run *' }).Count -ne 1) { + throw "empty Docker queries should run one exact Go cache GC probe; calls=$($dockerCalls -join '; ')" + } +} finally { + Remove-Item Function:\docker -ErrorAction SilentlyContinue +} +$retentionRoot = Join-Path ([System.IO.Path]::GetTempPath()) ('epar-native-cache-retention-' + [guid]::NewGuid().ToString('N')) +New-Item -ItemType Directory -Path $retentionRoot | Out-Null +try { + $keys = @('0', '1', '2', '3', '4', '5', '6') | ForEach-Object { [string]::new([char]$_, 64) } + foreach ($key in $keys[0..4]) { + $directory = Join-Path $retentionRoot $key + New-Item -ItemType Directory -Path $directory | Out-Null + [System.IO.File]::WriteAllText((Join-Path $directory 'ephemeral-action-runner.exe'), $key) + [System.IO.File]::WriteAllLines((Join-Path $directory 'controller-cache.manifest'), @( + 'schemaVersion=1', + "cacheKey=$key", + 'executable=ephemeral-action-runner.exe' + )) + } + $unmanifestedDirectory = Join-Path $retentionRoot $keys[5] + New-Item -ItemType Directory -Path $unmanifestedDirectory | Out-Null + [System.IO.File]::WriteAllText((Join-Path $unmanifestedDirectory 'ephemeral-action-runner.exe'), $keys[5]) + $mismatchedDirectory = Join-Path $retentionRoot $keys[6] + New-Item -ItemType Directory -Path $mismatchedDirectory | Out-Null + [System.IO.File]::WriteAllText((Join-Path $mismatchedDirectory 'ephemeral-action-runner.exe'), $keys[6]) + [System.IO.File]::WriteAllLines((Join-Path $mismatchedDirectory 'controller-cache.manifest'), @( + 'schemaVersion=1', + "cacheKey=$($keys[5])", + 'executable=ephemeral-action-runner.exe' + )) + $activeKey = $keys[4] + $activeDirectory = Join-Path $retentionRoot $activeKey + $process = Get-Process -Id $PID + [System.IO.File]::WriteAllLines((Join-Path $activeDirectory "lease-$PID-contract.txt"), @( + 'schemaVersion=1', + "host=$([Environment]::MachineName)", + "pid=$PID", + "processStartUtc=$($process.StartTime.ToUniversalTime().ToString('o'))", + "startedAtUtc=$([DateTime]::UtcNow.AddDays(-10).ToString('o'))" + )) + $oldest = [DateTime]::UtcNow.AddDays(-10) + for ($index = 0; $index -lt 5; $index++) { + (Get-Item -LiteralPath (Join-Path $retentionRoot $keys[$index])).LastWriteTimeUtc = $oldest.AddMinutes($index) + } + $staleBuild = Join-Path $retentionRoot '.build-stale' + New-Item -ItemType Directory -Path $staleBuild | Out-Null + [System.IO.File]::WriteAllLines((Join-Path $staleBuild 'lease-build-2147483646-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa.txt'), @( + 'schemaVersion=1', + "host=$([Environment]::MachineName)", + 'pid=2147483646', + "processStartUtc=$($oldest.ToString('o'))", + "startedAtUtc=$($oldest.ToString('o'))" + )) + (Get-Item -LiteralPath $staleBuild).LastWriteTimeUtc = $oldest + $activeBuild = Join-Path $retentionRoot '.build-active' + New-Item -ItemType Directory -Path $activeBuild | Out-Null + [System.IO.File]::WriteAllLines((Join-Path $activeBuild ("lease-build-{0}-bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb.txt" -f $PID)), @( + 'schemaVersion=1', + "host=$([Environment]::MachineName)", + "pid=$PID", + "processStartUtc=$($process.StartTime.ToUniversalTime().ToString('o'))", + "startedAtUtc=$($oldest.ToString('o'))" + )) + (Get-Item -LiteralPath $activeBuild).LastWriteTimeUtc = $oldest + $unmarkedBuild = Join-Path $retentionRoot '.build-unmarked' + New-Item -ItemType Directory -Path $unmarkedBuild | Out-Null + (Get-Item -LiteralPath $unmarkedBuild).LastWriteTimeUtc = $oldest + $malformedBuild = Join-Path $retentionRoot '.build-malformed' + New-Item -ItemType Directory -Path $malformedBuild | Out-Null + [System.IO.File]::WriteAllLines((Join-Path $malformedBuild 'lease-build-2147483646-cccccccccccccccccccccccccccccccc.txt'), @( + "host=$([Environment]::MachineName)", + 'pid=2147483646', + "processStartUtc=$($oldest.ToString('o'))", + "startedAtUtc=$($oldest.ToString('o'))" + )) + (Get-Item -LiteralPath $malformedBuild).LastWriteTimeUtc = $oldest + $foreignBuild = Join-Path $retentionRoot '.build-foreign' + New-Item -ItemType Directory -Path $foreignBuild | Out-Null + [System.IO.File]::WriteAllLines((Join-Path $foreignBuild 'lease-build-2147483646-dddddddddddddddddddddddddddddddd.txt'), @( + 'schemaVersion=1', + "host=foreign-$([Environment]::MachineName)", + 'pid=2147483646', + "processStartUtc=$($oldest.ToString('o'))", + "startedAtUtc=$([DateTime]::UtcNow.ToString('o'))" + )) + (Get-Item -LiteralPath $foreignBuild).LastWriteTimeUtc = $oldest + + Invoke-EparNativeControllerCacheRetention -CacheRoot $retentionRoot -CurrentCacheKey $keys[0] -KeepPrevious 2 -MaxBytes 1MB -GracePeriod ([TimeSpan]::Zero) -AbandonedBuildGracePeriod ([TimeSpan]::Zero) + + $remaining = @(Get-ChildItem -LiteralPath $retentionRoot -Directory | Where-Object { $_.Name -match '^[0-9a-f]{64}$' } | Select-Object -ExpandProperty Name | Sort-Object) + $expected = @($keys[0], $keys[2], $keys[3], $keys[4], $keys[5], $keys[6] | Sort-Object) + if (Compare-Object -ReferenceObject $expected -DifferenceObject $remaining) { + throw "native controller retention kept $($remaining -join ', '), want $($expected -join ', ')" + } + if (Test-Path -LiteralPath $staleBuild) { + throw 'native controller retention did not remove an abandoned build directory' + } + if (-not (Test-Path -LiteralPath $activeBuild)) { + throw 'native controller retention removed an active build directory' + } + if (-not (Test-Path -LiteralPath $unmarkedBuild)) { + throw 'native controller retention removed a markerless build directory without positive ownership evidence' + } + if (-not (Test-Path -LiteralPath $malformedBuild)) { + throw 'native controller retention treated a malformed build lease as ownership evidence' + } + if (-not (Test-Path -LiteralPath $foreignBuild)) { + throw 'native controller retention removed a recent foreign-host build lease' + } + + $policyRoot = Join-Path $retentionRoot 'policy-contract' + New-Item -ItemType Directory -Path $policyRoot | Out-Null + $policyKeys = @('a', 'b', 'c') | ForEach-Object { [string]::new([char]$_, 64) } + foreach ($key in $policyKeys) { + $directory = Join-Path $policyRoot $key + New-Item -ItemType Directory -Path $directory | Out-Null + [System.IO.File]::WriteAllText((Join-Path $directory 'ephemeral-action-runner.exe'), $(if ($key -eq $policyKeys[0]) { 'x' } else { [string]::new('x', 1024) })) + [System.IO.File]::WriteAllLines((Join-Path $directory 'controller-cache.manifest'), @( + 'schemaVersion=1', + "cacheKey=$key", + 'executable=ephemeral-action-runner.exe' + )) + } + (Get-Item -LiteralPath (Join-Path $policyRoot $policyKeys[1])).LastWriteTimeUtc = $oldest + Invoke-EparNativeControllerCacheRetention -CacheRoot $policyRoot -CurrentCacheKey $policyKeys[0] -KeepPrevious 5 -MaxBytes 512 -GracePeriod ([TimeSpan]::FromDays(7)) -AbandonedBuildGracePeriod ([TimeSpan]::Zero) + if (Test-Path -LiteralPath (Join-Path $policyRoot $policyKeys[1])) { + throw 'native controller retention kept an expired revision beyond the byte budget' + } + if (-not (Test-Path -LiteralPath (Join-Path $policyRoot $policyKeys[2]))) { + throw 'native controller retention removed a grace-protected revision beyond the byte budget' + } + $migrationRoot = Join-Path $retentionRoot 'migration-contract' + $migrationKey = [string]::new([char]'f', 64) + $migrationDirectory = Join-Path $migrationRoot $migrationKey + New-Item -ItemType Directory -Force -Path $migrationDirectory | Out-Null + [System.IO.File]::WriteAllText((Join-Path $migrationDirectory 'ephemeral-action-runner.exe'), 'legacy') + [System.IO.File]::WriteAllLines((Join-Path $migrationDirectory 'controller-cache.manifest'), @('schemaVersion=1', "cacheKey=$migrationKey", 'executable=ephemeral-action-runner.exe')) + Invoke-EparNativeControllerCacheRetention -CacheRoot $migrationRoot -CurrentCacheKey $migrationKey -KeepPrevious 0 -MaxBytes 1 -GracePeriod ([TimeSpan]::Zero) -AbandonedBuildGracePeriod ([TimeSpan]::Zero) -RemoveCurrent + if (Test-Path -LiteralPath $migrationDirectory) { throw 'native controller legacy migration left an inactive exact revision behind' } +} finally { + Remove-Item -LiteralPath $retentionRoot -Recurse -Force -ErrorAction SilentlyContinue +} +$retryClassifier = $builderAst.Find({ + param($node) + $node -is [System.Management.Automation.Language.FunctionDefinitionAst] -and $node.Name -eq 'Test-EparRetryableDockerContextMetadataDiagnostic' +}, $true) +if ($null -eq $retryClassifier) { + throw 'Docker context metadata retry classifier is missing' +} +Invoke-Expression $retryClassifier.Extent.Text +$buildInvoker = $builderAst.Find({ + param($node) + $node -is [System.Management.Automation.Language.FunctionDefinitionAst] -and $node.Name -eq 'Invoke-EparDockerBuild' +}, $true) +if ($null -eq $buildInvoker) { + throw 'Docker build retry function is missing' +} +Invoke-Expression $buildInvoker.Extent.Text +$retryable = 'ERROR: failed to build: failed to read metadata: open C:\Users\runner\.docker\contexts\meta\fe9c6bd7a66301f49ca9b6a70b217107cd1284598bfc254700c989b916da791e\meta.json: The process cannot access the file because it is being used by another process.' +if (-not (Test-EparRetryableDockerContextMetadataDiagnostic -Transcript $retryable)) { + throw 'known transient Docker context metadata sharing violation was not classified as retryable' +} +$wrappedRetryable = "docker.exe : $retryable`r`nAt line:1 char:1`r`n+ docker build`r`n+ ~~~~~~~~~~~~`r`n + CategoryInfo : NotSpecified: (ERROR: failed to build:String) [], RemoteException`r`n + FullyQualifiedErrorId : NativeCommandError" +if (-not (Test-EparRetryableDockerContextMetadataDiagnostic -Transcript $wrappedRetryable)) { + throw 'Windows PowerShell NativeCommandError-wrapped Docker context metadata sharing violation was not classified as retryable' +} +$nativeCaptureDirectory = Join-Path ([System.IO.Path]::GetTempPath()) ('epar-docker-context-retry-' + [guid]::NewGuid().ToString('N')) +New-Item -ItemType Directory -Path $nativeCaptureDirectory | Out-Null +try { + $fakeDocker = Join-Path $nativeCaptureDirectory 'docker.exe' + Copy-Item -LiteralPath $env:ComSpec -Destination $fakeDocker + $emitScript = Join-Path $nativeCaptureDirectory 'emit.cmd' + Set-Content -LiteralPath $emitScript -Encoding ASCII -Value @( + '@echo off' + "echo $retryable 1>&2" + 'exit /b 1' + ) + $capturedStderrPath = Join-Path $nativeCaptureDirectory 'stderr.txt' + $previousErrorActionPreference = $ErrorActionPreference + try { + $ErrorActionPreference = 'Continue' + & $fakeDocker /d /c $emitScript 2> $capturedStderrPath + $fakeDockerExitCode = $LASTEXITCODE + } finally { + $ErrorActionPreference = $previousErrorActionPreference + } + if ($fakeDockerExitCode -ne 1) { + throw "native stderr test shim exited $fakeDockerExitCode, want 1" + } + $capturedStderr = Get-Content -Raw -LiteralPath $capturedStderrPath + if (-not (Test-EparRetryableDockerContextMetadataDiagnostic -Transcript $capturedStderr)) { + throw "Windows PowerShell native stderr capture was not classified as retryable: $capturedStderr" + } +} finally { + Remove-Item -LiteralPath $nativeCaptureDirectory -Recurse -Force -ErrorAction SilentlyContinue +} +$retryHarnessDirectory = Join-Path ([System.IO.Path]::GetTempPath()) ('epar-docker-context-retry-loop-' + [guid]::NewGuid().ToString('N')) +New-Item -ItemType Directory -Path $retryHarnessDirectory | Out-Null +$previousPath = $env:PATH +try { + Set-Content -LiteralPath (Join-Path $retryHarnessDirectory 'retry-count.txt') -Encoding ASCII -Value '0' + Set-Content -LiteralPath (Join-Path $retryHarnessDirectory 'docker.cmd') -Encoding ASCII -Value @( + '@echo off' + 'set /p retry_count=<"%~dp0retry-count.txt"' + 'set /a retry_count=retry_count+1' + '> "%~dp0retry-count.txt" echo %retry_count%' + 'if %retry_count% LSS 3 (' + " echo $retryable 1>&2" + ' exit /b 1' + ')' + 'exit /b 0' + ) + $env:PATH = $retryHarnessDirectory + [System.IO.Path]::PathSeparator + $previousPath + $GoImage = 'contract-test-go-image' + $DevImage = 'contract-test-dev-image' + $RepoRoot = $ProjectRoot + $retryResult = Invoke-EparDockerBuild + $retryCount = [int] ((Get-Content -Raw -LiteralPath (Join-Path $retryHarnessDirectory 'retry-count.txt')).Trim()) + if ($retryResult -ne 0 -or $retryCount -ne 3) { + throw "Docker context retry loop result=$retryResult attempts=$retryCount, want success after 3 attempts" + } +} finally { + $env:PATH = $previousPath + Remove-Item -LiteralPath $retryHarnessDirectory -Recurse -Force -ErrorAction SilentlyContinue +} +if (Test-EparRetryableDockerContextMetadataDiagnostic -Transcript ($retryable + [Environment]::NewLine + 'real build failure')) { + throw 'mixed Docker context metadata stderr was incorrectly classified as retryable' +} +if (Test-EparRetryableDockerContextMetadataDiagnostic -Transcript ($wrappedRetryable + [Environment]::NewLine + 'real build failure')) { + throw 'mixed Windows PowerShell-wrapped Docker context metadata stderr was incorrectly classified as retryable' +} +if (Test-EparRetryableDockerContextMetadataDiagnostic -Transcript ($retryable -replace '\\contexts\\meta\\', '\buildx\')) { + throw 'unrelated Docker metadata sharing violation was incorrectly classified as retryable' +} + +$nativeBranch = $wrapperSource.IndexOf("if (`$env:EPAR_LEGACY_CONTROLLER_IN_DOCKER -ne '1')", [System.StringComparison]::Ordinal) +$legacyImage = $wrapperSource.IndexOf('$Image =', [System.StringComparison]::Ordinal) +if ($nativeBranch -lt 0 -or $legacyImage -lt 0 -or $nativeBranch -gt $legacyImage) { + throw 'native-controller dispatch must occur before the explicit legacy controller-in-Docker path' +} +foreach ($required in @("Join-Path `$PSScriptRoot 'build-native-controller.ps1'", 'exit $nativeExitCode')) { + if (-not $wrapperSource.Contains($required)) { + throw "native no-Go wrapper contract is missing: $required" + } +} +foreach ($required in @('$env:EPAR_INVOCATION = "start"', 'if ($ControllerArgs.Count -eq 0 -or $ControllerArgs[0].StartsWith("-"))', '$ControllerArgs = @("start") + $ControllerArgs', 'scripts\run-with-docker.ps1") @ControllerArgs', 'run ./cmd/ephemeral-action-runner @ControllerArgs')) { + if (-not $startSource.Contains($required)) { + throw "start command-forwarding contract is missing: $required" + } +} +if ($startSource.Contains('@StartArgs')) { + throw 'start command-forwarding contract still forces the start command' +} +foreach ($required in @('CGO_ENABLED=0', 'GOOS=windows', 'GOARCH=amd64', '.local\bin', 'dirty:sha256:', 'EPAR_NATIVE_CONTROLLER', 'Get-EparNativeSourceHash', 'activeControllerFingerprint', 'existingManifest.toolchainImageID', 'Invoke-EparNativeControllerCacheRetention', 'Test-EparNativeControllerBuildLeaseValid', 'controller-cache.manifest', 'ephemeral-action-runner.manifest', 'schemaVersion=2', 'lease-native-', 'golang:latest', 'Write-EparBootstrapAcquisitionJournal', 'Resolve-EparGoToolchainImage', 'previousDevImageID', '$previousDevImageID = Get-EparDockerImageID', '[System.IO.FileAttributes]::ReparsePoint', '$maximumAttempts = 5', 'Start-Sleep -Milliseconds', 'epar-native-controller-build.log', 'Invoke-EparTLSFailureDiagnostic', 'TLS verification was not disabled', 'Initialize-EparBootstrapBuildTrust', '--network none', 'GO111MODULE=off', 'GOTOOLCHAIN=local', 'SSL_CERT_FILE=/run/epar-bootstrap-ca.pem', 'scripts\bootstrap-trust', ':/run/epar-bootstrap-ca.pem:ro')) { + if (-not $builderSource.Contains($required)) { + throw "native controller build contract is missing: $required" + } +} + +Write-Output 'Windows no-Go native-controller source and transcript smoke passed' diff --git a/start b/start index 2e8a9eb..652cbca 100755 --- a/start +++ b/start @@ -1,14 +1,15 @@ #!/usr/bin/env bash set -euo pipefail -# One-command start: ./start or ./start --config .local/config.yml --instances 2 +# EPAR entry point: ./start, ./start --config .local/config.yml, or +# ./start storage status. # -# Uses local Go if present. Otherwise runs EPAR from source inside a -# containerized Go toolchain via `go run` (no local Go install needed, and -# no binary is built or left on disk). The containerized fallback requires Docker. -# See docs/advanced/no-go-install.md. +# Uses local Go if present. Otherwise a containerized Go toolchain builds a +# CGO-disabled native host controller cached under .local/bin, then runs it on +# the host. The no-Go fallback requires Docker. See docs/advanced/no-go-install.md. script_dir="$(cd -- "$(dirname -- "${BASH_SOURCE[0]:-$0}")" && pwd -P)" +export EPAR_INVOCATION=start case "$(uname -s)" in MINGW*|MSYS*|CYGWIN*) ps_script="$script_dir/start.ps1" @@ -24,6 +25,13 @@ EPAR_HOST_TRUST_HELPER="${script_dir}/scripts/host-trust/host-trust-feed.sh" GO_BIN="${EPAR_GO_BIN:-go}" USE_DOCKER_RUN="${EPAR_USE_DOCKER_RUN:-auto}" +epar_args=("$@") +if ((${#epar_args[@]} == 0)); then + epar_args=(start) +elif [[ "${epar_args[0]}" == -* ]]; then + epar_args=(start "${epar_args[@]}") +fi +controller_command="${epar_args[0]}" # A present-but-broken `go` (wrong architecture, ancient pre-modules install # left over on PATH, etc.) must not look "usable" just because it exists. @@ -32,8 +40,8 @@ go_usable() { } if [[ "${USE_DOCKER_RUN}" == "1" ]] || { [[ "${USE_DOCKER_RUN}" == "auto" ]] && ! go_usable "${GO_BIN}"; }; then - echo "Go not found or not runnable (or EPAR_USE_DOCKER_RUN=1); running with a containerized Go toolchain instead..." >&2 - exec "${script_dir}/scripts/run-with-docker.sh" start "$@" + echo "Go not found or not runnable (or EPAR_USE_DOCKER_RUN=1); building and running a cached native controller with the Docker toolchain..." >&2 + exec "${script_dir}/scripts/run-with-docker.sh" "${epar_args[@]}" fi if ! go_usable "${GO_BIN}"; then @@ -43,14 +51,25 @@ if ! go_usable "${GO_BIN}"; then exit 1 fi -epar_host_trust_prepare "${script_dir}" start "$@" +# The local controller reads the host trust stores itself. Feed variables are +# exclusively for the legacy controller-in-Docker bridge, so do not let a +# caller's bridge state bypass native first-run preflight or stale its build. +unset EPAR_BUILD_TRUST_FEED EPAR_HOST_TRUST_FEED EPAR_CONTROLLER_HOST_OS EPAR_HOST_TRUST_INIT_DEFERRED + +# The native controller resolves operational and runner trust directly. Keep +# the established explicit-init post-write verification, but never prepare a +# pre-wizard feed for the embedded first-run start continuation. +if [[ "$controller_command" != "init" ]]; then + exec "${GO_BIN}" run ./cmd/ephemeral-action-runner "${epar_args[@]}" +fi + +epar_host_trust_prepare "${script_dir}" "$controller_command" "${epar_args[@]}" trap epar_host_trust_cleanup EXIT INT TERM -if [[ -n "${EPAR_HOST_TRUST_FEED_DIR}" ]]; then - export EPAR_CONTROLLER_HOST_OS="$(epar_host_trust_host_os)" - export EPAR_HOST_TRUST_FEED="${EPAR_HOST_TRUST_FEED_DIR}/current.json" +status=0 +"${GO_BIN}" run ./cmd/ephemeral-action-runner "${epar_args[@]}" || status=$? +if [[ "$status" == "0" ]]; then + epar_host_trust_post_init "${script_dir}" || status=$? fi -"${GO_BIN}" run ./cmd/ephemeral-action-runner start "$@" -status=$? epar_host_trust_cleanup trap - EXIT INT TERM exit "$status" diff --git a/start.ps1 b/start.ps1 index 5dbd9f0..c37241b 100644 --- a/start.ps1 +++ b/start.ps1 @@ -3,24 +3,32 @@ param( [string[]] $EparArgs ) -# One-command start: .\start.ps1 or .\start.ps1 --config .local\config.yml --instances 2 +# EPAR entry point: .\start.ps1, .\start.ps1 --config .local\config.yml, or +# .\start.ps1 storage status. # -# Uses local Go if present and actually runnable. Otherwise runs EPAR from -# source inside a containerized Go toolchain via `go run` (no local Go -# install needed, and no binary is built or left on disk). The containerized -# fallback requires Docker. See docs/advanced/no-go-install.md. +# Uses local Go if present and actually runnable. Otherwise a containerized Go +# toolchain builds a CGO-disabled native host controller cached under +# .local/bin, then runs it on the host. The no-Go fallback requires Docker. +# See docs/advanced/no-go-install.md. $ErrorActionPreference = "Stop" $Root = Split-Path -Parent $MyInvocation.MyCommand.Path Set-Location -LiteralPath $Root . (Join-Path $Root "scripts\host-trust\wrapper-lib.ps1") +$OriginalInvocationExists = Test-Path Env:EPAR_INVOCATION +$OriginalInvocation = $env:EPAR_INVOCATION +$env:EPAR_INVOCATION = "start" $GoBin = if ($env:EPAR_GO_BIN) { $env:EPAR_GO_BIN } else { "go" } $UseDockerRun = if ($env:EPAR_USE_DOCKER_RUN) { $env:EPAR_USE_DOCKER_RUN } else { "auto" } -$StartArgs = @("start") -if ($null -ne $EparArgs -and $EparArgs.Count -gt 0) { - $StartArgs += $EparArgs +[string[]] $ControllerArgs = @() +if ($null -ne $EparArgs) { + $ControllerArgs = [string[]] @($EparArgs) } +if ($ControllerArgs.Count -eq 0 -or $ControllerArgs[0].StartsWith("-")) { + $ControllerArgs = @("start") + $ControllerArgs +} +$ControllerCommand = [string] $ControllerArgs[0] function Test-GoUsable { param([string]$GoBin) @@ -36,29 +44,31 @@ function Test-GoUsable { $goUsable = Test-GoUsable -GoBin $GoBin if ($UseDockerRun -eq "1" -or ($UseDockerRun -eq "auto" -and -not $goUsable)) { - Write-Warning "Go not found or not runnable (or EPAR_USE_DOCKER_RUN=1); running with a containerized Go toolchain instead..." - & (Join-Path $Root "scripts\run-with-docker.ps1") @StartArgs - exit $LASTEXITCODE + Write-Warning "Go not found or not runnable (or EPAR_USE_DOCKER_RUN=1); building and running a cached native controller with the Docker toolchain..." + try { + & (Join-Path $Root "scripts\run-with-docker.ps1") @ControllerArgs + $exitCode = $LASTEXITCODE + } finally { + if ($OriginalInvocationExists) { $env:EPAR_INVOCATION = $OriginalInvocation } else { Remove-Item Env:EPAR_INVOCATION -ErrorAction SilentlyContinue } + } + exit $exitCode } if (-not $goUsable) { + if ($OriginalInvocationExists) { $env:EPAR_INVOCATION = $OriginalInvocation } else { Remove-Item Env:EPAR_INVOCATION -ErrorAction SilentlyContinue } Write-Error "Go not found or not runnable: $GoBin`nInstall Go, set EPAR_GO_BIN, or set EPAR_USE_DOCKER_RUN=1 to run with a containerized Go toolchain instead.`nSee docs/advanced/no-go-install.md." exit 1 } -$bridge = Start-EparHostTrustBridge -ProjectRoot $Root -Command "start" -Arguments $EparArgs -$previousHostOS = $env:EPAR_CONTROLLER_HOST_OS -$previousFeed = $env:EPAR_HOST_TRUST_FEED +$bridge = if ($ControllerCommand -eq "init") { Start-EparHostTrustBridge -ProjectRoot $Root -Command $ControllerCommand -Arguments $ControllerArgs } else { $null } try { - if ($bridge.FeedDir) { - $env:EPAR_CONTROLLER_HOST_OS = Get-EparHostTrustHostOS - $env:EPAR_HOST_TRUST_FEED = Join-Path $bridge.FeedDir "current.json" - } - & $GoBin run ./cmd/ephemeral-action-runner @StartArgs + & $GoBin run ./cmd/ephemeral-action-runner @ControllerArgs $exitCode = $LASTEXITCODE + if ($exitCode -eq 0 -and $ControllerCommand -eq "init") { + Complete-EparHostTrustInit -ProjectRoot $Root -Bridge $bridge + } } finally { Stop-EparHostTrustBridge -Bridge $bridge - if ($null -eq $previousHostOS) { Remove-Item Env:EPAR_CONTROLLER_HOST_OS -ErrorAction SilentlyContinue } else { $env:EPAR_CONTROLLER_HOST_OS = $previousHostOS } - if ($null -eq $previousFeed) { Remove-Item Env:EPAR_HOST_TRUST_FEED -ErrorAction SilentlyContinue } else { $env:EPAR_HOST_TRUST_FEED = $previousFeed } + if ($OriginalInvocationExists) { $env:EPAR_INVOCATION = $OriginalInvocation } else { Remove-Item Env:EPAR_INVOCATION -ErrorAction SilentlyContinue } } exit $exitCode diff --git a/templates/docker-sandboxes/.dockerignore b/templates/docker-sandboxes/.dockerignore new file mode 100644 index 0000000..b0c8854 --- /dev/null +++ b/templates/docker-sandboxes/.dockerignore @@ -0,0 +1,21 @@ +* +!Dockerfile +!helpers.sha256 +!guest/ +!guest/*.sh +!guest/docker-daemon.json +!hook-launcher/ +!hook-launcher/*.go +!custom-install/ +!custom-install/run.sh +!inputs/ +!inputs/actions-runner.tar.gz +!inputs/tini +!host-trust-certificates/ +!host-trust-certificates/*.crt +!trusted-ca-certificates/ +!trusted-ca-certificates/*.crt +!host-trust-metadata/ +!host-trust-metadata/host-trust-generation.json +!profiles/ +!profiles/*.compatibility.json diff --git a/templates/docker-sandboxes/Dockerfile b/templates/docker-sandboxes/Dockerfile new file mode 100644 index 0000000..76d1799 --- /dev/null +++ b/templates/docker-sandboxes/Dockerfile @@ -0,0 +1,115 @@ +# syntax=docker/dockerfile:1.7.1@sha256:a57df69d0ea827fb7266491f2813635de6f17269be881f696fbfdf2d83dda33e + +ARG TEMPLATE_PLATFORM=linux/amd64 +ARG SOURCE_IMAGE=ghcr.io/catthehacker/ubuntu@sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac +ARG GO_BUILDER_IMAGE=docker.io/library/golang@sha256:9006890ecba0a168034d99516084099ae3114d9f2b7d6572c77f2dde57ebc980 + +FROM --platform=$BUILDPLATFORM ${GO_BUILDER_IMAGE} AS hook-builder +ARG TARGETOS +ARG TARGETARCH +ARG HOOK_LAUNCHER_SHA256=7fe07f10f484fa6888481a4165e81570187c0aeff422738d3ea5add6b95dd9b7 +WORKDIR /src +COPY hook-launcher/main.go /src/main.go +RUN echo "${HOOK_LAUNCHER_SHA256} /src/main.go" | sha256sum --check - \ + && install -d -m 0755 /out \ + && CGO_ENABLED=0 GOOS="${TARGETOS}" GOARCH="${TARGETARCH}" GOTOOLCHAIN=local go build -trimpath -buildvcs=false -ldflags='-s -w -buildid=' -o /out/epar-hook-bash /src/main.go + +FROM --platform=${TEMPLATE_PLATFORM} ${SOURCE_IMAGE} AS runner-template + +ARG BUILDKIT_SBOM_SCAN_STAGE=true +ARG TEMPLATE_PLATFORM +ARG SOURCE_IMAGE +ARG TARGETPLATFORM +ARG TARGETOS +ARG TARGETARCH +ARG SOURCE_PROFILE=act-22.04 +ARG SOURCE_INDEX_DIGEST=sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac +ARG SOURCE_MANIFEST_DIGEST=sha256:f3d493b10df1582ce631e0213bd90aa5f8196287c8a9f8ef546ecb44ca256655 +ARG SOURCE_REVISION=e2f8efe464c82732f78e967ee709c00b6af53643 +ARG TEMPLATE_VERSION=20260723-r4-amd64 +ARG COMPATIBILITY_FILE=act-22.04.amd64.compatibility.json +ARG ACTIONS_RUNNER_VERSION +ARG ACTIONS_RUNNER_SHA256 +ARG TINI_SHA256=sha256:93dcc18adc78c65a028a84799ecf8ad40c936fdfc5f2a57b1acda5a8117fa82c + +USER root +SHELL ["/bin/bash", "-euo", "pipefail", "-c"] + +COPY guest/ /opt/epar/ +COPY custom-install/ /opt/epar/custom-install/ +COPY host-trust-certificates/ /usr/local/share/ca-certificates/epar-host/ +COPY trusted-ca-certificates/ /usr/local/share/ca-certificates/epar/ +COPY host-trust-metadata/ /opt/epar/ +COPY helpers.sha256 /opt/epar/helpers.sha256 +COPY profiles/${COMPATIBILITY_FILE} /opt/epar/template-compatibility.json +COPY --from=hook-builder --chmod=0555 /out/epar-hook-bash /opt/epar/hook-bin/bash +COPY --chmod=0755 inputs/tini /usr/local/bin/tini +COPY inputs/actions-runner.tar.gz /tmp/actions-runner.tar.gz + +RUN [[ "${TARGETPLATFORM}" == "${TEMPLATE_PLATFORM}" ]] \ + && [[ "${TARGETOS}" == "linux" ]] \ + && [[ "${TARGETARCH}" == "amd64" || "${TARGETARCH}" == "arm64" ]] \ + && echo "${TINI_SHA256#sha256:} /usr/local/bin/tini" | sha256sum --check - \ + && echo "${ACTIONS_RUNNER_SHA256#sha256:} /tmp/actions-runner.tar.gz" | sha256sum --check - \ + && cd /opt/epar \ + && sha256sum --check helpers.sha256 \ + && chmod 0555 ./*.sh \ + && /opt/epar/prepare-template.sh \ + && /opt/epar/install-trusted-ca-certificates.sh \ + && rm -rf /opt/actions-runner \ + && install -d -m 0755 -o agent -g agent /opt/actions-runner /opt/actions-runner/_work /opt/actions-runner/_work/_tool /var/log/actions-runner \ + && tar -xzf /tmp/actions-runner.tar.gz -C /opt/actions-runner \ + && rm -f /tmp/actions-runner.tar.gz \ + && printf '%s\n' "${ACTIONS_RUNNER_VERSION}" > /opt/epar/actions-runner-version \ + && chown -R agent:agent /opt/actions-runner /var/log/actions-runner \ + && chmod 0555 /opt/epar/custom-install/run.sh \ + && /opt/epar/custom-install/run.sh \ + && sudo -u agent -H /opt/actions-runner/bin/Runner.Listener --version | grep -Fx "${ACTIONS_RUNNER_VERSION}" \ + && /usr/local/bin/tini --version 2>&1 | grep -F 'version 0.19.0' + +ENV HOME=/home/agent \ + USER=agent \ + LOGNAME=agent \ + SSH_AUTH_SOCK= \ + SSH_AUTH_SOCK_GATEWAY= \ + SSH_AGENT_PID= \ + XDG_CONFIG_HOME=/home/agent/.config \ + XDG_CACHE_HOME=/home/agent/.cache \ + XDG_DATA_HOME=/home/agent/.local/share \ + XDG_STATE_HOME=/home/agent/.local/state \ + XDG_RUNTIME_DIR=/run/user/1000 \ + DOCKER_CONFIG=/home/agent/.docker \ + RUNNER_TOOL_CACHE=/opt/actions-runner/_work/_tool \ + AGENT_TOOLSDIRECTORY=/opt/actions-runner/_work/_tool \ + DOTNET_INSTALL_DIR=/opt/actions-runner/_work/_tool/dotnet \ + EPAR_ACTIONS_RUNNER_DIR=/opt/actions-runner \ + EPAR_RUNNER_WORK_DIR=/opt/actions-runner \ + EPAR_TEMPLATE_PLATFORM=${TEMPLATE_PLATFORM} \ + DEBIAN_FRONTEND=dialog + +LABEL com.docker.sandboxes.start-docker=true \ + io.solutionforest.epar.template.schema-version="1" \ + io.solutionforest.epar.template.runner-execution="direct-actions-listener" \ + io.solutionforest.epar.template.docker-daemon-owner="docker-sandboxes-runtime" \ + io.solutionforest.epar.template.expected-docker-daemon-count="1" \ + io.solutionforest.epar.template.profile="${SOURCE_PROFILE}" \ + io.solutionforest.epar.template.platform="${TEMPLATE_PLATFORM}" \ + io.solutionforest.epar.template.runner-version="${ACTIONS_RUNNER_VERSION}" \ + io.solutionforest.epar.template.source-index-digest="${SOURCE_INDEX_DIGEST}" \ + io.solutionforest.epar.template.source-manifest-digest="${SOURCE_MANIFEST_DIGEST}" \ + org.opencontainers.image.base.name="${SOURCE_IMAGE}" \ + org.opencontainers.image.revision="${SOURCE_REVISION}" \ + org.opencontainers.image.version="${TEMPLATE_VERSION}" + +WORKDIR /home/agent +USER agent +ENTRYPOINT ["/usr/local/bin/tini", "-g", "--", "/opt/epar/template-entrypoint.sh"] +CMD ["sleep", "infinity"] + +FROM runner-template AS software-inventory +USER root +RUN install -d -m 0755 /out \ + && /opt/epar/collect-software-inventory.sh > /out/software-inventory.txt + +FROM scratch AS software-inventory-export +COPY --from=software-inventory /out/software-inventory.txt /software-inventory.txt diff --git a/templates/docker-sandboxes/custom-install/run.sh b/templates/docker-sandboxes/custom-install/run.sh new file mode 100644 index 0000000..9a5ee9a --- /dev/null +++ b/templates/docker-sandboxes/custom-install/run.sh @@ -0,0 +1,4 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Generated builds replace this no-op file with the selected project scripts. diff --git a/templates/docker-sandboxes/guest/check-host-trust-generation.sh b/templates/docker-sandboxes/guest/check-host-trust-generation.sh new file mode 100644 index 0000000..f0b6845 --- /dev/null +++ b/templates/docker-sandboxes/guest/check-host-trust-generation.sh @@ -0,0 +1,75 @@ +#!/usr/bin/env bash +set -euo pipefail + +if [[ "$#" -eq 0 ]]; then + marker="/opt/epar/host-trust-generation.json" + lease="/run/epar/host-trust-lease.json" +elif [[ "$#" -eq 2 ]]; then + marker="$1" + lease="$2" +else + echo "EPAR host-trust gate: invalid invocation" >&2 + exit 1 +fi + +if [[ ! -s "${marker}" ]]; then + echo "EPAR host-trust gate: image generation marker is missing" >&2 + exit 1 +fi +if [[ ! -s "${lease}" ]]; then + echo "EPAR host-trust gate: controller lease is missing" >&2 + exit 1 +fi +if [[ ! -x /usr/bin/python3 ]]; then + echo "EPAR host-trust gate: python3 is required" >&2 + exit 1 +fi + +/usr/bin/env -i PATH=/usr/bin:/bin LANG=C.UTF-8 /usr/bin/python3 -I -S - "${marker}" "${lease}" <<'PY' +import datetime +import json +import sys + +marker_path, lease_path = sys.argv[1:] + +def read_json(path, label): + try: + with open(path, "r", encoding="utf-8") as handle: + value = json.load(handle) + except Exception as exc: + raise SystemExit(f"EPAR host-trust gate: invalid {label}: {exc}") + if not isinstance(value, dict): + raise SystemExit(f"EPAR host-trust gate: {label} must be a JSON object") + return value + +marker = read_json(marker_path, "image marker") +lease = read_json(lease_path, "controller lease") + +for key in ("generation", "hostOS", "mode", "scopes"): + if marker.get(key) != lease.get(key): + raise SystemExit( + f"EPAR host-trust gate: {key} mismatch " + f"(image={marker.get(key)!r}, lease={lease.get(key)!r})" + ) + +if marker.get("mode") != "overlay" or not marker.get("generation") or not marker.get("scopes"): + raise SystemExit("EPAR host-trust gate: image trust policy is not an enabled overlay") + +expires = lease.get("expiresAt") +if not isinstance(expires, str) or not expires: + raise SystemExit("EPAR host-trust gate: lease expiry is missing") +try: + expires_at = datetime.datetime.fromisoformat(expires.replace("Z", "+00:00")) +except ValueError as exc: + raise SystemExit(f"EPAR host-trust gate: invalid lease expiry: {exc}") +if expires_at.tzinfo is None: + raise SystemExit("EPAR host-trust gate: lease expiry must include a timezone") +now = datetime.datetime.now(datetime.timezone.utc) +if expires_at <= now: + raise SystemExit( + "EPAR host-trust gate: lease expired at " + + expires_at.astimezone(datetime.timezone.utc).isoformat() + ) + +print("EPAR host-trust gate: generation and lease are current") +PY diff --git a/templates/docker-sandboxes/guest/check-runner.sh b/templates/docker-sandboxes/guest/check-runner.sh new file mode 100644 index 0000000..2b3c272 --- /dev/null +++ b/templates/docker-sandboxes/guest/check-runner.sh @@ -0,0 +1,30 @@ +#!/usr/bin/env bash +set -euo pipefail + +runner_dir="${EPAR_RUNNER_WORK_DIR:-/opt/actions-runner}" +pid_file="${EPAR_RUNNER_PID_FILE:-/var/run/actions-runner.pid}" +pid_start_file="${EPAR_RUNNER_PID_START_FILE:-${pid_file}.start}" +pid="$(cat "${pid_file}" 2>/dev/null || true)" +stored_start="$(cat "${pid_start_file}" 2>/dev/null || true)" + +if [[ ! "${pid}" =~ ^[1-9][0-9]*$ ]] || ! kill -0 "${pid}" >/dev/null 2>&1; then + echo "actions-runner process is not running" >&2 + exit 1 +fi +state="$(ps -p "${pid}" -o stat= 2>/dev/null | tr -d '[:space:]')" +if [[ -z "${state}" || "${state}" == Z* ]]; then + echo "actions-runner process ${pid} has invalid state ${state:-}" >&2 + exit 1 +fi +if [[ "$(readlink -f "/proc/${pid}/cwd" 2>/dev/null || true)" != "$(readlink -f "${runner_dir}")" ]]; then + echo "actions-runner process ${pid} has an unexpected working directory" >&2 + exit 1 +fi +stat_line="$(cat "/proc/${pid}/stat")" +stat_fields="${stat_line##*) }" +read -r -a fields <<<"${stat_fields}" +current_start="${fields[19]:-}" +if [[ ! "${stored_start}" =~ ^[0-9]+$ || "${current_start}" != "${stored_start}" ]]; then + echo "actions-runner process ${pid} does not match its stored start marker" >&2 + exit 1 +fi diff --git a/templates/docker-sandboxes/guest/collect-runner-diagnostics.sh b/templates/docker-sandboxes/guest/collect-runner-diagnostics.sh new file mode 100644 index 0000000..63190cd --- /dev/null +++ b/templates/docker-sandboxes/guest/collect-runner-diagnostics.sh @@ -0,0 +1,17 @@ +#!/usr/bin/env bash +set -u + +pid_file="${EPAR_RUNNER_PID_FILE:-/var/run/actions-runner.pid}" + +echo "=== EPAR Docker Sandboxes runner diagnostics ===" +pid="$(cat "${pid_file}" 2>/dev/null || true)" +if [[ ! "${pid}" =~ ^[1-9][0-9]*$ ]]; then + pid="" +fi +echo "runner_pid=${pid:-}" +if [[ -n "${pid}" ]]; then + ps -p "${pid}" -o pid=,ppid=,stat=,etime= 2>/dev/null || echo "runner_process=" +fi +echo "dockerd_processes=$(pgrep -x dockerd 2>/dev/null | wc -l | tr -d '[:space:]')" +docker info --format 'docker_server={{.ServerVersion}} driver={{.Driver}}' 2>/dev/null || echo "docker_server= driver=" +echo "=== end diagnostics ===" diff --git a/templates/docker-sandboxes/guest/collect-software-inventory.sh b/templates/docker-sandboxes/guest/collect-software-inventory.sh new file mode 100644 index 0000000..54f629b --- /dev/null +++ b/templates/docker-sandboxes/guest/collect-software-inventory.sh @@ -0,0 +1,21 @@ +#!/usr/bin/env bash +set -euo pipefail + +printf 'schemaVersion\t1\n' +printf 'platform\t%s\n' "${EPAR_TEMPLATE_PLATFORM:?EPAR_TEMPLATE_PLATFORM is required}" +printf '\n[os-release]\n' +LC_ALL=C sort /etc/os-release +printf '\n[dpkg]\n' +dpkg-query --show --showformat='${binary:Package}\t${Version}\t${Architecture}\n' | LC_ALL=C sort +printf '\n[tools]\n' +for tool_name in bash docker dockerd git go java node npm python3; do + tool_path="$(command -v "${tool_name}" 2>/dev/null || true)" + if [[ -n "${tool_path}" ]]; then + tool_version="$("${tool_path}" --version 2>&1 | head -n 1 || true)" + printf '%s\t%s\t%s\n' "${tool_name}" "${tool_path}" "${tool_version}" + else + printf '%s\t\t\n' "${tool_name}" + fi +done +printf 'actions-runner\t/opt/actions-runner/bin/Runner.Listener\t%s\n' "$(/opt/actions-runner/bin/Runner.Listener --version)" +printf 'tini\t/usr/local/bin/tini\t%s\n' "$(/usr/local/bin/tini --version 2>&1)" diff --git a/templates/docker-sandboxes/guest/configure-runner.sh b/templates/docker-sandboxes/guest/configure-runner.sh new file mode 100644 index 0000000..c65fa25 --- /dev/null +++ b/templates/docker-sandboxes/guest/configure-runner.sh @@ -0,0 +1,81 @@ +#!/usr/bin/env bash +set -euo pipefail +umask 077 + +if [[ "$(id -u)" != "0" ]]; then + echo "configure-runner.sh must run as root" >&2 + exit 1 +fi +unset SSH_AUTH_SOCK SSH_AUTH_SOCK_GATEWAY SSH_AGENT_PID + +: "${RUNNER_URL:?RUNNER_URL is required}" +: "${RUNNER_NAME:?RUNNER_NAME is required}" +: "${RUNNER_LABELS:?RUNNER_LABELS is required}" +RUNNER_EPHEMERAL="${RUNNER_EPHEMERAL:-true}" +RUNNER_GROUP="${RUNNER_GROUP:-}" +RUNNER_NO_DEFAULT_LABELS="${RUNNER_NO_DEFAULT_LABELS:-false}" +runner_dir="${EPAR_ACTIONS_RUNNER_DIR:-/opt/actions-runner}" + +install -d -m 0700 -o agent -g agent \ + /home/agent/.docker \ + /home/agent/.config \ + /home/agent/.cache \ + /home/agent/.local \ + /home/agent/.local/share \ + /home/agent/.local/state \ + /run/user/1000 + +if ! IFS= read -r runner_token || [[ -z "${runner_token}" ]]; then + echo "RUNNER_TOKEN must be provided as one nonempty line on stdin" >&2 + exit 1 +fi +if IFS= read -r extra_line; then + echo "RUNNER_TOKEN input must contain exactly one line" >&2 + exit 1 +fi + +cd "${runner_dir}" +if [[ -f .runner ]]; then + echo "refusing to configure a Docker Sandboxes template that already contains runner registration state" >&2 + exit 1 +fi + +args=( + --url "${RUNNER_URL}" + --token "${runner_token}" + --name "${RUNNER_NAME}" + --labels "${RUNNER_LABELS}" + --work _work + --unattended +) +if [[ "${RUNNER_EPHEMERAL}" == "true" ]]; then + args+=(--ephemeral) +fi +if [[ -n "${RUNNER_GROUP}" ]]; then + args+=(--runnergroup "${RUNNER_GROUP}") +fi +if [[ "${RUNNER_NO_DEFAULT_LABELS}" == "true" ]]; then + args+=(--no-default-labels) +fi + +configuration_environment=( + "HOME=/home/agent" + "USER=agent" + "LOGNAME=agent" + "XDG_CONFIG_HOME=/home/agent/.config" + "XDG_CACHE_HOME=/home/agent/.cache" + "XDG_DATA_HOME=/home/agent/.local/share" + "XDG_STATE_HOME=/home/agent/.local/state" + "XDG_RUNTIME_DIR=/run/user/1000" + "DOCKER_CONFIG=/home/agent/.docker" + "PATH=/opt/epar/hook-bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin" + "LANG=C.UTF-8" +) +for environment_name in SSL_CERT_FILE NODE_EXTRA_CA_CERTS REQUESTS_CA_BUNDLE JAVA_TOOL_OPTIONS NODE_USE_ENV_PROXY; do + if [[ -n "${!environment_name+x}" ]]; then + configuration_environment+=("${environment_name}=${!environment_name}") + fi +done + +sudo -u agent -H env -i "${configuration_environment[@]}" ./config.sh "${args[@]}" +unset runner_token args diff --git a/templates/docker-sandboxes/guest/docker-daemon.json b/templates/docker-sandboxes/guest/docker-daemon.json new file mode 100644 index 0000000..a19b36a --- /dev/null +++ b/templates/docker-sandboxes/guest/docker-daemon.json @@ -0,0 +1,7 @@ +{ + "proxies": { + "http-proxy": "http://gateway.docker.internal:3128", + "https-proxy": "http://gateway.docker.internal:3128", + "no-proxy": "*" + } +} diff --git a/templates/docker-sandboxes/guest/install-trusted-ca-certificates.sh b/templates/docker-sandboxes/guest/install-trusted-ca-certificates.sh new file mode 100644 index 0000000..fe6ded3 --- /dev/null +++ b/templates/docker-sandboxes/guest/install-trusted-ca-certificates.sh @@ -0,0 +1,22 @@ +#!/usr/bin/env bash +set -euo pipefail + +trust_dirs=( + "/usr/local/share/ca-certificates/epar" + "/usr/local/share/ca-certificates/epar-host" +) +has_certificates=false +for trust_dir in "${trust_dirs[@]}"; do + if [[ -d "${trust_dir}" ]] && find "${trust_dir}" -type f -name '*.crt' -print -quit | grep -q .; then + has_certificates=true + break + fi +done +if [[ "${has_certificates}" != "true" ]]; then + exit 0 +fi +if ! command -v update-ca-certificates >/dev/null 2>&1; then + echo "update-ca-certificates is required to install EPAR trusted CA certificates" >&2 + exit 1 +fi +update-ca-certificates diff --git a/templates/docker-sandboxes/guest/prepare-template.sh b/templates/docker-sandboxes/guest/prepare-template.sh new file mode 100644 index 0000000..1235f1a --- /dev/null +++ b/templates/docker-sandboxes/guest/prepare-template.sh @@ -0,0 +1,144 @@ +#!/usr/bin/env bash +set -euo pipefail + +if [[ "$(id -u)" != "0" ]]; then + echo "prepare-template.sh must run as root" >&2 + exit 1 +fi + +for command_name in bash cmp cut docker dockerd dpkg-query find getent grep groupadd groupmod head install jq nohup pgrep ps readlink seq sha256sum sort stat sudo tar tr useradd usermod wc; do + command -v "${command_name}" >/dev/null 2>&1 || { + echo "pinned source image is missing required command: ${command_name}" >&2 + exit 1 + } +done + +agent_uid="$(id -u agent 2>/dev/null || true)" +uid_1000_user="$(getent passwd 1000 | cut -d: -f1 || true)" + +if [[ -n "${agent_uid}" && "${agent_uid}" != "1000" ]]; then + echo "pinned source image already has an agent user with unexpected UID ${agent_uid}" >&2 + exit 1 +fi + +if [[ -z "${agent_uid}" ]]; then + if [[ -n "${uid_1000_user}" && "${uid_1000_user}" != "ubuntu" && "${uid_1000_user}" != "packer" ]]; then + echo "pinned source image assigns UID 1000 to unexpected user ${uid_1000_user}" >&2 + exit 1 + fi + + gid_1000_group="$(getent group 1000 | cut -d: -f1 || true)" + if [[ -n "${gid_1000_group}" && "${gid_1000_group}" != "ubuntu" && "${gid_1000_group}" != "packer" && "${gid_1000_group}" != "agent" ]]; then + echo "pinned source image assigns GID 1000 to unexpected group ${gid_1000_group}" >&2 + exit 1 + fi + if [[ "${gid_1000_group}" == "ubuntu" || "${gid_1000_group}" == "packer" ]]; then + groupmod --new-name agent "${gid_1000_group}" + elif [[ -z "${gid_1000_group}" ]]; then + groupadd --gid 1000 agent + fi + + if [[ "${uid_1000_user}" == "ubuntu" || "${uid_1000_user}" == "packer" ]]; then + usermod --login agent "${uid_1000_user}" + usermod --home /home/agent --move-home agent + else + useradd --create-home --uid 1000 --gid 1000 --shell /bin/bash agent + fi +fi + +if [[ "$(id -u agent)" != "1000" || "$(id -g agent)" != "1000" ]]; then + echo "agent identity must resolve to UID/GID 1000" >&2 + exit 1 +fi +if [[ "$(getent passwd agent | cut -d: -f6)" != "/home/agent" ]]; then + echo "agent home must resolve to /home/agent" >&2 + exit 1 +fi + +getent group docker >/dev/null 2>&1 || groupadd docker +getent group sudo >/dev/null 2>&1 || { + echo "pinned source image is missing the sudo group" >&2 + exit 1 +} +usermod --append --groups docker,sudo agent + +# Never carry registry credentials from the pinned source image into a reusable +# runner template. Scrub current and legacy Docker client configuration from +# every absolute passwd home, including source identities renamed to agent. +credential_homes=(/root /home/runner /home/agent) +if ! passwd_entries="$(getent passwd)" || [[ -z "${passwd_entries}" ]]; then + echo "failed to enumerate pinned source image passwd homes" >&2 + exit 1 +fi +while IFS=: read -r _ _ _ _ _ account_home _; do + if [[ -z "${account_home}" || "${account_home}" != /* ]]; then + echo "pinned source image contains an invalid passwd home ${account_home:-}" >&2 + exit 1 + fi + normalized_home="$(readlink -m -- "${account_home}")" + if [[ "${normalized_home}" == "/" ]]; then + continue + fi + credential_homes+=("${normalized_home}") +done <<<"${passwd_entries}" + +for root_home_docker_config in /.docker /.dockercfg; do + if [[ -e "${root_home_docker_config}" || -L "${root_home_docker_config}" ]]; then + echo "pinned source image contains Docker client configuration in a root-filesystem passwd home at ${root_home_docker_config}" >&2 + exit 1 + fi +done + +for credential_home in "${credential_homes[@]}"; do + rm -rf -- "${credential_home}/.docker" + rm -f -- "${credential_home}/.dockercfg" + for stale_docker_config in "${credential_home}/.docker" "${credential_home}/.dockercfg"; do + if [[ -e "${stale_docker_config}" || -L "${stale_docker_config}" ]]; then + echo "failed to scrub source Docker client configuration at ${stale_docker_config}" >&2 + exit 1 + fi + done +done + +install -d -m 0755 -o agent -g agent /home/agent +install -d -m 0700 -o agent -g agent \ + /home/agent/.docker \ + /home/agent/.docker/sandbox \ + /home/agent/.docker/sandbox/locks \ + /home/agent/.config \ + /home/agent/.cache \ + /home/agent/.local \ + /home/agent/.local/share \ + /home/agent/.local/state \ + /run/user/1000 +install -d -m 0755 /etc/sudoers.d /etc/apt/apt.conf.d +printf '%s\n' 'agent ALL=(ALL:ALL) NOPASSWD:ALL' > /etc/sudoers.d/epar-agent +rm -f /etc/sudoers.d/epar-proxy +chmod 0440 /etc/sudoers.d/epar-agent +printf '%s\n' 'APT::Periodic::Enable "0";' 'APT::Periodic::Update-Package-Lists "0";' 'APT::Periodic::Unattended-Upgrade "0";' > /etc/apt/apt.conf.d/99epar-disable-periodic +rm -f /etc/systemd/system/timers.target.wants/apt-daily.timer /etc/systemd/system/timers.target.wants/apt-daily-upgrade.timer + +# Docker Sandboxes supplies one private daemon and mounts its dedicated block +# volume here. Never preserve or preload a daemon data-root in the template. +rm -rf /var/lib/docker +install -d -m 0711 /var/lib/docker +if [[ -n "$(find /var/lib/docker -mindepth 1 -print -quit)" ]]; then + echo "/var/lib/docker must be empty in the template" >&2 + exit 1 +fi + +sudo -u agent -H true + +# Docker Sandboxes' forward proxy can replace registry Authorization headers +# with a host credential. Keep the sandbox-private daemon's registry traffic on +# the transparent, policy-enforced path so workflow-scoped Docker credentials +# remain authoritative. The Actions listener also starts from a clean +# environment without inherited proxy variables, so ordinary job traffic uses +# Docker Sandboxes' policy-enforced transparent path by default. +install -d -m 0755 -o root -g root /etc/docker +if [[ -e /etc/docker/daemon.json || -L /etc/docker/daemon.json ]]; then + echo "pinned source image unexpectedly supplies /etc/docker/daemon.json" >&2 + exit 1 +fi +install -m 0644 -o root -g root /opt/epar/docker-daemon.json /etc/docker/daemon.json +cmp -s /opt/epar/docker-daemon.json /etc/docker/daemon.json diff --git a/templates/docker-sandboxes/guest/run-runner.sh b/templates/docker-sandboxes/guest/run-runner.sh new file mode 100644 index 0000000..9e6cb11 --- /dev/null +++ b/templates/docker-sandboxes/guest/run-runner.sh @@ -0,0 +1,114 @@ +#!/usr/bin/env bash +set -euo pipefail +unset SSH_AUTH_SOCK SSH_AUTH_SOCK_GATEWAY SSH_AGENT_PID + +runner_dir="${EPAR_RUNNER_WORK_DIR:-/opt/actions-runner}" +tool_cache="${EPAR_RUNNER_TOOL_CACHE:-${runner_dir}/_work/_tool}" +pid_file="${EPAR_RUNNER_PID_FILE:-/var/run/actions-runner.pid}" +pid_start_file="${EPAR_RUNNER_PID_START_FILE:-${pid_file}.start}" +log_file="${EPAR_RUNNER_LOG_FILE:-/var/log/actions-runner/run.log}" +startup_check_seconds="${EPAR_RUNNER_STARTUP_CHECK_SECONDS:-1}" +agent_home="/home/agent" +agent_runtime_dir="/run/user/1000" + +process_start_time() { + local pid="$1" + local stat_line stat_fields + local -a fields + stat_line="$(cat "/proc/${pid}/stat" 2>/dev/null)" || return 1 + [[ "${stat_line}" == *") "* ]] || return 1 + stat_fields="${stat_line##*) }" + read -r -a fields <<<"${stat_fields}" + [[ "${fields[19]:-}" =~ ^[0-9]+$ ]] || return 1 + printf '%s\n' "${fields[19]}" +} + +install -d -m 0755 -o agent -g agent "$(dirname "${log_file}")" "${tool_cache}" "${tool_cache}/dotnet" +install -d -m 0700 -o agent -g agent \ + "${agent_home}/.docker" \ + "${agent_home}/.config" \ + "${agent_home}/.cache" \ + "${agent_home}/.local" \ + "${agent_home}/.local/share" \ + "${agent_home}/.local/state" \ + "${agent_runtime_dir}" +if [[ -e /home/runner/.docker || -L /home/runner/.docker ]]; then + echo "refusing to start the runner with stale Docker client configuration under /home/runner" >&2 + exit 1 +fi +old_pid="$(cat "${pid_file}" 2>/dev/null || true)" +if [[ "${old_pid}" =~ ^[1-9][0-9]*$ ]] && kill -0 "${old_pid}" >/dev/null 2>&1; then + echo "actions-runner is already running as PID ${old_pid}" >&2 + exit 1 +fi +rm -f "${pid_file}" "${pid_start_file}" + +[[ -s /opt/epar/host-trust-generation.json ]] +[[ -x /opt/epar/check-host-trust-generation.sh ]] +if [[ ! -x /usr/bin/python3 ]]; then + echo "EPAR runner trust policy: python3 is required" >&2 + exit 1 +fi +trust_mode="$(/usr/bin/env -i PATH=/usr/bin:/bin LANG=C.UTF-8 /usr/bin/python3 -I -S - /opt/epar/host-trust-generation.json <<'PY' +import json +import sys + +try: + with open(sys.argv[1], "r", encoding="utf-8") as handle: + marker = json.load(handle) +except Exception as exc: + raise SystemExit(f"EPAR runner trust policy: invalid image marker: {exc}") +if not isinstance(marker, dict) or marker.get("schemaVersion") != 1: + raise SystemExit("EPAR runner trust policy: unsupported image marker schema") +mode = marker.get("mode") +if mode == "disabled": + if marker.get("generation") != "disabled" or marker.get("hostOS") not in ("", None) or marker.get("scopes") != [] or marker.get("certificateCount") != 0: + raise SystemExit("EPAR runner trust policy: malformed disabled policy") +elif mode == "overlay": + if not isinstance(marker.get("generation"), str) or not marker["generation"] or not isinstance(marker.get("hostOS"), str) or not marker["hostOS"] or not isinstance(marker.get("scopes"), list) or not marker["scopes"] or not isinstance(marker.get("certificateCount"), int) or marker["certificateCount"] < 1: + raise SystemExit("EPAR runner trust policy: malformed overlay policy") +else: + raise SystemExit(f"EPAR runner trust policy: unknown mode {mode!r}") +print(mode) +PY +)" +runner_environment=( + "HOME=${agent_home}" + "USER=agent" + "LOGNAME=agent" + "XDG_CONFIG_HOME=${agent_home}/.config" + "XDG_CACHE_HOME=${agent_home}/.cache" + "XDG_DATA_HOME=${agent_home}/.local/share" + "XDG_STATE_HOME=${agent_home}/.local/state" + "XDG_RUNTIME_DIR=${agent_runtime_dir}" + "DOCKER_CONFIG=${agent_home}/.docker" + "EPAR_RUNNER_WORK_DIR=${runner_dir}" + "RUNNER_TOOL_CACHE=${tool_cache}" + "AGENT_TOOLSDIRECTORY=${tool_cache}" + "DOTNET_INSTALL_DIR=${tool_cache}/dotnet" + "PATH=/opt/epar/hook-bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin" + "LANG=C.UTF-8" +) +for environment_name in SSL_CERT_FILE NODE_EXTRA_CA_CERTS REQUESTS_CA_BUNDLE JAVA_TOOL_OPTIONS NODE_USE_ENV_PROXY; do + if [[ -n "${!environment_name+x}" ]]; then + runner_environment+=("${environment_name}=${!environment_name}") + fi +done +if [[ "${trust_mode}" == "overlay" ]]; then + runner_environment+=("ACTIONS_RUNNER_HOOK_JOB_STARTED=/opt/epar/check-host-trust-generation.sh") +fi +sudo -u agent -H env -i "${runner_environment[@]}" /bin/bash -c 'cd "$1" || exit 1; nohup ./run.sh >>"$2" 2>&1 "${pid_file}" +sleep "${startup_check_seconds}" +pid="$(cat "${pid_file}" 2>/dev/null || true)" +if [[ ! "${pid}" =~ ^[1-9][0-9]*$ ]] || ! kill -0 "${pid}" >/dev/null 2>&1; then + echo "actions-runner listener did not remain running" >&2 + exit 1 +fi +process_cwd="$(readlink -f "/proc/${pid}/cwd" 2>/dev/null || true)" +expected_cwd="$(readlink -f "${runner_dir}")" +if [[ "${process_cwd}" != "${expected_cwd}" ]]; then + echo "actions-runner PID ${pid} has unexpected working directory ${process_cwd:-}" >&2 + exit 1 +fi +process_start_time "${pid}" >"${pid_start_file}" +printf '%s\n' "${pid}" diff --git a/templates/docker-sandboxes/guest/template-entrypoint.sh b/templates/docker-sandboxes/guest/template-entrypoint.sh new file mode 100644 index 0000000..0e01fcf --- /dev/null +++ b/templates/docker-sandboxes/guest/template-entrypoint.sh @@ -0,0 +1,47 @@ +#!/usr/bin/env bash +set -euo pipefail + +if [[ "$(id -u)" != "1000" || "$(id -g)" != "1000" || "${HOME:-}" != "/home/agent" || "${USER:-}" != "agent" || "${LOGNAME:-}" != "agent" ]]; then + echo "EPAR Docker Sandboxes template: agent identity contract is not satisfied" >&2 + exit 1 +fi +if [[ -n "${SSH_AUTH_SOCK:-}" || -n "${SSH_AUTH_SOCK_GATEWAY:-}" || -n "${SSH_AGENT_PID:-}" || -e /run/ssh-agent.sock || -L /run/ssh-agent.sock ]]; then + echo "EPAR Docker Sandboxes template: host SSH-agent forwarding is not permitted; restart the Sandboxes daemon without SSH-agent variables" >&2 + exit 1 +fi +unset SSH_AUTH_SOCK SSH_AUTH_SOCK_GATEWAY SSH_AGENT_PID +unset http_proxy https_proxy no_proxy HTTP_PROXY HTTPS_PROXY NO_PROXY +sudo -n install -d -m 0700 -o agent -g agent /run/user/1000 +if [[ "${XDG_CONFIG_HOME:-}" != "/home/agent/.config" || "${XDG_CACHE_HOME:-}" != "/home/agent/.cache" || "${XDG_DATA_HOME:-}" != "/home/agent/.local/share" || "${XDG_STATE_HOME:-}" != "/home/agent/.local/state" || "${XDG_RUNTIME_DIR:-}" != "/run/user/1000" || "${DOCKER_CONFIG:-}" != "/home/agent/.docker" ]]; then + echo "EPAR Docker Sandboxes template: agent configuration-path contract is not satisfied" >&2 + exit 1 +fi +if [[ -e /home/runner/.docker || -L /home/runner/.docker ]]; then + echo "EPAR Docker Sandboxes template: stale Docker client configuration exists under /home/runner" >&2 + exit 1 +fi + +if [[ "${EPAR_SKIP_DOCKER_READY_CHECK:-0}" != "1" ]]; then + echo "EPAR Docker Sandboxes template: waiting for the sandbox-private Docker daemon" + for attempt in $(seq 1 120); do + daemon_count="$( (pgrep -x dockerd 2>/dev/null || true) | wc -l | tr -d '[:space:]')" + if [[ "${daemon_count}" == "1" ]] && docker info >/dev/null 2>&1; then + if [[ "$(docker info --format '{{.NoProxy}}')" != "*" ]]; then + echo "EPAR Docker Sandboxes template: sandbox-private Docker daemon is not using policy-enforced transparent egress" >&2 + exit 1 + fi + echo "EPAR Docker Sandboxes template: one sandbox-private Docker daemon is ready" + break + fi + if [[ "${attempt}" == "120" ]]; then + echo "EPAR Docker Sandboxes template: expected exactly one ready dockerd process; observed ${daemon_count}" >&2 + exit 1 + fi + sleep 1 + done +fi + +if [[ "$#" == "0" ]]; then + set -- sleep infinity +fi +exec "$@" diff --git a/templates/docker-sandboxes/guest/verify-template.sh b/templates/docker-sandboxes/guest/verify-template.sh new file mode 100644 index 0000000..20faab3 --- /dev/null +++ b/templates/docker-sandboxes/guest/verify-template.sh @@ -0,0 +1,85 @@ +#!/usr/bin/env bash +set -euo pipefail + +cd /opt/epar +sha256sum --check helpers.sha256 >/dev/null + +[[ "$(id -u agent)" == "1000" ]] +[[ "$(id -g agent)" == "1000" ]] +[[ "$(getent passwd agent | cut -d: -f6)" == "/home/agent" ]] +[[ "${HOME:-}" == "/home/agent" ]] +[[ "${USER:-}" == "agent" ]] +[[ "${LOGNAME:-}" == "agent" ]] +[[ -z "${SSH_AUTH_SOCK:-}" ]] +[[ -z "${SSH_AUTH_SOCK_GATEWAY:-}" ]] +[[ -z "${SSH_AGENT_PID:-}" ]] +[[ ! -e /run/ssh-agent.sock && ! -L /run/ssh-agent.sock ]] +[[ "${XDG_CONFIG_HOME:-}" == "/home/agent/.config" ]] +[[ "${XDG_CACHE_HOME:-}" == "/home/agent/.cache" ]] +[[ "${XDG_DATA_HOME:-}" == "/home/agent/.local/share" ]] +[[ "${XDG_STATE_HOME:-}" == "/home/agent/.local/state" ]] +[[ "${XDG_RUNTIME_DIR:-}" == "/run/user/1000" ]] +[[ "${DOCKER_CONFIG:-}" == "/home/agent/.docker" ]] +for private_directory in /home/agent/.docker /home/agent/.config /home/agent/.cache /home/agent/.local /home/agent/.local/share /home/agent/.local/state /run/user/1000; do + [[ "$(stat -c '%U:%G:%a' "${private_directory}")" == "agent:agent:700" ]] +done +[[ ! -e /home/agent/.docker/config.json && ! -L /home/agent/.docker/config.json ]] +if ! passwd_entries="$(getent passwd)" || [[ -z "${passwd_entries}" ]]; then + echo "failed to enumerate template passwd homes" >&2 + exit 1 +fi +while IFS=: read -r _ _ _ _ _ account_home _; do + [[ -n "${account_home}" && "${account_home}" == /* ]] + normalized_home="$(readlink -m -- "${account_home}")" + if [[ "${normalized_home}" == "/" ]]; then + sudo -n test ! -e /.docker + sudo -n test ! -L /.docker + sudo -n test ! -e /.dockercfg + sudo -n test ! -L /.dockercfg + continue + fi + sudo -n test ! -e "${normalized_home}/.dockercfg" + sudo -n test ! -L "${normalized_home}/.dockercfg" + if [[ "${normalized_home}" != "/home/agent" ]]; then + sudo -n test ! -e "${normalized_home}/.docker" + sudo -n test ! -L "${normalized_home}/.docker" + fi +done <<<"${passwd_entries}" +[[ ! -e /home/runner/.docker && ! -L /home/runner/.docker ]] +[[ ! -e /home/runner/.dockercfg && ! -L /home/runner/.dockercfg ]] +sudo -n test -f /etc/docker/daemon.json +sudo -n test ! -L /etc/docker/daemon.json +[[ "$(sudo -n stat -c '%U:%G:%a' /etc/docker/daemon.json)" == "root:root:644" ]] +sudo -n jq -e ' + .proxies == { + "http-proxy": "http://gateway.docker.internal:3128", + "https-proxy": "http://gateway.docker.internal:3128", + "no-proxy": "*" + } + and ((keys - ["proxies", "registry-mirrors"]) | length == 0) + and ((has("registry-mirrors") | not) or ((."registry-mirrors" | type) == "array" and all(."registry-mirrors"[]; type == "string"))) +' /etc/docker/daemon.json >/dev/null +id -nG agent | tr ' ' '\n' | grep -Fx docker >/dev/null +sudo -u agent -H sudo -n true +[[ "$(pgrep -x dockerd | wc -l | tr -d '[:space:]')" == "1" ]] +docker info >/dev/null +[[ "$(docker info --format '{{.NoProxy}}')" == "*" ]] +[[ -x /opt/actions-runner/bin/Runner.Listener ]] +[[ -x /opt/epar/check-host-trust-generation.sh ]] +[[ -x /opt/epar/hook-bin/bash ]] +[[ "$(PATH=/opt/epar/hook-bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin command -v bash)" == "/opt/epar/hook-bin/bash" ]] +[[ -x /usr/bin/python3 ]] +[[ -s /opt/epar/actions-runner-version ]] +[[ "$(sudo -u agent -H /opt/actions-runner/bin/Runner.Listener --version)" == "$(cat /opt/epar/actions-runner-version)" ]] +case "${EPAR_TEMPLATE_PLATFORM}" in + linux/amd64) + [[ "$(uname -m)" == "x86_64" ]] + ;; + linux/arm64) + [[ "$(uname -m)" == "aarch64" ]] + ;; + *) + echo "unsupported EPAR template platform: ${EPAR_TEMPLATE_PLATFORM}" >&2 + exit 1 + ;; +esac diff --git a/templates/docker-sandboxes/helpers.sha256 b/templates/docker-sandboxes/helpers.sha256 new file mode 100644 index 0000000..f882fa1 --- /dev/null +++ b/templates/docker-sandboxes/helpers.sha256 @@ -0,0 +1,11 @@ +5ce4fad927d8e8a21ad312ec49b0498b51b652cbb89e0c5ca58058b6a00fe91d ./check-host-trust-generation.sh +3f1cee32d05ffad64004377f0276ef8aa46a2848b32bca74ac35efb3d71fd1bf ./check-runner.sh +4fe8e512539f97d00db3c2856f452f017c4de6a04832f1c1af86f1994862df59 ./collect-runner-diagnostics.sh +9d659942a7a0958c07cb0f0f069c699ec865020f0f9302bc005680a88c10c8e6 ./collect-software-inventory.sh +99474651f2c675389f35e5ad9aedd32ddcd8dbc302b486f42f3b035414122952 ./configure-runner.sh +1fbc8c68c8d75f3982e23718ed3d5bd984c2afee21e26657b7a64a9f185a747f ./docker-daemon.json +32bc68fe28dbbe9eb6947a1023606c32fc9b20088efcb90b5c543d0fa3db1218 ./install-trusted-ca-certificates.sh +759a46fe38c27e9a64ffbd7d0eaafaffac1292e18ddf4a9527b0d71413aa8817 ./prepare-template.sh +746836a5cdac8da3d35e7914e60d293d0c605214beb498353168bafee08b09e7 ./run-runner.sh +71f328ef12301e4d79d285eb8cd5e1ba494e3c0be32a1e5569e25a442702fb48 ./template-entrypoint.sh +56068235384e118f0b200443f21e972f9225d839bda6239c7b75b468fa4c70d4 ./verify-template.sh diff --git a/templates/docker-sandboxes/hook-launcher/main.go b/templates/docker-sandboxes/hook-launcher/main.go new file mode 100644 index 0000000..cb7f043 --- /dev/null +++ b/templates/docker-sandboxes/hook-launcher/main.go @@ -0,0 +1,54 @@ +//go:build linux + +package main + +import ( + "fmt" + "os" + "strings" + "syscall" +) + +const ( + realBash = "/bin/bash" + hookPath = "/opt/epar/check-host-trust-generation.sh" +) + +func main() { + arguments := append([]string{"bash"}, os.Args[1:]...) + environment := os.Environ() + if isHostTrustHookInvocation(os.Args[1:]) { + arguments = append([]string{"bash", "-p"}, os.Args[1:]...) + environment = isolatedHookEnvironment(environment) + } + if err := syscall.Exec(realBash, arguments, environment); err != nil { + fmt.Fprintf(os.Stderr, "EPAR bash launcher: exec failed: %v\n", err) + os.Exit(126) + } +} + +func isHostTrustHookInvocation(arguments []string) bool { + for _, argument := range arguments { + if argument == hookPath { + return true + } + } + return false +} + +func isolatedHookEnvironment(environment []string) []string { + allowed := map[string]bool{ + "LANG": true, + "LC_ALL": true, + "TZ": true, + } + result := make([]string, 0, len(allowed)+2) + for _, entry := range environment { + name, _, found := strings.Cut(entry, "=") + if found && allowed[name] { + result = append(result, entry) + } + } + result = append(result, "PATH=/usr/bin:/bin", "EPAR_HOOK_LAUNCHER=isolated-v1") + return result +} diff --git a/templates/docker-sandboxes/hook-launcher/main_test.go b/templates/docker-sandboxes/hook-launcher/main_test.go new file mode 100644 index 0000000..36ac9d2 --- /dev/null +++ b/templates/docker-sandboxes/hook-launcher/main_test.go @@ -0,0 +1,51 @@ +//go:build linux + +package main + +import ( + "slices" + "testing" +) + +func TestHostTrustHookInvocationRequiresExactArgument(t *testing.T) { + if !isHostTrustHookInvocation([]string{"--noprofile", "--norc", hookPath}) { + t.Fatal("exact hook path was not recognized") + } + for _, arguments := range [][]string{ + {hookPath + ".bak"}, + {"echo " + hookPath}, + {"/tmp/check-host-trust-generation.sh"}, + } { + if isHostTrustHookInvocation(arguments) { + t.Fatalf("non-exact hook invocation was recognized: %q", arguments) + } + } +} + +func TestIsolatedHookEnvironmentDropsWorkflowStartupAndSecretInputs(t *testing.T) { + got := isolatedHookEnvironment([]string{ + "HOME=/home/agent", + "LANG=C.UTF-8", + "LC_ALL=C", + "TZ=UTC", + "PATH=/tmp/attacker:/usr/bin", + "BASH_ENV=/tmp/attack.sh", + "ENV=/tmp/attack.sh", + "PYTHONPATH=/tmp/attacker", + "PYTHONSTARTUP=/tmp/attack.py", + "LD_PRELOAD=/tmp/attack.so", + "DOTNET_STARTUP_HOOKS=/tmp/attack.dll", + "ACTIONS_RUNTIME_TOKEN=sentinel", + "GITHUB_TOKEN=sentinel", + }) + want := []string{ + "LANG=C.UTF-8", + "LC_ALL=C", + "TZ=UTC", + "PATH=/usr/bin:/bin", + "EPAR_HOOK_LAUNCHER=isolated-v1", + } + if !slices.Equal(got, want) { + t.Fatalf("isolated environment mismatch\n got: %q\nwant: %q", got, want) + } +} diff --git a/templates/docker-sandboxes/profiles/act-22.04.amd64.compatibility.json b/templates/docker-sandboxes/profiles/act-22.04.amd64.compatibility.json new file mode 100644 index 0000000..ee25861 --- /dev/null +++ b/templates/docker-sandboxes/profiles/act-22.04.amd64.compatibility.json @@ -0,0 +1,27 @@ +{ + "schemaVersion": 2, + "templateSchemaVersion": 1, + "profile": "act-22.04", + "validationStatus": "planned", + "platform": "linux/amd64", + "source": { + "reference": "ghcr.io/catthehacker/ubuntu@sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac", + "indexDigest": "sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac", + "manifestDigest": "sha256:f3d493b10df1582ce631e0213bd90aa5f8196287c8a9f8ef546ecb44ca256655" + }, + "runner": { + "execution": "direct-actions-listener", + "version": "2.332.0", + "user": "agent", + "uid": 1000, + "gid": 1000, + "home": "/home/agent", + "directory": "/opt/actions-runner" + }, + "docker": { + "daemonOwner": "docker-sandboxes-runtime", + "expectedDaemonCount": 1, + "imagePreloadsVarLibDocker": false, + "buildRequiresPrivilegedBuildkit": false + } +} diff --git a/templates/docker-sandboxes/profiles/act-22.04.arm64.compatibility.json b/templates/docker-sandboxes/profiles/act-22.04.arm64.compatibility.json new file mode 100644 index 0000000..c18bb05 --- /dev/null +++ b/templates/docker-sandboxes/profiles/act-22.04.arm64.compatibility.json @@ -0,0 +1,27 @@ +{ + "schemaVersion": 2, + "templateSchemaVersion": 1, + "profile": "act-22.04", + "validationStatus": "unvalidated", + "platform": "linux/arm64", + "source": { + "reference": "ghcr.io/catthehacker/ubuntu@sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac", + "indexDigest": "sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac", + "manifestDigest": "sha256:72b9ec71ee5972e02df5053f0000d34dbd2a3d0165b912bf25bbeabd72fba160" + }, + "runner": { + "execution": "direct-actions-listener", + "version": "2.332.0", + "user": "agent", + "uid": 1000, + "gid": 1000, + "home": "/home/agent", + "directory": "/opt/actions-runner" + }, + "docker": { + "daemonOwner": "docker-sandboxes-runtime", + "expectedDaemonCount": 1, + "imagePreloadsVarLibDocker": false, + "buildRequiresPrivilegedBuildkit": false + } +} diff --git a/templates/docker-sandboxes/profiles/full.amd64.compatibility.json b/templates/docker-sandboxes/profiles/full.amd64.compatibility.json new file mode 100644 index 0000000..8fda5da --- /dev/null +++ b/templates/docker-sandboxes/profiles/full.amd64.compatibility.json @@ -0,0 +1,27 @@ +{ + "schemaVersion": 2, + "templateSchemaVersion": 1, + "profile": "full", + "validationStatus": "planned", + "platform": "linux/amd64", + "source": { + "reference": "ghcr.io/catthehacker/ubuntu@sha256:76581ac3f31aa1ad7cb558b47c3e836b9cbcd82dc08fc69349f77e3967bea50c", + "indexDigest": "sha256:76581ac3f31aa1ad7cb558b47c3e836b9cbcd82dc08fc69349f77e3967bea50c", + "manifestDigest": "sha256:58314fa8cbf0f0e5384a37b3444811033320038816ef7c16f30b3e841ed65e51" + }, + "runner": { + "execution": "direct-actions-listener", + "version": "2.332.0", + "user": "agent", + "uid": 1000, + "gid": 1000, + "home": "/home/agent", + "directory": "/opt/actions-runner" + }, + "docker": { + "daemonOwner": "docker-sandboxes-runtime", + "expectedDaemonCount": 1, + "imagePreloadsVarLibDocker": false, + "buildRequiresPrivilegedBuildkit": false + } +} diff --git a/templates/docker-sandboxes/profiles/full.arm64.compatibility.json b/templates/docker-sandboxes/profiles/full.arm64.compatibility.json new file mode 100644 index 0000000..f4e13c7 --- /dev/null +++ b/templates/docker-sandboxes/profiles/full.arm64.compatibility.json @@ -0,0 +1,27 @@ +{ + "schemaVersion": 2, + "templateSchemaVersion": 1, + "profile": "full", + "validationStatus": "unvalidated", + "platform": "linux/arm64", + "source": { + "reference": "ghcr.io/catthehacker/ubuntu@sha256:76581ac3f31aa1ad7cb558b47c3e836b9cbcd82dc08fc69349f77e3967bea50c", + "indexDigest": "sha256:76581ac3f31aa1ad7cb558b47c3e836b9cbcd82dc08fc69349f77e3967bea50c", + "manifestDigest": "sha256:245c8981fbf4ac268db015463c6c446b9411481f7e0001537128dc384d46dd0c" + }, + "runner": { + "execution": "direct-actions-listener", + "version": "2.332.0", + "user": "agent", + "uid": 1000, + "gid": 1000, + "home": "/home/agent", + "directory": "/opt/actions-runner" + }, + "docker": { + "daemonOwner": "docker-sandboxes-runtime", + "expectedDaemonCount": 1, + "imagePreloadsVarLibDocker": false, + "buildRequiresPrivilegedBuildkit": false + } +} diff --git a/templates/docker-sandboxes/sources.lock.json b/templates/docker-sandboxes/sources.lock.json new file mode 100644 index 0000000..ffd2ad9 --- /dev/null +++ b/templates/docker-sandboxes/sources.lock.json @@ -0,0 +1,113 @@ +{ + "schemaVersion": 2, + "defaultPlatform": "linux/amd64", + "supportedPlatforms": ["linux/amd64", "linux/arm64"], + "dockerfileFrontend": { + "inspectionReference": "docker.io/docker/dockerfile@sha256:a57df69d0ea827fb7266491f2813635de6f17269be881f696fbfdf2d83dda33e", + "reference": "docker.io/docker/dockerfile:1.7.1@sha256:a57df69d0ea827fb7266491f2813635de6f17269be881f696fbfdf2d83dda33e", + "indexDigest": "sha256:a57df69d0ea827fb7266491f2813635de6f17269be881f696fbfdf2d83dda33e" + }, + "sbomGenerator": { + "inspectionReference": "docker.io/docker/buildkit-syft-scanner@sha256:79e7b013cbec16bbb436f312819a49a4a57752b2270c1a9332ae1a10fcc82a68", + "indexDigest": "sha256:79e7b013cbec16bbb436f312819a49a4a57752b2270c1a9332ae1a10fcc82a68" + }, + "goBuilder": { + "version": "1.25.12", + "inspectionReference": "docker.io/library/golang@sha256:9006890ecba0a168034d99516084099ae3114d9f2b7d6572c77f2dde57ebc980", + "indexDigest": "sha256:9006890ecba0a168034d99516084099ae3114d9f2b7d6572c77f2dde57ebc980" + }, + "hookLauncher": { + "sha256": "7fe07f10f484fa6888481a4165e81570187c0aeff422738d3ea5add6b95dd9b7" + }, + "tini": { + "version": "0.19.0" + }, + "platforms": { + "linux/amd64": { + "architecture": "amd64", + "dockerfileFrontendManifestDigest": "sha256:b5f3b260a9678e1d83d2fce86eeddf79420b79147eaba2a25986f47133d73720", + "goBuilderManifestDigest": "sha256:12e171e33ce7ade87ac8ab2bbe65cea9371527285bdab43ca02780a9e6ac60e5", + "goBuilderReference": "docker.io/library/golang@sha256:12e171e33ce7ade87ac8ab2bbe65cea9371527285bdab43ca02780a9e6ac60e5", + "sbomGeneratorManifestDigest": "sha256:13864237fb990943433f89d698590aad1de38d4a7e13d38e7b12f2488c1952e7", + "sbomGeneratorReference": "docker.io/docker/buildkit-syft-scanner@sha256:13864237fb990943433f89d698590aad1de38d4a7e13d38e7b12f2488c1952e7", + "tini": { + "url": "https://github.com/krallin/tini/releases/download/v0.19.0/tini-amd64", + "sha256": "93dcc18adc78c65a028a84799ecf8ad40c936fdfc5f2a57b1acda5a8117fa82c" + } + }, + "linux/arm64": { + "architecture": "arm64", + "dockerfileFrontendManifestDigest": "sha256:c8678869a83fab70232869ba24acc1c0be661f4d65135c0eeacb6a8e78420fdd", + "goBuilderManifestDigest": "sha256:afe53a4752b49f57ddebc97501a99394e2f7715236b4241efa830d54efb44434", + "goBuilderReference": "docker.io/library/golang@sha256:afe53a4752b49f57ddebc97501a99394e2f7715236b4241efa830d54efb44434", + "sbomGeneratorManifestDigest": "sha256:860305b3d1667c35142f11f6e9485e322c1c6173702a0831dc68739a34847f2d", + "sbomGeneratorReference": "docker.io/docker/buildkit-syft-scanner@sha256:860305b3d1667c35142f11f6e9485e322c1c6173702a0831dc68739a34847f2d", + "tini": { + "url": "https://github.com/krallin/tini/releases/download/v0.19.0/tini-arm64", + "sha256": "07952557df20bfd2a95f9bef198b445e006171969499a1d361bd9e6f8e5e0e81" + } + } + }, + "supersededRecords": { + "linux/amd64": { + "act-22.04": { + "authoritative": false, + "reason": "Predates the current runner-template helper and architecture changes", + "manifestDigest": "sha256:f3d493b10df1582ce631e0213bd90aa5f8196287c8a9f8ef546ecb44ca256655", + "templateTag": "epar-docker-sandboxes-catthehacker-act-22.04:20260723-r3-amd64" + }, + "full": { + "authoritative": false, + "reason": "Predates the current runner-template helper and architecture changes", + "manifestDigest": "sha256:58314fa8cbf0f0e5384a37b3444811033320038816ef7c16f30b3e841ed65e51", + "templateTag": "epar-docker-sandboxes-catthehacker-full:20260723-r1-amd64" + } + } + }, + "profiles": { + "act-22.04": { + "sourceRepository": "ghcr.io/catthehacker/ubuntu", + "observedTagReference": "ghcr.io/catthehacker/ubuntu:act-22.04", + "inspectionReference": "ghcr.io/catthehacker/ubuntu@sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac", + "immutableReference": "ghcr.io/catthehacker/ubuntu@sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac", + "indexDigest": "sha256:b40b8af93baee90b83f29c834440873300c8478809535786dbf79fa836c086ac", + "sourceRevision": "e2f8efe464c82732f78e967ee709c00b6af53643", + "platforms": { + "linux/amd64": { + "validationStatus": "planned", + "manifestDigest": "sha256:f3d493b10df1582ce631e0213bd90aa5f8196287c8a9f8ef546ecb44ca256655", + "templateTag": "epar-docker-sandboxes-catthehacker-act-22.04:20260723-r4-amd64", + "compatibilityFile": "act-22.04.amd64.compatibility.json" + }, + "linux/arm64": { + "validationStatus": "unvalidated", + "manifestDigest": "sha256:72b9ec71ee5972e02df5053f0000d34dbd2a3d0165b912bf25bbeabd72fba160", + "templateTag": "epar-docker-sandboxes-catthehacker-act-22.04:20260723-r4-arm64", + "compatibilityFile": "act-22.04.arm64.compatibility.json" + } + } + }, + "full": { + "sourceRepository": "ghcr.io/catthehacker/ubuntu", + "observedTagReference": "ghcr.io/catthehacker/ubuntu:full-latest", + "inspectionReference": "ghcr.io/catthehacker/ubuntu@sha256:76581ac3f31aa1ad7cb558b47c3e836b9cbcd82dc08fc69349f77e3967bea50c", + "immutableReference": "ghcr.io/catthehacker/ubuntu@sha256:76581ac3f31aa1ad7cb558b47c3e836b9cbcd82dc08fc69349f77e3967bea50c", + "indexDigest": "sha256:76581ac3f31aa1ad7cb558b47c3e836b9cbcd82dc08fc69349f77e3967bea50c", + "sourceRevision": "96c58e2540a8c11351aed1269df0553663a1b8d7", + "platforms": { + "linux/amd64": { + "validationStatus": "planned", + "manifestDigest": "sha256:58314fa8cbf0f0e5384a37b3444811033320038816ef7c16f30b3e841ed65e51", + "templateTag": "epar-docker-sandboxes-catthehacker-full:20260723-r2-amd64", + "compatibilityFile": "full.amd64.compatibility.json" + }, + "linux/arm64": { + "validationStatus": "unvalidated", + "manifestDigest": "sha256:245c8981fbf4ac268db015463c6c446b9411481f7e0001537128dc384d46dd0c", + "templateTag": "epar-docker-sandboxes-catthehacker-full:20260723-r2-arm64", + "compatibilityFile": "full.arm64.compatibility.json" + } + } + } + } +}