diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index caa0e4e88..a7c116dc0 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1520,6 +1520,143 @@ jobs: name: e2e-report path: tests/e2e-cucumber/results/ + # The same suite on a GitHub-hosted Windows runner. The self-hosted GPU lanes + # (e2e-selfhosted.yml) run only the scenarios that need their GPU + # (E2E_GPU_ONLY), so this is where everything else meets native Windows — + # path handling, the Windows probes, the PowerShell surfaces. Built with + # `rocm/e2e-test-hooks` by `cargo xtask e2e`, like the Linux lane above and + # unlike windows-build-and-test's lifecycle run, which must ship exactly what + # a release ships. Scenarios on a simulated machine are `@requires-os:linux` + # and skip here: the Windows probes ask the registry, not a filesystem. + e2e-windows: + name: E2E tests (Windows) + timeout-minutes: 60 + runs-on: windows-latest + # Only `changes`: unlike the Linux `e2e` job this does not wait for the + # Linux `build-and-test`, which tells it nothing about the Windows build and + # would put that job's duration in front of this one's. + needs: [changes] + # Same compile cache as windows-build-and-test (see build-and-test for the + # design: Azure backend, written only by push and merge_group), under its + # own prefix — this build carries `rocm/e2e-test-hooks`, so its entries + # would never match that job's anyway. + env: + CARGO_INCREMENTAL: "0" + RUSTC_WRAPPER: sccache + SCCACHE_AZURE_CONNECTION_STRING: ${{ secrets.CACHE_AZURE_CONNECTION_STRING }} + SCCACHE_AZURE_BLOB_CONTAINER: cache + SCCACHE_AZURE_KEY_PREFIX: rocm-cli/e2e-windows + SCCACHE_AZURE_RW_MODE: ${{ (github.event_name == 'push' || github.event_name == 'merge_group') && 'READ_WRITE' || 'READ_ONLY' }} + if: >- + always() + && needs.changes.result == 'success' + && (github.event_name != 'workflow_dispatch' + || inputs.platform == 'all' || inputs.platform == 'mock') + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + + # Whether the real work runs this time: on a dispatch, or when the change + # is heavy (`merge_group` forces heavy). The job itself always runs, so the + # check is always reported — required checks that never report stall the + # merge queue. + - name: Decide whether to run + id: gate + shell: bash + run: | + if [ "${{ github.event_name }}" = "workflow_dispatch" ] \ + || [ "${{ needs.changes.outputs.heavy }}" = "true" ]; then + echo "run=true" >> "$GITHUB_OUTPUT" + else + echo "run=false" >> "$GITHUB_OUTPUT" + fi + + # Installs sccache. Runs BEFORE setup-rust-toolchain, whose rust-cache step + # runs `cargo metadata` and honours RUSTC_WRAPPER. `version` is pinned so + # the compiler wrapper does not float with the latest sccache release. + # continue-on-error for the same reason as build-and-test: a failed install + # is handled by the probe, not by failing the job. + - name: Set up sccache + if: steps.gate.outputs.run == 'true' + continue-on-error: true + uses: mozilla-actions/sccache-action@fc920bf0ec8de6ee65d409111f7ec508035751ba # v0.0.11 + with: + version: v0.17.0 + + # Keep the remote backend only if it answers — see the equivalent step in + # build-and-test for why this is a probe, and why failure clears the Azure + # variables rather than the wrapper. + - name: Probe sccache cache backend + id: sccache_probe + if: steps.gate.outputs.run == 'true' + shell: pwsh + run: | + # This script decides whether to use the cache; a non-zero status from + # the commands it runs is an answer, not a job failure. Pin the native + # error preference off so a future runner default cannot turn a failed + # probe into a failed step. + $PSNativeCommandUseErrorActionPreference = $false + # Every Out-File below passes -Encoding utf8 explicitly. PowerShell 7 + # (`shell: pwsh`) already defaults to UTF-8, but Windows PowerShell 5.1 + # defaults to UTF-16LE, which the runner cannot parse — so an edit that + # switched this step to `shell: powershell` would silently corrupt the + # environment file rather than fail visibly. + function Disable-Backend { + "SCCACHE_AZURE_CONNECTION_STRING=" | Out-File -FilePath $env:GITHUB_ENV -Append -Encoding utf8 + "SCCACHE_AZURE_BLOB_CONTAINER=" | Out-File -FilePath $env:GITHUB_ENV -Append -Encoding utf8 + } + if (-not (Get-Command sccache -ErrorAction SilentlyContinue)) { + Write-Output "::warning::sccache is not installed; compiling without it." + "RUSTC_WRAPPER=" | Out-File -FilePath $env:GITHUB_ENV -Append -Encoding utf8 + # Nothing will read the credential now; don't leave it in the job env. + Disable-Backend + exit 0 + } + # sccache is the compiler wrapper from here on, remote backend or not, + # so there are statistics worth reporting either way. + "active=true" | Out-File -FilePath $env:GITHUB_OUTPUT -Append -Encoding utf8 + if (-not $env:SCCACHE_AZURE_CONNECTION_STRING) { + Write-Output "::notice::No cache credential available (expected on forks); compiling without a shared cache." + Disable-Backend + exit 0 + } + sccache --start-server + if ($LASTEXITCODE -ne 0) { + Write-Output "::warning::sccache could not reach the cache backend; compiling without a shared cache." + # Leave no half-started server behind for the cargo steps to find. + sccache --stop-server *> $null + Disable-Backend + } + exit 0 + + - uses: actions-rust-lang/setup-rust-toolchain@166cdcfd11aee3cb47222f9ddb555ce30ddb9659 # v1.17.0 + if: steps.gate.outputs.run == 'true' + with: + cache-on-failure: true + + - name: Run E2E tests + if: steps.gate.outputs.run == 'true' + shell: pwsh + run: cargo xtask e2e + + - name: sccache stats + if: always() && steps.sccache_probe.outputs.active == 'true' + shell: pwsh + run: sccache --show-stats + + # Its own grid in the run Summary; see e2e-report for why the consolidated + # (required) report does not wait for this lane. + - name: Report summary + if: always() && steps.gate.outputs.run == 'true' + shell: pwsh + run: cargo xtask e2e-report --artifacts-dir tests/e2e-cucumber/results --html-out tests/e2e-cucumber/results/consolidated.html >> $env:GITHUB_STEP_SUMMARY + + - name: Upload E2E report + if: always() && steps.gate.outputs.run == 'true' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: e2e-windows-report + path: tests/e2e-cucumber/results/ + # Consolidate the mock (GitHub-hosted) platform's report into the cross-platform # report: a platform × tier matrix in the run Summary plus a single merged HTML # artifact. `if: always()` so a failing platform still appears; runs on @@ -1527,8 +1664,11 @@ jobs: # # The self-hosted GPU platforms are consolidated by e2e-selfhosted.yml's own # report job (they moved there so an offline runner can't stall this workflow's - # required checks). This job only depends on the mock `e2e` lane; the `*-report` - # glob still auto-includes any new GitHub-hosted platform added here. + # required checks). This job only depends on the mock `e2e` lane, and downloads + # only its artifact. `e2e-windows` summarizes itself instead: waiting on it here + # would put a cold Windows release build in front of this REQUIRED check, and + # matching it by glob would include it or not depending on which job finished + # first. e2e-report: name: E2E consolidated report runs-on: ubuntu-latest @@ -1549,9 +1689,9 @@ jobs: - uses: actions-rust-lang/setup-rust-toolchain@166cdcfd11aee3cb47222f9ddb555ce30ddb9659 # v1.17.0 - # Pull every e2e artifact matching the glob. `xtask e2e-report` turns each - # into one labeled platform and the `*-report` glob makes new platforms - # appear automatically. Layout note: download-artifact@v8 extracts a match + # Pull the mock lane's artifact (named exactly; see the job comment for why + # not a glob). `xtask e2e-report` turns it into one labeled platform. + # Layout note: download-artifact@v8 extracts a match # into e2e-artifacts// when several match, but flattens # straight into e2e-artifacts/ when EXACTLY ONE matches (its source picks # the root path when `artifacts.length === 1`). After the ci.yml ⇄ @@ -1561,7 +1701,7 @@ jobs: - name: Download all E2E reports uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: - pattern: '*-report' + pattern: 'e2e-report' path: e2e-artifacts - name: Build consolidated report + step summary diff --git a/.github/workflows/e2e-selfhosted.yml b/.github/workflows/e2e-selfhosted.yml index fde24b1d1..fd055cc9b 100644 --- a/.github/workflows/e2e-selfhosted.yml +++ b/.github/workflows/e2e-selfhosted.yml @@ -181,14 +181,25 @@ jobs: # a single nightly scenario (e.g. the 27B serve) without the full nightly run. E2E_INCLUDE_NIGHTLY: "${{ inputs.include_nightly && '1' || '' }}" # Opt-in @merge-queue serves: the heavy real serves (default-engine + - # readiness) run only in the merge queue, where a cheaper per-engine canary - # (scenarios 5 vLLM / 7 lemonade) has already guarded the PR. This keeps the - # per-PR GPU run short while still exercising the full serve matrix before a - # change lands. Only set on the `merge_group` event. + # readiness) run only in the merge queue, where the cheaper per-engine + # `@gpu-smoke` canaries have already guarded the PR. Only set on the + # `merge_group` event. E2E_MERGE_QUEUE: "${{ github.event_name == 'merge_group' && '1' || '' }}" + # On a pull request and in the merge queue, of the scenarios that need a + # real GPU only the `@gpu-smoke` canaries (and, in the queue, the + # @merge-queue serves) run. Every GPU scenario that only needs the CLI to + # see a GPU runs on a simulated machine on the hosted `E2E tests` lane. The + # full real-GPU suite runs on pushes to main/release, on dispatch, and + # nightly. Keeps the GPU lanes, which set merge latency, short. + E2E_GPU_SMOKE_ONLY: "${{ (github.event_name == 'merge_group' || github.event_name == 'pull_request') && '1' || '' }}" # A real GPU lane: run the `@requires-real-gpu` scenarios that a simulated # host cannot stand in for. See HardwareMode in src/expectation.rs. E2E_HARDWARE: real + # ...and ONLY those, on every event: everything that needs no GPU runs on + # the GitHub-hosted Linux and Windows lanes in ci.yml (`E2E tests`, + # `E2E tests (Windows)`), so a GPU runner's time is not spent on it. See + # restrict_to_lane in src/expectation.rs. + E2E_GPU_ONLY: "1" steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -421,8 +432,11 @@ jobs: E2E_INCLUDE_NIGHTLY: "${{ inputs.include_nightly && '1' || '' }}" # Heavy @merge-queue serves run only in the merge queue; see e2e-gpu. E2E_MERGE_QUEUE: "${{ github.event_name == 'merge_group' && '1' || '' }}" + # GPU smoke test only on pull requests and in the merge queue; see the e2e-gpu lane. + E2E_GPU_SMOKE_ONLY: "${{ (github.event_name == 'merge_group' || github.event_name == 'pull_request') && '1' || '' }}" # See the e2e-gpu lane. E2E_HARDWARE: real + E2E_GPU_ONLY: "1" steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -676,8 +690,11 @@ jobs: E2E_INCLUDE_NIGHTLY: "${{ inputs.include_nightly && '1' || '' }}" # Heavy @merge-queue serves run only in the merge queue; see e2e-gpu. E2E_MERGE_QUEUE: "${{ github.event_name == 'merge_group' && '1' || '' }}" + # GPU smoke test only on pull requests and in the merge queue; see the e2e-gpu lane. + E2E_GPU_SMOKE_ONLY: "${{ (github.event_name == 'merge_group' || github.event_name == 'pull_request') && '1' || '' }}" # See the e2e-gpu lane. E2E_HARDWARE: real + E2E_GPU_ONLY: "1" steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -964,6 +981,8 @@ jobs: E2E_INCLUDE_NIGHTLY: "${{ inputs.include_nightly && '1' || '' }}" # Heavy @merge-queue serves run only in the merge queue; see e2e-gpu. E2E_MERGE_QUEUE: "${{ github.event_name == 'merge_group' && '1' || '' }}" + # GPU smoke test only on pull requests and in the merge queue; see the e2e-gpu lane. + E2E_GPU_SMOKE_ONLY: "${{ (github.event_name == 'merge_group' || github.event_name == 'pull_request') && '1' || '' }}" # See the e2e-gpu lane. E2E_HARDWARE: real # apps/rocm/build.rs falls back to GitHub's own GITHUB_HEAD_REF/ @@ -1364,7 +1383,7 @@ jobs: # Forward the job-level E2E_* env vars into the guest: `wsl.exe` # does not share the calling process's environment unless a # variable is named in WSLENV. - $env:WSLENV = 'E2E_SERVE_TIMEOUT_SECS:E2E_TUI_TIMEOUT_SECS:E2E_INCLUDE_NIGHTLY:E2E_MERGE_QUEUE:E2E_HARDWARE:ROCM_CLI_RELEASE_REF' + $env:WSLENV = 'E2E_SERVE_TIMEOUT_SECS:E2E_TUI_TIMEOUT_SECS:E2E_INCLUDE_NIGHTLY:E2E_MERGE_QUEUE:E2E_GPU_SMOKE_ONLY:E2E_HARDWARE:ROCM_CLI_RELEASE_REF' & "$env:GITHUB_WORKSPACE\.github\scripts\Invoke-WslBash.ps1" -PipeFail -Script @' export PATH="$HOME/.cargo/bin:$PATH" cd /root/work/rocm-cli @@ -1489,8 +1508,11 @@ jobs: E2E_INCLUDE_NIGHTLY: "${{ inputs.include_nightly && '1' || '' }}" # Full serve matrix at the merge-queue gate, like every other lane (see e2e-gpu). E2E_MERGE_QUEUE: "${{ github.event_name == 'merge_group' && '1' || '' }}" + # GPU smoke test only on pull requests and in the merge queue; see the e2e-gpu lane. + E2E_GPU_SMOKE_ONLY: "${{ (github.event_name == 'merge_group' || github.event_name == 'pull_request') && '1' || '' }}" # See the e2e-gpu lane. E2E_HARDWARE: real + E2E_GPU_ONLY: "1" steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -1619,8 +1641,11 @@ jobs: E2E_INCLUDE_NIGHTLY: "${{ inputs.include_nightly && '1' || '' }}" # Full serve matrix at the merge-queue gate, like every other lane (see e2e-gpu). E2E_MERGE_QUEUE: "${{ github.event_name == 'merge_group' && '1' || '' }}" + # GPU smoke test only on pull requests and in the merge queue; see the e2e-gpu lane. + E2E_GPU_SMOKE_ONLY: "${{ (github.event_name == 'merge_group' || github.event_name == 'pull_request') && '1' || '' }}" # See the e2e-gpu lane. E2E_HARDWARE: real + E2E_GPU_ONLY: "1" steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 diff --git a/crates/e2e-report/src/consolidated.rs b/crates/e2e-report/src/consolidated.rs index 91acbd436..803a7dac3 100644 --- a/crates/e2e-report/src/consolidated.rs +++ b/crates/e2e-report/src/consolidated.rs @@ -67,6 +67,8 @@ fn parse_descriptor(name: &str) -> Descriptor { let (platform, os) = match core { // The bare mock expect-pass artifact is `e2e-report` → core "report" or "". "" | "report" => ("Mock", "Linux"), + // The GitHub-hosted Windows lane: no GPU, native Windows. + "windows" => ("Mock", "Windows"), "gpu" => ("MI300X", "Linux"), "gpu-rad3" => ("R9700", "Linux"), "gpu-mi350p" => ("MI350P", "Linux"), @@ -1758,6 +1760,7 @@ mod tests { fn parse_descriptor_maps_known_artifacts() { for (name, platform, os) in [ ("e2e-report", "Mock", "Linux"), + ("e2e-windows-report", "Mock", "Windows"), ("e2e-gpu-report", "MI300X", "Linux"), ("e2e-gpu-rad3-report", "R9700", "Linux"), ("e2e-gpu-mi350p-report", "MI350P", "Linux"), diff --git a/docs/ci-hardware-testing.md b/docs/ci-hardware-testing.md index 2c581431b..7b1d96814 100644 --- a/docs/ci-hardware-testing.md +++ b/docs/ci-hardware-testing.md @@ -32,6 +32,7 @@ separate tier flag or tag filter to maintain. | Job | Workflow | Platform | Runner labels | |---|---|---|---| | `e2e` | `ci.yml` | Mock (no GPU) | GitHub-hosted `ubuntu-latest` | +| `e2e-windows` | `ci.yml` | Mock (no GPU) on Windows | GitHub-hosted `windows-latest` | | `e2e-gpu` | `e2e-selfhosted.yml` | MI300X (AMD Instinct, bare-metal Linux) | self-hosted `[self-hosted, linux, mi300x]` | | `e2e-gpu-strix-ubuntu` | `e2e-selfhosted.yml` | Strix Halo (gfx1151) on Ubuntu | self-hosted `[self-hosted, linux, devlab-dispatch, strix-halo]` | | `e2e-gpu-strix-windows` | `e2e-selfhosted.yml` | Strix Halo (gfx1151) on Windows 11 | self-hosted `[self-hosted, windows, devlab-dispatch, strix-halo]` | @@ -78,12 +79,21 @@ named for. `e2e` is the blocking, GitHub-hosted mock job: `@requires-gpu` scenarios resolve to skip here, and known bugs resolve to xfail from -`expectations.toml`. It is a required check and must stay green. +`expectations.toml`. It is a required check and must stay green. It also runs +every scenario that plants a simulated machine (see `docs/testing.md`). +`e2e-windows` runs the same suite on GitHub-hosted Windows, the platform where +the non-GPU scenarios meet the Windows probes and paths; it is advisory until +it has a record of being green. The self-hosted jobs (`e2e-gpu`, `e2e-gpu-strix-ubuntu`, `e2e-gpu-strix-windows`, -`e2e-wsl`, `e2e-gpu-rad3`, and `e2e-gpu-mi350p`) run on AMD GPU systems, so they -exercise host/GPU detection, engine `detect`/`capabilities`, and live serving -scenarios that the mock job cannot. GPU availability is advisory in the WSL lane, as +`e2e-wsl`, `e2e-gpu-rad3`, and `e2e-gpu-mi350p`) run on AMD GPU systems. All +but `e2e-wsl` run ONLY the scenarios that need a real GPU (`E2E_GPU_ONLY=1`): live serving, SDK installs, +and detection on the real machine. Everything else runs on the two hosted jobs +above, so a GPU runner's time is not spent on it. On a pull request and in the +merge queue they narrow further to the `@gpu-smoke` canaries (plus, in the +queue, the `@merge-queue` serves); the full real-GPU set runs on pushes to +`main` and `release/**`, on dispatch, and nightly. `e2e-wsl` has no usable GPU +and runs the whole non-GPU suite inside a real WSL2 distribution. GPU availability is advisory in the WSL lane, as described below. `e2e-wsl` runs on the DevLab Dispatch pool. Its WSL2/Ubuntu-24.04 distro is @@ -146,7 +156,7 @@ covers the mock platform; reports — including partial or failed runs — by scenario id into one HTML report and GitHub step summary. -The lane artifacts are named canonically (`e2e-report`, `e2e-gpu-report`, +The lane artifacts are named canonically (`e2e-report`, `e2e-windows-report`, `e2e-gpu-report`, `e2e-gpu-rad3-report`, `e2e-gpu-mi350p-report`, `e2e-gpu-strix-ubuntu-report`, `e2e-gpu-strix-windows-report`, `e2e-gpu-strix-wsl-report`) in `ci.yml` and `e2e-selfhosted.yml`, because the report derives each platform's name and OS diff --git a/docs/testing.md b/docs/testing.md index 465abcb5e..b89d40d62 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -182,12 +182,22 @@ The cucumber suite runs in one of two hardware modes, chosen by usable AMD GPU. Any other value aborts the run, so a misspelt `real` cannot silently skip the -real-GPU scenarios. To run them locally on a GPU host: +real-GPU scenarios. + +To run the real-GPU scenarios locally on a GPU host: ```bash E2E_HARDWARE=real cargo xtask e2e ``` +The self-hosted GPU lanes run only the scenarios that need their GPU +(`E2E_GPU_ONLY=1`); everything else runs on the GitHub-hosted `E2E tests` +(Linux) and `E2E tests (Windows)` jobs, and inside a real WSL2 distribution on +the WSL lane. On a pull request and in the merge queue the GPU lanes narrow +further, to the cheap per-engine `@gpu-smoke` canaries (and, in the queue, the +`@merge-queue` serves) with `E2E_GPU_SMOKE_ONLY=1`; the full real-GPU set runs +nightly and on pushes to `main` and `release/**`. + A scenario describes its machine with a Given step — `a machine with an AMD Instinct GPU`, `a machine with eight AMD Instinct GPUs`, `a bare-metal Linux machine with an AMD GPU`, `a WSL machine with an AMD GPU passed through`, `a diff --git a/tests/e2e-cucumber/README.md b/tests/e2e-cucumber/README.md index ac7c95324..306891c13 100644 --- a/tests/e2e-cucumber/README.md +++ b/tests/e2e-cucumber/README.md @@ -145,6 +145,8 @@ Scenarios carry stable-id and capability tags: | `@requires-os:` | Premise is OS-specific; skip on other OSes. | | `@serve-timeout:` | Lengthen the serve-readiness wait for a genuinely slow serve (e.g. a large model). | | `@nightly` | Expensive scenario skipped by default; included when `E2E_INCLUDE_NIGHTLY=1`. | +| `@merge-queue` | A heavy real-GPU serve (default-engine endpoint, inference, readiness) that runs only in the merge queue, which opts in with `E2E_MERGE_QUEUE=1`; the `@gpu-smoke` canaries cover the same ground on the pull request. | +| `@gpu-smoke` | A cheap per-engine real-GPU canary. On a pull request and in the merge queue the self-hosted GPU lanes set `E2E_GPU_SMOKE_ONLY=1`, under which these (and, in the queue, the `@merge-queue` serves) are the only real-GPU scenarios that run; every other one resolves to **skip** with that reason. Scenarios that need no GPU never run on a GPU lane: those lanes set `E2E_GPU_ONLY=1` and leave them to the hosted Linux and Windows jobs. | | `@lifecycle` | Expensive, OS-mutating release-lifecycle scenario (packaging + real installer + install/uninstall). Skipped by default; included when `E2E_INCLUDE_LIFECYCLE=1`. `E2E_ONLY_LIFECYCLE=1` selects only this set without bypassing expectation resolution. | Known bugs are **not** tagged in the `.feature` files — they live in @@ -164,7 +166,15 @@ serve does not compete with the first for device memory, and the failure quotes the service log tail plus the device's free-VRAM state, which is where the engine's own reason for the stall is recorded. -CI runs one job per platform, each executing the full suite. The mock job lives +CI runs one job per platform. The two GitHub-hosted jobs, `e2e` on Linux and +`e2e-windows` on Windows, run every scenario that needs no GPU on every pull +request — on Linux including every GPU scenario that plants a simulated +machine. The self-hosted GPU jobs run only the scenarios that need their GPU +(`E2E_GPU_ONLY=1`), and on a pull request and in the merge queue only the +`@gpu-smoke` canaries of those (plus, in the queue, the `@merge-queue` serves); +they run the full real-GPU set on pushes to `main` and `release/**`, on +dispatch, and nightly. The WSL job has no GPU and runs the whole non-GPU suite +inside a real WSL2 distribution. The mock job lives in the `CI` workflow (`ci.yml`); the self-hosted GPU jobs live in a separate `E2E self-hosted` workflow (`e2e-selfhosted.yml`) so a job queued on an offline self-hosted runner can never stall `ci.yml`'s merge-required checks: @@ -172,6 +182,7 @@ self-hosted runner can never stall `ci.yml`'s merge-required checks: | Job | Workflow | Platform | Blocking | |---|---|---|---| | `e2e` | `ci.yml` | Mock (no GPU, GitHub-hosted) | yes | +| `e2e-windows` | `ci.yml` | Mock / Windows (no GPU, GitHub-hosted) | no | | `e2e-gpu` | `e2e-selfhosted.yml` | MI300X (self-hosted) | no | | `e2e-gpu-strix-ubuntu` | `e2e-selfhosted.yml` | Strix Halo / Ubuntu (self-hosted) | no | | `e2e-gpu-strix-windows` | `e2e-selfhosted.yml` | Strix Halo / Windows (self-hosted) | no | diff --git a/tests/e2e-cucumber/features/model_serving.feature b/tests/e2e-cucumber/features/model_serving.feature index e5f59fb4e..fa364ab1d 100644 --- a/tests/e2e-cucumber/features/model_serving.feature +++ b/tests/e2e-cucumber/features/model_serving.feature @@ -32,11 +32,11 @@ Feature: Model serving # vLLM serve + inference (safetensors model). Engine coverage: vLLM. This is the # deliberate vLLM half of a per-engine pair with `serve-lemonade-inference` # below, so it stays pinned to vLLM (the slug names the engine). It is also the - # vLLM per-PR canary: one real vLLM serve runs on every PR so a broken serve is - # caught before merge, while the heavier `@merge-queue` serves + # vLLM per-PR canary (`@gpu-smoke`): one real vLLM serve runs on every PR so a + # broken serve is caught before merge, while the heavier `@merge-queue` serves # (`serve-default-engine-working-endpoint`, `serve-default-engine-inference`, # and `serve-readiness-contract`) run only in the merge queue. - @id:serve-vllm-inference @requires-real-gpu @requires-engine:vllm + @id:serve-vllm-inference @requires-real-gpu @requires-engine:vllm @gpu-smoke Scenario: serve-05 - A served model responds to inference requests on vLLM Given a managed runtime is active And a model is being served on GPU @@ -59,9 +59,9 @@ Feature: Model serving And the response identifies the correct model # Lemonade serve + inference (GGUF model). Engine coverage: Lemonade. The - # lemonade per-PR canary: one real lemonade serve runs on every PR (the - # counterpart to the vLLM canary above). - @id:serve-lemonade-inference @requires-real-gpu @requires-engine:lemonade + # lemonade per-PR canary (`@gpu-smoke`): one real lemonade serve runs on every + # PR (the counterpart to the vLLM canary above). + @id:serve-lemonade-inference @requires-real-gpu @requires-engine:lemonade @gpu-smoke Scenario: serve-07 - A model served on lemonade responds to inference requests Given a managed runtime is active And a GGUF model is being served on lemonade @@ -77,7 +77,7 @@ Feature: Model serving # unrelated EAI-7423 xfail. Reuses the same small Qwen3-0.6B-GGUF checkpoint as # `serve-lemonade-inference` (cache-shared, no extra download) so this stays a # fast per-PR canary rather than needing the @nightly large-checkpoint path. - @id:serve-hf-checkpoint-inference @requires-real-gpu @requires-engine:lemonade + @id:serve-hf-checkpoint-inference @requires-real-gpu @requires-engine:lemonade @gpu-smoke Scenario: serve-08 - A canonical Hugging Face checkpoint serves and responds to inference Given a managed runtime is active And a canonical Hugging Face GGUF checkpoint is being served on lemonade diff --git a/tests/e2e-cucumber/src/capability.rs b/tests/e2e-cucumber/src/capability.rs index a7fe4dbfa..931c1cf0d 100644 --- a/tests/e2e-cucumber/src/capability.rs +++ b/tests/e2e-cucumber/src/capability.rs @@ -640,7 +640,12 @@ fn derive_platform_slug( }; } if !has_amd_gpu { - return "mock".to_owned(); + // The hosted Windows lane has no GPU either; give it its own column. + return if os_normalized(os_family) == "windows" { + "mock-windows".to_owned() + } else { + "mock".to_owned() + }; } match gfx_target { Some(t) => { @@ -950,6 +955,11 @@ Local model engines #[test] fn platform_slug_derivation() { assert_eq!(derive_platform_slug(false, None, "other", false), "mock"); + // The hosted Windows lane must not collide with the hosted Linux one. + assert_eq!( + derive_platform_slug(false, None, "windows", false), + "mock-windows" + ); // gfx950 normalizes to a `-dcgpu` family like gfx94x does, but it is a // different part with its own lane and report column — it must not be // slugged as mi300x. diff --git a/tests/e2e-cucumber/src/expectation.rs b/tests/e2e-cucumber/src/expectation.rs index 9c2c2d35c..af67d0f21 100644 --- a/tests/e2e-cucumber/src/expectation.rs +++ b/tests/e2e-cucumber/src/expectation.rs @@ -36,6 +36,7 @@ const SERVE_TIMEOUT_PREFIX: &str = "serve-timeout:"; const NIGHTLY_TAG: &str = "nightly"; const LIFECYCLE_TAG: &str = "lifecycle"; const MERGE_QUEUE_TAG: &str = "merge-queue"; +const GPU_SMOKE_TAG: &str = "gpu-smoke"; // `@serial` deliberately has no entry here, and `from_tags` below silently // ignores it like any other unrecognized tag: it isn't an expectation- @@ -210,6 +211,10 @@ pub struct ScenarioDecl { /// `E2E_MERGE_QUEUE`. Keeps the PR feedback loop short while still exercising /// the full serve matrix before a change lands. pub merge_queue: bool, + /// `@gpu-smoke`: one of the cheap per-engine real-GPU canaries that still + /// run on a pull request and in the merge queue, where a GPU lane runs no + /// other real-GPU scenario (`E2E_GPU_SMOKE_ONLY`, see [`restrict_to_smoke`]). + pub gpu_smoke: bool, } impl ScenarioDecl { @@ -231,6 +236,7 @@ impl ScenarioDecl { let mut nightly = false; let mut lifecycle = false; let mut merge_queue = false; + let mut gpu_smoke = false; for tag in tags { let tag = tag .as_ref() @@ -269,6 +275,8 @@ impl ScenarioDecl { lifecycle = true; } else if tag == MERGE_QUEUE_TAG { merge_queue = true; + } else if tag == GPU_SMOKE_TAG { + gpu_smoke = true; } } Self { @@ -287,6 +295,7 @@ impl ScenarioDecl { nightly, lifecycle, merge_queue, + gpu_smoke, } } @@ -669,6 +678,62 @@ pub fn resolve( Expectation::ExpectPass } +/// Which part of the suite a lane runs, beyond what the host can run at all. +/// +/// Both default to `false`: a lane runs everything its host can. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct LaneSelection { + /// `E2E_GPU_ONLY`: a self-hosted GPU lane runs only the scenarios that need + /// the real hardware it has. Everything else runs on the GitHub-hosted + /// Linux and Windows lanes, where a GPU runner's time is not spent on it. + pub gpu_only: bool, + /// `E2E_GPU_SMOKE_ONLY`: on a pull request and in the merge queue, a GPU + /// lane's real-GPU scenarios narrow further to the `@gpu-smoke` canaries + /// (and, in the queue, the `@merge-queue` serves). + pub smoke_only: bool, +} + +/// Whether a scenario needs the real GPU of the machine it runs on — a usable +/// device, or a detected chip name — as opposed to running anywhere, simulated +/// machines included. +const fn needs_real_gpu(decl: &ScenarioDecl) -> bool { + decl.requires_gpu || decl.requires_gfx_target +} + +/// Narrow a scenario's resolution to what this lane runs (see [`LaneSelection`]). +/// +/// A scenario outside the lane's share resolves to `Skip` with the reason +/// recorded, so the report shows it as not applicable on that lane rather than +/// as a result that never arrived. An expectation that already resolved to +/// `Skip` keeps its own, more precise reason. +#[must_use] +pub fn restrict_to_lane( + decl: &ScenarioDecl, + expectation: Expectation, + lane: LaneSelection, +) -> Expectation { + if matches!(expectation, Expectation::Skip { .. }) { + return expectation; + } + let real_gpu = needs_real_gpu(decl); + if lane.gpu_only && !real_gpu { + return Expectation::Skip { + reason: "this GPU lane runs only scenarios that need its GPU (E2E_GPU_ONLY); \ + the rest run on the GitHub-hosted Linux and Windows lanes" + .to_owned(), + }; + } + if lane.smoke_only && real_gpu && !decl.gpu_smoke && !decl.merge_queue { + return Expectation::Skip { + reason: "this GPU lane runs only the @gpu-smoke real-GPU canaries on a pull \ + request (E2E_GPU_SMOKE_ONLY); the full GPU suite runs nightly and on \ + pushes" + .to_owned(), + }; + } + expectation +} + /// Tiny glob: `*` matches any run of chars. Used only for `therock_family`. fn glob_match(pattern: &str, text: &str) -> bool { let pattern = pattern.to_ascii_lowercase(); @@ -1979,4 +2044,78 @@ flaky = true } } } + + #[test] + fn a_gpu_lane_runs_only_its_gpu_scenarios_narrowed_to_the_smoke_test_on_prs() { + let xfail = || Expectation::ExpectXfail { + bug: "B-1".to_owned(), + reason: "known".to_owned(), + flaky: false, + }; + let canary = decl(&["id:serve-vllm", "requires-real-gpu", "gpu-smoke"]); + let queue_serve = decl(&["id:serve-default", "requires-real-gpu", "merge-queue"]); + let heavy = decl(&["id:runtime-install", "requires-real-gpu"]); + let gfx = decl(&["id:runtime-resolve", "requires-gfx-target"]); + let no_gpu = decl(&["id:examine-version"]); + let simulated = decl(&["id:examine-detects-wsl", "requires-os:linux"]); + let skip_reason = |e: Expectation| match e { + Expectation::Skip { reason } => reason, + other => panic!("expected skip, got {other:?}"), + }; + + let gpu_lane = LaneSelection { + gpu_only: true, + smoke_only: false, + }; + // A GPU lane leaves everything that needs no GPU, simulated machines + // included, to the hosted lanes — and says so. + for d in [&no_gpu, &simulated] { + assert!( + skip_reason(restrict_to_lane(d, Expectation::ExpectPass, gpu_lane)) + .contains("E2E_GPU_ONLY") + ); + } + // It keeps every scenario that needs its hardware, resolution intact. + for d in [&canary, &queue_serve, &heavy, &gfx] { + assert_eq!(restrict_to_lane(d, xfail(), gpu_lane), xfail()); + } + + let pr_gpu_lane = LaneSelection { + gpu_only: true, + smoke_only: true, + }; + // On a pull request only the canaries and the queue serves remain. + assert_eq!(restrict_to_lane(&canary, xfail(), pr_gpu_lane), xfail()); + assert_eq!( + restrict_to_lane(&queue_serve, Expectation::ExpectPass, pr_gpu_lane), + Expectation::ExpectPass + ); + for d in [&heavy, &gfx] { + assert!( + skip_reason(restrict_to_lane(d, xfail(), pr_gpu_lane)) + .contains("E2E_GPU_SMOKE_ONLY") + ); + } + + // A lane without a GPU (the WSL lane sets only the smoke switch) keeps + // its non-GPU scenarios. + let smoke_only = LaneSelection { + gpu_only: false, + smoke_only: true, + }; + assert_eq!( + restrict_to_lane(&no_gpu, Expectation::ExpectPass, smoke_only), + Expectation::ExpectPass + ); + // A skip that already happened keeps its own, more precise reason. + let skip = Expectation::Skip { + reason: "requires os 'linux'".to_owned(), + }; + assert_eq!(restrict_to_lane(&no_gpu, skip.clone(), pr_gpu_lane), skip); + // A lane with no selection changes nothing. + assert_eq!( + restrict_to_lane(&heavy, Expectation::ExpectPass, LaneSelection::default()), + Expectation::ExpectPass + ); + } } diff --git a/tests/e2e-cucumber/tests/e2e.rs b/tests/e2e-cucumber/tests/e2e.rs index 1df20ed4f..929805a5d 100644 --- a/tests/e2e-cucumber/tests/e2e.rs +++ b/tests/e2e-cucumber/tests/e2e.rs @@ -1340,6 +1340,14 @@ async fn main() { // per-engine canary covers them on the PR fast path); set by // e2e-selfhosted.yml on the `merge_group` event. let include_merge_queue = std::env::var_os("E2E_MERGE_QUEUE").is_some_and(|v| v == "1"); + // Which share of the suite this lane runs: a self-hosted GPU lane runs only + // what needs its GPU, narrowed to the `@gpu-smoke` canaries on a pull + // request; see `restrict_to_lane`. Selected inside the filter, like + // `E2E_ONLY_LIFECYCLE`, so expectation resolution still runs. + let lane = e2e_cucumber::expectation::LaneSelection { + gpu_only: std::env::var_os("E2E_GPU_ONLY").is_some_and(|v| v == "1"), + smoke_only: std::env::var_os("E2E_GPU_SMOKE_ONLY").is_some_and(|v| v == "1"), + }; // `@requires-real-gpu` scenarios run only where a lane on real hardware opts // in; everywhere else GPU behaviour is exercised against a simulated host. // An unrecognised value aborts the run instead of silently skipping them. @@ -1464,6 +1472,8 @@ async fn main() { hardware, }, ); + let expectation = + e2e_cucumber::expectation::restrict_to_lane(&decl, expectation, lane); let run = (!only_lifecycle || decl.lifecycle) && !matches!(expectation, Expectation::Skip { .. }); if let Some(id) = &decl.id { diff --git a/xtask/src/e2e_report.rs b/xtask/src/e2e_report.rs index 07ed4b928..3d4b0e44d 100644 --- a/xtask/src/e2e_report.rs +++ b/xtask/src/e2e_report.rs @@ -130,6 +130,7 @@ fn label_for_root_report(dir: &Path) -> String { }); match slug.as_deref() { Some("mock") => "e2e-report".to_owned(), + Some("mock-windows") => "e2e-windows-report".to_owned(), Some("mi300x") => "e2e-gpu-report".to_owned(), Some("gfx1201") => "e2e-gpu-rad3-report".to_owned(), Some("mi350p") => "e2e-gpu-mi350p-report".to_owned(), @@ -198,6 +199,7 @@ mod tests { /// claiming an identity it has no business asserting. const CANONICAL_REPORT_ARTIFACTS: &[&str] = &[ "e2e-report", + "e2e-windows-report", "e2e-gpu-report", "e2e-gpu-rad3-report", "e2e-gpu-mi350p-report", @@ -330,6 +332,7 @@ mod tests { ("strix-halo-windows", "e2e-gpu-strix-windows-report"), ("strix-halo-wsl", "e2e-gpu-strix-wsl-report"), ("mock", "e2e-report"), + ("mock-windows", "e2e-windows-report"), ] { let tmp = tempfile::tempdir().expect("tempdir"); let root = tmp.path(); diff --git a/xtask/src/workflow_contract.rs b/xtask/src/workflow_contract.rs index 89c0a7758..8e7d54438 100644 --- a/xtask/src/workflow_contract.rs +++ b/xtask/src/workflow_contract.rs @@ -816,24 +816,93 @@ trigger-a-workflow#triggering-a-workflow-from-a-workflow" ); } + /// On a pull request and in the merge queue every self-hosted GPU lane + /// narrows its real-GPU scenarios to the `@gpu-smoke` canaries, while the + /// heavy `@merge-queue` serves stay opted in for the merge queue alone. + /// Pushes, dispatches and nightly set neither, so they run the full suite. #[test] - fn self_hosted_wsl_enables_merge_queue_scenarios_only_for_merge_group() { + fn self_hosted_gpu_lanes_run_only_the_gpu_smoke_test_on_prs_and_in_the_merge_queue() { + const SMOKE_ONLY: &str = "${{ (github.event_name == 'merge_group' || github.event_name == 'pull_request') && '1' || '' }}"; + const QUEUE_ONLY: &str = "${{ github.event_name == 'merge_group' && '1' || '' }}"; let sh = read_workflow("e2e-selfhosted.yml"); - let wsl = job_block(&sh, "e2e-wsl"); - let env = job_mapping(wsl, "env"); - assert_eq!( - env.get("E2E_MERGE_QUEUE").map(String::as_str), - Some("${{ github.event_name == 'merge_group' && '1' || '' }}"), - "e2e-wsl's job-level env must opt into @merge-queue scenarios for merge_group only" - ); + for lane in [ + "e2e-gpu", + "e2e-gpu-strix-ubuntu", + "e2e-gpu-strix-windows", + "e2e-wsl", + "e2e-gpu-rad3", + "e2e-gpu-mi350p", + ] { + let env = job_mapping(job_block(&sh, lane), "env"); + assert_eq!( + env.get("E2E_GPU_SMOKE_ONLY").map(String::as_str), + Some(SMOKE_ONLY), + "{lane} must narrow its real-GPU scenarios to the smoke test on pull \ + requests and in the merge queue, and nowhere else" + ); + assert_eq!( + env.get("E2E_MERGE_QUEUE").map(String::as_str), + Some(QUEUE_ONLY), + "{lane} must opt into the @merge-queue serves for merge_group only" + ); + // A GPU lane leaves everything that needs no GPU to the hosted Linux + // and Windows lanes. The WSL lane has no GPU, so it is the one + // self-hosted lane that must keep running those. + let gpu_only = env.get("E2E_GPU_ONLY").map(String::as_str); + if lane == "e2e-wsl" { + assert_eq!( + gpu_only, None, + "e2e-wsl has no GPU and must run the non-GPU suite" + ); + } else { + assert_eq!( + gpu_only, + Some("1"), + "{lane} must run only its GPU scenarios" + ); + } + } let nightly = read_workflow("nightly.yml"); let nightly_wsl = job_block(&nightly, "e2e-wsl-nightly"); let nightly_env = job_mapping(nightly_wsl, "env"); - assert!( - !nightly_env.contains_key("E2E_MERGE_QUEUE"), - "the nightly WSL job cannot receive merge_group events and must not opt into @merge-queue scenarios" - ); + for key in ["E2E_MERGE_QUEUE", "E2E_GPU_SMOKE_ONLY"] { + assert!( + !nightly_env.contains_key(key), + "the nightly WSL job runs the full suite and must not set {key}" + ); + } + } + + /// The WSL lane runs the suite inside the distro, which sees only the + /// variables WSLENV names. One left out there is silently unset: dropping + /// `E2E_GPU_SMOKE_ONLY` would run the full GPU suite on every pull request, + /// dropping `E2E_HARDWARE` would skip every real-GPU scenario, and neither + /// would ever turn the lane red. + #[test] + fn wsl_lanes_forward_every_e2e_variable_into_the_distro() { + for (workflow, job) in [ + ("e2e-selfhosted.yml", "e2e-wsl"), + ("nightly.yml", "e2e-wsl-nightly"), + ] { + let text = read_workflow(workflow); + let block = job_block(&text, job); + let forwarded: Vec<&str> = block + .lines() + .find_map(|line| line.trim().strip_prefix("$env:WSLENV = '")) + .and_then(|rest| rest.strip_suffix('\'')) + .unwrap_or_else(|| panic!("{workflow} job {job} sets no WSLENV")) + .split(':') + .collect(); + for key in job_mapping(block, "env").keys() { + if key.starts_with("E2E_") { + assert!( + forwarded.contains(&key.as_str()), + "{workflow} job {job} sets {key} but does not forward it through WSLENV" + ); + } + } + } } #[test] @@ -1807,10 +1876,11 @@ esac "the E2E README must name every nightly self-hosted lane, in workflow order" ); - // The README's CI job table is the per-PR view: the blocking mock job - // from ci.yml, then one row per self-hosted lane. Only the self-hosted - // rows are derivable, so the mock row is matched by workflow and the - // rest are compared against e2e-selfhosted.yml. + // The README's CI job table is the per-PR view: the two GitHub-hosted + // jobs from ci.yml (the blocking Linux mock job and the Windows one), + // then one row per self-hosted lane. Only the self-hosted rows are + // derivable, so the hosted rows are matched by workflow and the rest + // are compared against e2e-selfhosted.yml. let readme_rows = markdown_table_rows(&readme, "| Job | Workflow | Platform | Blocking |"); let (mock_rows, self_hosted_rows): (Vec<_>, Vec<_>) = readme_rows .into_iter() @@ -1820,8 +1890,8 @@ esac .iter() .map(|row| row[0].clone()) .collect::>(), - vec!["`e2e`".to_owned()], - "the README CI job table must carry exactly one blocking mock row" + vec!["`e2e`".to_owned(), "`e2e-windows`".to_owned()], + "the README CI job table must carry exactly the two hosted E2E rows" ); assert_eq!( self_hosted_rows