diff --git a/.github/actions/validate-runner/action.yml b/.github/actions/validate-runner/action.yml index cd78a73a..61b643ba 100644 --- a/.github/actions/validate-runner/action.yml +++ b/.github/actions/validate-runner/action.yml @@ -12,6 +12,13 @@ inputs: artifacts must be in their build paths required: false default: "false" + warp-probe: + description: >- + Also run the guest warp probe (doctor H6) on CI's schedule, for jobs + that boot guests without the microVM scenarios' own probes; requires + verify-openvmm + required: false + default: "false" runs: using: composite @@ -160,23 +167,20 @@ runs: throw "Hyper-V SVGA firmware was not found: $svgaFirmware" } - # Host time. On Linux, the nonstop_tsc gate stays fail-closed in front of - # the time ABI host qualification (doc/design/time-abi.md, "Host - # qualification"), which then checks the backend, the CPU fingerprint and + # Host time. The time ABI host qualification (doc/design/time-abi.md, + # "Host qualification") checks the backend, the CPU fingerprint and # generation, and the TSC rate stability on its short CI schedule, and, # in jobs that downloaded OpenVMM, its CPU profile check (H2) and - # preflight (H3); other jobs qualify without OpenVMM. These checks are - # host-level, and the guest warp probe (H6) runs only in the microVM - # scenarios, so the gate keeps a host whose guests warp (#265) out of - # every other job. - - name: Validate host TSC + # preflight (H3); other jobs qualify without OpenVMM. Jobs whose guests + # run without the microVM scenarios' warp probes add the guest warp probe + # (H6). The host's invariant-TSC flags are evidence only: the guest warp + # probe measures the cross-vCPU skew that a host without one can cause + # (#265). + - name: Report host TSC if: runner.os != 'Windows' shell: bash run: | set -euo pipefail - # Guests on a host without an invariant TSC intermittently see - # cross-vCPU TSC warps, which make Linux mark the guest TSC unstable - # (#211). echo "Kernel: $(uname -r)" echo "CPU: $(awk -F': ' '/^model name/ { print $2; exit }' /proc/cpuinfo)" echo "Clocksource: $(cat /sys/devices/system/clocksource/clocksource0/current_clocksource 2>/dev/null || echo unknown)" @@ -190,8 +194,7 @@ runs: case "${flags}" in *" nonstop_tsc "*) ;; *) - echo "::error::Runner host does not expose an invariant TSC (nonstop_tsc); redeploy the VM on a host that does" >&2 - exit 1 + echo "::notice::Runner host does not expose an invariant TSC (nonstop_tsc); the guest warp probe measures the cross-vCPU skew that this can cause (#265)" ;; esac @@ -202,6 +205,7 @@ runs: python3 scripts/nvx.py doctor --backend "${{ inputs.backend }}" --checks H1 H2 H4 + ${{ inputs.warp-probe == 'true' && 'H6' || '' }} ${{ inputs.verify-openvmm == 'true' && 'H3 --cpu-fingerprint "${RUNNER_TEMP}/nvx-cpu-fingerprint.json"' || '--no-openvmm' }} --ci-schedule --summary "${GITHUB_STEP_SUMMARY}" @@ -213,6 +217,7 @@ runs: python scripts\nvx.py doctor --backend "${{ inputs.backend }}" --checks H1 H2 H4 + ${{ inputs.warp-probe == 'true' && 'H6' || '' }} ${{ inputs.verify-openvmm == 'true' && 'H3 --cpu-fingerprint "$env:RUNNER_TEMP\nvx-cpu-fingerprint.json"' || '--no-openvmm' }} --ci-schedule --summary "$env:GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/copilot-setup-steps.yml b/.github/workflows/copilot-setup-steps.yml index 0164d551..3685acbd 100644 --- a/.github/workflows/copilot-setup-steps.yml +++ b/.github/workflows/copilot-setup-steps.yml @@ -180,7 +180,7 @@ jobs: fi ls -l /dev/kvm if ! grep -qw nonstop_tsc /proc/cpuinfo; then - echo "::notice::The host CPU has no invariant TSC; snapshot restore tests may be unstable" + echo "::notice::The host CPU has no invariant TSC; the microVM scenarios' guest warp probe measures the cross-vCPU skew that this can cause" fi - name: Restore guest artifacts built by CI diff --git a/.github/workflows/run-platform.yml b/.github/workflows/run-platform.yml index eb448eab..3170c5ea 100644 --- a/.github/workflows/run-platform.yml +++ b/.github/workflows/run-platform.yml @@ -108,12 +108,15 @@ jobs: shell: bash run: chmod +x openvmm/target/release/openvmm - # Qualification verifies the downloaded OpenVMM's time ABI preflight. + # Qualification verifies the downloaded OpenVMM's time ABI preflight and, + # because the benchmarks boot guests without the microVM scenarios' warp + # probes, runs the guest warp probe. - name: Validate runner uses: ./.github/actions/validate-runner with: backend: ${{ inputs.backend }} verify-openvmm: "true" + warp-probe: "true" - name: Run benchmark uses: ./.github/actions/run-benchmark diff --git a/doc/ci.md b/doc/ci.md index 6972e67e..d624fd75 100644 --- a/doc/ci.md +++ b/doc/ci.md @@ -167,29 +167,32 @@ builds the debug kernel beside the shared `artifacts` job own key, so a kernel rebuild delays only the debug jobs, and a failed debug kernel build fails the required status check and blocks the release. -Every job that uses the `validate-runner` action first requires an invariant -TSC on a Linux runner (`nonstop_tsc` in `/proc/cpuinfo`) and fails without -one, as before the time ABI. It then qualifies the runner for the time ABI -with `nvx.py doctor --checks H1 H2 H4 --ci-schedule` (see [Host -qualification](#host-qualification)): the backend, the CPU fingerprint and -generation, and the TSC rate stability on its short schedule. The microVM and -platform jobs run it after downloading OpenVMM and add OpenVMM's CPU profile -check (H2) and preflight (H3), and keep the fingerprint as an artifact when -the profile check fails; the other jobs pass `--no-openvmm`. This takes a few -seconds. It reports the CPU generation, the CPU profile that `auto` selects, -and the measured TSC rate in the log and the job summary, and fails the job -with a stable code when the runner is not qualified, for example -`E_PROFILE_HOST_UNKNOWN` on an unknown CPU generation. The doctor gates only -on measured properties, alike on every backend: it records the host OS's -invariant-TSC flags and clocksource as evidence, and the guest warp probe in -the microVM scenarios measures the skew that a host without an invariant TSC -causes. On such an MSHV runner VM, never-restored guests hit cross-vCPU TSC -warps when an idle host CPU woke (#211, #265), which is why the probe schedule -includes idle gaps. The `nonstop_tsc` gate stays in front of it until every -job that runs guests also runs the warp probe: only the microVM jobs do, while -the OpenVMM vmm-tests and the platform benchmarks run guests after host-level -checks that such a host can pass. Runner labels do not encode the generation; -per-PR CI captures and restores on one runner, so generations never mix. +Every job that uses the `validate-runner` action first reports the Linux +runner's kernel, CPU, clocksource, and TSC flags. It then qualifies the runner +for the time ABI with `nvx.py doctor --checks H1 H2 H4 --ci-schedule` (see +[Host qualification](#host-qualification)): the backend, the CPU fingerprint +and generation, and the TSC rate stability on its short schedule. The microVM +and platform jobs run it after downloading OpenVMM and add OpenVMM's CPU +profile check (H2) and preflight (H3), and keep the fingerprint as an artifact +when the profile check fails; the other jobs pass `--no-openvmm`. This takes a +few seconds. It reports the CPU generation, the CPU profile that `auto` +selects, and the measured TSC rate in the log and the job summary, and fails +the job with a stable code when the runner is not qualified, for example +`E_PROFILE_HOST_UNKNOWN` on an unknown CPU generation. The platform jobs also +run the guest warp probe (H6) on CI's schedule, which adds about 25 s, because +their benchmarks boot guests without the microVM scenarios' probes. The doctor +gates only on measured properties, alike on every backend: it records the host +OS's invariant-TSC flags and clocksource as evidence, and the guest warp probe +measures the skew that a host without an invariant TSC causes. Before the time +ABI, never-restored guests on such an MSHV runner VM, an Ice Lake 8370C, hit +cross-vCPU TSC warps when an idle host CPU woke (#211, #265), which is why the +probe schedules include idle gaps. Under the time ABI, its guests' warp probes +measure at most 36 ns, so `validate-runner` only notes a missing +`nonstop_tsc`, and the Linux runner setup script only warns. The OpenVMM +vmm-tests run no warp probe: they boot OpenVMM's own test guests, whose +verdicts don't depend on the skew bound. Runner labels do not encode the +generation; per-PR CI captures and restores on one runner, so generations +never mix. The warp probe replaced the `restore-tsc-sync` scenario, its test-only `clearcpuid=tsc_adjust` kernel option, the guest's scan of the kernel log for @@ -375,16 +378,16 @@ cannot run. The doctor gates on measured properties, alike on every backend: the guest warp probe at 1 µs with its idle gaps (H6), the TSC rate stability (H4), and the CPU profile (H2 and H3). The host OS's invariant-TSC flags and clocksource -are evidence only in the doctor. In CI, `validate-runner` keeps the -`nonstop_tsc` gate on Linux runners in front of it, because only the microVM -jobs run H6, and runs the cheap host-level checks H1, H2, and H4 in every job -that uses it, on the short `--ci-schedule`. The microVM and platform jobs run -it after downloading OpenVMM and the guest artifacts and add H2's CPU profile -check and H3, so that H4 also compares the measured rate with the backend's -native rate; the other jobs pass `--no-openvmm`. The guest warp probe needs a -time ABI boot, so the microVM boot and restore scenarios run CI's warp -schedule after every boot and restore and assert its verdict. H5 and H7 remain -available for interactive qualification. +are evidence only, in the doctor and in CI. In CI, `validate-runner` runs the +cheap host-level checks H1, H2, and H4 in every job that uses it, on the short +`--ci-schedule`. The microVM and platform jobs run it after downloading +OpenVMM and the guest artifacts and add H2's CPU profile check and H3, so that +H4 also compares the measured rate with the backend's native rate; the other +jobs pass `--no-openvmm`. The guest warp probe needs a time ABI boot: the +platform jobs add H6 on CI's schedule before their benchmarks, and the microVM +boot and restore scenarios run CI's warp schedule after every boot and restore +and assert its verdict. H5 and H7 remain available for interactive +qualification. ## Rust toolchain diff --git a/doc/design/time-abi.md b/doc/design/time-abi.md index f82673bd..f2936349 100644 --- a/doc/design/time-abi.md +++ b/doc/design/time-abi.md @@ -1789,17 +1789,17 @@ failure by the status together with its `NVX-TIME-ABI-VIOLATION` event. `nvx.py doctor --backend ` qualifies a host by running every check, with `H4` and `H6` on their long schedules. The `validate-runner` action runs -`H1`, `H2`, and the short `H4` before every CI job and `H3` in microVM jobs, -and every CI microVM boot and restore scenario runs the CI warp schedule; -`H5` and `H7` run in `doctor` only. Both tools print one -`NVX-DOCTOR: check= status= detail=...` line per check and -fail if any check fails. In CI, `validate-runner` first requires -`nonstop_tsc` in `/proc/cpuinfo` on Linux runners and fails without it, as -before the time ABI, and the Linux runner setup script requires it at -provisioning. This gate stays until every job that runs guests also runs -`H6`: only the microVM jobs do, while the OpenVMM vmm-tests and the platform -benchmarks run guests after host-level checks, which a host without an -invariant TSC can pass. +`H1`, `H2`, and the short `H4` before every CI job, `H3` in microVM and +platform jobs, and `H6` on the CI schedule in platform jobs, whose benchmarks +boot guests without the scenarios' probes; every CI microVM boot and restore +scenario runs the CI warp schedule. `H5` and `H7` run in `doctor` only. Both +tools print one `NVX-DOCTOR: check= status= detail=...` line +per check and fail if any check fails. In CI, `validate-runner` reports +`nonstop_tsc` in `/proc/cpuinfo` on Linux runners as evidence, and the Linux +runner setup script warns without it; neither gates on it (see [Qualification +gates](#qualification-gates)). The OpenVMM vmm-tests run no warp probe: they +boot OpenVMM's own test guests, whose verdicts don't depend on the cross-vCPU +skew bound. | ID | Check | | --- | --- | @@ -1900,15 +1900,17 @@ of the warps in #265: passes the boot check, and runs the probe five times, with idle gaps of 0.1, 1, 5, and 1 s; a 1-vCPU microVM then boots and runs it once. - CI: every microVM boot and restore scenario runs the probe twice, with a - 1 s idle gap, after boot and after every restore. + 1 s idle gap, after boot and after every restore. The platform jobs run + `H6` on this schedule before their benchmarks: the larger microVM runs the + probe twice, 1 s apart, and the 1-vCPU microVM runs it once. ### Qualification gates The doctor qualifies alike on every backend, on measured properties only: the CPU profile (`H2`, `H3`), rate stability (`H4`), and the idle-scheduled warp -probe (`H6`, and the CI schedule in every microVM job). It records the host -OS's invariant-TSC bit and the host clocksource as evidence and never gates -on them, on any backend: +probe (`H6`, and the CI schedule in every microVM and platform job). It +records the host OS's invariant-TSC bit and the host clocksource as evidence +and never gates on them, on any backend: - On Azure, WHP and nested MSHV cannot offer the invariant-TSC bit to their guests through their feature banks, although the host OS sees an invariant @@ -1924,10 +1926,14 @@ on them, on any backend: cause. That 8370C MSHV runner, whose host OS lacks the bit and which showed the -#265 warps, is out of rotation and unqualified because our account cannot run -guests there, not because of the doctor's rules; CI's `nonstop_tsc` gate -would also reject it. Its host-level warp probe saw backward steps of at -most 2.1 ns. +#265 warps before the time ABI, qualifies. With `H4` and `H6` on their long +schedules, `H1` to `H7` pass, and `H6` measures at most 26 ns; its CPU profile +check passes with the same surface digest before and after the MSHV probe +partition fix (#411). Its guests' warp probes, over the full microVM suite on +the production and debug kernels and ten more rounds of `smp`, +`smp-snapshot`, and `restore-processors` (288 runs), measured at most 36 ns of +offset and 4 ns of backward step. Its host-level warp probe saw backward steps +of at most 2.1 ns. ### Generations and runner placement @@ -2354,9 +2360,9 @@ Migration impact: profile of their generation pins. - Hosts that fail qualification cannot run microVMs on a pinned profile until replaced; an Intel development host can boot on a host profile - (`--cpu-profile host`) instead. - One 8370C MSHV runner stays out of rotation and unqualified because our - account cannot run guests there. + (`--cpu-profile host`) instead. A host without an invariant TSC, such as + the 8370C MSHV runner, qualifies when its guests stay within the skew + bound (see [Qualification gates](#qualification-gates)). - Per-PR CI captures and restores on the same runner. Same-generation cross-VM restore is validated by the fleet restore matrix on WHP only; no usable KVM or MSHV pair of one generation exists, so the simulated host diff --git a/doc/usage.md b/doc/usage.md index e288d6d5..1ccf285b 100644 --- a/doc/usage.md +++ b/doc/usage.md @@ -303,7 +303,7 @@ check and the subset that CI runs. | `--ci-schedule` | off | Run `H4` and `H6` on CI's short schedules: 3 rate samples 1 s apart instead of 13 samples 10 s apart, and two warp-probe runs instead of five. | | `--probe-dir PATH` | `$RUNNER_TOOL_CACHE/nvx-host-time-probe`, or `build/host-time-probe` | Select the cache directory for the host probe, which the doctor builds with `rustc`. | | `--summary PATH` | none | Append a Markdown summary, for example to `$GITHUB_STEP_SUMMARY`. | -| `--timeout SECONDS` | `120` | Set the seconds allowed for each probe or guest. | +| `--timeout SECONDS` | `120` | Set the seconds allowed for each probe or guest. `H4` gets its sampling window on top. | ### `test-aci-edge-sandboxes` diff --git a/scripts/nvx_tools/doctor.py b/scripts/nvx_tools/doctor.py index 78a3ad28..cf4b7f26 100644 --- a/scripts/nvx_tools/doctor.py +++ b/scripts/nvx_tools/doctor.py @@ -203,16 +203,20 @@ def build_probe(directory: Path) -> Path: def run_probe( - context: DoctorContext, *arguments: str + context: DoctorContext, *arguments: str, duration: float = 0.0 ) -> list[tuple[str, dict[str, str]]]: - """Run the host probe and parse its ``NVX-HOST-TIME-PROBE`` records.""" + """Run the host probe and parse its ``NVX-HOST-TIME-PROBE`` records. + + ``duration`` is the time the probe spends sleeping by design, which the + timeout allows on top of ``context.timeout``. + """ if context.probe is None: context.probe = build_probe(context.probe_directory) completed = subprocess.run( [os.fspath(context.probe), *arguments], capture_output=True, text=True, - timeout=context.timeout, + timeout=context.timeout + duration, check=False, ) if completed.returncode != 0: @@ -702,6 +706,8 @@ def check_rate(context: DoctorContext) -> CheckResult: clocksource = host_clocksource() context.facts["host_clocksource"] = clocksource schedule = context.schedule + # The probe sleeps between samples for the whole window, 120 s on the + # qualification schedule, so the timeout covers the window as well. records = run_probe( context, "rate", @@ -709,6 +715,7 @@ def check_rate(context: DoctorContext) -> CheckResult: str(schedule.rate_samples), "--interval-ms", str(schedule.rate_interval_ms), + duration=(schedule.rate_samples - 1) * schedule.rate_interval_ms / 1000, ) clocks = {fields.get("role"): fields for kind, fields in records if kind == "clock"} for role in ("rate", "stability"): @@ -1158,7 +1165,8 @@ def configure_parser(parser: argparse.ArgumentParser) -> None: "--timeout", type=float, default=120.0, - help="seconds allowed for each probe or guest (default: 120)", + help="seconds allowed for each probe or guest, on top of H4's sampling " + "window (default: 120)", ) parser.add_argument( "--openvmm-arg", diff --git a/scripts/setup/README.md b/scripts/setup/README.md index 57ea9249..91b48063 100644 --- a/scripts/setup/README.md +++ b/scripts/setup/README.md @@ -73,10 +73,13 @@ Docker on a GitHub-hosted runner instead. Linux provisioning runs through the SSH administrator, but the listener and workflow jobs run as the dedicated `nvx-runner` account, which has neither sudo nor Docker access. -Linux provisioning and check mode stop unless the host CPU exposes an invariant -TSC (`nonstop_tsc` in `/proc/cpuinfo`). Guests on an Azure VM without one hit -cross-vCPU TSC warps during CPU activation, so redeploy such a VM instead of -registering it. +Linux provisioning and check mode warn, but don't stop, when the host CPU +doesn't expose an invariant TSC (`nonstop_tsc` in `/proc/cpuinfo`). The flag +is evidence only. Before the time ABI, guests on such an Azure VM hit +cross-vCPU TSC warps during CPU activation (#265); under it, the guest warp +probe (H6) checks their skew against the ABI's 1 µs bound, so qualify such a +VM with `python3 scripts/nvx.py doctor --backend ` before +registering it. CI's microVM and platform jobs run the probe on every runner. Persistent runners execute pushes and same-repository pull requests only. Fork pull requests remain on GitHub-hosted jobs until a maintainer stages the change on a trusted repository branch. diff --git a/scripts/setup/setup-linux-runner.sh b/scripts/setup/setup-linux-runner.sh index 79959a2f..dacc7070 100644 --- a/scripts/setup/setup-linux-runner.sh +++ b/scripts/setup/setup-linux-runner.sh @@ -60,11 +60,11 @@ version_at_least() { [ "$(printf '%s\n%s\n' "$2" "$1" | sort -V | head -n 1)" = "$2" ] } -require_invariant_tsc() { - # Guests on a host without an invariant TSC intermittently see cross-vCPU - # TSC warps, which make Linux mark the guest TSC unstable (#211). +report_invariant_tsc() { + # The invariant-TSC flag is evidence only: the doctor's guest warp probe + # measures the cross-vCPU skew that a host without one can cause (#265). grep -Eq '^flags[[:space:]]*:.*[[:space:]]nonstop_tsc([[:space:]]|$)' "$1" || - die "host does not expose an invariant TSC (nonstop_tsc); redeploy the VM on a host that does" + printf '%s\n' "warning: host does not expose an invariant TSC (nonstop_tsc); qualify it with nvx.py doctor, whose guest warp probe measures the cross-vCPU skew that this can cause" >&2 } validate_runner_name() { @@ -818,7 +818,7 @@ case "$backend" in kvm | mshv) ;; *) die "--backend must be kvm or mshv" ;; esac -require_invariant_tsc /proc/cpuinfo +report_invariant_tsc /proc/cpuinfo [ "$(id -u)" -ne 0 ] || die "run this script as the SSH administrator, not root" require_command sudo sudo -n true || die "passwordless sudo is required" diff --git a/scripts/test_nvx_tools.py b/scripts/test_nvx_tools.py index 39eb0ec6..4bbc69c3 100644 --- a/scripts/test_nvx_tools.py +++ b/scripts/test_nvx_tools.py @@ -3750,7 +3750,7 @@ def test_windows_runner_requires_inbox_pcat_firmware(self): ): self.assertIn(firmware, configuration) - def test_linux_runners_require_an_invariant_tsc(self): + def test_linux_runners_report_the_invariant_tsc_as_evidence(self): validate_runner = ( BuildConstants.REPO_ROOT / ".github" @@ -3762,18 +3762,21 @@ def test_linux_runners_require_an_invariant_tsc(self): BuildConstants.REPO_ROOT / "scripts" / "setup" / "setup-linux-runner.sh" ).read_text(encoding="utf-8") - step = validate_runner.split(" - name: Validate host TSC\n", 1)[1] + # Neither CI nor provisioning rejects a host without an invariant + # TSC: the guest warp probe measures what its guests observe (#265). + step = validate_runner.split(" - name: Report host TSC\n", 1)[1] step = step.split("\n\n - name: ", 1)[0] self.assertIn(" if: runner.os != 'Windows'\n", step) check = "\n".join( line.removeprefix(" ") for line in step.split(" run: |\n", 1)[1].splitlines() ) - function = linux_setup.split("require_invariant_tsc() {\n", 1)[1] - function = "require_invariant_tsc() {\n" + function.split("\n}\n", 1)[0] + function = linux_setup.split("report_invariant_tsc() {\n", 1)[1] + function = "report_invariant_tsc() {\n" + function.split("\n}\n", 1)[0] function += "\n}\n" + self.assertNotRegex(function, r"\bdie\b") self.assertLess( - linux_setup.index("require_invariant_tsc /proc/cpuinfo\n"), + linux_setup.index("report_invariant_tsc /proc/cpuinfo\n"), linux_setup.index('sudo -n true || die "passwordless sudo is required"'), ) if os.name != "posix": @@ -3798,28 +3801,37 @@ def test_linux_runners_require_an_invariant_tsc(self): timeout=10, check=False, ) - self.assertEqual(result.returncode == 0, invariant, result.stderr) + self.assertEqual(result.returncode, 0, result.stderr) self.assertIn("CPU: Test CPU", result.stdout) + self.assertIn( + "TSC flag nonstop_tsc: " + + ("present" if invariant else "absent"), + result.stdout, + ) self.assertEqual( - "::error::Runner host does not expose an invariant TSC" - in result.stderr, + "::notice::Runner host does not expose an invariant TSC" + in result.stdout, not invariant, ) + self.assertNotIn("::error::", result.stdout + result.stderr) with self.subTest(flags=flags, check="setup-linux-runner"): result = subprocess.run( [ "sh", "-c", - 'die() { echo "$*" >&2; exit 1; }\n' - f"{function}" - f"require_invariant_tsc '{cpuinfo}'\n", + f"set -eu\n{function}report_invariant_tsc '{cpuinfo}'\n", ], capture_output=True, text=True, timeout=10, check=False, ) - self.assertEqual(result.returncode == 0, invariant, result.stderr) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertEqual( + "warning: host does not expose an invariant TSC" + in result.stderr, + not invariant, + ) def test_runners_qualify_their_host_time_before_every_job(self): validate_runner = ( @@ -3830,12 +3842,13 @@ def test_runners_qualify_their_host_time_before_every_job(self): / "action.yml" ).read_text(encoding="utf-8") - # The doctor adds to the nonstop_tsc gate, which stays first and - # fail-closed on Linux: the backend, the CPU fingerprint and - # generation, and the TSC rate stability. Jobs that haven't downloaded - # OpenVMM skip its CPU profile check (H2) and preflight (H3). + # The doctor follows the host TSC report on Linux: the backend, the CPU + # fingerprint and generation, and the TSC rate stability. Jobs that + # haven't downloaded OpenVMM skip its CPU profile check (H2) and + # preflight (H3), and only jobs that ask for it run the guest warp + # probe (H6). self.assertLess( - validate_runner.index(" - name: Validate host TSC\n"), + validate_runner.index(" - name: Report host TSC\n"), validate_runner.index(" - name: Qualify host time on Linux\n"), ) for name, shell, command, summary in ( @@ -3859,6 +3872,9 @@ def test_runners_qualify_their_host_time_before_every_job(self): self.assertIn(command, step) self.assertIn('--backend "${{ inputs.backend }}"', step) self.assertIn("--checks H1 H2 H4\n", step) + self.assertIn( + "${{ inputs.warp-probe == 'true' && 'H6' || '' }}\n", step + ) fingerprint = ( '"${RUNNER_TEMP}/nvx-cpu-fingerprint.json"' if shell == "bash" @@ -3877,6 +3893,11 @@ def test_runners_qualify_their_host_time_before_every_job(self): ) self.assertEqual(args.checks, ["H1", "H2", "H4", "H3"]) self.assertEqual(args.cpu_fingerprint, Path("fingerprint.json")) + args = nvx.parse_args( + ["doctor", "--backend", "kvm", "--checks", "H1", "H2", "H4", "H6", "H3"] + + ["--cpu-fingerprint", "fingerprint.json", "--ci-schedule"] + ) + self.assertEqual(args.checks, ["H1", "H2", "H4", "H6", "H3"]) args = nvx.parse_args( ["doctor", "--backend", "kvm", "--checks", "H1", "H2", "H4"] + ["--no-openvmm", "--ci-schedule"] @@ -3885,6 +3906,10 @@ def test_runners_qualify_their_host_time_before_every_job(self): self.assertTrue(args.ci_schedule) self.assertIn(" verify-openvmm:\n", validate_runner) self.assertIn(' default: "false"\n', validate_runner) + warp_probe = validate_runner.split(" warp-probe:\n", 1)[1] + warp_probe = warp_probe.split("\nruns:\n", 1)[0] + self.assertIn("requires\n verify-openvmm\n", warp_probe) + self.assertIn(' default: "false"\n', warp_probe) # A failed CPU profile check keeps OpenVMM's fingerprint. upload = validate_runner.split(" - name: Upload the CPU fingerprint\n")[1] upload = upload.split("\n\n", 1)[0] @@ -3910,6 +3935,12 @@ def test_runners_qualify_their_host_time_before_every_job(self): self.assertIn( 'verify-openvmm: "true"', workflow[validate : validate + 220] ) + # The microVM scenarios probe after every boot and restore; + # the benchmarks boot guests without such probes. + self.assertEqual( + 'warp-probe: "true"' in workflow[validate : validate + 260], + workflow_name == "run-platform.yml", + ) for step in ( " - name: Download guest artifacts\n", " - name: Download OpenVMM executable\n", diff --git a/scripts/test_time_abi.py b/scripts/test_time_abi.py index 9cf949cc..851110dd 100644 --- a/scripts/test_time_abi.py +++ b/scripts/test_time_abi.py @@ -2025,6 +2025,16 @@ def check( str(context.schedule.rate_interval_ms), ), ) + # The probe sleeps through its sampling window, which the timeout + # allows on top of --timeout (120 s on the qualification schedule). + self.assertEqual( + run.call_args.kwargs, + { + "duration": (context.schedule.rate_samples - 1) + * context.schedule.rate_interval_ms + / 1000 + }, + ) return result def ci_context(backend: str = "kvm") -> doctor.DoctorContext: @@ -2494,8 +2504,11 @@ def test_parses_probe_records(self): "noise\nNVX-HOST-TIME-PROBE rate index=1 tsc_hz=1.5\n" "NVX-HOST-TIME-PROBE skew pairs=1 cpus=0-1\n" ) - with patch.object(doctor.subprocess, "run", return_value=completed(output)): + with patch.object( + doctor.subprocess, "run", return_value=completed(output) + ) as run: records = doctor.run_probe(context, "skew") + self.assertEqual(run.call_args.kwargs["timeout"], context.timeout) self.assertEqual( records, [ @@ -2503,6 +2516,11 @@ def test_parses_probe_records(self): ("skew", {"pairs": "1", "cpus": "0-1"}), ], ) + with patch.object( + doctor.subprocess, "run", return_value=completed(output) + ) as run: + doctor.run_probe(context, "rate", duration=120.0) + self.assertEqual(run.call_args.kwargs["timeout"], context.timeout + 120.0) with patch.object( doctor.subprocess, "run", return_value=completed("", 3, "no CPU") ):