Skip to content

Preserve semantic errors in null checks #12588

Preserve semantic errors in null checks

Preserve semantic errors in null checks #12588

Workflow file for this run

name: Codspeed Benchmarking
# Concurrency control:
# - PRs: new commits on a feature branch will cancel in-progress (outdated) runs.
# - Push to develop: every commit gets its own group, so baseline runs never cancel and never
# queue behind each other. Serialising them meant a burst of merges left later commits without
# a finished baseline, so CodSpeed fell back to an older comparison base and reported changes
# unrelated to the PR being tested.
# - `workflow_dispatch`: groups by branch and queues if run on develop.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event_name == 'push' && github.sha || '' }}
cancel-in-progress: ${{ github.ref != 'refs/heads/develop' }}
on:
push:
branches: [develop]
pull_request: { }
workflow_dispatch: { }
permissions:
contents: read
env:
CARGO_TERM_COLOR: always
RUST_BACKTRACE: 1
NIGHTLY_TOOLCHAIN: nightly-2026-09-10
jobs:
changes:
name: "Detect CUDA changes"
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
pull-requests: read
outputs:
run-cuda-benchmarks: ${{ github.event_name != 'pull_request' || steps.filter.outputs.cuda == 'true' }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: dorny/paths-filter@ceb8a2b8f2d89434be7ff52d3de7ec3738c5cc9d # v4
id: filter
if: github.event_name == 'pull_request'
with:
filters: |
cuda:
- "vortex-cuda/**"
# Only this workflow defines the CUDA benchmark jobs.
- ".github/workflows/codspeed.yml"
bench-codspeed:
strategy:
matrix:
include:
# Distribute both compilation and simulation work across shards. Rebalance
# these groups as new benchmarks are added, using CI build and run durations.
- { shard: 1, name: "Array compute", packages: "vortex-array", array-group: "compute", features: "--features _test-harness" }
- { shard: 2, name: "Array selection", packages: "vortex-array", array-group: "selection", features: "--features _test-harness" }
- { shard: 3, name: "Array dictionaries", packages: "vortex-array", array-group: "dictionary", features: "--features _test-harness" }
- { shard: 4, name: "Array builders & views", packages: "vortex-array", array-group: "builders", features: "--features _test-harness" }
- { shard: 5, name: "Core & encodings", packages: "vortex-buffer vortex-error vortex-mask vortex-compute vortex-file vortex-alp vortex-bytebool vortex-datetime-parts" }
- { shard: 6, name: "Main library & storage", packages: "vortex vortex-btrblocks vortex-compressor vortex-row" }
- { shard: 7, name: "FastLanes & decimals", packages: "vortex-decimal-byte-parts vortex-fastlanes", features: "--features _test-harness" }
- { shard: 8, name: "Encodings & layout", packages: "vortex-pco vortex-runend vortex-sequence vortex-sparse vortex-zigzag vortex-zstd vortex-layout" }
- { shard: 9, name: "Strings, tensor & spatial", packages: "vortex-fsst vortex-tensor vortex-spatial", features: "--features _test-harness" }
name: "Benchmark with Codspeed (Shard #${{ matrix.shard }} - ${{ matrix.name }})"
timeout-minutes: 30
runs-on: >-
${{ github.repository == 'vortex-data/vortex'
&& format('runs-on={0}/runner=amd64-medium/image=ubuntu24-full-x64-pre-v2/extras=s3-cache/tag=bench-codspeed-{1}', github.run_id, matrix.shard)
|| 'ubuntu-latest' }}
steps:
- uses: runs-on/action@v2
if: github.repository == 'vortex-data/vortex'
with:
sccache: s3
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: ./.github/actions/setup-prebuild
with:
enable-sccache: ${{ github.repository == 'vortex-data/vortex' && 'true' || 'false' }}
- uses: ./.github/actions/system-info
- name: Install Codspeed
uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995
with:
tool: cargo-codspeed
- name: Select array benchmarks
if: matrix.array-group
id: select
env:
ARRAY_GROUP: ${{ matrix.array-group }}
run: |
python3 - <<'EOF' >> "$GITHUB_OUTPUT"
import os, pathlib, tomllib
groups = {
"dictionary": ("dict_", "chunked_dict_", "patches_"),
"selection": ("take_", "filter_", "piecewise_sequence_"),
"compute": ("aggregate_", "cast_", "expr_", "binary_ops", "compare", "kleene_bool", "like", "list_sum"),
}
manifest = tomllib.loads(pathlib.Path("vortex-array/Cargo.toml").read_text())
benches = []
for bench in manifest["bench"]:
name = bench["name"]
# Unclassified and newly added targets stay covered by the builders shard.
group = next((group for group, prefixes in groups.items() if name.startswith(prefixes)), "builders")
if group == os.environ["ARRAY_GROUP"]:
benches.append(name)
if not benches:
raise SystemExit("array shard has no benchmark targets")
print("benches=" + " ".join(f"--bench {name}" for name in benches))
EOF
- name: Build benchmarks
env:
RUSTFLAGS: "-C target-feature=+avx2 -C force-frame-pointers=yes -C link-arg=-fuse-ld=mold"
# Benchmarks carrying `#[cpu_features]` belong to a walltime leg below and are
# skipped here. Untagged ones run as they always have, under bare names, so this
# job's CodSpeed history is unaffected.
VORTEX_BENCH_VARIANT: simulation
run: cargo codspeed build --locked ${{ matrix.features }} $(printf -- '-p %s ' ${{ matrix.packages }}) ${{ steps.select.outputs.benches }} --profile bench
- name: Run benchmarks
uses: CodSpeedHQ/action@373d6868929f444bc08d901fd0eb0ad52a8875ea # v5
with:
run: cargo codspeed run
token: ${{ secrets.CODSPEED_TOKEN }}
mode: "simulation"
# Exclude deferred allocator cleanup from the affected array shards.
# Allocation-focused benchmarks in the core shard still measure allocator time.
exclude-allocations: ${{ matrix.array-group == 'dictionary' || matrix.array-group == 'selection' }}
# Cached package restoration fails on dangling libc6-dbg documentation symlinks.
cache-instruments: "false"
# Compile on smaller VMs from the same CPU family as the measurement hosts. Globally
# enabled features also affect build scripts, so the build host must support them.
bench-codspeed-cpu-features-build:
if: github.repository == 'vortex-data/vortex'
strategy:
fail-fast: false
matrix: &cpu-feature-matrix
include:
# avx2 and avx512 share a family so the only difference between the two series is
# the build flags, not the silicon. c7i.metal-24xl is the smaller of the two c7i
# metal sizes: Sapphire Rapids, and AVX-512 capable.
- features: avx2
family: c7i.metal-24xl
build-family: c7i.4xlarge
image: ubuntu24-full-x64-pre-v2
rustflags: "-C target-feature=+avx2 -C force-frame-pointers=yes"
# Every AVX-512 extension Sapphire Rapids implements, not just the two the current
# `cfg(target_feature)` gates test for. Those gates decide which kernel the code
# under test selects, but the rest of the build — anything the compiler
# auto-vectorizes, the scalar baselines included — sees the whole feature set, and
# a two-feature build is not what anything ships on.
- features: avx512
family: c7i.metal-24xl
build-family: c7i.4xlarge
image: ubuntu24-full-x64-pre-v2
rustflags: >-
-C target-feature=+avx512f,+avx512bw,+avx512cd,+avx512dq,+avx512vl,+avx512ifma,+avx512vbmi,+avx512vbmi2,+avx512vnni,+avx512bitalg,+avx512vpopcntdq,+avx512bf16,+avx512fp16
-C force-frame-pointers=yes
# Graviton3 supports SVE, so this family can also host a future SVE leg.
- features: neon
family: c7g.metal
build-family: c7g.4xlarge
image: ubuntu24-full-arm64-pre-v2
rustflags: "-C target-feature=+neon -C force-frame-pointers=yes"
name: "Build Codspeed CPU benchmarks (${{ matrix.features }})"
timeout-minutes: 30
runs-on: >-
runs-on=${{ github.run_id }}/runner=bench-dedicated/family=${{ matrix.build-family }}/image=${{ matrix.image }}/extras=s3-cache/tag=bench-codspeed-cpu-features-build-${{ matrix.features }}
steps:
- uses: runs-on/action@v2
with:
sccache: s3
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: ./.github/actions/setup-prebuild
with:
enable-sccache: "true"
- uses: ./.github/actions/system-info
- name: Install Codspeed
uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995
with:
tool: cargo-codspeed
# Select individual targets: selecting packages also links their untagged binaries,
# including nearly all of vortex-array, even though this job never measures them.
- name: Select tagged benchmark targets
id: select
run: |
python3 - <<'EOF' >> "$GITHUB_OUTPUT"
import json, pathlib, subprocess
meta = json.loads(subprocess.check_output(
["cargo", "metadata", "--locked", "--no-deps", "--format-version", "1"]))
tagged = sorted(
(package["name"], target["name"])
for package in meta["packages"]
for target in package["targets"]
if "bench" in target["kind"]
and "cpu_features]" in pathlib.Path(target["src_path"]).read_text()
)
if not tagged:
raise SystemExit("no benchmark carries `#[cpu_features]`; this job has nothing to measure")
print("packages=" + " ".join(f"-p {name}" for name in sorted({package for package, _ in tagged})))
print("benches=" + " ".join(f"--bench {name}" for name in sorted({bench for _, bench in tagged})))
EOF
- name: Build benchmarks
env:
RUSTFLAGS: ${{ matrix.rustflags }}
VORTEX_BENCH_VARIANT: ${{ matrix.features }}
VORTEX_BENCH_PREFIX: "${{ matrix.features }}::"
VORTEX_BENCH_SUFFIX: "_${{ matrix.features }}"
run: |
cargo codspeed build --locked -m walltime --profile bench \
${{ steps.select.outputs.packages }} ${{ steps.select.outputs.benches }}
- name: Package benchmark executables
run: tar -C target -cf codspeed-cpu-benchmarks.tar codspeed
- name: Upload benchmark executables
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: codspeed-cpu-benchmarks-${{ matrix.features }}
path: codspeed-cpu-benchmarks.tar
retention-days: 1
if-no-files-found: error
# Only measurements need metal. Reuse the build matrix to keep each artifact paired
# with the matching CPU family, image, and benchmark-name filter.
bench-codspeed-cpu-features:
if: github.repository == 'vortex-data/vortex'
needs: [bench-codspeed-cpu-features-build]
strategy:
fail-fast: false
matrix: *cpu-feature-matrix
name: "Benchmark with Codspeed (${{ matrix.features }})"
timeout-minutes: 30
runs-on: >-
runs-on=${{ github.run_id }}/runner=bench-dedicated/family=${{ matrix.family }}/image=${{ matrix.image }}/extras=s3-cache/tag=bench-codspeed-cpu-features-${{ matrix.features }}
steps:
# Initialise the RunsOn artifact proxy used by download-artifact.
- uses: runs-on/action@v2
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: ./.github/actions/setup-prebuild
- uses: ./.github/actions/system-info
- name: Install Codspeed
uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995
with:
tool: cargo-codspeed
- name: Download benchmark executables
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
name: codspeed-cpu-benchmarks-${{ matrix.features }}
# The tar archive preserves executable permissions lost by upload-artifact.
- name: Unpack benchmark executables
run: |
mkdir -p target
tar -C target -xf codspeed-cpu-benchmarks.tar
# Pinning clocks and reserving CPUs only pays off for walltime measurements, so this
# runs here but not in the simulation job. It must come after setup, which itself
# spawns processes that would otherwise be pinned to the housekeeping CPUs.
- name: Setup benchmark environment
run: sudo bash scripts/setup-benchmark.sh
# The trailing filter is what keeps a leg to the tagged benchmarks. Unlike a simulation
# build, a walltime one does not set `--cfg codspeed`, so benchmarks kept out of
# CodSpeed that way are compiled in here and divan would otherwise measure them.
#
# divan matches the filter's `::`-separated components against the benchmark path's,
# so this selects paths of the form `<bench target>::<features>::<name>`. That middle
# component only exists because `#[cpu_features]` puts it there: an untagged benchmark has one
# component fewer and cannot match, whatever it is called.
- name: Run benchmarks
uses: CodSpeedHQ/action@373d6868929f444bc08d901fd0eb0ad52a8875ea # v5
env:
# Use more samples to reduce noise between walltime runs.
DIVAN_SAMPLE_COUNT: "1000"
with:
run: bash scripts/bench-taskset.sh cargo codspeed run -- '.*::${{ matrix.features }}::'
token: ${{ secrets.CODSPEED_TOKEN }}
mode: "walltime"
# Getting a GPU box is slow, in the future we can build on a box without one and only run
# on GPU machines.
bench-codspeed-cuda-build:
needs: [changes]
if: >-
!cancelled() && github.repository == 'vortex-data/vortex' &&
needs.changes.outputs.run-cuda-benchmarks == 'true'
name: "Build Codspeed CUDA benchmarks"
timeout-minutes: 30
runs-on: >-
runs-on=${{ github.run_id }}/family=g5/cpu=8/image=ubuntu24-gpu-x64/extras=s3-cache/tag=bench-codspeed-cuda-build
steps:
- uses: runs-on/action@v2
with:
sccache: s3
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: ./.github/actions/setup-rust
with:
repo-token: ${{ secrets.GITHUB_TOKEN }}
enable-sccache: "true"
- name: Install Codspeed
uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995
with:
tool: cargo-codspeed
- name: Build benchmarks
# cargo-codspeed forces -Cdebuginfo=2 -Cstrip=none
# https://github.com/CodSpeedHQ/codspeed-rust/issues/190
# CARGO_ENCODED_RUSTFLAGS allows us to override it since we need
# only debug = limited
env:
CARGO_ENCODED_RUSTFLAGS: "-Cforce-frame-pointers=yes"
run: |
cargo codspeed build --locked \
-m walltime \
--bench bitpacked_cuda \
--bench dynamic_dispatch_cuda \
--bench alp_cuda \
--bench date_time_parts_cuda \
--bench dict_cuda \
--bench fsst_cuda \
--bench runend_cuda \
--profile bench
- name: Package CUB shared library
run: |
find target/release/build -path '*/out/libvortex_cub.so' \
-exec cp {} target/codspeed/walltime/vortex-cuda/libvortex_cub.so \;
test -f target/codspeed/walltime/vortex-cuda/libvortex_cub.so
- name: Upload benchmark executables
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: codspeed-cuda-benchmarks
path: target/codspeed/
retention-days: 1
if-no-files-found: error
bench-codspeed-cuda:
if: github.repository == 'vortex-data/vortex'
needs: [bench-codspeed-cuda-build]
strategy:
matrix:
include:
- { shard: 1, name: "Bitpacked", benches: "bitpacked_cuda" }
- { shard: 2, name: "Dynamic dispatch", benches: "dynamic_dispatch_cuda" }
- { shard: 3, name: "Standalone kernels", benches: "alp_cuda date_time_parts_cuda delta_cuda dict_cuda fsst_cuda runend_cuda" }
name: "Benchmark with Codspeed (CUDA Shard #${{ matrix.shard }} - ${{ matrix.name }})"
timeout-minutes: 30
runs-on: >-
runs-on=${{ github.run_id }}/family=g5/cpu=8/image=ubuntu24-gpu-x64/extras=s3-cache/tag=bench-codspeed-cuda-${{ matrix.shard }}
steps:
- uses: runs-on/action@v2
with:
sccache: s3
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: ./.github/actions/setup-rust
with:
repo-token: ${{ secrets.GITHUB_TOKEN }}
enable-sccache: "true"
- name: Display NVIDIA SMI details
run: |
nvidia-smi
nvidia-smi -L
nvidia-smi -q -d Memory
- name: Install Codspeed
uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995
with:
tool: cargo-codspeed
- name: Download benchmark executables
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
name: codspeed-cuda-benchmarks
path: target/codspeed
- name: Restore executable permissions
run: find target/codspeed -type f -exec chmod +x {} +
- name: Run benchmarks
uses: CodSpeedHQ/action@373d6868929f444bc08d901fd0eb0ad52a8875ea # v5
env:
CARGO_MANIFEST_DIR: ${{ github.workspace }}/vortex-cuda
with:
run: cargo codspeed run $(printf -- '--bench %s ' ${{ matrix.benches }})
token: ${{ secrets.CODSPEED_TOKEN }}
mode: "walltime"