Repository navigation
Make opening a Vortex file with a cached footer cheaper #12593
Workflow file for this run
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Codspeed Benchmarking | |
| # Concurrency control: | |
| # - PRs: new commits on a feature branch will cancel in-progress (outdated) runs. | |
| # - Push to develop: every commit gets its own group, so baseline runs never cancel and never | |
| # queue behind each other. Serialising them meant a burst of merges left later commits without | |
| # a finished baseline, so CodSpeed fell back to an older comparison base and reported changes | |
| # unrelated to the PR being tested. | |
| # - `workflow_dispatch`: groups by branch and queues if run on develop. | |
| concurrency: | |
| group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event_name == 'push' && github.sha || '' }} | |
| cancel-in-progress: ${{ github.ref != 'refs/heads/develop' }} | |
| on: | |
| push: | |
| branches: [develop] | |
| pull_request: { } | |
| workflow_dispatch: { } | |
| permissions: | |
| contents: read | |
| env: | |
| CARGO_TERM_COLOR: always | |
| RUST_BACKTRACE: 1 | |
| NIGHTLY_TOOLCHAIN: nightly-2026-09-10 | |
| jobs: | |
| changes: | |
| name: "Detect CUDA changes" | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 10 | |
| permissions: | |
| pull-requests: read | |
| outputs: | |
| run-cuda-benchmarks: ${{ github.event_name != 'pull_request' || steps.filter.outputs.cuda == 'true' }} | |
| steps: | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 | |
| - uses: dorny/paths-filter@ceb8a2b8f2d89434be7ff52d3de7ec3738c5cc9d # v4 | |
| id: filter | |
| if: github.event_name == 'pull_request' | |
| with: | |
| filters: | | |
| cuda: | |
| - "vortex-cuda/**" | |
| # Only this workflow defines the CUDA benchmark jobs. | |
| - ".github/workflows/codspeed.yml" | |
| bench-codspeed: | |
| strategy: | |
| matrix: | |
| include: | |
| # Distribute both compilation and simulation work across shards. Rebalance | |
| # these groups as new benchmarks are added, using CI build and run durations. | |
| - { shard: 1, name: "Array compute", packages: "vortex-array", array-group: "compute", features: "--features _test-harness" } | |
| - { shard: 2, name: "Array selection", packages: "vortex-array", array-group: "selection", features: "--features _test-harness" } | |
| - { shard: 3, name: "Array dictionaries", packages: "vortex-array", array-group: "dictionary", features: "--features _test-harness" } | |
| - { shard: 4, name: "Array builders & views", packages: "vortex-array", array-group: "builders", features: "--features _test-harness" } | |
| - { shard: 5, name: "Core & encodings", packages: "vortex-buffer vortex-error vortex-mask vortex-compute vortex-file vortex-alp vortex-bytebool vortex-datetime-parts" } | |
| - { shard: 6, name: "Main library & storage", packages: "vortex vortex-btrblocks vortex-compressor vortex-row" } | |
| - { shard: 7, name: "FastLanes & decimals", packages: "vortex-decimal-byte-parts vortex-fastlanes", features: "--features _test-harness" } | |
| - { shard: 8, name: "Encodings & layout", packages: "vortex-pco vortex-runend vortex-sequence vortex-sparse vortex-zigzag vortex-zstd vortex-layout" } | |
| - { shard: 9, name: "Strings, tensor & spatial", packages: "vortex-fsst vortex-tensor vortex-spatial", features: "--features _test-harness" } | |
| name: "Benchmark with Codspeed (Shard #${{ matrix.shard }} - ${{ matrix.name }})" | |
| timeout-minutes: 30 | |
| runs-on: >- | |
| ${{ github.repository == 'vortex-data/vortex' | |
| && format('runs-on={0}/runner=amd64-medium/image=ubuntu24-full-x64-pre-v2/extras=s3-cache/tag=bench-codspeed-{1}', github.run_id, matrix.shard) | |
| || 'ubuntu-latest' }} | |
| steps: | |
| - uses: runs-on/action@v2 | |
| if: github.repository == 'vortex-data/vortex' | |
| with: | |
| sccache: s3 | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 | |
| - uses: ./.github/actions/setup-prebuild | |
| with: | |
| enable-sccache: ${{ github.repository == 'vortex-data/vortex' && 'true' || 'false' }} | |
| - uses: ./.github/actions/system-info | |
| - name: Install Codspeed | |
| uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995 | |
| with: | |
| tool: cargo-codspeed | |
| - name: Select array benchmarks | |
| if: matrix.array-group | |
| id: select | |
| env: | |
| ARRAY_GROUP: ${{ matrix.array-group }} | |
| run: | | |
| python3 - <<'EOF' >> "$GITHUB_OUTPUT" | |
| import os, pathlib, tomllib | |
| groups = { | |
| "dictionary": ("dict_", "chunked_dict_", "patches_"), | |
| "selection": ("take_", "filter_", "piecewise_sequence_"), | |
| "compute": ("aggregate_", "cast_", "expr_", "binary_ops", "compare", "kleene_bool", "like", "list_sum"), | |
| } | |
| manifest = tomllib.loads(pathlib.Path("vortex-array/Cargo.toml").read_text()) | |
| benches = [] | |
| for bench in manifest["bench"]: | |
| name = bench["name"] | |
| # Unclassified and newly added targets stay covered by the builders shard. | |
| group = next((group for group, prefixes in groups.items() if name.startswith(prefixes)), "builders") | |
| if group == os.environ["ARRAY_GROUP"]: | |
| benches.append(name) | |
| if not benches: | |
| raise SystemExit("array shard has no benchmark targets") | |
| print("benches=" + " ".join(f"--bench {name}" for name in benches)) | |
| EOF | |
| - name: Build benchmarks | |
| env: | |
| RUSTFLAGS: "-C target-feature=+avx2 -C force-frame-pointers=yes -C link-arg=-fuse-ld=mold" | |
| # Benchmarks carrying `#[cpu_features]` belong to a walltime leg below and are | |
| # skipped here. Untagged ones run as they always have, under bare names, so this | |
| # job's CodSpeed history is unaffected. | |
| VORTEX_BENCH_VARIANT: simulation | |
| run: cargo codspeed build --locked ${{ matrix.features }} $(printf -- '-p %s ' ${{ matrix.packages }}) ${{ steps.select.outputs.benches }} --profile bench | |
| - name: Run benchmarks | |
| uses: CodSpeedHQ/action@373d6868929f444bc08d901fd0eb0ad52a8875ea # v5 | |
| with: | |
| run: cargo codspeed run | |
| token: ${{ secrets.CODSPEED_TOKEN }} | |
| mode: "simulation" | |
| # Exclude deferred allocator cleanup from the affected array shards. | |
| # Allocation-focused benchmarks in the core shard still measure allocator time. | |
| exclude-allocations: ${{ matrix.array-group == 'dictionary' || matrix.array-group == 'selection' }} | |
| # Cached package restoration fails on dangling libc6-dbg documentation symlinks. | |
| cache-instruments: "false" | |
| # Compile on smaller VMs from the same CPU family as the measurement hosts. Globally | |
| # enabled features also affect build scripts, so the build host must support them. | |
| bench-codspeed-cpu-features-build: | |
| if: github.repository == 'vortex-data/vortex' | |
| strategy: | |
| fail-fast: false | |
| matrix: &cpu-feature-matrix | |
| include: | |
| # avx2 and avx512 share a family so the only difference between the two series is | |
| # the build flags, not the silicon. c7i.metal-24xl is the smaller of the two c7i | |
| # metal sizes: Sapphire Rapids, and AVX-512 capable. | |
| - features: avx2 | |
| family: c7i.metal-24xl | |
| build-family: c7i.4xlarge | |
| image: ubuntu24-full-x64-pre-v2 | |
| rustflags: "-C target-feature=+avx2 -C force-frame-pointers=yes" | |
| # Every AVX-512 extension Sapphire Rapids implements, not just the two the current | |
| # `cfg(target_feature)` gates test for. Those gates decide which kernel the code | |
| # under test selects, but the rest of the build — anything the compiler | |
| # auto-vectorizes, the scalar baselines included — sees the whole feature set, and | |
| # a two-feature build is not what anything ships on. | |
| - features: avx512 | |
| family: c7i.metal-24xl | |
| build-family: c7i.4xlarge | |
| image: ubuntu24-full-x64-pre-v2 | |
| rustflags: >- | |
| -C target-feature=+avx512f,+avx512bw,+avx512cd,+avx512dq,+avx512vl,+avx512ifma,+avx512vbmi,+avx512vbmi2,+avx512vnni,+avx512bitalg,+avx512vpopcntdq,+avx512bf16,+avx512fp16 | |
| -C force-frame-pointers=yes | |
| # Graviton3 supports SVE, so this family can also host a future SVE leg. | |
| - features: neon | |
| family: c7g.metal | |
| build-family: c7g.4xlarge | |
| image: ubuntu24-full-arm64-pre-v2 | |
| rustflags: "-C target-feature=+neon -C force-frame-pointers=yes" | |
| name: "Build Codspeed CPU benchmarks (${{ matrix.features }})" | |
| timeout-minutes: 30 | |
| runs-on: >- | |
| runs-on=${{ github.run_id }}/runner=bench-dedicated/family=${{ matrix.build-family }}/image=${{ matrix.image }}/extras=s3-cache/tag=bench-codspeed-cpu-features-build-${{ matrix.features }} | |
| steps: | |
| - uses: runs-on/action@v2 | |
| with: | |
| sccache: s3 | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 | |
| - uses: ./.github/actions/setup-prebuild | |
| with: | |
| enable-sccache: "true" | |
| - uses: ./.github/actions/system-info | |
| - name: Install Codspeed | |
| uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995 | |
| with: | |
| tool: cargo-codspeed | |
| # Select individual targets: selecting packages also links their untagged binaries, | |
| # including nearly all of vortex-array, even though this job never measures them. | |
| - name: Select tagged benchmark targets | |
| id: select | |
| run: | | |
| python3 - <<'EOF' >> "$GITHUB_OUTPUT" | |
| import json, pathlib, subprocess | |
| meta = json.loads(subprocess.check_output( | |
| ["cargo", "metadata", "--locked", "--no-deps", "--format-version", "1"])) | |
| tagged = sorted( | |
| (package["name"], target["name"]) | |
| for package in meta["packages"] | |
| for target in package["targets"] | |
| if "bench" in target["kind"] | |
| and "cpu_features]" in pathlib.Path(target["src_path"]).read_text() | |
| ) | |
| if not tagged: | |
| raise SystemExit("no benchmark carries `#[cpu_features]`; this job has nothing to measure") | |
| print("packages=" + " ".join(f"-p {name}" for name in sorted({package for package, _ in tagged}))) | |
| print("benches=" + " ".join(f"--bench {name}" for name in sorted({bench for _, bench in tagged}))) | |
| EOF | |
| - name: Build benchmarks | |
| env: | |
| RUSTFLAGS: ${{ matrix.rustflags }} | |
| VORTEX_BENCH_VARIANT: ${{ matrix.features }} | |
| VORTEX_BENCH_PREFIX: "${{ matrix.features }}::" | |
| VORTEX_BENCH_SUFFIX: "_${{ matrix.features }}" | |
| run: | | |
| cargo codspeed build --locked -m walltime --profile bench \ | |
| ${{ steps.select.outputs.packages }} ${{ steps.select.outputs.benches }} | |
| - name: Package benchmark executables | |
| run: tar -C target -cf codspeed-cpu-benchmarks.tar codspeed | |
| - name: Upload benchmark executables | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 | |
| with: | |
| name: codspeed-cpu-benchmarks-${{ matrix.features }} | |
| path: codspeed-cpu-benchmarks.tar | |
| retention-days: 1 | |
| if-no-files-found: error | |
| # Only measurements need metal. Reuse the build matrix to keep each artifact paired | |
| # with the matching CPU family, image, and benchmark-name filter. | |
| bench-codspeed-cpu-features: | |
| if: github.repository == 'vortex-data/vortex' | |
| needs: [bench-codspeed-cpu-features-build] | |
| strategy: | |
| fail-fast: false | |
| matrix: *cpu-feature-matrix | |
| name: "Benchmark with Codspeed (${{ matrix.features }})" | |
| timeout-minutes: 30 | |
| runs-on: >- | |
| runs-on=${{ github.run_id }}/runner=bench-dedicated/family=${{ matrix.family }}/image=${{ matrix.image }}/extras=s3-cache/tag=bench-codspeed-cpu-features-${{ matrix.features }} | |
| steps: | |
| # Initialise the RunsOn artifact proxy used by download-artifact. | |
| - uses: runs-on/action@v2 | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 | |
| - uses: ./.github/actions/setup-prebuild | |
| - uses: ./.github/actions/system-info | |
| - name: Install Codspeed | |
| uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995 | |
| with: | |
| tool: cargo-codspeed | |
| - name: Download benchmark executables | |
| uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 | |
| with: | |
| name: codspeed-cpu-benchmarks-${{ matrix.features }} | |
| # The tar archive preserves executable permissions lost by upload-artifact. | |
| - name: Unpack benchmark executables | |
| run: | | |
| mkdir -p target | |
| tar -C target -xf codspeed-cpu-benchmarks.tar | |
| # Pinning clocks and reserving CPUs only pays off for walltime measurements, so this | |
| # runs here but not in the simulation job. It must come after setup, which itself | |
| # spawns processes that would otherwise be pinned to the housekeeping CPUs. | |
| - name: Setup benchmark environment | |
| run: sudo bash scripts/setup-benchmark.sh | |
| # The trailing filter is what keeps a leg to the tagged benchmarks. Unlike a simulation | |
| # build, a walltime one does not set `--cfg codspeed`, so benchmarks kept out of | |
| # CodSpeed that way are compiled in here and divan would otherwise measure them. | |
| # | |
| # divan matches the filter's `::`-separated components against the benchmark path's, | |
| # so this selects paths of the form `<bench target>::<features>::<name>`. That middle | |
| # component only exists because `#[cpu_features]` puts it there: an untagged benchmark has one | |
| # component fewer and cannot match, whatever it is called. | |
| - name: Run benchmarks | |
| uses: CodSpeedHQ/action@373d6868929f444bc08d901fd0eb0ad52a8875ea # v5 | |
| env: | |
| # Use more samples to reduce noise between walltime runs. | |
| DIVAN_SAMPLE_COUNT: "1000" | |
| with: | |
| run: bash scripts/bench-taskset.sh cargo codspeed run -- '.*::${{ matrix.features }}::' | |
| token: ${{ secrets.CODSPEED_TOKEN }} | |
| mode: "walltime" | |
| # Getting a GPU box is slow, in the future we can build on a box without one and only run | |
| # on GPU machines. | |
| bench-codspeed-cuda-build: | |
| needs: [changes] | |
| if: >- | |
| !cancelled() && github.repository == 'vortex-data/vortex' && | |
| needs.changes.outputs.run-cuda-benchmarks == 'true' | |
| name: "Build Codspeed CUDA benchmarks" | |
| timeout-minutes: 30 | |
| runs-on: >- | |
| runs-on=${{ github.run_id }}/family=g5/cpu=8/image=ubuntu24-gpu-x64/extras=s3-cache/tag=bench-codspeed-cuda-build | |
| steps: | |
| - uses: runs-on/action@v2 | |
| with: | |
| sccache: s3 | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 | |
| - uses: ./.github/actions/setup-rust | |
| with: | |
| repo-token: ${{ secrets.GITHUB_TOKEN }} | |
| enable-sccache: "true" | |
| - name: Install Codspeed | |
| uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995 | |
| with: | |
| tool: cargo-codspeed | |
| - name: Build benchmarks | |
| # cargo-codspeed forces -Cdebuginfo=2 -Cstrip=none | |
| # https://github.com/CodSpeedHQ/codspeed-rust/issues/190 | |
| # CARGO_ENCODED_RUSTFLAGS allows us to override it since we need | |
| # only debug = limited | |
| env: | |
| CARGO_ENCODED_RUSTFLAGS: "-Cforce-frame-pointers=yes" | |
| run: | | |
| cargo codspeed build --locked \ | |
| -m walltime \ | |
| --bench bitpacked_cuda \ | |
| --bench dynamic_dispatch_cuda \ | |
| --bench alp_cuda \ | |
| --bench date_time_parts_cuda \ | |
| --bench dict_cuda \ | |
| --bench fsst_cuda \ | |
| --bench runend_cuda \ | |
| --profile bench | |
| - name: Package CUB shared library | |
| run: | | |
| find target/release/build -path '*/out/libvortex_cub.so' \ | |
| -exec cp {} target/codspeed/walltime/vortex-cuda/libvortex_cub.so \; | |
| test -f target/codspeed/walltime/vortex-cuda/libvortex_cub.so | |
| - name: Upload benchmark executables | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7 | |
| with: | |
| name: codspeed-cuda-benchmarks | |
| path: target/codspeed/ | |
| retention-days: 1 | |
| if-no-files-found: error | |
| bench-codspeed-cuda: | |
| if: github.repository == 'vortex-data/vortex' | |
| needs: [bench-codspeed-cuda-build] | |
| strategy: | |
| matrix: | |
| include: | |
| - { shard: 1, name: "Bitpacked", benches: "bitpacked_cuda" } | |
| - { shard: 2, name: "Dynamic dispatch", benches: "dynamic_dispatch_cuda" } | |
| - { shard: 3, name: "Standalone kernels", benches: "alp_cuda date_time_parts_cuda delta_cuda dict_cuda fsst_cuda runend_cuda" } | |
| name: "Benchmark with Codspeed (CUDA Shard #${{ matrix.shard }} - ${{ matrix.name }})" | |
| timeout-minutes: 30 | |
| runs-on: >- | |
| runs-on=${{ github.run_id }}/family=g5/cpu=8/image=ubuntu24-gpu-x64/extras=s3-cache/tag=bench-codspeed-cuda-${{ matrix.shard }} | |
| steps: | |
| - uses: runs-on/action@v2 | |
| with: | |
| sccache: s3 | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 | |
| - uses: ./.github/actions/setup-rust | |
| with: | |
| repo-token: ${{ secrets.GITHUB_TOKEN }} | |
| enable-sccache: "true" | |
| - name: Display NVIDIA SMI details | |
| run: | | |
| nvidia-smi | |
| nvidia-smi -L | |
| nvidia-smi -q -d Memory | |
| - name: Install Codspeed | |
| uses: taiki-e/cache-cargo-install-action@66c9585ef5ca780ee69399975a5e911f47905995 | |
| with: | |
| tool: cargo-codspeed | |
| - name: Download benchmark executables | |
| uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 | |
| with: | |
| name: codspeed-cuda-benchmarks | |
| path: target/codspeed | |
| - name: Restore executable permissions | |
| run: find target/codspeed -type f -exec chmod +x {} + | |
| - name: Run benchmarks | |
| uses: CodSpeedHQ/action@373d6868929f444bc08d901fd0eb0ad52a8875ea # v5 | |
| env: | |
| CARGO_MANIFEST_DIR: ${{ github.workspace }}/vortex-cuda | |
| with: | |
| run: cargo codspeed run $(printf -- '--bench %s ' ${{ matrix.benches }}) | |
| token: ${{ secrets.CODSPEED_TOKEN }} | |
| mode: "walltime" |