ci: add performance regression check for PRs / merges #5
Workflow file for this run
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # Licensed to the Apache Software Foundation (ASF) under one | |
| # or more contributor license agreements. See the NOTICE file | |
| # distributed with this work for additional information | |
| # regarding copyright ownership. The ASF licenses this file | |
| # to you under the Apache License, Version 2.0 (the | |
| # "License"); you may not use this file except in compliance | |
| # with the License. You may obtain a copy of the License at | |
| # | |
| # http://www.apache.org/licenses/LICENSE-2.0 | |
| # | |
| # Unless required by applicable law or agreed to in writing, | |
| # software distributed under the License is distributed on an | |
| # "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY | |
| # KIND, either express or implied. See the License for the | |
| # specific language governing permissions and limitations | |
| # under the License. | |
| # Catches performance regressions by running TPC-H SF1 for the base branch and | |
| # for the PR merged into it, then failing when the PR is slower than the | |
| # configured limits allow. | |
| # | |
| # The two binaries are built by two jobs, on a runner each, and both are then | |
| # measured by a third job on one machine: building is embarrassingly parallel, | |
| # while comparing timings taken on different machines is meaningless. | |
| name: Benchmarks | |
| concurrency: | |
| group: ${{ github.repository }}-${{ github.head_ref || github.sha }}-${{ github.workflow }} | |
| cancel-in-progress: true | |
| # The builds and the benchmark runs add up to roughly half an hour of runner | |
| # time, and benchmarks on shared runners are noisy, so this is opt-in: add the | |
| # `performance` label to a PR, or start it by hand from the Actions tab. | |
| on: | |
| pull_request: | |
| types: [opened, synchronize, reopened, labeled] | |
| workflow_dispatch: | |
| inputs: | |
| iterations: | |
| description: 'Iterations per query (the fastest one is compared)' | |
| type: string | |
| default: '5' | |
| query_regression: | |
| description: 'Fail if a single query is more than this much slower' | |
| type: string | |
| default: '1.20' | |
| total_regression: | |
| description: 'Fail if the total time is more than this much slower' | |
| type: string | |
| default: '1.05' | |
| profile: | |
| description: 'Cargo profile to build both sides with' | |
| type: choice | |
| options: | |
| - release-nonlto | |
| - release | |
| default: release-nonlto | |
| permissions: | |
| contents: read | |
| env: | |
| # `release-nonlto` is `release` with `lto = false` and 16 codegen units. Fat | |
| # LTO with a single codegen unit roughly doubles the build, and both sides are | |
| # built identically, so the ratio the gate looks at still holds. Dispatch with | |
| # `release` when a change is expected to interact with cross-crate inlining, | |
| # or to get numbers comparable with locally posted `bench.sh` results. | |
| CARGO_PROFILE: ${{ inputs.profile || 'release-nonlto' }} | |
| ITERATIONS: ${{ inputs.iterations || '5' }} | |
| QUERY_REGRESSION: ${{ inputs.query_regression || '1.20' }} | |
| TOTAL_REGRESSION: ${{ inputs.total_regression || '1.05' }} | |
| # `benchmark_runner` finds `sql_benchmarks` through the CARGO_MANIFEST_DIR | |
| # baked into it at compile time, so a binary built in one job only works in | |
| # another if its tree sits at the same absolute path there. Every job below | |
| # puts the two trees under this root, which is why it is a fixed path rather | |
| # than something derived from the workspace or the runner. | |
| BENCH_ROOT: /tmp/df-bench | |
| # Same cargo network settings as .github/actions/setup-rust-runtime, without | |
| # its RUSTFLAGS: benchmark binaries are built with the defaults. | |
| CARGO_HTTP_MULTIPLEXING: "false" | |
| CARGO_NET_RETRY: "10" | |
| CARGO_HTTP_RETRY: "10" | |
| jobs: | |
| # Resolved once, so the three jobs below cannot disagree about what "base" is. | |
| resolve: | |
| name: resolve base commit | |
| if: github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'performance') | |
| runs-on: ${{ vars.USE_RUNS_ON == 'true' && format('runs-on={0},family=c8a+m8a,cpu=2,image=ubuntu24-full-x64,extras=s3-cache,tag=datafusion', github.run_id) || 'ubuntu-latest' }} | |
| timeout-minutes: 15 | |
| outputs: | |
| base_sha: ${{ steps.base.outputs.sha }} | |
| steps: | |
| - uses: runs-on/action@46910bf61b41721b0579f237e186afb35477007a # v2.3.0 | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 | |
| with: | |
| # For `pull_request` this is the PR already merged into the base | |
| # branch; depth 2 is enough to also reach the merge's first parent. | |
| fetch-depth: 2 | |
| - name: Resolve base commit | |
| id: base | |
| env: | |
| BASE_REF: ${{ github.event.pull_request.base.ref || github.event.repository.default_branch }} | |
| run: | | |
| set -euo pipefail | |
| if [ "$(git rev-list --parents -n 1 HEAD | wc -w)" -ge 3 ]; then | |
| # HEAD is the PR merged into the base branch, so its first parent | |
| # is the base branch tip that the merge used -- the commit this PR | |
| # would actually land on. | |
| base_sha=$(git rev-parse "HEAD^1") | |
| else | |
| # Manual run on a branch: compare it as-is against the base tip. | |
| git fetch --no-tags --depth 1 origin "$BASE_REF" | |
| base_sha=$(git rev-parse FETCH_HEAD) | |
| echo "HEAD is not a merge commit, comparing it as-is against $BASE_REF" | |
| fi | |
| echo "sha=${base_sha}" >> "$GITHUB_OUTPUT" | |
| echo "base: ${base_sha} $(git log -1 --format=%s "${base_sha}")" | |
| echo "candidate: $(git rev-parse HEAD) $(git log -1 --format=%s HEAD)" | |
| # One runner per side, so the two builds really do run at the same time. | |
| build: | |
| name: build ${{ matrix.side }} runner | |
| needs: resolve | |
| runs-on: ${{ vars.USE_RUNS_ON == 'true' && format('runs-on={0},family=c8a+m8a,cpu=32,image=ubuntu24-full-x64,extras=s3-cache,disk=large,tag=datafusion', github.run_id) || 'ubuntu-latest' }} | |
| timeout-minutes: 90 | |
| strategy: | |
| # No point in building one side if the other one is broken. | |
| fail-fast: true | |
| matrix: | |
| side: [base, pr] | |
| steps: | |
| - uses: runs-on/action@46910bf61b41721b0579f237e186afb35477007a # v2.3.0 | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 | |
| with: | |
| fetch-depth: 2 | |
| - name: Free Disk Space (Ubuntu) | |
| uses: jlumbroso/free-disk-space@54081f138730dfa15788a46383842cd2f914a1be # v1.3.1 | |
| - name: Install Rust | |
| run: | | |
| if ! command -v rustup > /dev/null; then | |
| curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y --default-toolchain none | |
| echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" | |
| export PATH="$HOME/.cargo/bin:$PATH" | |
| fi | |
| # installs the channel pinned by rust-toolchain.toml | |
| rustup toolchain install | |
| - name: Check out the ${{ matrix.side }} tree | |
| env: | |
| SIDE: ${{ matrix.side }} | |
| BASE_SHA: ${{ needs.resolve.outputs.base_sha }} | |
| run: | | |
| if [ "$SIDE" = "base" ]; then | |
| git worktree add --detach "$BENCH_ROOT/$SIDE" "$BASE_SHA" | |
| else | |
| git worktree add --detach "$BENCH_ROOT/$SIDE" HEAD | |
| fi | |
| - name: Build benchmark_runner | |
| env: | |
| SIDE: ${{ matrix.side }} | |
| CARGO_TARGET_DIR: ${{ runner.temp }}/target | |
| run: | | |
| cd "$BENCH_ROOT/$SIDE" | |
| cargo build --profile "$CARGO_PROFILE" -p datafusion-benchmarks --bin benchmark_runner | |
| cp "$CARGO_TARGET_DIR/$CARGO_PROFILE/benchmark_runner" "$RUNNER_TEMP/benchmark_runner" | |
| - name: Upload benchmark_runner | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: benchmark_runner-${{ matrix.side }} | |
| path: ${{ runner.temp }}/benchmark_runner | |
| retention-days: 1 | |
| # Both sides are measured here, on this one machine, back to back. | |
| benchmark: | |
| name: TPC-H SF1 (PR vs base) | |
| needs: [resolve, build] | |
| runs-on: ${{ vars.USE_RUNS_ON == 'true' && format('runs-on={0},family=c8a+m8a,cpu=16,image=ubuntu24-full-x64,extras=s3-cache,disk=large,tag=datafusion', github.run_id) || 'ubuntu-latest' }} | |
| timeout-minutes: 90 | |
| steps: | |
| - uses: runs-on/action@46910bf61b41721b0579f237e186afb35477007a # v2.3.0 | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 | |
| with: | |
| fetch-depth: 2 | |
| - name: Configure paths | |
| run: | | |
| echo "DATA_DIR=$RUNNER_TEMP/bench-data" >> "$GITHUB_ENV" | |
| echo "RESULTS_DIR=$RUNNER_TEMP/results" >> "$GITHUB_ENV" | |
| - name: Check out both trees | |
| env: | |
| BASE_SHA: ${{ needs.resolve.outputs.base_sha }} | |
| run: | | |
| # The same paths the build jobs used, so each binary finds the | |
| # `sql_benchmarks` directory of the tree it was built from. | |
| git worktree add --detach "$BENCH_ROOT/base" "$BASE_SHA" | |
| git worktree add --detach "$BENCH_ROOT/pr" HEAD | |
| - name: Download base benchmark_runner | |
| uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 | |
| with: | |
| name: benchmark_runner-base | |
| path: ${{ runner.temp }}/bin/base | |
| - name: Download PR benchmark_runner | |
| uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 | |
| with: | |
| name: benchmark_runner-pr | |
| path: ${{ runner.temp }}/bin/pr | |
| - name: Check both runners see their queries | |
| run: | | |
| # Artifacts do not carry the executable bit, and a binary whose tree | |
| # is missing at the path baked into it would discover no benchmark at | |
| # all -- fail here rather than three steps later. | |
| for side in base pr; do | |
| binary="$RUNNER_TEMP/bin/$side/benchmark_runner" | |
| chmod +x "$binary" | |
| if ! "$binary" --list | grep -qE '^[[:space:]]+tpch[[:space:]]'; then | |
| echo "::error::the $side runner does not see the tpch suite at $BENCH_ROOT/$side/benchmarks" | |
| "$binary" --list | |
| exit 1 | |
| fi | |
| done | |
| - name: Install Rust | |
| # Only so tpchgen-cli can be built from source if no prebuilt binary | |
| # matches this runner; nothing here is compiled otherwise. | |
| run: | | |
| if ! command -v rustup > /dev/null; then | |
| curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y --default-toolchain none | |
| echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" | |
| export PATH="$HOME/.cargo/bin:$PATH" | |
| fi | |
| rustup toolchain install | |
| - name: Install uv | |
| uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 | |
| - name: Install tpchgen-cli | |
| uses: taiki-e/install-action@7f4eb899022d8fe70b20c4f3de697aa85c309026 # v2.85.11 | |
| with: | |
| # pinned, so the generated data does not change under the benchmark | |
| tool: tpchgen-cli@3.0.0 | |
| - name: Generate TPC-H SF1 data | |
| # Same generator settings as `bench.sh data tpch`, inlined because that | |
| # path also downloads the expected answers through `docker run -it`, | |
| # which needs a TTY and is only used for result validation. | |
| run: | | |
| mkdir -p "$DATA_DIR/tpch_sf1" "$RESULTS_DIR/base" "$RESULTS_DIR/pr" | |
| tpchgen-cli \ | |
| --scale-factor 1 \ | |
| --format parquet \ | |
| --parquet-compression 'ZSTD(1)' \ | |
| --parts=1 \ | |
| --output-dir "$DATA_DIR/tpch_sf1" | |
| # Each side runs from its own tree and reads the one generated dataset. | |
| - name: Benchmark base | |
| run: | | |
| cd "$BENCH_ROOT/base/benchmarks" | |
| "$RUNNER_TEMP/bin/base/benchmark_runner" tpch \ | |
| --scale-factor 1 \ | |
| --format parquet \ | |
| --iterations "$ITERATIONS" \ | |
| --path "$DATA_DIR" \ | |
| --output "$RESULTS_DIR/base/tpch_sf1.json" | |
| - name: Benchmark PR | |
| run: | | |
| cd "$BENCH_ROOT/pr/benchmarks" | |
| "$RUNNER_TEMP/bin/pr/benchmark_runner" tpch \ | |
| --scale-factor 1 \ | |
| --format parquet \ | |
| --iterations "$ITERATIONS" \ | |
| --path "$DATA_DIR" \ | |
| --output "$RESULTS_DIR/pr/tpch_sf1.json" | |
| - name: Compare | |
| run: | | |
| set -uo pipefail | |
| status=0 | |
| uv run --no-project --with rich python3 benchmarks/compare.py \ | |
| "$RESULTS_DIR/base/tpch_sf1.json" \ | |
| "$RESULTS_DIR/pr/tpch_sf1.json" \ | |
| --fail-threshold "$QUERY_REGRESSION" \ | |
| --fail-total-threshold "$TOTAL_REGRESSION" \ | |
| > "$RESULTS_DIR/comparison.txt" 2>&1 || status=$? | |
| cat "$RESULTS_DIR/comparison.txt" | |
| { | |
| echo "### TPC-H SF1: base (${{ needs.resolve.outputs.base_sha }}) vs PR" | |
| echo | |
| echo "\`$ITERATIONS\` iterations per query; the fastest of each is compared, to keep runner noise out of the ratio." | |
| echo "Fails above \`${QUERY_REGRESSION}x\` for a single query or \`${TOTAL_REGRESSION}x\` in total." | |
| echo "Both sides built with the \`$CARGO_PROFILE\` profile." | |
| echo | |
| echo '```' | |
| cat "$RESULTS_DIR/comparison.txt" | |
| echo '```' | |
| } >> "$GITHUB_STEP_SUMMARY" | |
| if [ "$status" -ne 0 ]; then | |
| echo "::error::TPC-H SF1 got slower than the configured limits allow, see the job summary" | |
| fi | |
| exit "$status" | |
| - name: Upload benchmark results | |
| if: always() | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: tpch-sf1-comparison | |
| path: ${{ runner.temp }}/results | |
| retention-days: 7 |