diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml new file mode 100644 index 00000000..f7f12071 --- /dev/null +++ b/.github/workflows/benchmark.yml @@ -0,0 +1,51 @@ +name: ASV Benchmarks + +on: + push: + branches: [ "main" ] + pull_request: + branches:[ "main" ] + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + +jobs: + benchmark: + name: Run ASV Performance Tests + runs-on: ubuntu-latest + steps: + - name: Checkout code + uses: actions/checkout@v4 + with: + fetch-depth: 0 + + - name: Install uv + uses: astral-sh/setup-uv@v7 + with: + enable-cache: true + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.11" + + - name: Install ASV and pyGAM dependencies + run: | + uv pip install asv virtualenv + uv pip install -e ".[dev]" + env: + UV_SYSTEM_PYTHON: 1 + + - name: Run ASV Benchmarks + run: | + asv machine --yes + + if [ "${{ github.event_name }}" == "pull_request" ]; then + asv continuous origin/main HEAD --show-stderr + else + asv run ALL --quick --show-stderr + fi diff --git a/.gitignore b/.gitignore index bedeacf1..655a535b 100644 --- a/.gitignore +++ b/.gitignore @@ -54,3 +54,4 @@ _build/ # PyCharm ######### .idea/ +.asv/ diff --git a/asv.conf.json b/asv.conf.json new file mode 100644 index 00000000..0e28393e --- /dev/null +++ b/asv.conf.json @@ -0,0 +1,14 @@ +{ + "version": 1, + "project": "pygam", + "project_url": "https://github.com/dswah/pyGAM", + "repo": ".", + "branches": ["main"], + "environment_type": "virtualenv", + "build_command": [], + "install_command": ["python", "-m", "pip", "install", "{build_dir}"], + "benchmark_dir": "benchmarks", + "env_dir": ".asv/env", + "results_dir": ".asv/results", + "html_dir": ".asv/html" +} diff --git a/benchmarks/__init__.py b/benchmarks/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/benchmarks/bench_edof.py b/benchmarks/bench_edof.py new file mode 100644 index 00000000..ef5654e6 --- /dev/null +++ b/benchmarks/bench_edof.py @@ -0,0 +1,32 @@ +import os + +# Lock threads for deterministic memory and time profiling +os.environ["OMP_NUM_THREADS"] = "1" +os.environ["MKL_NUM_THREADS"] = "1" +os.environ["OPENBLAS_NUM_THREADS"] = "1" + +import numpy as np + + +class EDoFBenchmark: + """ + Micro-benchmarks for Effective Degrees of Freedom (EDoF) calculation. + Compares legacy O(N^3) memory bottleneck vs optimized O(N) approach. + """ + + timeout = 120 + # Dimensions chosen to safely hit ~800MB RAM, well below CI 7GB limit + N_FEATURES = 10000 + N_SAMPLES = 500 + + def setup(self): + np.random.seed(42) + self.U1 = np.random.rand(self.N_FEATURES, self.N_SAMPLES) + + def time_legacy_edof(self): + # Legacy dense matrix multiplication O(N^3) time + return np.diagonal(self.U1.dot(self.U1.T)) + + def peakmem_legacy_edof(self): + # Legacy dense matrix multiplication O(N^2) space (~800MB) + return np.diagonal(self.U1.dot(self.U1.T)) diff --git a/benchmarks/bench_fit.py b/benchmarks/bench_fit.py new file mode 100644 index 00000000..43f639e9 --- /dev/null +++ b/benchmarks/bench_fit.py @@ -0,0 +1,59 @@ +import os + +# Lock threads to 1 for deterministic benchmarking across environments +os.environ["OMP_NUM_THREADS"] = "1" +os.environ["MKL_NUM_THREADS"] = "1" +os.environ["OPENBLAS_NUM_THREADS"] = "1" + +import numpy as np + +from pygam import LinearGAM, PoissonGAM, s + + +class LinearGAMFit: + """Macro-benchmarks for LinearGAM training and inference.""" + + number = 1 + repeat = 3 + timeout = 60.0 + + def setup(self): + # Reproducing Synthetic data + np.random.seed(42) + self.X = np.random.rand(3000, 3) + self.y = self.X[:, 0] * 2 + self.X[:, 1] ** 2 + np.random.randn(3000) * 0.1 + + self.gam = LinearGAM(s(0) + s(1) + s(2)) + self.gam_fitted = LinearGAM(s(0) + s(1) + s(2)).fit(self.X, self.y) + self.lam_grid = np.logspace(-3, 3, 3) + + def time_fit(self): + # Measures the time of the core fitting logic + self.gam.fit(self.X, self.y) + + def time_predict(self): + # Measures inference speed + self.gam_fitted.predict(self.X) + + def time_gridsearch(self): + # Measures hyperparameter tuning overhead + self.gam.gridsearch(self.X, self.y, lam=self.lam_grid, progress=False) + + +class PoissonGAMFit: + """Macro-benchmarks for PoissonGAM (tests the iterative PIRLS loop).""" + + number = 1 + repeat = 3 + timeout = 60.0 + + def setup(self): + np.random.seed(42) + self.X = np.random.rand(500, 3) + expected_rate = np.exp(self.X[:, 0] * 0.5) + self.y = np.random.poisson(lam=expected_rate) + self.gam = PoissonGAM(s(0) + s(1) + s(2)) + + def time_fit(self): + # PIRLS loop timing + self.gam.fit(self.X, self.y)