Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,6 @@ jobs:
with:
enable-cache: true
- name: Install
run: uv sync --python 3.12 --extra onnx-cpu --group dev
run: uv sync --python 3.12 --extra onnx-cpu --extra openvino --group dev
- name: Test
run: uv run pytest -q
15 changes: 15 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,18 @@ SAB uses [uv](https://docs.astral.sh/uv/). Each host installs the extra for its
uv run python -m sab.models.benchmark_rfdetr <path to coco val dir> <path to coco val annotations>
```

Other hosts install the extra for each runtime they need:

| Extra | Adds | Host |
|---|---|---|
| `openvino` | `OpenVINORuntime` | Any CPU host. |

The `nvidia` extra now includes `openvino`, so `benchmark_all` also runs the OpenVINO CPU rows. These rows take most of the run time. To run only the TensorRT and ONNX Runtime rows, add `--runtimes=tensorrt,onnxruntime`.

```bash
uv sync --python 3.12 --extra openvino
```

### Options

`benchmark_all` and each script accept these flags:
Expand All @@ -71,6 +83,9 @@ Each runtime times the smallest call that runs the full graph. Preprocessing and
| TensorRT | CUDA graph replay, or `execute_async_v3` | Not included. The input is already in GPU memory. |
| ONNX Runtime (GPU) | `run_with_iobinding` | Not included. IOBinding binds GPU buffers. |
| ONNX Runtime (CPU) | `run_with_iobinding` | None (CPU). |
| OpenVINO | `InferRequest.infer()` | None (CPU). |

OpenVINO rows read the existing `.onnx` files and compile them on the host, as TensorRT rows do. They set `INFERENCE_PRECISION_HINT` from the precision of the row. On a CPU with no native fp16, the fp16 rows are skipped, because OpenVINO compiles them in f32.

### Throttle detection

Expand Down
4 changes: 3 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -30,8 +30,9 @@ tensorrt = [
]
onnx-cpu = ["onnxruntime>=1.22.0"]
onnx-gpu = ["onnxruntime-gpu>=1.22.0"]
openvino = ["openvino>=2025.0"]
# One extra for each host.
nvidia = ["single-artifact-benchmarking[tensorrt,onnx-gpu]"]
nvidia = ["single-artifact-benchmarking[tensorrt,onnx-gpu,openvino]"]

[dependency-groups]
dev = ["pytest>=8"]
Expand All @@ -56,6 +57,7 @@ testpaths = ["tests"]
markers = [
"tensorrt: needs TensorRT and an NVIDIA GPU",
"onnxruntime: needs onnxruntime",
"openvino: needs openvino",
]

# PyPI holds only stub source packages for TensorRT 10; the wheels are on NVIDIA's index.
Expand Down
35 changes: 30 additions & 5 deletions sab/models/benchmark_rfdetr.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
from sab.results import pretty_print_results
from sab.runner import run_benchmark_on_artifacts
from sab.runtimes.onnxruntime import ONNXRuntime
from sab.runtimes.openvino import OpenVINORuntime
from sab.runtimes.tensorrt import TRTRuntime


Expand Down Expand Up @@ -52,29 +53,40 @@ def postprocess(self, outputs: dict[str, torch.Tensor], metadata: dict) -> tuple
return postprocess_output(outputs, metadata)


# rf-detr-large.onnx in the bucket is the deprecated large model. The current large model is rf-detr-large-new.onnx.
ONNX_FILES = (
"rf-detr-nano.onnx",
"rf-detr-small.onnx",
"rf-detr-medium.onnx",
"rf-detr-large-new.onnx",
"rf-detr-xlarge.onnx",
"rf-detr-xxlarge.onnx",
)


def build_requests(buffer_time: float = 0.0) -> list[ArtifactBenchmarkRequest]:
return [
onnx_rows = [
request
for size in ("nano", "small", "medium")
for onnx_file in ONNX_FILES
for request in (
ArtifactBenchmarkRequest(
artifact_path=f"rf-detr-{size}.onnx",
artifact_path=onnx_file,
runtime=TRTRuntime,
processor=RFDETRProcessor,
device="gpu",
precision="fp32",
buffer_time=buffer_time,
),
ArtifactBenchmarkRequest(
artifact_path=f"rf-detr-{size}.onnx",
artifact_path=onnx_file,
runtime=TRTRuntime,
processor=RFDETRProcessor,
device="gpu",
precision="fp16",
buffer_time=buffer_time,
),
ArtifactBenchmarkRequest(
artifact_path=f"rf-detr-{size}.onnx",
artifact_path=onnx_file,
runtime=ONNXRuntime,
processor=RFDETRProcessor,
device="cpu",
Expand All @@ -83,6 +95,19 @@ def build_requests(buffer_time: float = 0.0) -> list[ArtifactBenchmarkRequest]:
),
)
]
openvino_rows = [
ArtifactBenchmarkRequest(
artifact_path=onnx_file,
runtime=OpenVINORuntime,
processor=RFDETRProcessor,
device="cpu",
precision=precision,
buffer_time=buffer_time,
)
for onnx_file in ONNX_FILES
for precision in ("fp32", "fp16")
]
return onnx_rows + openvino_rows


def main(
Expand Down
17 changes: 16 additions & 1 deletion sab/models/benchmark_yolo26.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
from sab.results import pretty_print_results
from sab.runner import run_benchmark_on_artifacts
from sab.runtimes.onnxruntime import ONNXRuntime
from sab.runtimes.openvino import OpenVINORuntime
from sab.runtimes.tensorrt import TRTRuntime


Expand All @@ -22,7 +23,7 @@ class YOLO26Processor(YOLOv11Processor):


def build_requests(buffer_time: float = 0.0) -> list[ArtifactBenchmarkRequest]:
return [
onnx_rows = [
request
for size in ("n", "s", "m", "l", "x")
for request in (
Expand Down Expand Up @@ -55,6 +56,20 @@ def build_requests(buffer_time: float = 0.0) -> list[ArtifactBenchmarkRequest]:
),
)
]
openvino_rows = [
ArtifactBenchmarkRequest(
artifact_path=f"yolo26{size}.onnx",
runtime=OpenVINORuntime,
processor=YOLO26Processor,
device="cpu",
precision=precision,
buffer_time=buffer_time,
needs_class_remapping=True,
)
for size in ("n", "s", "m", "l", "x")
for precision in ("fp32", "fp16")
]
return onnx_rows + openvino_rows


def main(
Expand Down
17 changes: 16 additions & 1 deletion sab/models/benchmark_yolov11.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
from sab.results import pretty_print_results
from sab.runner import run_benchmark_on_artifacts
from sab.runtimes.onnxruntime import ONNXRuntime
from sab.runtimes.openvino import OpenVINORuntime
from sab.runtimes.tensorrt import TRTRuntime

TRT_WITHOUT_CUDA_GRAPH = partial(TRTRuntime, use_cuda_graph=False)
Expand Down Expand Up @@ -96,7 +97,7 @@ def postprocess(self, outputs: dict[str, torch.Tensor], metadata: dict) -> tuple


def build_requests(buffer_time: float = 0.0) -> list[ArtifactBenchmarkRequest]:
return [
onnx_rows = [
request
for size in ("n", "s", "m", "l", "x")
for request in (
Expand Down Expand Up @@ -129,6 +130,20 @@ def build_requests(buffer_time: float = 0.0) -> list[ArtifactBenchmarkRequest]:
),
)
]
openvino_rows = [
ArtifactBenchmarkRequest(
artifact_path=f"yolo11{size}_nms_conf_0.01.onnx",
runtime=OpenVINORuntime,
processor=YOLOv11Processor,
device="cpu",
precision=precision,
buffer_time=buffer_time,
needs_class_remapping=True,
)
for size in ("n", "s", "m", "l", "x")
for precision in ("fp32", "fp16")
]
return onnx_rows + openvino_rows


def main(
Expand Down
17 changes: 16 additions & 1 deletion sab/models/benchmark_yolov8.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,14 +4,15 @@
from sab.request import ArtifactBenchmarkRequest
from sab.results import pretty_print_results
from sab.runner import run_benchmark_on_artifacts
from sab.runtimes.openvino import OpenVINORuntime


class YOLOv8Processor(YOLOv11Processor):
"""YOLOv8 exports use the same letterbox and NMS output layout as YOLOv11."""


def build_requests(buffer_time: float = 0.0) -> list[ArtifactBenchmarkRequest]:
return [
onnx_rows = [
ArtifactBenchmarkRequest(
artifact_path=f"yolov8{size}_nms_conf_0.01.onnx",
runtime=TRT_WITHOUT_CUDA_GRAPH,
Expand All @@ -23,6 +24,20 @@ def build_requests(buffer_time: float = 0.0) -> list[ArtifactBenchmarkRequest]:
)
for size in ("n", "s", "m")
]
openvino_rows = [
ArtifactBenchmarkRequest(
artifact_path=f"yolov8{size}_nms_conf_0.01.onnx",
runtime=OpenVINORuntime,
processor=YOLOv8Processor,
device="cpu",
precision=precision,
buffer_time=buffer_time,
needs_class_remapping=True,
)
for size in ("n", "s", "m")
for precision in ("fp32", "fp16")
]
return onnx_rows + openvino_rows


def main(
Expand Down
8 changes: 6 additions & 2 deletions sab/runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@
from sab.processors import Pipeline
from sab.request import ArtifactBenchmarkRequest
from sab.results import load_results, result_key, save_results
from sab.runtimes.base import runtime_class
from sab.runtimes.base import UnavailableOnHost, runtime_class


def parse_filter(value: str | Iterable[str] | None) -> set[str] | None:
Expand Down Expand Up @@ -139,7 +139,11 @@ def store(row: dict):
rows.append(stored_rows[stored_index[key]])
continue

row = run_benchmark_on_artifact(request, images_dir, annotations_file_path)
try:
row = run_benchmark_on_artifact(request, images_dir, annotations_file_path)
except UnavailableOnHost as reason:
print(f"Skipping {request.artifact_path} ({request.runtime_name}, {request.device}, {request.precision}): {reason}.")
continue
print(row)
store(row)
rows.append(row)
Expand Down
2 changes: 2 additions & 0 deletions sab/runtimes/__init__.py
Original file line number Diff line number Diff line change
@@ -1,12 +1,14 @@
from sab.runtimes.base import DEVICES, PRECISIONS, InputSpec, Runtime, RuntimeFactory, runtime_class
from sab.runtimes.onnxruntime import ONNXRuntime
from sab.runtimes.openvino import OpenVINORuntime
from sab.runtimes.tensorrt import TRTRuntime

__all__ = [
"DEVICES",
"PRECISIONS",
"InputSpec",
"ONNXRuntime",
"OpenVINORuntime",
"Runtime",
"RuntimeFactory",
"TRTRuntime",
Expand Down
4 changes: 4 additions & 0 deletions sab/runtimes/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,10 @@ class InputSpec:
shape: tuple[int, ...]


class UnavailableOnHost(RuntimeError):
"""The artifact loaded, but this host cannot run the row as requested. The runner skips the row."""


def pick_image_input(input_names: list[str], image_input_name: str | None) -> str:
"""The name of the image input: the given name, the only input, or the input named "images"."""
if image_input_name is not None:
Expand Down
Loading
Loading