Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion docs/commands/perf.md
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@ $ winml perf [options]

| Flag | Short | Type | Default | Description |
|---|---|---|---|---|
| `--model` | `-m` | `TEXT` | — | HuggingFace model ID or path to a local `.onnx` file. With `--runtime winml-runtime`, also accepts a prebuilt CGC `.mlir` file. Required. With `--runtime ort-genai`, also accepts a prebuilt genai **bundle directory**, or a HuggingFace model ID that is auto-built into a bundle on demand. |
| `--model` | `-m` | `TEXT` | — | HuggingFace model ID, local `.onnx` file, or ModelKit build output directory containing `model.onnx` plus either `winml_build_config.json` or `build_manifest.json`. With `--runtime winml-runtime`, also accepts a prebuilt CGC `.mlir` file. Required. With `--runtime ort-genai`, also accepts a prebuilt genai **bundle directory**, or a HuggingFace model ID that is auto-built into a bundle on demand. |
| `--runtime` | | `auto\|winml-ort\|ort-genai\|winml-runtime` | `auto` | Inference runtime. `auto` selects `ort-genai` for local folders containing `genai_config.json`, `winml-runtime` for `.mlir` files, otherwise `winml-ort`; `winml-ort` benchmarks single-shot ONNX inference; `ort-genai` benchmarks an onnxruntime-genai bundle (LLM generation: time-to-first-token + decode tokens/sec); `winml-runtime` runs ONNX or prebuilt CGC MLIR through Windows ML Runtime. With `ort-genai`, a model ID that is not a bundle directory is auto-built into one before benchmarking. An explicit `--ep` or `--device` selects both the transformer build and runtime target; without an override, the auto-build defaults to QNN/NPU. Bundles are cached under `~/.cache/winml/`, separately for each explicit EP/device target. GenAI cache controls are tracked in issue #1275. |
| `--task` | | `TEXT` | auto-detected | Explicit task override (e.g., `image-classification`). Inferred from the model if omitted. |
| `--iterations` | | `INTEGER` | `100` (`10` with `--op-tracing`) | Number of timed inference iterations used to compute statistics. Explicit values override the op-tracing default. |
Expand Down
16 changes: 16 additions & 0 deletions src/winml/modelkit/commands/perf.py
Original file line number Diff line number Diff line change
Expand Up @@ -1348,6 +1348,12 @@ def _load_model(self) -> None:

model_id = self.config.model_id
model_path = Path(model_id)
from ..utils.model_input import resolve_build_output

build_onnx = resolve_build_output(model_path)
if build_onnx is not None:
model_path = build_onnx
model_id = str(model_path)
is_onnx = model_path.suffix.lower() == ".onnx"
is_mlir = model_path.suffix.lower() == ".mlir"
if (is_onnx or is_mlir) and not model_path.exists():
Expand Down Expand Up @@ -3346,6 +3352,16 @@ def perf(
model_input = classify_model_input(hf_model)
if model_input.kind is ModelInputKind.INVALID:
raise click.UsageError(model_input.error or f"Invalid model input: {hf_model}")
if model_input.kind is ModelInputKind.FOLDER:
from ..utils.model_input import resolve_build_output

try:
build_onnx = resolve_build_output(Path(hf_model))
except FileNotFoundError as exc:
raise click.UsageError(str(exc)) from exc
if build_onnx is not None:
hf_model = str(build_onnx)
model_input = classify_model_input(hf_model)
is_onnx = model_input.kind is ModelInputKind.ONNX_FILE
if is_onnx and model_input.local_path and not Path(model_input.local_path).exists():
raise click.UsageError(f"ONNX file not found: {hf_model}")
Expand Down
12 changes: 12 additions & 0 deletions src/winml/modelkit/utils/model_input.py
Original file line number Diff line number Diff line change
Expand Up @@ -92,6 +92,18 @@ def _classify_local_dir(path: Path, raw: str) -> ModelInput:
return ModelInput(kind=ModelInputKind.FOLDER, raw=raw, local_path=str(path))


def resolve_build_output(path: Path) -> Path | None:
"""Return the final ONNX from a plain ModelKit build directory, if identified."""
if not path.is_dir() or not any(
(path / marker).is_file() for marker in ("winml_build_config.json", "build_manifest.json")
):
return None
onnx_path = path / "model.onnx"
if not onnx_path.is_file():
raise FileNotFoundError(f"Build output has no final model.onnx: {path}")
return onnx_path


def classify_model_input(value: str) -> ModelInput:
r"""Classify a ``-m/--model`` value without any network I/O.

Expand Down
60 changes: 60 additions & 0 deletions tests/unit/commands/test_perf_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -790,6 +790,41 @@ def test_onnx_load_model_calls_from_onnx(self, tmp_path: Path) -> None:
assert kwargs.kwargs["force_rebuild"] is False
assert benchmark._model is mock_model

@pytest.mark.parametrize("marker", ["winml_build_config.json", "build_manifest.json"])
def test_build_directory_loads_final_onnx(self, tmp_path: Path, marker: str) -> None:
(tmp_path / "model.onnx").write_bytes(b"fake onnx")
(tmp_path / marker).write_text("{}")
benchmark = PerfBenchmark(BenchmarkConfig(model_id=str(tmp_path), device="cpu"))

with (
patch(
"winml.modelkit.models.auto.WinMLAutoModel.from_onnx",
return_value=MagicMock(),
) as mock_from_onnx,
patch(
"winml.modelkit.models.auto.WinMLAutoModel.from_pretrained",
side_effect=AssertionError("build directory must not use HF loading"),
),
):
benchmark._load_model()

assert mock_from_onnx.call_args.kwargs["onnx_path"] == tmp_path / "model.onnx"
assert mock_from_onnx.call_args.kwargs["skip_build"] is True

def test_hf_directory_still_loads_from_pretrained(self, tmp_path: Path) -> None:
(tmp_path / "config.json").write_text("{}")
benchmark = PerfBenchmark(
BenchmarkConfig(model_id=str(tmp_path), task="image-classification", device="cpu")
)
with patch(
"winml.modelkit.models.auto.WinMLAutoModel.from_pretrained",
return_value=MagicMock(),
) as mock_from_pretrained:
benchmark._load_model()

mock_from_pretrained.assert_called_once()
assert mock_from_pretrained.call_args.args[0] == str(tmp_path)

def test_hf_load_model_calls_from_pretrained(self) -> None:
"""HF model input should use WinMLAutoModel.from_pretrained in _load_model."""
config = BenchmarkConfig(
Expand Down Expand Up @@ -974,6 +1009,31 @@ def test_cli_onnx_routes_through_perf_benchmark(
assert result.exit_code == 0, result.output
mock_perf_cls.assert_called_once()

def test_cli_build_directory_routes_as_onnx(self, runner: CliRunner, tmp_path: Path) -> None:
(tmp_path / "model.onnx").write_bytes(b"fake onnx")
(tmp_path / "winml_build_config.json").write_text("{}")
with (
patch("winml.modelkit.commands.perf.PerfBenchmark") as mock_perf_cls,
patch("winml.modelkit.commands.perf.display_console_report"),
patch("winml.modelkit.commands.perf.write_json_report"),
):
mock_perf_cls.return_value.run.return_value = MagicMock()
result = runner.invoke(
perf, ["-m", str(tmp_path), "-o", str(tmp_path / "out.json")], obj={}
)

assert result.exit_code == 0, result.output
assert mock_perf_cls.call_args.args[0].model_id == str(tmp_path / "model.onnx")

def test_cli_incomplete_build_directory_reports_missing_model(
self, runner: CliRunner, tmp_path: Path
) -> None:
(tmp_path / "winml_build_config.json").write_text("{}")
result = runner.invoke(perf, ["-m", str(tmp_path)], obj={})

assert result.exit_code == 2
assert "no final model.onnx" in result.output

def test_cli_onnx_winml_runtime_preserves_explicit_target(
self, runner: CliRunner, tmp_path: Path
) -> None:
Expand Down