diff --git a/docs/commands/perf.md b/docs/commands/perf.md index bc8045e88..b0f5fc731 100644 --- a/docs/commands/perf.md +++ b/docs/commands/perf.md @@ -16,7 +16,7 @@ $ winml perf [options] | Flag | Short | Type | Default | Description | |---|---|---|---|---| -| `--model` | `-m` | `TEXT` | — | HuggingFace model ID or path to a local `.onnx` file. With `--runtime winml-runtime`, also accepts a prebuilt CGC `.mlir` file. Required. With `--runtime ort-genai`, also accepts a prebuilt genai **bundle directory**, or a HuggingFace model ID that is auto-built into a bundle on demand. | +| `--model` | `-m` | `TEXT` | — | HuggingFace model ID, local `.onnx` file, or ModelKit build output directory containing `model.onnx` plus either `winml_build_config.json` or `build_manifest.json`. With `--runtime winml-runtime`, also accepts a prebuilt CGC `.mlir` file. Required. With `--runtime ort-genai`, also accepts a prebuilt genai **bundle directory**, or a HuggingFace model ID that is auto-built into a bundle on demand. | | `--runtime` | | `auto\|winml-ort\|ort-genai\|winml-runtime` | `auto` | Inference runtime. `auto` selects `ort-genai` for local folders containing `genai_config.json`, `winml-runtime` for `.mlir` files, otherwise `winml-ort`; `winml-ort` benchmarks single-shot ONNX inference; `ort-genai` benchmarks an onnxruntime-genai bundle (LLM generation: time-to-first-token + decode tokens/sec); `winml-runtime` runs ONNX or prebuilt CGC MLIR through Windows ML Runtime. With `ort-genai`, a model ID that is not a bundle directory is auto-built into one before benchmarking. An explicit `--ep` or `--device` selects both the transformer build and runtime target; without an override, the auto-build defaults to QNN/NPU. Bundles are cached under `~/.cache/winml/`, separately for each explicit EP/device target. GenAI cache controls are tracked in issue #1275. | | `--task` | | `TEXT` | auto-detected | Explicit task override (e.g., `image-classification`). Inferred from the model if omitted. | | `--iterations` | | `INTEGER` | `100` (`10` with `--op-tracing`) | Number of timed inference iterations used to compute statistics. Explicit values override the op-tracing default. | diff --git a/src/winml/modelkit/commands/perf.py b/src/winml/modelkit/commands/perf.py index ee4bf2cec..559104a93 100644 --- a/src/winml/modelkit/commands/perf.py +++ b/src/winml/modelkit/commands/perf.py @@ -1348,6 +1348,12 @@ def _load_model(self) -> None: model_id = self.config.model_id model_path = Path(model_id) + from ..utils.model_input import resolve_build_output + + build_onnx = resolve_build_output(model_path) + if build_onnx is not None: + model_path = build_onnx + model_id = str(model_path) is_onnx = model_path.suffix.lower() == ".onnx" is_mlir = model_path.suffix.lower() == ".mlir" if (is_onnx or is_mlir) and not model_path.exists(): @@ -3346,6 +3352,16 @@ def perf( model_input = classify_model_input(hf_model) if model_input.kind is ModelInputKind.INVALID: raise click.UsageError(model_input.error or f"Invalid model input: {hf_model}") + if model_input.kind is ModelInputKind.FOLDER: + from ..utils.model_input import resolve_build_output + + try: + build_onnx = resolve_build_output(Path(hf_model)) + except FileNotFoundError as exc: + raise click.UsageError(str(exc)) from exc + if build_onnx is not None: + hf_model = str(build_onnx) + model_input = classify_model_input(hf_model) is_onnx = model_input.kind is ModelInputKind.ONNX_FILE if is_onnx and model_input.local_path and not Path(model_input.local_path).exists(): raise click.UsageError(f"ONNX file not found: {hf_model}") diff --git a/src/winml/modelkit/utils/model_input.py b/src/winml/modelkit/utils/model_input.py index bf48ad8b5..2b2a7a15b 100644 --- a/src/winml/modelkit/utils/model_input.py +++ b/src/winml/modelkit/utils/model_input.py @@ -92,6 +92,18 @@ def _classify_local_dir(path: Path, raw: str) -> ModelInput: return ModelInput(kind=ModelInputKind.FOLDER, raw=raw, local_path=str(path)) +def resolve_build_output(path: Path) -> Path | None: + """Return the final ONNX from a plain ModelKit build directory, if identified.""" + if not path.is_dir() or not any( + (path / marker).is_file() for marker in ("winml_build_config.json", "build_manifest.json") + ): + return None + onnx_path = path / "model.onnx" + if not onnx_path.is_file(): + raise FileNotFoundError(f"Build output has no final model.onnx: {path}") + return onnx_path + + def classify_model_input(value: str) -> ModelInput: r"""Classify a ``-m/--model`` value without any network I/O. diff --git a/tests/unit/commands/test_perf_cli.py b/tests/unit/commands/test_perf_cli.py index 8a4a994e2..7c9bdafa6 100644 --- a/tests/unit/commands/test_perf_cli.py +++ b/tests/unit/commands/test_perf_cli.py @@ -790,6 +790,41 @@ def test_onnx_load_model_calls_from_onnx(self, tmp_path: Path) -> None: assert kwargs.kwargs["force_rebuild"] is False assert benchmark._model is mock_model + @pytest.mark.parametrize("marker", ["winml_build_config.json", "build_manifest.json"]) + def test_build_directory_loads_final_onnx(self, tmp_path: Path, marker: str) -> None: + (tmp_path / "model.onnx").write_bytes(b"fake onnx") + (tmp_path / marker).write_text("{}") + benchmark = PerfBenchmark(BenchmarkConfig(model_id=str(tmp_path), device="cpu")) + + with ( + patch( + "winml.modelkit.models.auto.WinMLAutoModel.from_onnx", + return_value=MagicMock(), + ) as mock_from_onnx, + patch( + "winml.modelkit.models.auto.WinMLAutoModel.from_pretrained", + side_effect=AssertionError("build directory must not use HF loading"), + ), + ): + benchmark._load_model() + + assert mock_from_onnx.call_args.kwargs["onnx_path"] == tmp_path / "model.onnx" + assert mock_from_onnx.call_args.kwargs["skip_build"] is True + + def test_hf_directory_still_loads_from_pretrained(self, tmp_path: Path) -> None: + (tmp_path / "config.json").write_text("{}") + benchmark = PerfBenchmark( + BenchmarkConfig(model_id=str(tmp_path), task="image-classification", device="cpu") + ) + with patch( + "winml.modelkit.models.auto.WinMLAutoModel.from_pretrained", + return_value=MagicMock(), + ) as mock_from_pretrained: + benchmark._load_model() + + mock_from_pretrained.assert_called_once() + assert mock_from_pretrained.call_args.args[0] == str(tmp_path) + def test_hf_load_model_calls_from_pretrained(self) -> None: """HF model input should use WinMLAutoModel.from_pretrained in _load_model.""" config = BenchmarkConfig( @@ -974,6 +1009,31 @@ def test_cli_onnx_routes_through_perf_benchmark( assert result.exit_code == 0, result.output mock_perf_cls.assert_called_once() + def test_cli_build_directory_routes_as_onnx(self, runner: CliRunner, tmp_path: Path) -> None: + (tmp_path / "model.onnx").write_bytes(b"fake onnx") + (tmp_path / "winml_build_config.json").write_text("{}") + with ( + patch("winml.modelkit.commands.perf.PerfBenchmark") as mock_perf_cls, + patch("winml.modelkit.commands.perf.display_console_report"), + patch("winml.modelkit.commands.perf.write_json_report"), + ): + mock_perf_cls.return_value.run.return_value = MagicMock() + result = runner.invoke( + perf, ["-m", str(tmp_path), "-o", str(tmp_path / "out.json")], obj={} + ) + + assert result.exit_code == 0, result.output + assert mock_perf_cls.call_args.args[0].model_id == str(tmp_path / "model.onnx") + + def test_cli_incomplete_build_directory_reports_missing_model( + self, runner: CliRunner, tmp_path: Path + ) -> None: + (tmp_path / "winml_build_config.json").write_text("{}") + result = runner.invoke(perf, ["-m", str(tmp_path)], obj={}) + + assert result.exit_code == 2 + assert "no final model.onnx" in result.output + def test_cli_onnx_winml_runtime_preserves_explicit_target( self, runner: CliRunner, tmp_path: Path ) -> None: