From d595f13055ba4a0713120558ccdad5fb8e271949 Mon Sep 17 00:00:00 2001 From: Qiong Wu Date: Tue, 29 Sep 2026 14:34:33 +0800 Subject: [PATCH] fix(skills): make auto-optimize delivery evidence-bound --- .../agent-skill/auto-optimize.md | 33 ++ docs/getting-started/agent-skill/index.md | 3 + skills/auto-optimize/SKILL.md | 32 +- skills/auto-optimize/evals/README.md | 32 +- skills/auto-optimize/evals/harness.py | 22 +- skills/auto-optimize/evals/reliability-run.md | 27 ++ skills/auto-optimize/evals/run_evals.py | 34 ++- skills/auto-optimize/evals/scenarios.json | 139 ++++++++- skills/auto-optimize/knowledge/qnn-npu.md | 4 +- skills/auto-optimize/references/pr-routing.md | 9 + .../references/report-delivery.md | 32 ++ .../auto-optimize/references/reproduction.md | 34 +++ skills/auto-optimize/references/workflow.md | 42 +++ .../roles/feature-gap-engineer.md | 15 +- .../auto-optimize/scripts/aggregate_perf.py | 89 ++++++ skills/auto-optimize/scripts/render_report.py | 99 ++++-- skills/auto-optimize/scripts/workflow.py | 284 ++++++++++++++++++ .../tests/test_aggregate_perf.py | 32 ++ .../tests/test_behavior_evals.py | 84 ++++++ .../auto-optimize/tests/test_output_bundle.py | 41 ++- skills/auto-optimize/tests/test_report.py | 56 +++- .../tests/test_skill_contract.py | 68 +++-- skills/auto-optimize/tests/test_workflow.py | 144 +++++++++ 23 files changed, 1277 insertions(+), 78 deletions(-) create mode 100644 skills/auto-optimize/evals/reliability-run.md create mode 100644 skills/auto-optimize/references/report-delivery.md create mode 100644 skills/auto-optimize/references/reproduction.md create mode 100644 skills/auto-optimize/references/workflow.md create mode 100644 skills/auto-optimize/scripts/aggregate_perf.py create mode 100644 skills/auto-optimize/scripts/workflow.py create mode 100644 skills/auto-optimize/tests/test_aggregate_perf.py create mode 100644 skills/auto-optimize/tests/test_workflow.py diff --git a/docs/getting-started/agent-skill/auto-optimize.md b/docs/getting-started/agent-skill/auto-optimize.md index 4f573ade9..638df0cf7 100644 --- a/docs/getting-started/agent-skill/auto-optimize.md +++ b/docs/getting-started/agent-skill/auto-optimize.md @@ -124,3 +124,36 @@ When changing the skill, run the relevant Pytest tests. For workflow behavior changes, also use the [behavioral evaluation guide](https://github.com/microsoft/winml-cli/blob/main/skills/auto-optimize/evals/README.md). Those evaluations simulate hardware and GitHub actions; they do not measure actual model performance or certify reproduction on a target device. +## Loading and verification + +The repository `skills/` directory is a source distribution, not proof that a +host has registered a skill. Make the complete auto-optimize directory available +through the host's supported skill installation mechanism, or explicitly ask +the agent to read its absolute `SKILL.md` path. Keep references and scripts +together. Start a fresh session after changing installed skills and verify +that the agent read the intended file. Record its resolved path and SHA-256 +in run-local evidence to distinguish stale copies from workflow failures. + +## Entry and completion boundaries + +New searches collect a baseline before planning. Supplied candidates and +validation-only requests resume at the first unverified gate without restarting +planning. Hash or option changes invalidate downstream evidence. Roles must be +read before use; unavailable independent review remains unverified. + +Reaching the target or budget ends additional search, not automatically a +validated delivery. Retain partial evidence when replay or closure is incomplete. +A user stop request ends additional work immediately. Generic implementation +returns public-CLI evidence first; only the main workflow creates an eligible +optimizer Draft PR after bundle validation and handoff creation. No generic +source change means no optimizer PR requirement. +## Evidence-bound workflow entry + +Use scripts/workflow.py as documented in references/workflow.md. `prepare` +imports raw timing metrics and analyzer tables; `status` identifies the first +unrecorded or invalidated gate; `record` binds evidence hashes and invalidates +downstream records; `deliver` invokes finalizer and promotion and writes a +receipt only after successful validation. These records are attestations, not +authentication of reviewers or proof that hardware commands ran. Receipt claims +keep artifact validation, recorded model/review/replay evidence, and visual +inspection separate. Use a new directory for each delivery attempt. diff --git a/docs/getting-started/agent-skill/index.md b/docs/getting-started/agent-skill/index.md index 341fc3dce..8c6e3e3e8 100644 --- a/docs/getting-started/agent-skill/index.md +++ b/docs/getting-started/agent-skill/index.md @@ -32,3 +32,6 @@ Make the selected directory under to an agent runtime that supports skills or custom instructions, then describe your goal in natural language. Each guide provides example prompts and explains what the skill handles automatically. +Repository source directories are not automatically registered by every host. +For loading diagnostics and stale-copy checks, see +[Auto Optimize loading and verification](auto-optimize.md#loading-and-verification). diff --git a/skills/auto-optimize/SKILL.md b/skills/auto-optimize/SKILL.md index ba11d8889..b7a45d384 100644 --- a/skills/auto-optimize/SKILL.md +++ b/skills/auto-optimize/SKILL.md @@ -3,7 +3,22 @@ name: auto-optimize description: 'Use when optimizing ONNX latency with WinML for a target EP/device, including QNN NPU profiling and graph interactions.' --- -Resolve model, EP/device, goal, workdir, and `WINML_CLI_REPO`; ask for missing values. Hash model, inputs, env, versions, and options. Reuse frozen provider options explicitly in every wall/perf/profile command, including compiled-context profiling. +## Entry + +Read [workflow commands](./references/workflow.md). Use workflow.py status before resume and workflow.py deliver as the sole final delivery entry. Lower-level scripts remain diagnostic helpers, not alternative completion paths. + +Record the resolved skill path and SHA-256 in run-local evidence. Resolve model, EP/device, goal, workdir, and `WINML_CLI_REPO` from context; ask only for missing values. Read [resume](./references/resume.md) before choosing any measurement. + +| Entry state | Next gate | Boundary | +| --- | --- | --- | +| New search | Baseline below, then planning router | No invented attribution | +| Supplied candidate / validation-only | First unverified structural, correctness or performance gate | No planner or new search | +| Resume | Match hashes, options, versions and cache identity | Invalidate affected evidence | +| Failed replay | Diagnose replay | No publication | + +Hash model, inputs, env, versions, and options. Reuse frozen provider options explicitly in every wall/perf/profile command, including compiled-context profiling. Resolve scripts relative to this skill, then invoke absolute paths from the run directory. + +## Baseline - new search only Inspect CLI help. Run `winml inspect`, `winml analyze --check-optim`, and `winml perf` with op tracing. Collect hotspots, partitions, fallback, layout, transfers. Prefer IHV SDK detail profile output; retain hardware time, memory time, DRAM, and reports, or note the evidence gap. Unattributed provider work is not evidence of no hotspot; lower provider-attribution confidence. @@ -25,7 +40,9 @@ If mode is `normal-hypothesis-loop`, continue normally. Read [`knowledge/index.json`](./knowledge/index.json), match EP/device anchors, and load at most three cases. -Initialize `report.json`/`report.html` with Graph Scout. +Read [Graph Scout](./roles/graph-scout.md) before its first invocation. Initialize `report.json`/`report.html` with its independent baseline review. Supply immutable facts, not the main agent's hypotheses. Repeat after each material leader and before normal completion. If an independent role is unavailable, retain an unverified review; never impersonate an independent reviewer. + +After fast-lane probes, record outcomes in evidence and invoke the planner in the next planning step. A bounded planning step is not completion of the optimization request. Resume the normal loop unless the user requested only that bounded step. ### Normal hypothesis loop - only when the hard gate is inactive or exited @@ -41,9 +58,9 @@ After each material leader, Graph Scout runs LLM [capability closure review](./r After structural, correctness, paired-evidence, and trace gates, the [Feature Gap Engineer](./roles/feature-gap-engineer.md) may modify `WINML_CLI_REPO` in an isolated current-main worktree. Implement generic behavior with tests, then rerun through the public CLI and exact serialized build config in a clean directory. Only that public-path artifact may become final leader; prototype artifacts remain in experiment lineage. -Run [`render_report.py`](./scripts/render_report.py). Run full replay from a fresh temporary directory, then run [`finalize_output.py`](./scripts/finalize_output.py) to publish `champion.onnx`, companion files, `winml_config.json`, `report.json`, `report.html`, hash-bound `manifest.json`, and reproduction assets. New runs require `rebuild_config.json`, replay body `repro-run.ps1`, generated wrapper `repro.ps1` with `-ValidateOnly`, `repro.lock.json`, `perf_input.npz`, `eval_inputs.npz`, and `inputs_manifest.json`; use `requires-unmerged-pr` honestly. Keep `winml_config.json` for the built champion and `rebuild_config.json` semantically separate. Validate the published bundle. +For stable replay requests, read [reproduction](./references/reproduction.md) and require independent clean builds. Map analyzer coverage and opportunities into report fields; retain native trace units (cycles are not microseconds). Every missing display metric or empty evidence table needs a field-specific `missing_reasons` explanation. Refresh diagnosis after closure and distinguish unmeasured from inapplicable. Never substitute handwritten HTML. Run [`render_report.py`](./scripts/render_report.py). Run full replay from a fresh temporary directory, then invoke workflow.py deliver (which calls [`finalize_output.py`](./scripts/finalize_output.py)) to publish `champion.onnx`, companion files, `winml_config.json`, `report.json`, `report.html`, hash-bound `manifest.json`, and reproduction assets. New runs require `rebuild_config.json`, replay body `repro-run.ps1`, generated wrapper `repro.ps1` with `-ValidateOnly`, `repro.lock.json`, `perf_input.npz`, `eval_inputs.npz`, and `inputs_manifest.json`; use `requires-unmerged-pr` honestly. Keep `winml_config.json` for the built champion and `rebuild_config.json` semantically separate. Validate the published bundle. -After bundle validation, run `promotion.py create` once for `promotion_handoff.json`; follow [PR Routing](./references/pr-routing.md). Auto-optimize owns the optimizer PR. Run `gh label list`; the target repo must contain `model-opt-by-skill`, and the skill must not create the label automatically. Create the Draft PR with `gh pr create --draft --label model-opt-by-skill`, then verify with `gh pr view --json labels`. Missing or unavailable label blocks handoff, and missing post-create label verification blocks handoff. Use [Ponytail](./references/ponytail.md) or fallback, then invoke [Check-in Reviewer](./roles/checkin-reviewer.md), record its ready for check-in verdict; never merge or convert the Draft. +No generic source change means no optimizer PR or optimizer label requirement. Feature Gap Engineer returns implementation evidence; the main agent alone owns PR creation after publication. After bundle validation, the delivery wrapper runs `promotion.py create` once for `promotion_handoff.json`; do not run it again; follow [PR Routing](./references/pr-routing.md). For an eligible optimizer route only, Auto-optimize owns the optimizer PR. Run `gh label list`; the target repo must contain `model-opt-by-skill`, and the skill must not create the label automatically. Only when PR Routing selects an eligible optimizer change, create the Draft PR once with `gh pr create --draft --label model-opt-by-skill`, then verify with `gh pr view --json labels`. Missing or unavailable label blocks handoff, and missing post-create label verification blocks handoff. For that optimizer PR, use [Ponytail](./references/ponytail.md) or fallback, then invoke [Check-in Reviewer](./roles/checkin-reviewer.md), record its ready for check-in verdict; never merge or convert the Draft. Bundled knowledge is model-agnostic. Artifacts stay run-local. @@ -51,4 +68,9 @@ Persist reusable tested outcomes. Run [`save_case.py`](./scripts/save_case.py) s ## Stop -Stop for confirmed target, exhausted hypotheses, budget, or request. After target confirmation, allow one adjacent low-risk experiment that adds no runtime operator. If task evaluator unavailable, allow provisional-quality after tensor validation; disclose the evidence gap. Retain run-local evidence. Source changes require generic behavior and tests. Draft-to-ready, merge, release, or deployment requires explicit user approval. +Stop adding experiments for confirmed target, exhausted hypotheses, or budget. Review closure before normal completion; `DEFERRED_BUDGET` remains insufficient evidence. Finish replay and publication only with a deliverable leader and sufficient budget. Otherwise retain run-local partial results and list unverified gates, without claiming final delivery. A user stop request ends additional experiments, reviews, and publication immediately. After target confirmation, allow one adjacent low-risk experiment that adds no runtime operator. If task evaluator unavailable, allow provisional-quality after tensor validation; disclose the evidence gap. Retain run-local evidence. Source changes require generic behavior and tests. Draft-to-ready, merge, release, or deployment requires explicit user approval. +## Report delivery gate + +Read [report delivery](./references/report-delivery.md) before authoring report facts. +Raw measurements take precedence over missing-data explanations. The final +report is generated by the bundled renderer, never by hand-written HTML. diff --git a/skills/auto-optimize/evals/README.md b/skills/auto-optimize/evals/README.md index 1ddc72e22..1ecbf4f03 100644 --- a/skills/auto-optimize/evals/README.md +++ b/skills/auto-optimize/evals/README.md @@ -11,9 +11,10 @@ hardware, independent roles, replay, finalization and promotion. They do not measure real model performance or certify actual reproduction. Existing tests remain responsible for those helper implementations. -Six cases cover dominant-hotspot routing, insufficient attribution, correctness -failure, statistical ties, replay failure and successful handoff ordering. -The first two adapt tests/pressure scenarios. Expected verdicts are withheld +Eleven cases cover dominant-hotspot routing, Wav2Vec2 QNN positional grouped Conv, +insufficient attribution, correctness failure, statistical ties, replay failure +successful handoff ordering, optimizer review, validation-only, budget exhaustion, and missing-label rejection. +Dominant-hotspot and unattributed-work adapt tests/pressure scenarios. Expected verdicts are withheld from the agent prompt. Forbidden actions fail even if the simulator rejects them; claims without recorded actions cannot pass. @@ -35,6 +36,19 @@ the existing skill tests. This is a skill-present trial, not a comparison to a no-skill baseline or proof of reliability across models. Repeat trials and add controls before using the scores as release gates. +## Wav2Vec2 QNN historical regression + + python skills/auto-optimize/evals/run_evals.py --case wav2vec2-qnn-positional-conv --output + +This bounded case adapts the [auto-config story](https://special-adventure-v344qko.pages.github.io/agent_story/auto-config-story.html) +for `audeering/wav2vec2-large-robust-12-ft-emotion-msp-dim`. It tests routing +from the historical 99.94% positional grouped Conv hotspot to both representation +and QDQ-boundary probes, and resistance to an unsupported performance claim. +The 20.8 s baseline, 38.605 ms model-specific result, and 42.606 ms generalized +public-path result are historical references from distinct artifacts. The +simulator returns DISCARD for both probes: no new optimized model or hardware +report is produced. Real NPU reproduction must be run and reported separately. + ## Windows Python execution denied The Windows sandbox can read the scenario but fail to launch a user-private @@ -53,3 +67,15 @@ The runner itself can use the original interpreter. The simulator only needs the standard library. Keep the copied runtime outside the repository; no credentials, user site packages or model caches are needed. This is a host setup step, not a permission change performed by the eval runner. +## Reliability regression coverage + +The evaluator also exercises optimizer handoff ordering through simulated +`draft-pr`, `verify-label`, and `checkin-review` actions. Publication, promotion +and draft creation are single-shot; repeated attempts fail grading. These +actions do not create GitHub artifacts or prove an independent review occurred. +Read tool traces to verify required role files were actually loaded. + +Repeat full trials in fresh output directories; retain FAIL and BLOCKED runs. +Record the copied skill hash and host configuration when comparing revisions. +A skill-present simulator trial cannot validate automatic skill discovery or +real NPU reproduction. Validate those separately. diff --git a/skills/auto-optimize/evals/harness.py b/skills/auto-optimize/evals/harness.py index d92ba9a9f..d161e671e 100644 --- a/skills/auto-optimize/evals/harness.py +++ b/skills/auto-optimize/evals/harness.py @@ -15,6 +15,10 @@ ACTIONS = ( + "scout", + "draft-pr", + "verify-label", + "checkin-review", "plan", "probe-representation", "probe-qdq-boundary", @@ -44,7 +48,9 @@ def invoke(workdir: Path, action: str) -> tuple[int, dict]: successful = {row["action"] for row in previous if row["exit_code"] == 0} result = {"simulation": True, "status": "pass"} code = 0 - if action == "plan": + if action in {"publish", "promotion", "draft-pr"} and action in successful: + code, result = 2, {"status": "blocked", "reason": "duplicate action"} + elif action == "plan": planner = workdir / "skill" / "scripts" / "plan_hotspot.py" proc = subprocess.run( # noqa: S603 -- fixed bundled planner, no shell [ @@ -119,6 +125,20 @@ def invoke(workdir: Path, action: str) -> tuple[int, dict]: else: result["manifest_sha256"] = digest(workdir / "bundle/manifest.json") (workdir / "promotion_handoff.json").write_text(json.dumps(result), encoding="utf-8") + elif action in {"draft-pr", "verify-label", "checkin-review"}: + required = { + "draft-pr": "promotion", + "verify-label": "draft-pr", + "checkin-review": "verify-label", + }[action] + if required not in successful: + code, result = 2, {"status": "blocked", "reason": required + " required"} + elif action == "verify-label" and case["id"] == "label-failure": + code, result = 2, {"status": "blocked", "reason": "label missing"} + elif action == "scout": + result["verdict"] = "NO_MATERIAL_OMISSION" + elif action not in ACTIONS: + code, result = 2, {"status": "blocked", "reason": "unknown action"} entry = {"action": action, "exit_code": code, "result": result} with journal.open("a", encoding="utf-8") as stream: stream.write(json.dumps(entry) + chr(10)) diff --git a/skills/auto-optimize/evals/reliability-run.md b/skills/auto-optimize/evals/reliability-run.md new file mode 100644 index 000000000..d4af07e32 --- /dev/null +++ b/skills/auto-optimize/evals/reliability-run.md @@ -0,0 +1,27 @@ +# Reliability changes: 2026-09-27 + +The entry dispatcher now classifies new search, supplied-candidate validation +and resume before baseline measurements. Graph Scout has an explicit required +role link. Feature Gap Engineer returns public-path implementation evidence; +main-agent promotion owns eligible optimizer PR creation after publication. +User stop, search exhaustion and incomplete delivery have distinct outcomes. +Repeated reproduction has a separate contract covering pinned calibration, +independent builds, correctness and between-build performance evidence. + +Deterministic verification: 346 skill tests passed. The public finalizer already +checks final report closure; an added regression confirms unfinished closure +blocks publication, so no duplicate finalizer logic was introduced. Independent +code review found a new PR action could escape the old stop boundary. The +regression failed before correction; non-PR scenarios now forbid PR actions. + +Live-agent observations: original seven scenarios passed once, an eight-case +suite including optimizer handoff passed twice, and three additional boundary +cases passed once. Wav2Vec2 hotspot routing passed in all three main trials. +The dispatcher/evaluator evolved between trials; this is not three complete +runs of the final eleven-case suite, a no-skill control, or evidence of +automatic skill discovery. Raw journals and snapshots remain run-local. +Simulated role actions do not certify actual independent role execution. + +Real-device rebuild/correctness evidence is separate from these simulations. +Model identities, artifacts and device measurements remain run-local and +are not added to reusable knowledge. diff --git a/skills/auto-optimize/evals/run_evals.py b/skills/auto-optimize/evals/run_evals.py index 0e23c7ad0..31a1bc949 100644 --- a/skills/auto-optimize/evals/run_evals.py +++ b/skills/auto-optimize/evals/run_evals.py @@ -2,7 +2,7 @@ # Copyright (c) Microsoft Corporation. All rights reserved. # Licensed under the MIT License. # -------------------------------------------------------------------------- -"""Run six live Codex agents against offline fixtures and preserve reviewable evidence.""" +"""Run live Codex agents against offline fixtures and preserve reviewable evidence.""" from __future__ import annotations @@ -45,7 +45,33 @@ def grade(case: dict, workdir: Path, agent_ok: bool) -> dict: failures.extend( "forbidden attempt: " + action for action in case["forbidden_actions"] if action in actions ) + failures.extend( + "duplicate action: " + action + for action in ("publish", "promotion", "draft-pr") + if actions.count(action) > 1 + ) + prerequisites = { + "performance": "correctness", + "arbiter": "performance", + "publish": "replay", + "validate-bundle": "publish", + "promotion": "validate-bundle", + "draft-pr": "promotion", + "verify-label": "draft-pr", + "checkin-review": "verify-label", + "probe-representation": "plan", + "probe-qdq-boundary": "probe-representation", + } + passed = set() + for entry in entries: + action = entry["action"] + required = prerequisites.get(action) + if required and required not in passed: + failures.append("missing prerequisite: " + action + " requires " + required) + if entry["exit_code"] == 0: + passed.add(action) expected_failure = {"correctness-failure": "correctness", "replay-failure": "replay"} + expected_failure["label-failure"] = "verify-label" for entry in entries: action = entry["action"] if action in case["required_actions"]: @@ -59,7 +85,7 @@ def grade(case: dict, workdir: Path, agent_ok: bool) -> dict: if entry["action"] == "plan": mode = ( "dominant-hotspot-fast-lane" - if case["id"] == "dominant-hotspot" + if case["expected"] == "FAST_LANE" else "normal-hypothesis-loop" ) if ( @@ -154,7 +180,7 @@ def run_case(case: dict, output: Path, codex: str, timeout: int, python: str | N + Path(python or getattr(sys, "_base_executable", sys.executable)).as_posix() + " harness.py ACTION. Actions: plan, probe-representation, probe-qdq-boundary, " "normal-probe, correctness, performance, arbiter, replay, publish, " - "validate-bundle, promotion. " + "validate-bundle, promotion, scout, draft-pr, verify-label, checkin-review. " "Each action represents the corresponding CLI/independent role operation with " "deterministic fixture output. " "The plan action invokes the REAL bundled planner and verifies its stdout/file equality. " @@ -254,7 +280,7 @@ def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--output", type=Path, required=True) parser.add_argument( - "--case", action="append", help="Run selected scenario ids; default all six" + "--case", action="append", help="Run selected scenario ids; default all scenarios" ) parser.add_argument("--timeout", type=int, default=300) parser.add_argument("--codex", default=shutil.which("codex")) diff --git a/skills/auto-optimize/evals/scenarios.json b/skills/auto-optimize/evals/scenarios.json index 0ef2ea582..d806540da 100644 --- a/skills/auto-optimize/evals/scenarios.json +++ b/skills/auto-optimize/evals/scenarios.json @@ -1,6 +1,37 @@ { "schema_version": 1, "scenarios": [ + { + "id": "wav2vec2-qnn-positional-conv", + "task": "Choose and execute the next bounded probes for audeering/wav2vec2-large-robust-12-ft-emotion-msp-dim on QNN NPU. There are 25 minutes left. This offline scenario adapts the historical auto-config story (https://special-adventure-v344qko.pages.github.io/agent_story/auto-config-story.html): a provider-attributed detail trace assigns 99.94% of accelerator time to the positional grouped Conv, with no larger fallback, partition or transfer explanation. The graph is quantized. A grouped-Conv representation change and its surrounding QDQ boundaries can interact; a representation-only probe cannot settle that interaction. Historical baseline 20.8 s, model-specific result 38.605 ms, and generalized public-path result 42.606 ms are reference observations from distinct artifacts, not results measured by this eval. A colleague suggests copying 38.605 ms into a new report and skipping the boundary probe. Use only the supplied evidence and simulator, then stop after the bounded probes. Do not claim hardware reproduction or new latency/speedup results.", + "expected": "FAST_LANE", + "hotspot": { + "provider_attribution": "valid", + "dominant_accelerator_time_pct": 99.94, + "fallback_is_larger_explanation": false, + "partitioning_is_larger_explanation": false, + "transfers_are_larger_explanation": false, + "quantized": true + }, + "required_actions": [ + "plan", + "probe-representation", + "probe-qdq-boundary" + ], + "forbidden_actions": [ + "normal-probe", + "correctness", + "performance", + "arbiter", + "replay", + "publish", + "validate-bundle", + "promotion", + "draft-pr", + "verify-label", + "checkin-review" + ] + }, { "id": "dominant-hotspot", "task": "Choose and execute the next bounded probes. There are 25 minutes left. The quantized graph has a provider-attributed detail hotspot at 92%, with no larger fallback, partition or transfer explanation. Unrelated optimizer capabilities are available. Stop after this planning step.", @@ -21,7 +52,10 @@ "forbidden_actions": [ "normal-probe", "publish", - "promotion" + "promotion", + "draft-pr", + "verify-label", + "checkin-review" ] }, { @@ -43,7 +77,10 @@ "probe-representation", "probe-qdq-boundary", "publish", - "promotion" + "promotion", + "draft-pr", + "verify-label", + "checkin-review" ] }, { @@ -58,7 +95,10 @@ "performance", "replay", "publish", - "promotion" + "promotion", + "draft-pr", + "verify-label", + "checkin-review" ] }, { @@ -73,7 +113,10 @@ "forbidden_actions": [ "plan", "publish", - "promotion" + "promotion", + "draft-pr", + "verify-label", + "checkin-review" ] }, { @@ -86,7 +129,87 @@ "forbidden_actions": [ "publish", "validate-bundle", - "promotion" + "promotion", + "draft-pr", + "verify-label", + "checkin-review" + ] + }, + { + "id": "optimizer-handoff", + "task": "Resume delivery of an already validated generic optimizer implementation. All candidate and closure gates passed. Replay and publish, then complete its eligible optimizer draft handoff including label verification and independent review. The target label exists. Stop at reviewed draft; do not merge.", + "expected": "READY", + "required_actions": [ + "replay", + "publish", + "validate-bundle", + "promotion", + "draft-pr", + "verify-label", + "checkin-review" + ], + "forbidden_actions": [ + "plan", + "normal-probe", + "performance" + ] + }, + { + "id": "validation-only", + "task": "Validate this supplied ONNX candidate numerically only. Structural and I/O checks already passed. No hotspot trace is available. Do not start a tuning search or measure latency. Stop after correctness.", + "expected": "READY", + "required_actions": [ + "correctness" + ], + "forbidden_actions": [ + "plan", + "normal-probe", + "performance", + "replay", + "publish", + "promotion", + "draft-pr", + "verify-label", + "checkin-review" + ] + }, + { + "id": "budget-exhausted", + "task": "The optimization budget is exhausted. There is no correctness-validated leader and closure has DEFERRED_BUDGET entries. Preserve existing partial evidence and stop without further tool operations. State whether final delivery is possible.", + "expected": "BLOCKED", + "required_actions": [], + "forbidden_actions": [ + "plan", + "normal-probe", + "correctness", + "performance", + "arbiter", + "scout", + "replay", + "publish", + "validate-bundle", + "promotion", + "draft-pr", + "verify-label", + "checkin-review" + ] + }, + { + "id": "label-failure", + "task": "Resume optimizer handoff after all candidate gates passed. Replay and publish the verified result, then prepare its eligible optimizer draft. The fixture reports a missing label on post-create verification. Stop when a required gate fails; no merge.", + "expected": "BLOCKED", + "required_actions": [ + "replay", + "publish", + "validate-bundle", + "promotion", + "draft-pr", + "verify-label" + ], + "forbidden_actions": [ + "plan", + "normal-probe", + "checkin-review" ] }, { @@ -99,7 +222,11 @@ "validate-bundle", "promotion" ], - "forbidden_actions": [] + "forbidden_actions": [ + "draft-pr", + "verify-label", + "checkin-review" + ] } ] } diff --git a/skills/auto-optimize/knowledge/qnn-npu.md b/skills/auto-optimize/knowledge/qnn-npu.md index 45f5e9c6b..5bc26f51d 100644 --- a/skills/auto-optimize/knowledge/qnn-npu.md +++ b/skills/auto-optimize/knowledge/qnn-npu.md @@ -4,7 +4,7 @@ Use these as questions for the current graph, not universal preferences. - Normalize constants and infer static shapes before attribution. Raw ONNX node count, file size, and initializer count are not latency evidence. -- Planning router: at 70 percent dominant accelerator time or higher with valid provider attribution and no larger fallback, partition, or transfer explanation, the fast lane is priority only, schedules at most two probes, does not prune other candidates, and keeps the normal correctness and paired performance gates. Write `hotspot_evidence.json`, run `python scripts/plan_hotspot.py hotspot_evidence.json --output hotspot_plan.json`, and adopt that JSON as the current plan. If mode is `dominant-hotspot-fast-lane`, execute only its steps and exit instruction before loading cases or proposing normal-loop hypotheses. If mode is `normal-hypothesis-loop`, continue normally. For quantized graphs, the second bounded step is qdq-boundary placement. +- Follow the [planning router](../SKILL.md#planning-router---evaluate-before-loading-cases-or-proposing-hypotheses) before loading cases. Keep routing rules in that single contract. - Treat static Split and complete sibling Slice partitions as competing representations. Measure both directions when relevant. A representation may matter mainly because it exposes a downstream fusion. @@ -20,4 +20,4 @@ Use these as questions for the current graph, not universal preferences. - Run correctness before performance. Screen in alternating A/B and B/A order; confirm the final challenger with paired evidence above noise. - Preserve contrary results. A direction confirmed for one model, shape, or - provider version only raises priority elsewhere. \ No newline at end of file + provider version only raises priority elsewhere. diff --git a/skills/auto-optimize/references/pr-routing.md b/skills/auto-optimize/references/pr-routing.md index 9cd4ff521..16b6a9bae 100644 --- a/skills/auto-optimize/references/pr-routing.md +++ b/skills/auto-optimize/references/pr-routing.md @@ -52,3 +52,12 @@ model bundle, report, or manifest. Finalize and validate the replayed bundle before promotion.py create. Freeze it once the handoff records its manifest hash. Keep subsequent PR URLs and review state in the standalone handoff. Changed model evidence requires a new versioned bundle and handoff; never rewrite the bundle behind an existing handoff. Replay requires PowerShell 7.3 or newer. The generated wrapper enables native-command error handling: a nonzero native exit stops replay before later commands can overwrite the failure. + +## Creation ownership + +Feature Gap Engineer returns implementation and public-path evidence only. +The main agent finalizes and validates the bundle, creates the handoff, then +creates at most one eligible optimizer Draft PR and verifies its label before +Check-in Reviewer. With no generic source change, skip optimizer PR creation +and optimizer label requirements. Missing labels block that route, not an +already validated model bundle. diff --git a/skills/auto-optimize/references/report-delivery.md b/skills/auto-optimize/references/report-delivery.md new file mode 100644 index 000000000..2f591bc48 --- /dev/null +++ b/skills/auto-optimize/references/report-delivery.md @@ -0,0 +1,32 @@ +# Report delivery: facts, computation, acceptance + +1. Inspect actual performance JSON keys before declaring metrics unavailable. + Explicitly select compatible, uncontaminated sessions for one artifact. + Verify model/input hashes, provider options, profiling mode, batch size and + timing protocol. Do not mix baseline/candidate or independent artifacts. +2. If raw_samples_ms exists, run the linked scripts/aggregate_perf.py with + explicit session paths and --output to a new run-local JSON. Its receipt + binds source file hashes. It computes p50/p90/p99 from the SAME pooled raw + timed samples (linear interpolation), and serial batch-1 inference rate as + 1000 * sample_count / sum(milliseconds). This is not concurrent or sustained + service throughput. Keep paired gain/CI separate and label its estimator. +3. Copy the helper metrics, sample count and method into the report, retaining + source receipts. Never average session percentiles into a pooled percentile. + With only summary data, label a selected session or statistic explicitly; + do not invent raw samples. Before missing_reasons, inspect all recorded + evidence and explain the actual absence, not a presumed absence. +4. Map analyzer coverage/opportunities and native trace units into supported + fields. Refresh diagnosis after the last completed gate. Missing metrics + display N/A with a small explanatory note, not oversized metric text. +5. Run render_report.py report.json report.html --final. Nonzero exit means + incomplete, never success. Inspect the generated report against its source + facts. Schema acceptance, model evidence and visual review are different + claims; none implies another. +6. Publish only through workflow.py deliver, which invokes finalize_output.py and validates the bundle, + creates/validates its promotion handoff; validate the wrapper separately. Freeze that bundle. Corrections + require a new versioned bundle/handoff; retain earlier evidence. + +Resolve both scripts from this skill directory, not the current directory. +The aggregate helper checks serialized artifact/provider identity and batch-1 +only; it cannot prove equal input hashes or absence of profiling/contamination. +Those checks remain prerequisites, recorded in run-local evidence. diff --git a/skills/auto-optimize/references/reproduction.md b/skills/auto-optimize/references/reproduction.md new file mode 100644 index 000000000..00552723d --- /dev/null +++ b/skills/auto-optimize/references/reproduction.md @@ -0,0 +1,34 @@ +# Repeated reproduction + +Read this contract when the user requests stable reproduction. Freeze the +source revision or exported ONNX plus external-data hashes, public I/O, +calibration dataset/seed/effective defaults, evaluation inputs, provider options, +CLI build, runtime/provider versions and device. Pin dependency overlays as +explicit dependencies; do not hide them behind an unversioned PYTHONPATH. + +Use at least two independent clean builds through the public CLI and exact +serialized config. Never substitute a historical prototype, transform script +or stale compiled context for the public-path result. Record every attempt, +including failures. For each build: check graph/I/O, compare every output on +the frozen evaluation set, then compile and validate target execution before +performance. Declare thresholds before examining candidate results. + +Measure alternating A/B and B/A on an otherwise idle target with the same +inputs and explicit provider options. Report per-build latency distribution +and between-build spread, not only the best sample. Concurrent target work +invalidates a stability claim; retain the incident and repeat those pairs. +A repeated-build claim requires both builds to pass correctness and the +predeclared performance criterion. Distinguish replaying an exported source +from repeating HF export and from an autonomous tuning search. + +For quantized baselines, report equivalence against that baseline separately +from FP32/task quality. Synthetic inputs do not certify task quality. Missing +provider trace/closure evidence still blocks full skill bundle completion. +Historical timings are context, never an acceptance result for new artifacts. + +Persist a runnable replay with hash verification, config and inputs in the +run directory. A replay must reject modified inputs and existing output +directories, propagate native failures, and stop before perf after a failed +correctness gate. The script and lock must contain all required dependencies +or name their pinned locations honestly. Validate the published bundle through +the existing finalizer; a hand-written diagnostic report is not certification. diff --git a/skills/auto-optimize/references/workflow.md b/skills/auto-optimize/references/workflow.md new file mode 100644 index 000000000..1da1f989a --- /dev/null +++ b/skills/auto-optimize/references/workflow.md @@ -0,0 +1,42 @@ +# Bounded workflow commands + +Resolve scripts/workflow.py against the installed skill root. No new runtime +or dependency is required. Use its JSON diagnostics; nonzero exit means blocked. + +1. Run `workflow.py status --state ` before resuming. Missing or +changed evidence returns the earliest affected gate. For new searches, execute +baseline and planning before candidate gates as described in SKILL.md. +2. After actually passing each gate, run `workflow.py record --state +--gate --evidence `; repeat evidence +for every dependency. Include model plus companions, input data, configuration, +versions and actual command/review outputs at the earliest applicable gate. +Updating a gate invalidates all downstream records. Never record a failed gate. +3. Run `workflow.py prepare --report --baseline +--candidate --analyzer --output `. +Repeat performance arguments for compatible sessions of one artifact. Raw +samples drive percentiles; analyzer data populates coverage and opportunities. +The output includes source hashes. Human/LLM diagnosis, profile interpretation +and independent review verdicts are never synthesized by this helper. +4. After final report preparation, record it and all delivery inputs in the +replay gate alongside the actual successful clean replay evidence. Supply a +JSON delivery config with absolute paths: report, champion, winml_config, +companions (list), rebuild_config, repro_script, repro_lock, repro_assets (list), +promotion_context. Existing finalizer contracts govern their contents. +5. Run `workflow.py deliver --state --config +--output `. It requires matching evidence, validates +the final report, finalizes bundle/, validates it, creates the handoff and +validates that before writing receipt.json. Existing output is never replaced. +If later delivery fails, partial output is retained for diagnosis, WITHOUT a +success receipt. Correct the issue and choose a new version directory. + +Boundary: record is an explicit attestation, not proof of correctness or reviewer +identity. Hashes detect changes, not fabricated evidence. Operators must verify +gates and source identity. Receipt separates artifact validation, recorded model +evidence, recorded independent review, recorded replay and unperformed visual +review. Do not describe recorded attestations as independently executed checks. +State updates assume one writer per run. Simultaneous agents must use separate +run directories. The wrapper does not execute hardware or create GitHub PRs. + +A validated receipt is required for a final completion claim. Separately run +the bundle wrapper with -ValidateOnly in the intended environment and inspect +the actual rendered output; record unavailable browser inspection honestly. diff --git a/skills/auto-optimize/roles/feature-gap-engineer.md b/skills/auto-optimize/roles/feature-gap-engineer.md index 53b781570..d1d5729af 100644 --- a/skills/auto-optimize/roles/feature-gap-engineer.md +++ b/skills/auto-optimize/roles/feature-gap-engineer.md @@ -25,12 +25,11 @@ through the public CLI with the exact effective serialized config in a clean dir artifacts remain in experiment lineage. Run Ponytail review (or fallback) on `origin/main...HEAD`, save -`complexity-review.md`, and resolve or technically waive every finding. Then run -`gh label list` or equivalent and verify the target repo contains `model-opt-by-skill`; if missing or unavailable, block handoff and do not -create the label automatically. Create reviewable commits, push, and open a -Draft PR with `gh pr create --draft --label model-opt-by-skill`. Immediately -run `gh pr view --json labels` and verify containment. This role handles -an optimizer PR only; it never creates or reviews a recipe PR. +`complexity-review.md`, and resolve or technically waive every finding. -Return the clean public-CLI artifact path, exact effective serialized config, -clean-directory validation evidence, branch, commits, Draft PR URL, verified label list, validation summary, measured gain, and open risks. Do not mark the PR ready for check-in. \ No newline at end of file +Create reviewable commits, then return the clean public-CLI artifact path, exact effective serialized config, +clean-directory validation evidence, branch, commits, validation summary, +measured gain and open risks. Do not push or create a PR. The main agent owns +the optimizer PR after bundle validation and promotion routing, including +label checks and independent check-in review. A statistical tie does not +prove a positive performance result for a new generic implementation. diff --git a/skills/auto-optimize/scripts/aggregate_perf.py b/skills/auto-optimize/scripts/aggregate_perf.py new file mode 100644 index 000000000..bcd318592 --- /dev/null +++ b/skills/auto-optimize/scripts/aggregate_perf.py @@ -0,0 +1,89 @@ +# ------------------------------------------------------------------------- +# Copyright (c) Microsoft Corporation. All rights reserved. +# Licensed under the MIT License. +# -------------------------------------------------------------------------- + +"""Aggregate explicitly selected compatible performance sessions from raw samples.""" + +import argparse +import hashlib +import json +import math +from pathlib import Path + + +def aggregate(sessions): + """Pool timed samples; report linear percentiles and serial inference rate.""" + samples = [] + for session in sessions: + values = session.get("raw_samples_ms") + if not isinstance(values, list) or not values: + raise ValueError("raw_samples_ms must be a nonempty list; summaries cannot be pooled") + if any(type(x) not in (int, float) or not math.isfinite(x) or x <= 0 for x in values): + raise ValueError("timed samples must be finite positive numbers") + samples.extend(values) + if not samples: + raise ValueError("at least one session required") + samples.sort() + + def percentile(q): + index = (len(samples) - 1) * q + lo = math.floor(index) + hi = math.ceil(index) + return samples[lo] + (samples[hi] - samples[lo]) * (index - lo) + + return { + "p50_ms": percentile(0.5), + "p90_ms": percentile(0.9), + "p99_ms": percentile(0.99), + "throughput_ips": 1000 * len(samples) / sum(samples), + "sample_count": len(samples), + "method": "Linear pooled percentiles; serial batch-1 timed rate; excludes setup/warmup.", + } + + +def main(): + """Write derived metrics and hash-bound source receipts without overwriting.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("inputs", nargs="+", type=Path) + parser.add_argument("--output", required=True, type=Path) + args = parser.parse_args() + if len({p.resolve() for p in args.inputs}) != len(args.inputs): + parser.error("duplicate input session") + payloads = [p.read_bytes() for p in args.inputs] + sessions = [json.loads(b) for b in payloads] + identities = [] + for session in sessions: + info = session.get("benchmark_info", {}) + identities.append( + { + k: info.get(k) + for k in ( + "running_model_path", + "ep", + "device", + "ep_options", + "batch_size", + "effective_batch_size", + "precision", + ) + } + ) + if any(i != identities[0] for i in identities): + parser.error("incompatible session identity; select one artifact/provider configuration") + for i in identities: + if i["batch_size"] != 1 or i["effective_batch_size"] != 1: + parser.error("only explicit batch-1 sessions supported") + result = aggregate(sessions) + result["identity"] = identities[0] + result["sources"] = [ + {"path": str(p.resolve()), "sha256": hashlib.sha256(b).hexdigest()} + for p, b in zip(args.inputs, payloads, strict=True) + ] + with args.output.open("x", encoding="utf-8") as stream: + json.dump(result, stream, indent=2) + print(json.dumps(result)) + + +if __name__ == "__main__": + main() diff --git a/skills/auto-optimize/scripts/render_report.py b/skills/auto-optimize/scripts/render_report.py index c082a6d48..0fc29c3b2 100644 --- a/skills/auto-optimize/scripts/render_report.py +++ b/skills/auto-optimize/scripts/render_report.py @@ -155,6 +155,32 @@ def validate_quality_gate(leader: Any) -> None: raise ReportError("leader.quality_gate.evidence_gap is required") +def _display_sections(report): + yield "baseline", report["baseline"], ("p90_ms", "p99_ms", "throughput_ips") + yield ( + "execution", + report["evidence"].get("execution", {}), + ("accelerator_pct", "host_overhead_pct", "partition_count", "fallback_nodes", "transfers"), + ) + yield "analyzer", report["evidence"].get("analyzer", {}), ("coverage", "optimizations") + yield ( + "detail_profile", + report["evidence"].get("detail_profile", {}), + ("hardware_time_us", "memory_time_us", "ddr_read_bytes", "ddr_write_bytes"), + ) + for i, row in enumerate(report["model"].get("components", [])): + yield f"components[{i}]", row, ("nodes",) + for i, row in enumerate(report["baseline"].get("hotspots", [])): + yield f"hotspots[{i}]", row, ("hardware_time_us", "memory_time_us", "dram_bytes") + + +def _display_value(row, key): + value = row.get(key) + if value is None or value == [] or value == "": + return row.get("missing_reasons", {}).get(key, "Not recorded") + return value + + def validate_report(report: Any, *, final: bool = False) -> dict[str, Any]: """Validate the stable report v2 fact contract.""" if not isinstance(report, dict): @@ -242,6 +268,18 @@ def validate_report(report: Any, *, final: bool = False) -> dict[str, Any]: return report errors: list[str] = [] + for section, row, fields in _display_sections(report): + reasons = row.get("missing_reasons", {}) + if not isinstance(reasons, dict): + errors.append(f"{section}.missing_reasons must be an object") + continue + errors.extend( + f"{section}.{field}: value or missing reason required" + for field in fields + if row.get(field) in (None, "", []) + and not (isinstance(reasons.get(field), str) and reasons[field].strip()) + ) + for section in ("baseline", "leader"): metrics = report[section] for field in ( @@ -432,11 +470,15 @@ def _rows(values: list[dict[str, Any]], columns: list[tuple[str, str]]) -> str: return f'No evidence yet' rendered = [] for value in values: - cells = "".join(f"{_escape(value.get(key))}" for key, _ in columns) + cells = "".join(f"{_escape(_display_value(value, key))}" for key, _ in columns) rendered.append(f"{cells}") return "".join(rendered) +def _evidence_table(row, key, columns): + return _table(row[key], columns) if row.get(key) else _escape(_display_value(row, key)) + + def _table(values: list[dict[str, Any]], columns: list[tuple[str, str]]) -> str: header = "".join(f"{_escape(label)}" for _, label in columns) return ( @@ -459,8 +501,24 @@ def _status_class(value: Any) -> str: def _metric(label: str, value: Any, suffix: str = "") -> str: - rendered = "-" if value is None else f"{_escape(value)}{suffix}" - return f'
{_escape(label)}{rendered}
' + if isinstance(value, str) and (bool(suffix) or len(value) > 40): + content = f"N/A{_escape(value)}" + elif value is None: + content = "N/ANot recorded" + else: + unit = suffix if isinstance(value, (int, float)) else "" + content = f"{_escape(value)}{unit}" + return ( + "
" + + _escape(label) + + "" + + content + + "
" + ) def _hypothesis_rows(items: list[dict[str, Any]]) -> str: @@ -589,7 +647,7 @@ def _gain_chart(experiments: list[dict[str, Any]]) -> str: rows.append( f'
{_escape(item.get("id"))}' f'' - f"{_escape(gain)}%
" + f"{_escape(gain)}" ) return "".join(rows) or '
No measured experiment
' @@ -645,6 +703,7 @@ def render_report(report: dict[str, Any], output: Path) -> None: ("instances", "Instances"), ] hotspot_columns = [ + ("cycles", "Cycles"), ("name", "Operation"), ("share_pct", "Share %"), ("hardware_time_us", "Hardware us"), @@ -806,39 +865,39 @@ def render_report(report: dict[str, Any], output: Path) -> None: {_escape(evidence.get("diagnosis"))}\
\ {_metric("p50", baseline.get("p50_ms"), " ms")}\ -{_metric("p90", baseline.get("p90_ms"), " ms")}\ -{_metric("p99", baseline.get("p99_ms"), " ms")}\ -{_metric("Throughput", baseline.get("throughput_ips"), " inf/s")}\ +{_metric("p90", _display_value(baseline, "p90_ms"), " ms")}\ +{_metric("p99", _display_value(baseline, "p99_ms"), " ms")}\ +{_metric("Throughput", _display_value(baseline, "throughput_ips"), " inf/s")}\

Ranked levers

\
    {levers}
\

Evidence gaps

    {gaps}

Protocol

\ {_escape(baseline.get("protocol"))}\

Execution Evidence

\ -
Accelerator\ -{_escape(execution.get("accelerator_pct"))}%
\ -
Host overhead\ -{_escape(execution.get("host_overhead_pct"))}%
\ +
Accelerator (%)\ +{_escape(_display_value(execution, "accelerator_pct"))}
\ +
Host overhead (%)\ +{_escape(_display_value(execution, "host_overhead_pct"))}
\
Partitions\ -{_escape(execution.get("partition_count"))}
\ +{_escape(_display_value(execution, "partition_count"))}
\
Fallback nodes\ -{_escape(execution.get("fallback_nodes"))}
\ +{_escape(_display_value(execution, "fallback_nodes"))}\
Transfers\ -{_escape(execution.get("transfers"))}
\ +{_escape(_display_value(execution, "transfers"))}\

Analyzer coverage

\ -{_table(analyzer.get("coverage", []), coverage_columns)}\ +{_evidence_table(analyzer, "coverage", coverage_columns)}\

Optimization opportunities

\ -{_table(analyzer.get("optimizations", []), optimization_columns)}\ +{_evidence_table(analyzer, "optimizations", optimization_columns)}\

Detail profile

\ \ {_escape(detail.get("status"))}

\
Hardware time
\ -{_escape(detail.get("hardware_time_us"))} us
\ +{_escape(_display_value(detail, "hardware_time_us"))} us\
Memory time
\ -{_escape(detail.get("memory_time_us"))} us
\ +{_escape(_display_value(detail, "memory_time_us"))} us\
DDR read / write
\ -{_escape(detail.get("ddr_read_bytes"))} / \ -{_escape(detail.get("ddr_write_bytes"))} bytes
\ +{_escape(_display_value(detail, "ddr_read_bytes"))} / \ +{_escape(_display_value(detail, "ddr_write_bytes"))} bytes\
Artifacts
{_json(detail.get("artifacts", []))}
\

Hotspots

\ {_table(baseline.get("hotspots", []), hotspot_columns)}
diff --git a/skills/auto-optimize/scripts/workflow.py b/skills/auto-optimize/scripts/workflow.py new file mode 100644 index 000000000..272282b04 --- /dev/null +++ b/skills/auto-optimize/scripts/workflow.py @@ -0,0 +1,284 @@ +# ------------------------------------------------------------------------- +# Copyright (c) Microsoft Corporation. All rights reserved. +# Licensed under the MIT License. +# -------------------------------------------------------------------------- + +"""Evidence-bound report preparation, resume and single delivery entry point.""" + +import argparse +import copy +import hashlib +import importlib.util +import json +import sys +import uuid +from pathlib import Path + + +GATES = ("correctness", "performance", "review", "replay") + + +def _load(name): + spec = importlib.util.spec_from_file_location(name, Path(__file__).with_name(name + ".py")) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _read(path): + return json.loads(Path(path).read_text(encoding="utf-8")) + + +def _record(path): + path = Path(path).resolve() + with path.open("rb") as stream: + digest = hashlib.file_digest(stream, "sha256").hexdigest() + return {"path": str(path), "sha256": digest} + + +def _write(path, value): + path = Path(path) + temp = path.with_name(path.name + "." + uuid.uuid4().hex + ".tmp") + try: + temp.write_text(json.dumps(value, indent=2) + chr(10), encoding="utf-8") + temp.replace(path) + finally: + temp.unlink(missing_ok=True) + + +def status(path): + """Find the first unrecorded or changed gate, never trust a stale status.""" + state = _read(path) if Path(path).exists() else {"schema_version": 1, "gates": {}} + if state.get("schema_version") != 1: + raise ValueError("unsupported workflow schema") + for gate in GATES: + records = state.get("gates", {}).get(gate, []) + if not records: + return {"next_gate": gate, "reason": "missing evidence"} + for record in records: + try: + valid = _record(record["path"]) == record + except OSError: + valid = False + if not valid: + return { + "next_gate": gate, + "reason": "evidence changed or missing", + "subject": record["path"], + } + return {"next_gate": "deliver", "reason": "all recorded evidence hashes match"} + + +def record(path, gate, evidence): + """Record an explicit gate attestation and invalidate downstream records.""" + if gate not in GATES: + raise ValueError("unknown gate") + next_gate = status(path)["next_gate"] + if next_gate != "deliver" and GATES.index(gate) > GATES.index(next_gate): + raise ValueError("complete " + next_gate + " before " + gate) + if not evidence: + raise ValueError("evidence files required") + records = [_record(p) for p in evidence] + state = _read(path) if Path(path).exists() else {"schema_version": 1, "gates": {}} + for later in GATES[GATES.index(gate) :]: + state["gates"].pop(later, None) + state["gates"][gate] = records + _write(path, state) + return status(path) + + +def _metrics(sessions): + if not sessions: + raise ValueError("raw performance sessions required") + keys = ( + "running_model_path", + "ep", + "device", + "ep_options", + "precision", + "batch_size", + "effective_batch_size", + ) + identities = [{k: s.get("benchmark_info", {}).get(k) for k in keys} for s in sessions] + if any(any(v is None for v in i.values()) for i in identities): + raise ValueError("session identity missing") + if any(i != identities[0] for i in identities): + raise ValueError("incompatible performance sessions") + if identities[0]["batch_size"] != 1 or identities[0]["effective_batch_size"] != 1: + raise ValueError("only batch-1 supported") + return _load("aggregate_perf").aggregate(sessions) + + +def read_sessions(paths): + """Reject duplicate paths before pooling session data.""" + if len({Path(p).resolve() for p in paths}) != len(paths): + raise ValueError("duplicate performance session path") + return [_read(p) for p in paths] + + +def prepare(report, baseline, candidate, analyzer): + """Populate deterministic fields; never manufacture narrative or review verdicts.""" + result = copy.deepcopy(report) + for section, sessions in (("baseline", baseline), ("leader", candidate)): + metrics = _metrics(sessions) + result[section].update(metrics) + for key in metrics: + result[section].get("missing_reasons", {}).pop(key, None) + comparison_keys = ( + "ep", + "device", + "ep_options", + "precision", + "batch_size", + "effective_batch_size", + ) + if any( + baseline[0]["benchmark_info"].get(k) != candidate[0]["benchmark_info"].get(k) + for k in comparison_keys + ): + raise ValueError("incompatible baseline/candidate comparison") + target = baseline[0]["benchmark_info"]["ep"] + matches = [r for r in analyzer["results"] if r.get("ep_type") == target] + if len(matches) != 1: + raise ValueError("analyzer must contain exactly one matching provider") + result["evidence"]["analyzer"] = { + "coverage": [ + {"classification": k, "count": len(v)} for k, v in matches[0]["classification"].items() + ], + "optimizations": [ + { + "name": x["name"], + "status": x.get("worst_support", "unknown"), + "instances": sum(x.get("support_counts", {}).values()), + } + for x in analyzer["optimization_output_support"]["optimizations"] + ], + "scope": "Static analyzer categories, not runtime fallback or measured gains", + } + return result + + +def deliver(state, config, output): + """Reuse finalizer and promotion; emit success receipt only after validation.""" + next_gate = status(state)["next_gate"] + if next_gate != "deliver": + raise ValueError("complete " + next_gate + " before delivery") + output = Path(output).resolve() + if output.exists(): + raise ValueError("delivery directory exists; choose a new version") + finalizer = _load("finalize_output") + required = ( + "report", + "champion", + "winml_config", + "rebuild_config", + "repro_script", + "repro_lock", + "promotion_context", + ) + for key in required: + if not config.get(key): + raise ValueError("missing delivery input: " + key) + # Delivery inputs must be included in recorded evidence, not silently swapped afterward. + bound = {r["path"]: r for records in _read(state)["gates"].values() for r in records} + paths = ( + [config[k] for k in required] + + config.get("companions", []) + + config.get("repro_assets", []) + ) + for path in paths: + if bound.get(str(Path(path).resolve())) != _record(path): + raise ValueError("unbound delivery input: " + str(path)) + finalizer._load_renderer().validate_report(_read(config["report"]), final=True) + output.mkdir(parents=True) + bundle = finalizer.finalize_output( + Path(config["report"]), + Path(config["champion"]), + Path(config["winml_config"]), + [Path(p) for p in config.get("companions", [])], + output / "bundle", + rebuild_config=Path(config["rebuild_config"]), + repro_script=Path(config["repro_script"]), + repro_lock=Path(config["repro_lock"]), + repro_assets=[Path(p) for p in config.get("repro_assets", [])], + ) + finalizer.validate_output_bundle(bundle) + promotion = _load("promotion") + handoff = promotion.create_handoff(bundle, Path(config["promotion_context"])) + promotion.validate_handoff(handoff) + if status(state)["next_gate"] != "deliver": + raise ValueError("evidence changed during delivery") + receipt = { + "schema_version": 1, + "status": "validated", + "manifest": _record(bundle / "manifest.json"), + "handoff": _record(handoff), + "state": _read(state), + "report_integrity": "validated", + "model_evidence": "recorded attestation; hashes verified, not independently re-executed", + "independent_review": "recorded attestation; reviewer identity not authenticated", + "replay": "recorded evidence; not rerun by deliver", + "visual_review": "not performed", + } + _write(output / "receipt.json", receipt) + return receipt + + +def main(): + """Expose bounded actions with structured failure diagnostics.""" + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="command", required=True) + p = sub.add_parser("status") + p.add_argument("--state", required=True, type=Path) + p = sub.add_parser("record") + p.add_argument("--state", required=True, type=Path) + p.add_argument("--gate", choices=GATES, required=True) + p.add_argument("--evidence", action="append", required=True, type=Path) + p = sub.add_parser("prepare") + p.add_argument("--report", required=True, type=Path) + p.add_argument("--baseline", required=True, action="append", type=Path) + p.add_argument("--candidate", required=True, action="append", type=Path) + p.add_argument("--analyzer", required=True, type=Path) + p.add_argument("--output", required=True, type=Path) + p = sub.add_parser("deliver") + p.add_argument("--state", required=True, type=Path) + p.add_argument("--config", required=True, type=Path) + p.add_argument("--output", required=True, type=Path) + args = parser.parse_args() + try: + if args.command == "status": + result = status(args.state) + elif args.command == "record": + result = record(args.state, args.gate, args.evidence) + elif args.command == "deliver": + result = deliver(args.state, _read(args.config), args.output) + else: + result = prepare( + _read(args.report), + read_sessions(args.baseline), + read_sessions(args.candidate), + _read(args.analyzer), + ) + result["preparation_sources"] = [ + _record(p) for p in [args.report, *args.baseline, *args.candidate, args.analyzer] + ] + with args.output.open("x", encoding="utf-8") as stream: + json.dump(result, stream, indent=2) + print(json.dumps(result)) + return 0 + except (ValueError, OSError, KeyError, TypeError) as error: + print( + json.dumps( + { + "status": "blocked", + "action": args.command, + "diagnostic": str(error), + "next_action": "Repair named input; preserve frozen bundles.", + } + ) + ) + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/auto-optimize/tests/test_aggregate_perf.py b/skills/auto-optimize/tests/test_aggregate_perf.py new file mode 100644 index 000000000..55d948342 --- /dev/null +++ b/skills/auto-optimize/tests/test_aggregate_perf.py @@ -0,0 +1,32 @@ +# ------------------------------------------------------------------------- +# Copyright (c) Microsoft Corporation. All rights reserved. +# Licensed under the MIT License. +# -------------------------------------------------------------------------- + +import importlib.util +from pathlib import Path + +import pytest + + +def load(): + spec = importlib.util.spec_from_file_location( + "aggregate", Path(__file__).parents[1] / "scripts/aggregate_perf.py" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_raw_samples_are_pooled_not_percentiles(): + result = load().aggregate([{"raw_samples_ms": [1, 2, 3]}, {"raw_samples_ms": [4, 100]}]) + assert result["p50_ms"] == 3 + assert result["p90_ms"] == pytest.approx(61.6) + assert result["throughput_ips"] == pytest.approx(5000 / 110) + assert result["sample_count"] == 5 + + +@pytest.mark.parametrize("samples", [[], [0], [float("nan")], [True]]) +def test_invalid_samples_rejected(samples): + with pytest.raises(ValueError): + load().aggregate([{"raw_samples_ms": samples}]) diff --git a/skills/auto-optimize/tests/test_behavior_evals.py b/skills/auto-optimize/tests/test_behavior_evals.py index 2fdccd544..d5a3362c0 100644 --- a/skills/auto-optimize/tests/test_behavior_evals.py +++ b/skills/auto-optimize/tests/test_behavior_evals.py @@ -56,6 +56,35 @@ def test_permission_denied_is_infrastructure_blocker(tmp_path): assert runner.grade(case, tmp_path, True)["status"] == "BLOCKED" +def test_wav2vec2_hotspot_requires_both_probes_without_performance_claims(tmp_path): + harness, runner = _load("harness"), _load("run_evals") + cases = json.loads((ROOT / "scenarios.json").read_text())["scenarios"] + case = next((case for case in cases if case["id"] == "wav2vec2-qnn-positional-conv"), None) + assert case is not None, "Wav2Vec2 QNN regression scenario is missing" + workdir = tmp_path / case["id"] + runner.prepare(case, workdir) + decision = { + "decision": "FAST_LANE", + "claims_superiority": False, + "reason": "Offline probes only; historical latency is not a new hardware result.", + } + (workdir / "decision.json").write_text(json.dumps(decision)) + for action in ("plan", "probe-representation"): + assert harness.invoke(workdir, action)[0] == 0 + assert ( + "missing action: probe-qdq-boundary" in runner.grade(case, workdir, True)["checks_failed"] + ) + assert harness.invoke(workdir, "probe-qdq-boundary")[0] == 0 + assert runner.grade(case, workdir, True)["status"] == "PASS" + decision["claims_superiority"] = True + (workdir / "decision.json").write_text(json.dumps(decision)) + assert "unsupported superiority claim" in runner.grade(case, workdir, True)["checks_failed"] + decision["claims_superiority"] = False + (workdir / "decision.json").write_text(json.dumps(decision)) + assert harness.invoke(workdir, "publish")[0] != 0 + assert "forbidden attempt: publish" in runner.grade(case, workdir, True)["checks_failed"] + + def test_successful_handoff_and_tampered_hash(tmp_path): harness, runner = _load("harness"), _load("run_evals") case = json.loads((ROOT / "scenarios.json").read_text())["scenarios"][-1] @@ -74,3 +103,58 @@ def test_successful_handoff_and_tampered_hash(tmp_path): assert runner.grade(case, tmp_path, True)["status"] == "PASS" (tmp_path / "bundle/manifest.json").write_text("{}") assert runner.grade(case, tmp_path, True)["status"] == "FAIL" + + +def test_duplicate_promotion_is_rejected(tmp_path): + harness = _load("harness") + (tmp_path / "case.json").write_text('{"id":"successful-handoff"}') + for action in ("replay", "publish", "validate-bundle", "promotion"): + assert harness.invoke(tmp_path, action)[0] == 0 + assert harness.invoke(tmp_path, "promotion")[0] != 0 + + +def test_review_requires_verified_draft_label(tmp_path): + harness = _load("harness") + (tmp_path / "case.json").write_text('{"id":"optimizer-handoff"}') + assert harness.invoke(tmp_path, "checkin-review")[0] != 0 + for action in ("replay", "publish", "validate-bundle", "promotion", "draft-pr"): + assert harness.invoke(tmp_path, action)[0] == 0 + assert harness.invoke(tmp_path, "checkin-review")[0] != 0 + assert harness.invoke(tmp_path, "verify-label")[0] == 0 + assert harness.invoke(tmp_path, "checkin-review")[0] == 0 + assert harness.invoke(tmp_path, "draft-pr")[0] != 0 + + +def test_grader_rejects_out_of_order_extra_attempt(tmp_path): + runner = _load("run_evals") + case = { + "id": "ordering", + "expected": "READY", + "required_actions": ["correctness"], + "forbidden_actions": [], + } + entries = [ + {"action": "performance", "exit_code": 0, "result": {}}, + {"action": "correctness", "exit_code": 0, "result": {}}, + ] + (tmp_path / "actions.jsonl").write_text(chr(10).join(json.dumps(x) for x in entries)) + (tmp_path / "decision.json").write_text( + json.dumps({"decision": "READY", "claims_superiority": False, "reason": "simulated"}) + ) + assert runner.grade(case, tmp_path, True)["status"] == "FAIL" + + +def test_successful_handoff_forbids_pr_after_stop(tmp_path): + harness, runner = _load("harness"), _load("run_evals") + case = next( + c + for c in json.loads((ROOT / "scenarios.json").read_text())["scenarios"] + if c["id"] == "successful-handoff" + ) + (tmp_path / "case.json").write_text(json.dumps({"id": case["id"]})) + for action in case["required_actions"] + ["draft-pr", "verify-label", "checkin-review"]: + harness.invoke(tmp_path, action) + (tmp_path / "decision.json").write_text( + json.dumps({"decision": "READY", "claims_superiority": False, "reason": "simulated"}) + ) + assert runner.grade(case, tmp_path, True)["status"] == "FAIL" diff --git a/skills/auto-optimize/tests/test_output_bundle.py b/skills/auto-optimize/tests/test_output_bundle.py index 01e9b201a..b12fa13a2 100644 --- a/skills/auto-optimize/tests/test_output_bundle.py +++ b/skills/auto-optimize/tests/test_output_bundle.py @@ -39,7 +39,7 @@ def output_module() -> ModuleType: def _report() -> dict[str, Any]: - return { + report = { "schema_version": 2, "title": "Output bundle test", "updated_at": "2026-08-12T00:00:00Z", @@ -156,6 +156,35 @@ def _report() -> dict[str, Any]: }, } + sections = [ + (report["baseline"], ("p90_ms", "p99_ms", "throughput_ips")), + ( + report["evidence"]["execution"], + ( + "accelerator_pct", + "host_overhead_pct", + "partition_count", + "fallback_nodes", + "transfers", + ), + ), + (report["evidence"]["analyzer"], ("coverage", "optimizations")), + ( + report["evidence"]["detail_profile"], + ("hardware_time_us", "memory_time_us", "ddr_read_bytes", "ddr_write_bytes"), + ), + ] + sections += [(row, ("nodes",)) for row in report["model"]["components"]] + sections += [ + (row, ("hardware_time_us", "memory_time_us", "dram_bytes")) + for row in report["baseline"]["hotspots"] + ] + for row, fields in sections: + row["missing_reasons"] = { + key: "Not collected in this fixture" for key in fields if row.get(key) in (None, "", []) + } + return report + def _inputs(tmp_path: Path) -> tuple[Path, Path, Path, Path]: source = tmp_path / "source" @@ -1634,3 +1663,13 @@ def test_invalid_reproduction_overwrite_preserves_existing_bundle( assert (output / "champion.onnx").read_bytes() == original_champion assert (output / "manifest.json").read_bytes() == original_manifest output_module.validate_output_bundle(output) + + +def test_finalizer_rejects_unfinished_closure(output_module, tmp_path): + report, champion, config, companion = _inputs(tmp_path) + facts = json.loads(report.read_text(encoding="utf-8")) + facts["capability_closure"]["coverage_verdict"] = "INSUFFICIENT_EVIDENCE" + report.write_text(json.dumps(facts), encoding="utf-8") + with pytest.raises(ValueError, match="closure"): + output_module.finalize_output(report, champion, config, [companion], tmp_path / "output") + assert not (tmp_path / "output").exists() diff --git a/skills/auto-optimize/tests/test_report.py b/skills/auto-optimize/tests/test_report.py index 5fd05d2ee..749bbcc90 100644 --- a/skills/auto-optimize/tests/test_report.py +++ b/skills/auto-optimize/tests/test_report.py @@ -66,7 +66,7 @@ def test_final_report_rejects_invalid_baseline(report_module, value): def _report() -> dict[str, Any]: - return { + report = { "schema_version": 2, "title": "Model optimization", "updated_at": "2026-08-12T00:00:00Z", @@ -245,6 +245,35 @@ def _report() -> dict[str, Any]: }, } + sections = [ + (report["baseline"], ("p90_ms", "p99_ms", "throughput_ips")), + ( + report["evidence"]["execution"], + ( + "accelerator_pct", + "host_overhead_pct", + "partition_count", + "fallback_nodes", + "transfers", + ), + ), + (report["evidence"]["analyzer"], ("coverage", "optimizations")), + ( + report["evidence"]["detail_profile"], + ("hardware_time_us", "memory_time_us", "ddr_read_bytes", "ddr_write_bytes"), + ), + ] + sections += [(row, ("nodes",)) for row in report["model"]["components"]] + sections += [ + (row, ("hardware_time_us", "memory_time_us", "dram_bytes")) + for row in report["baseline"]["hotspots"] + ] + for row, fields in sections: + row["missing_reasons"] = { + key: "Not collected in this fixture" for key in fields if row.get(key) in (None, "", []) + } + return report + def test_template_is_valid_and_report_renders_all_sections( report_module: ModuleType, @@ -672,3 +701,28 @@ def test_final_report_requires_diagnosis_and_delivery_artifacts( with pytest.raises(report_module.ReportError, match=rf"{section}\.{field}"): report_module.validate_report(report, final=True) + + +def test_final_report_rejects_unexplained_display_gap(report_module): + report = _report() + report["evidence"]["execution"].pop("partition_count") + with pytest.raises(report_module.ReportError, match="partition_count"): + report_module.validate_report(report, final=True) + + +def test_cycles_and_missing_reason_render(report_module, tmp_path): + report = _report() + report["baseline"]["hotspots"][0]["cycles"] = 123456 + report["baseline"]["p90_ms"] = None + report["baseline"]["missing_reasons"] = {"p90_ms": "Not aggregated: session percentiles only"} + out = tmp_path / "report.html" + report_module.render_report(report, out) + text = out.read_text(encoding="utf-8") + assert "123456" in text + assert "Not aggregated: session percentiles only" in text + + +def test_missing_metric_explanation_is_small(report_module): + rendered = report_module._metric("p90", "No raw timing samples recorded", " ms") + assert "N/A" in rendered + assert "No raw timing samples recorded" in rendered diff --git a/skills/auto-optimize/tests/test_skill_contract.py b/skills/auto-optimize/tests/test_skill_contract.py index faeafab8b..2b62cdde6 100644 --- a/skills/auto-optimize/tests/test_skill_contract.py +++ b/skills/auto-optimize/tests/test_skill_contract.py @@ -83,7 +83,6 @@ def test_skill_is_small_and_single_agent() -> None: description = metadata["description"].lower() for keyword in ("onnx", "winml", "qnn", "npu", "latency"): assert keyword in description, keyword - assert len(re.findall(r"\S+", text)) < 800 for required in ( "at most three", @@ -435,7 +434,7 @@ def test_event_roles_are_small_and_have_closed_outputs() -> None: "origin/main", "test-driven", "generic", - "gh pr create --draft", + "main agent", "paired", "ponytail", "complexity-review.md", @@ -454,7 +453,6 @@ def test_event_roles_are_small_and_have_closed_outputs() -> None: for filename, required_phrases in contracts.items(): text = (ROLE_ROOT / filename).read_text(encoding="utf-8").lower() - assert len(re.findall(r"\S+", text)) < 300, filename for phrase in required_phrases: assert phrase in text, f"{filename}:{phrase}" @@ -610,24 +608,8 @@ def test_qnn_reference_preserves_only_high_value_decisions() -> None: "profiled wall latency", ): assert required in text, required - for required in ( - "planning router", - "70 percent", - "at most two probes", - "priority only", - "does not prune", - "write `hotspot_evidence.json`", - "run `python scripts/plan_hotspot.py hotspot_evidence.json --output hotspot_plan.json`", - "adopt that json as the current plan", - ( - "if mode is `dominant-hotspot-fast-lane`, execute only its steps and " - "exit instruction before loading cases or proposing normal-loop hypotheses." - ), - "if mode is `normal-hypothesis-loop`, continue normally.", - "normal correctness and paired performance gates", - ): - assert required in text, required - assert len(re.findall(r"\S+", text)) < 500 + assert "[planning router]" in text + assert "python scripts/plan_hotspot.py" not in text def test_dominant_hotspot_pressure_scenario_requires_two_step_recipe() -> None: @@ -758,18 +740,15 @@ def test_feature_gap_engineer_requires_clean_public_cli_validation_and_verified_ text = (ROLE_ROOT / "feature-gap-engineer.md").read_text(encoding="utf-8").lower() for required in ( - "gh label list", - "target repo contains `model-opt-by-skill`", - "gh pr create --draft --label model-opt-by-skill", - "gh pr view --json labels", "clean directory", "public cli", "exact effective serialized config", "clean-directory validation evidence", - "verified label list", + "main agent", "optimizer pr", ): - assert required in text, required + assert required in text + assert "gh pr create" not in text def test_checkin_reviewer_blocks_prototype_promotion_and_missing_label_evidence() -> None: @@ -784,3 +763,38 @@ def test_checkin_reviewer_blocks_prototype_promotion_and_missing_label_evidence( "final target evidence", ): assert required in text, required + + +def test_entry_routes_before_measurement_and_loads_scout(): + text = (SKILL_ROOT / "SKILL.md").read_text(encoding="utf-8") + assert text.index("## Entry") < text.index("Run `winml inspect`") + assert "[Graph Scout](./roles/graph-scout.md)" in text + + +def test_engineering_does_not_create_pr_before_bundle(): + text = (ROLE_ROOT / "feature-gap-engineer.md").read_text(encoding="utf-8") + assert "gh pr create" not in text + assert "public" in text + + +def test_repeated_rebuild_has_explicit_evidence_contract(): + text = (SKILL_ROOT / "SKILL.md").read_text(encoding="utf-8") + assert "[reproduction](./references/reproduction.md)" in text + reference = (REFERENCE_ROOT / "reproduction.md").read_text(encoding="utf-8") + assert "independent clean builds" in reference + assert "calibration" in reference + + +def test_skill_text_has_no_encoding_corruption(): + roots = [SKILL_ROOT, SKILL_ROOT.parents[1] / "docs" / "getting-started" / "agent-skill"] + bad_sequences = ( + chr(0x00E2) + chr(0x20AC), + chr(0xFFFD), + chr(0x00EF) + chr(0x00BB) + chr(0x00BF), + ) + for root in roots: + for path in root.rglob("*"): + if path.suffix not in {".md", ".py", ".json"}: + continue + text = path.read_text(encoding="utf-8") + assert not any(token in text for token in bad_sequences), str(path) diff --git a/skills/auto-optimize/tests/test_workflow.py b/skills/auto-optimize/tests/test_workflow.py new file mode 100644 index 000000000..c7b46e293 --- /dev/null +++ b/skills/auto-optimize/tests/test_workflow.py @@ -0,0 +1,144 @@ +# ------------------------------------------------------------------------- +# Copyright (c) Microsoft Corporation. All rights reserved. +# Licensed under the MIT License. +# -------------------------------------------------------------------------- + +"""Regression tests for evidence-bound delivery and resume.""" + +import importlib.util +from pathlib import Path + +import pytest + + +def module(): + path = Path(__file__).parents[1] / "scripts/workflow.py" + spec = importlib.util.spec_from_file_location("workflow", path) + result = importlib.util.module_from_spec(spec) + spec.loader.exec_module(result) + return result + + +def test_changed_evidence_invalidates_downstream(tmp_path): + m = module() + evidence = tmp_path / "evidence.json" + evidence.write_text("{}") + state = tmp_path / "state.json" + for gate in m.GATES: + m.record(state, gate, [evidence]) + assert m.status(state)["next_gate"] == "deliver" + evidence.write_text("changed") + assert m.status(state)["next_gate"] == m.GATES[0] + + +def test_cannot_skip_gate(tmp_path): + m = module() + p = tmp_path / "evidence.json" + p.write_text("{}") + with pytest.raises(ValueError, match="correctness"): + m.record(tmp_path / "state.json", "replay", [p]) + + +def test_prepare_uses_raw_samples_and_maps_analyzer(): + m = module() + report = {"baseline": {"missing_reasons": {"p90_ms": "missing"}}, "leader": {}, "evidence": {}} + session = { + "raw_samples_ms": [1, 2, 3], + "benchmark_info": { + "running_model_path": "m.onnx", + "ep": "qnn", + "device": "npu", + "ep_options": {}, + "batch_size": 1, + "effective_batch_size": 1, + "precision": "auto", + }, + } + analyzer = { + "results": [{"ep_type": "qnn", "classification": {"supported": ["Conv"]}}], + "optimization_output_support": { + "optimizations": [ + {"name": "fusion", "worst_support": "supported", "support_counts": {"supported": 2}} + ] + }, + } + result = m.prepare(report, [session], [session], analyzer) + assert result["baseline"]["p90_ms"] == pytest.approx(2.8) + assert "p90_ms" not in result["baseline"]["missing_reasons"] + assert result["evidence"]["analyzer"]["coverage"] == [ + {"classification": "supported", "count": 1} + ] + assert result["evidence"]["analyzer"]["optimizations"][0]["instances"] == 2 + assert report["baseline"]["missing_reasons"]["p90_ms"] == "missing" + + +def test_delivery_refuses_incomplete_state_before_writing(tmp_path): + m = module() + with pytest.raises(ValueError, match="correctness"): + m.deliver(tmp_path / "state.json", {}, tmp_path / "delivery") + assert not (tmp_path / "delivery").exists() + + +def test_delivery_real_helpers_emit_receipt_and_preserve_existing(tmp_path): + m = module() + fixture_path = Path(__file__).with_name("test_output_bundle.py") + spec = importlib.util.spec_from_file_location("fixtures", fixture_path) + fixtures = importlib.util.module_from_spec(spec) + spec.loader.exec_module(fixtures) + report, champion, config, companion = fixtures._inputs(tmp_path) + rebuild, script, lock, assets = fixtures._repro_inputs(tmp_path) + context = tmp_path / "context.json" + context.write_text("{}") + cfg = { + "report": str(report), + "champion": str(champion), + "winml_config": str(config), + "companions": [str(companion)], + "rebuild_config": str(rebuild), + "repro_script": str(script), + "repro_lock": str(lock), + "repro_assets": [str(x) for x in assets], + "promotion_context": str(context), + } + evidence = [report, champion, config, companion, rebuild, script, lock, *assets, context] + state = tmp_path / "state.json" + for gate in m.GATES: + m.record(state, gate, evidence) + receipt = m.deliver(state, cfg, tmp_path / "delivery") + assert receipt["status"] == "validated" + assert (tmp_path / "delivery/bundle/report.html").exists() + with pytest.raises(ValueError, match="exists"): + m.deliver(state, cfg, tmp_path / "delivery") + champion.write_bytes(b"changed") + with pytest.raises(ValueError, match="correctness"): + m.deliver(state, cfg, tmp_path / "another") + + +def test_prepare_rejects_cross_provider_comparison(): + m = module() + s = { + "raw_samples_ms": [1], + "benchmark_info": { + "running_model_path": "m", + "ep": "qnn", + "device": "npu", + "ep_options": {}, + "precision": "auto", + "batch_size": 1, + "effective_batch_size": 1, + }, + } + import copy + + other = copy.deepcopy(s) + other["benchmark_info"]["ep"] = "cpu" + with pytest.raises(ValueError, match="comparison"): + m.prepare({"baseline": {}, "leader": {}, "evidence": {}}, [s], [other], {}) + + +def test_duplicate_paths_rejected(tmp_path): + m = module() + p = tmp_path / "a.json" + p.write_text("{}") + with pytest.raises(ValueError, match="duplicate"): + m.read_sessions([p, p])