diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json index 622e59c..c5a32d6 100644 --- a/reports/benchmark_reproducibility.json +++ b/reports/benchmark_reproducibility.json @@ -3,24 +3,21 @@ "ok": true, "generated_at": "2026-06-17", "skill_dir": ".", - "commit": "3961b60341a893464c1ea96f9921f804809f37f9", + "commit": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "git_status": { "available": true, "dirty": true, - "changed_file_count": 18, + "changed_file_count": 9, "sample": [ - " M reports/adaptation_proposals.json", - " M reports/adaptation_proposals.md", " M reports/benchmark_reproducibility.json", " M reports/benchmark_reproducibility.md", - " M reports/context_budget.json", - " M reports/context_budget_summary.json", " M reports/evidence_consistency.json", - " M reports/python_compatibility.json", - " M reports/python_compatibility.md", - " M reports/review-studio.html", - " M reports/review-studio.json", - " M reports/review-viewer.json" + " M reports/evidence_consistency.md", + " M reports/review-viewer.json", + " M reports/skill-interpretation.html", + " M reports/skill-interpretation.json", + " M reports/skill-overview.html", + " M reports/skill-overview.json" ], "scope": "generation-time status before this report is written" }, @@ -30,8 +27,8 @@ "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "678c3839e1d41692b4fb991e542e7d6ef60fbca5248dab2e3f313e138aedde77", - "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", + "evidence_bundle_sha256": "c31f7210165c76ab9b288c6c7d5d228fd46c8c8627cd86ca095cc1033607e34f", + "source_contract_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -47,12 +44,12 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, "public_claim_blocker_count": 5, "working_tree_dirty": true, - "changed_file_count": 18 + "changed_file_count": 9 }, "public_claim": { "ready": false, @@ -62,13 +59,13 @@ "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", - "world-class source checks are not all accepted (6/13 pass, 7 blocked)" + "world-class source checks are not all accepted (7/13 pass, 6 blocked)" ], "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { "ready": false, - "commit": "3961b60341a893464c1ea96f9921f804809f37f9", + "commit": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "status_scope": "generation-time status before this report is written", "reason": "working tree was dirty at generation time" }, @@ -78,7 +75,7 @@ "existing_count": 25, "missing_count": 0, "missing_paths": [], - "sha256": "678c3839e1d41692b4fb991e542e7d6ef60fbca5248dab2e3f313e138aedde77" + "sha256": "c31f7210165c76ab9b288c6c7d5d228fd46c8c8627cd86ca095cc1033607e34f" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -151,8 +148,8 @@ "label": "output_execution", "path": "reports/output_execution_runs.json", "exists": true, - "bytes": 7966, - "sha256": "2c9409158a128e6d2ad4c96c499dd412591c7b287d907e8d88987132a0592633" + "bytes": 7965, + "sha256": "6334c2f0261c2c978ebcb518cdac4ddf409f41e3732b08998c8e4b80529e446d" }, { "label": "blind_review", @@ -186,22 +183,22 @@ "label": "trust_report", "path": "reports/security_trust_report.json", "exists": true, - "bytes": 129520, - "sha256": "0d60cae961055a68ac84d83cc23d6d8456cc4b42b1a103dd963f3343a20987ff" + "bytes": 129571, + "sha256": "5b05d1110b96a6a72300cc5913140a9d4c0bdf099b1232df8c3bce6e2d023514" }, { "label": "python_compatibility", "path": "reports/python_compatibility.json", "exists": true, "bytes": 28298, - "sha256": "b223212ec521b782c479ace75a0e93866e09aff7f2637fdd181e8906bb4b0376" + "sha256": "8b48340618cc702a8d8f5aaf6ce6d05a1a5463a8008d06a28809671632b212da" }, { "label": "registry_audit", "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, - "sha256": "e75a341d15e4d89d46bded0cdb5c1ba378b5674df84d9182eef57c5fb6204c9e" + "sha256": "dfd07c1825de99435bd661e2663ae74251bbe03cb5ddda66d268efb6a872eb31" }, { "label": "package_verification", @@ -214,71 +211,71 @@ "label": "install_simulation", "path": "reports/install_simulation.json", "exists": true, - "bytes": 8604, - "sha256": "e29ae26bc97c94ec0a9da3ab60f28ef90e23c9d0c53b777810165224e725b452" + "bytes": 8758, + "sha256": "5269d6e6364aaa39e935cbf02df7569a6a88d2653120bc6e3e34d3684a73639d" }, { "label": "skill_os2_audit", "path": "reports/skill_os2_audit.json", "exists": true, "bytes": 14310, - "sha256": "f31dc78072fe38192f25148e11fd902c902e883e230585314172dcf1a3241f32" + "sha256": "390d0bb42652e9f84fcab195de1da3c9cda213c254fd9319d31c54dd8429aee7" }, { "label": "world_class_evidence_plan", "path": "reports/world_class_evidence_plan.json", "exists": true, "bytes": 20828, - "sha256": "0a836a86ad6c648d8c442a16cae7cdb17d3640a0ba6ec69c1290825ede6d152a" + "sha256": "9b74155fb9bfe44ba838d64b1a31768b83978b154bdd8eaa606f494c816f189f" }, { "label": "world_class_evidence_ledger", "path": "reports/world_class_evidence_ledger.json", "exists": true, - "bytes": 22288, - "sha256": "fe81833eab69292525975b347177bc7fcda6410f493fc60bd840a20e86ca0039" + "bytes": 22285, + "sha256": "f6e1a69b0359c28c88e9e9e69ff66763c721ac8a112a43a673ce98cd9d3bce41" }, { "label": "world_class_evidence_intake", "path": "reports/world_class_evidence_intake.json", "exists": true, "bytes": 19319, - "sha256": "4506c836338c65a3a01b15b1f26a1fdd0a8ef520a1e5f1574ee89ce9d3bbeb29" + "sha256": "46f801051ec075725059e4e2682ef2162fd892b5e7255c8d5121d7b24dcc7f2c" }, { "label": "world_class_evidence_preflight", "path": "reports/world_class_evidence_preflight.json", "exists": true, - "bytes": 41208, - "sha256": "725d10a6f951ab1e64f4e6a1cfa907404b51f598a753b5a0351e1381fef7cd9e" + "bytes": 41202, + "sha256": "b7fd72d7e75486612f65a9bb73b1ab59106103ebf3ad10dd446cd80903088454" }, { "label": "world_class_submission_review", "path": "reports/world_class_submission_review.json", "exists": true, - "bytes": 13658, - "sha256": "7693638c2e8369a8a179b3420c11090668f8f104f452dc370082dc9ebafe882d" + "bytes": 13655, + "sha256": "9fbd4cd437986fca9a043cbab24aec7d575aaf166a8d9365c79c56f093cab5b8" }, { "label": "world_class_operator_runbook", "path": "reports/world_class_operator_runbook.json", "exists": true, - "bytes": 24899, - "sha256": "f90d59c3e5866da33138869c1c5299350a43389d81e84ba466df794befb494fa" + "bytes": 24835, + "sha256": "2d32a2e917b7215afe7a7a93ca8bacffa540a245953aa39e2ee47eea1b169d17" }, { "label": "world_class_operator_runbook_markdown", "path": "reports/world_class_operator_runbook.md", "exists": true, - "bytes": 15479, - "sha256": "be79ee0f70a3ceed473c481562c5fddaee6cfb36c960ba15ae10c846d5809650" + "bytes": 15424, + "sha256": "00d8ebabb91e14a06cacdb212f448668aa1e47b72587448e30681fc53acaad9a" }, { "label": "world_class_operator_runbook_html", "path": "reports/world_class_operator_runbook.html", "exists": true, - "bytes": 21409, - "sha256": "9a7be02a6990245c9dd1f72ced5928d80a3aaebea4647737c2aaf0b0a2f798df" + "bytes": 21348, + "sha256": "ca72c65b437b2211d08b9d463c9a33a7993b4e388bb75f62757ead4dfcf81a49" }, { "label": "world_class_claim_guard", diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md index 5e34bb1..c9e742c 100644 --- a/reports/benchmark_reproducibility.md +++ b/reports/benchmark_reproducibility.md @@ -1,9 +1,9 @@ # Benchmark Reproducibility Generated at: `2026-06-17` -Commit: `3961b60341a893464c1ea96f9921f804809f37f9` +Commit: `7476c7ccbfaa094932809d405accf3aca9bc16ea` Working tree dirty at generation: `true` -Evidence bundle SHA256: `678c3839e1d41692b4fb991e542e7d6ef60fbca5248dab2e3f313e138aedde77` +Evidence bundle SHA256: `c31f7210165c76ab9b288c6c7d5d228fd46c8c8627cd86ca095cc1033607e34f` ## Summary @@ -12,7 +12,7 @@ Evidence bundle SHA256: `678c3839e1d41692b4fb991e542e7d6ef60fbca5248dab2e3f313e1 - methodology complete: `true` - required artifacts: `25` - missing artifacts: `0` -- source contract sha256: `d50f6ac9714b` +- source contract sha256: `9cb9a19e7745` - archive sha256: `5802e5f52255` - output cases: `5` - disclosed failure cases: `3` @@ -20,10 +20,10 @@ Evidence bundle SHA256: `678c3839e1d41692b4fb991e542e7d6ef60fbca5248dab2e3f313e1 - provider evidence complete: `false` - human review complete: `false` - world-class ready: `false` -- world-class source checks: `6` pass / `13` total; `7` blocked +- world-class source checks: `7` pass / `13` total; `6` blocked - public claim ready: `false` - public claim blockers: `5` -- changed files at generation: `18` +- changed files at generation: `9` This report proves local benchmark reproducibility only. It keeps external provider and human-review gaps visible instead of counting them as complete. The git commit is generation-time context; the evidence bundle SHA is the durable anchor for the artifacts listed below. @@ -39,7 +39,7 @@ This report proves local benchmark reproducibility only. It keeps external provi | provider-backed model holdout evidence is incomplete | | human blind-review adjudication is incomplete | | world-class evidence is not accepted yet (4 open gaps, 4 ledger pending) | -| world-class source checks are not all accepted (6/13 pass, 7 blocked) | +| world-class source checks are not all accepted (7/13 pass, 6 blocked) | ## Release Lock @@ -51,7 +51,7 @@ This report proves local benchmark reproducibility only. It keeps external provi - algorithm: `sha256(path,label,exists,artifact_sha256)` - artifacts: `25` / `25` -- sha256: `678c3839e1d41692b4fb991e542e7d6ef60fbca5248dab2e3f313e138aedde77` +- sha256: `c31f7210165c76ab9b288c6c7d5d228fd46c8c8627cd86ca095cc1033607e34f` ## Methodology Sections @@ -73,25 +73,25 @@ This report proves local benchmark reproducibility only. It keeps external provi | output_cases | `evals/output/cases.jsonl` | present | `a6ae96857116` | | output_schema | `evals/output/schema.json` | present | `8ee340c95064` | | output_scorecard | `reports/output_quality_scorecard.json` | present | `0806258a8e08` | -| output_execution | `reports/output_execution_runs.json` | present | `2c9409158a12` | +| output_execution | `reports/output_execution_runs.json` | present | `6334c2f0261c` | | blind_review | `reports/output_blind_review_pack.json` | present | `bbe2db8ec277` | | review_adjudication | `reports/output_review_adjudication.json` | present | `bb8c72a9291e` | | trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` | | runtime_conformance | `reports/conformance_matrix.json` | present | `97f9ba949c23` | -| trust_report | `reports/security_trust_report.json` | present | `0d60cae96105` | -| python_compatibility | `reports/python_compatibility.json` | present | `b223212ec521` | -| registry_audit | `reports/registry_audit.json` | present | `e75a341d15e4` | +| trust_report | `reports/security_trust_report.json` | present | `5b05d1110b96` | +| python_compatibility | `reports/python_compatibility.json` | present | `8b48340618cc` | +| registry_audit | `reports/registry_audit.json` | present | `dfd07c1825de` | | package_verification | `reports/package_verification.json` | present | `a27941fdb865` | -| install_simulation | `reports/install_simulation.json` | present | `e29ae26bc97c` | -| skill_os2_audit | `reports/skill_os2_audit.json` | present | `f31dc78072fe` | -| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `0a836a86ad6c` | -| world_class_evidence_ledger | `reports/world_class_evidence_ledger.json` | present | `fe81833eab69` | -| world_class_evidence_intake | `reports/world_class_evidence_intake.json` | present | `4506c836338c` | -| world_class_evidence_preflight | `reports/world_class_evidence_preflight.json` | present | `725d10a6f951` | -| world_class_submission_review | `reports/world_class_submission_review.json` | present | `7693638c2e83` | -| world_class_operator_runbook | `reports/world_class_operator_runbook.json` | present | `f90d59c3e586` | -| world_class_operator_runbook_markdown | `reports/world_class_operator_runbook.md` | present | `be79ee0f70a3` | -| world_class_operator_runbook_html | `reports/world_class_operator_runbook.html` | present | `9a7be02a6990` | +| install_simulation | `reports/install_simulation.json` | present | `5269d6e6364a` | +| skill_os2_audit | `reports/skill_os2_audit.json` | present | `390d0bb42652` | +| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `9b74155fb9bf` | +| world_class_evidence_ledger | `reports/world_class_evidence_ledger.json` | present | `f6e1a69b0359` | +| world_class_evidence_intake | `reports/world_class_evidence_intake.json` | present | `46f801051ec0` | +| world_class_evidence_preflight | `reports/world_class_evidence_preflight.json` | present | `b7fd72d7e754` | +| world_class_submission_review | `reports/world_class_submission_review.json` | present | `9fbd4cd43798` | +| world_class_operator_runbook | `reports/world_class_operator_runbook.json` | present | `2d32a2e917b7` | +| world_class_operator_runbook_markdown | `reports/world_class_operator_runbook.md` | present | `00d8ebabb91e` | +| world_class_operator_runbook_html | `reports/world_class_operator_runbook.html` | present | `ca72c65b437b` | | world_class_claim_guard | `reports/world_class_claim_guard.json` | present | `abe7f7d60c00` | ## Reproduction Commands diff --git a/reports/evidence_consistency.json b/reports/evidence_consistency.json index 7dac20a..05c8b48 100644 --- a/reports/evidence_consistency.json +++ b/reports/evidence_consistency.json @@ -4,14 +4,14 @@ "generated_at": "2026-06-17", "skill_dir": ".", "summary": { - "check_count": 35, - "pass_count": 35, + "check_count": 36, + "pass_count": 36, "warn_count": 0, "fail_count": 0, "decision": "consistent" }, "status_counts": { - "pass": 35, + "pass": 36, "warn": 0, "fail": 0 }, @@ -189,12 +189,12 @@ "status": "pass", "expected": { "status": "pass", - "detail": "initial load 990/1000; deferred 495216/120000; top deferred scripts 435118; resource governance governed; quality density 131.3", + "detail": "initial load 990/1000; deferred 495718/120000; top deferred scripts 435620; resource governance governed; quality density 131.3", "evidence": "reports/context_budget.json" }, "actual": { "status": "pass", - "detail": "initial load 990/1000; deferred 495216/120000; top deferred scripts 435118; resource governance governed; quality density 131.3", + "detail": "initial load 990/1000; deferred 495718/120000; top deferred scripts 435620; resource governance governed; quality density 131.3", "evidence": "reports/context_budget.json" }, "paths": [ @@ -207,13 +207,28 @@ "key": "benchmark-release-lock-self-consistency", "label": "Benchmark release lock matches git dirty state", "status": "pass", - "expected": true, - "actual": true, + "expected": false, + "actual": false, "paths": [ "reports/benchmark_reproducibility.json" ], "detail": "The benchmark release lock must reflect the generation-time git dirty flag." }, + { + "key": "benchmark-clean-worktree-release-lock", + "label": "Clean worktree keeps a clean benchmark release lock", + "status": "pass", + "expected": "checked only when git is available and the current worktree is clean", + "actual": { + "available": true, + "clean": false, + "changed_file_count": 10 + }, + "paths": [ + "reports/benchmark_reproducibility.json" + ], + "detail": "Dirty or non-git worktrees cannot prove final release-lock freshness, so this check is advisory until the final clean-lock pass." + }, { "key": "skill-ir-evidence-path-contract", "label": "Human-facing reports expose the canonical Skill IR artifact", @@ -248,8 +263,8 @@ "key": "overview-benchmark-commit", "label": "overview embeds the benchmark commit", "status": "pass", - "expected": "3961b60341a893464c1ea96f9921f804809f37f9", - "actual": "3961b60341a893464c1ea96f9921f804809f37f9", + "expected": "7476c7ccbfaa094932809d405accf3aca9bc16ea", + "actual": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-overview.json" @@ -261,30 +276,30 @@ "label": "overview embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 25, "missing_artifact_count": 0, - "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", + "source_contract_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "actual": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 25, "missing_artifact_count": 0, - "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", + "source_contract_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "paths": [ "reports/benchmark_reproducibility.json", @@ -298,13 +313,13 @@ "status": "pass", "expected": { "event_count": 1, - "adoption_sample_count": 0, - "activation_count": 0, - "accepted_count": 0, - "adoption_rate": 0, + "adoption_sample_count": 1, + "activation_count": 1, + "accepted_count": 1, + "adoption_rate": 100.0, "risk_band": "low", "event_types": { - "review_event": 1 + "skill_activation": 1 }, "source_types": { "manual": 1 @@ -312,13 +327,13 @@ }, "actual": { "event_count": 1, - "adoption_sample_count": 0, - "activation_count": 0, - "accepted_count": 0, - "adoption_rate": 0, + "adoption_sample_count": 1, + "activation_count": 1, + "accepted_count": 1, + "adoption_rate": 100.0, "risk_band": "low", "event_types": { - "review_event": 1 + "skill_activation": 1 }, "source_types": { "manual": 1 @@ -341,8 +356,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "ready_to_claim_world_class": false, "decision": "evidence-pending" }, @@ -353,8 +368,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "ready_to_claim_world_class": false, "decision": "evidence-pending" }, @@ -374,7 +389,7 @@ "pending_count": 4, "accepted_count": 0, "source_check_count": 13, - "source_pass_count": 6 + "source_pass_count": 7 }, "actual": { "ready": false, @@ -382,7 +397,7 @@ "pending_count": 4, "accepted_count": 0, "source_check_count": 13, - "source_pass_count": 6 + "source_pass_count": 7 }, "paths": [ "reports/world_class_evidence_ledger.json", @@ -394,8 +409,8 @@ "key": "interpretation-benchmark-commit", "label": "interpretation embeds the benchmark commit", "status": "pass", - "expected": "3961b60341a893464c1ea96f9921f804809f37f9", - "actual": "3961b60341a893464c1ea96f9921f804809f37f9", + "expected": "7476c7ccbfaa094932809d405accf3aca9bc16ea", + "actual": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-interpretation.json" @@ -407,30 +422,30 @@ "label": "interpretation embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 25, "missing_artifact_count": 0, - "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", + "source_contract_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "actual": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 25, "missing_artifact_count": 0, - "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", + "source_contract_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "paths": [ "reports/benchmark_reproducibility.json", @@ -444,13 +459,13 @@ "status": "pass", "expected": { "event_count": 1, - "adoption_sample_count": 0, - "activation_count": 0, - "accepted_count": 0, - "adoption_rate": 0, + "adoption_sample_count": 1, + "activation_count": 1, + "accepted_count": 1, + "adoption_rate": 100.0, "risk_band": "low", "event_types": { - "review_event": 1 + "skill_activation": 1 }, "source_types": { "manual": 1 @@ -458,13 +473,13 @@ }, "actual": { "event_count": 1, - "adoption_sample_count": 0, - "activation_count": 0, - "accepted_count": 0, - "adoption_rate": 0, + "adoption_sample_count": 1, + "activation_count": 1, + "accepted_count": 1, + "adoption_rate": 100.0, "risk_band": "low", "event_types": { - "review_event": 1 + "skill_activation": 1 }, "source_types": { "manual": 1 @@ -487,8 +502,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "ready_to_claim_world_class": false, "decision": "evidence-pending" }, @@ -499,8 +514,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "ready_to_claim_world_class": false, "decision": "evidence-pending" }, @@ -520,7 +535,7 @@ "pending_count": 4, "accepted_count": 0, "source_check_count": 13, - "source_pass_count": 6 + "source_pass_count": 7 }, "actual": { "ready": false, @@ -528,7 +543,7 @@ "pending_count": 4, "accepted_count": 0, "source_check_count": 13, - "source_pass_count": 6 + "source_pass_count": 7 }, "paths": [ "reports/world_class_evidence_ledger.json", @@ -1327,7 +1342,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 6, + "source_pass_count": 7, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -1386,8 +1401,7 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events", - "Adoption sample" + "External events" ] } ] @@ -1401,7 +1415,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 6, + "source_pass_count": 7, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -1460,8 +1474,7 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events", - "Adoption sample" + "External events" ] } ] @@ -1811,15 +1824,15 @@ "expected": { "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false }, "actual": { "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false }, "paths": [ @@ -1835,8 +1848,8 @@ "expected": { "pending_count": 4, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "ready_to_claim_world_class": false, "preflight_counts_as_evidence": false, "credential_value_exposed": false @@ -1844,8 +1857,8 @@ "actual": { "pending_count": 4, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "ready_to_claim_world_class": false, "preflight_counts_as_evidence": false, "credential_value_exposed": false @@ -2099,8 +2112,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "plan_keys": [ "human-adjudication", "native-client-telemetry", @@ -2248,8 +2261,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "plan_keys": [ "human-adjudication", "native-client-telemetry", diff --git a/reports/evidence_consistency.md b/reports/evidence_consistency.md index 6127ac0..26627b6 100644 --- a/reports/evidence_consistency.md +++ b/reports/evidence_consistency.md @@ -5,8 +5,8 @@ Generated at: `2026-06-17` ## Summary - decision: `consistent` -- checks: `35` -- pass: `35` +- checks: `36` +- pass: `36` - warn: `0` - fail: `0` @@ -20,6 +20,7 @@ This gate compares generated evidence reports against each other. It does not cr | Release evidence flow covers first-class reports | `pass` | Release refresh and clean-lock instructions must regenerate every first-class report before evidence consistency can be trusted. | `AGENTS.md`, `reports/output_execution_runs.json`, `reports/install_simulation.json`, `reports/security_trust_report.json`, `reports/registry_audit.json`, `reports/package_verification.json`, `reports/upgrade_check.json`, `reports/adoption_drift_report.json`, `reports/architecture_maintainability.json`, `reports/python_compatibility.json`, `reports/runtime_permission_probes.json`, `reports/review_waivers.json`, `reports/review_annotations.json`, `reports/skill_atlas.json`, `reports/skill_os2_audit.json`, `reports/skill_os2_coverage.json`, `reports/context_budget.json`, `reports/context_budget_summary.json`, `reports/benchmark_reproducibility.json`, `reports/skill-overview.json`, `reports/skill-interpretation.json`, `reports/review-viewer.json`, `reports/world_class_evidence_preflight.json`, `reports/skillops/daily`, `reports/skillops/weekly`, `reports/review-studio.json`, `reports/evidence_consistency.json` | | Review Studio mirrors context budget governance | `pass` | Review Studio must not keep stale context warnings after context reports prove large deferred resources are governed. | `reports/context_budget.json`, `reports/review-studio.json` | | Benchmark release lock matches git dirty state | `pass` | The benchmark release lock must reflect the generation-time git dirty flag. | `reports/benchmark_reproducibility.json` | +| Clean worktree keeps a clean benchmark release lock | `pass` | Dirty or non-git worktrees cannot prove final release-lock freshness, so this check is advisory until the final clean-lock pass. | `reports/benchmark_reproducibility.json` | | Human-facing reports expose the canonical Skill IR artifact | `pass` | Skill IR is the 2.0 platform-neutral semantic source, so user-facing reports must link to the artifact that actually exists. | `reports/skill-overview.json`, `reports/skill-interpretation.json`, `reports/review-studio.json`, `skill-ir/examples/yao-meta-skill.json` | | overview embeds the benchmark commit | `pass` | Human-facing reports must point to the same benchmark release-lock commit. | `reports/benchmark_reproducibility.json`, `reports/skill-overview.json` | | overview embeds benchmark summary fields | `pass` | Selected summary fields must match exactly across generated reports. | `reports/benchmark_reproducibility.json`, `reports/skill-overview.json` | diff --git a/reports/review-studio.json b/reports/review-studio.json index 056b946..00a8a2d 100644 --- a/reports/review-studio.json +++ b/reports/review-studio.json @@ -1985,9 +1985,9 @@ "public_claim_ready": false, "public_claim_blocker_count": 5, "working_tree_dirty": true, - "changed_file_count": 33 + "changed_file_count": 9 }, - "commit": "108f4d8b8736c5cda3be52697b44efaf4e785b42", + "commit": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -5650,24 +5650,21 @@ "ok": true, "generated_at": "2026-06-17", "skill_dir": ".", - "commit": "108f4d8b8736c5cda3be52697b44efaf4e785b42", + "commit": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "git_status": { "available": true, "dirty": true, - "changed_file_count": 33, + "changed_file_count": 9, "sample": [ - " M reports/architecture_maintainability.json", - " M reports/architecture_maintainability.md", - " M reports/context_budget.json", - " M reports/context_budget.md", - " M reports/context_budget_summary.json", + " M reports/benchmark_reproducibility.json", + " M reports/benchmark_reproducibility.md", " M reports/evidence_consistency.json", " M reports/evidence_consistency.md", - " M reports/python_compatibility.json", - " M reports/python_compatibility.md", - " M reports/review-studio.html", - " M reports/review-studio.json", - " M reports/security_trust_report.json" + " M reports/review-viewer.json", + " M reports/skill-interpretation.html", + " M reports/skill-interpretation.json", + " M reports/skill-overview.html", + " M reports/skill-overview.json" ], "scope": "generation-time status before this report is written" }, @@ -5699,7 +5696,7 @@ "public_claim_ready": false, "public_claim_blocker_count": 5, "working_tree_dirty": true, - "changed_file_count": 33 + "changed_file_count": 9 }, "public_claim": { "ready": false, @@ -5715,7 +5712,7 @@ }, "release_lock": { "ready": false, - "commit": "108f4d8b8736c5cda3be52697b44efaf4e785b42", + "commit": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "status_scope": "generation-time status before this report is written", "reason": "working tree was dirty at generation time" }, diff --git a/reports/review-viewer.json b/reports/review-viewer.json index 9ea2cdd..d1afed9 100644 --- a/reports/review-viewer.json +++ b/reports/review-viewer.json @@ -412,7 +412,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 6, + "source_pass_count": 7, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -471,8 +471,7 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events", - "Adoption sample" + "External events" ] } ] @@ -1001,8 +1000,8 @@ "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "678c3839e1d41692b4fb991e542e7d6ef60fbca5248dab2e3f313e138aedde77", - "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", + "evidence_bundle_sha256": "c31f7210165c76ab9b288c6c7d5d228fd46c8c8627cd86ca095cc1033607e34f", + "source_contract_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1018,14 +1017,14 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, "public_claim_blocker_count": 5, "working_tree_dirty": true, - "changed_file_count": 18 + "changed_file_count": 9 }, - "commit": "3961b60341a893464c1ea96f9921f804809f37f9", + "commit": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1130,7 +1129,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 231, - "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" + "package_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b" }, "skill_atlas": { "skill_count": 12, @@ -1168,7 +1167,7 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "a163b6e288c5342e683875b47fdfe62989d3f09241c0c8fd86ae615c6d5da7ad", + "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b" }, "compatibility": { @@ -1205,7 +1204,7 @@ "install_simulated": true, "install_simulation": "reports/install_simulation.json" }, - "generated_at": "2026-06-16" + "generated_at": "2026-06-17" }, "failures": [], "warnings": [] @@ -1228,7 +1227,7 @@ "ok": true, "summary": { "archive_present": true, - "archive_entry_count": 658, + "archive_entry_count": 677, "archive_extracted": true, "entrypoint_loaded": true, "manifest_loaded": true, @@ -1300,7 +1299,7 @@ { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "a163b6e288c5342e683875b47fdfe62989d3f09241c0c8fd86ae615c6d5da7ad" + "to": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" } ] }, @@ -1317,14 +1316,14 @@ "ok": true, "summary": { "event_count": 1, - "adoption_sample_count": 0, - "activation_count": 0, - "accepted_count": 0, + "adoption_sample_count": 1, + "activation_count": 1, + "accepted_count": 1, "edited_count": 0, "rejected_count": 0, "missed_count": 0, "failed_count": 0, - "adoption_rate": 0, + "adoption_rate": 100.0, "missed_trigger_count": 0, "wrong_trigger_count": 0, "bad_output_count": 0, @@ -1333,7 +1332,7 @@ "review_overdue_count": 0, "risk_band": "low", "event_types": { - "review_event": 1 + "skill_activation": 1 }, "failure_types": {}, "source_types": { @@ -1600,7 +1599,7 @@ "status": "external_required", "category": "external", "owner": "Browser/Chrome/IDE/provider client integrator", - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1653,8 +1652,8 @@ "missing_submission_count": 4, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "submitted_but_pending_count": 0, "source_accepted_without_valid_submission_count": 0, "overclaim_guard_active": true, @@ -2004,7 +2003,7 @@ "status": "pending", "source_status": "external_required", "source_accepted": false, - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -2041,7 +2040,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 0, + "adoption_sample_count": 1, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -2062,8 +2061,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 0, - "status": "blocked", + "actual": 1, + "status": "pass", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -2079,8 +2078,8 @@ } ], "source_check_count": 3, - "source_pass_count": 1, - "source_blocked_count": 2, + "source_pass_count": 2, + "source_blocked_count": 1, "submission_state": { "status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json", diff --git a/reports/skill-interpretation.html b/reports/skill-interpretation.html index 0c4638f..d3c8e09 100644 --- a/reports/skill-interpretation.html +++ b/reports/skill-interpretation.html @@ -910,7 +910,7 @@ -

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks6 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
  • 采用样本Adoption sample
+

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks7 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
diff --git a/reports/skill-interpretation.json b/reports/skill-interpretation.json index 158ad24..921ed3f 100644 --- a/reports/skill-interpretation.json +++ b/reports/skill-interpretation.json @@ -412,7 +412,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 6, + "source_pass_count": 7, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -471,8 +471,7 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events", - "Adoption sample" + "External events" ] } ] @@ -1005,8 +1004,8 @@ "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "678c3839e1d41692b4fb991e542e7d6ef60fbca5248dab2e3f313e138aedde77", - "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", + "evidence_bundle_sha256": "c31f7210165c76ab9b288c6c7d5d228fd46c8c8627cd86ca095cc1033607e34f", + "source_contract_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1022,14 +1021,14 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, "public_claim_blocker_count": 5, "working_tree_dirty": true, - "changed_file_count": 18 + "changed_file_count": 9 }, - "commit": "3961b60341a893464c1ea96f9921f804809f37f9", + "commit": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1134,7 +1133,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 231, - "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" + "package_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b" }, "skill_atlas": { "skill_count": 12, @@ -1172,7 +1171,7 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "a163b6e288c5342e683875b47fdfe62989d3f09241c0c8fd86ae615c6d5da7ad", + "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b" }, "compatibility": { @@ -1209,7 +1208,7 @@ "install_simulated": true, "install_simulation": "reports/install_simulation.json" }, - "generated_at": "2026-06-16" + "generated_at": "2026-06-17" }, "failures": [], "warnings": [] @@ -1232,7 +1231,7 @@ "ok": true, "summary": { "archive_present": true, - "archive_entry_count": 658, + "archive_entry_count": 677, "archive_extracted": true, "entrypoint_loaded": true, "manifest_loaded": true, @@ -1304,7 +1303,7 @@ { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "a163b6e288c5342e683875b47fdfe62989d3f09241c0c8fd86ae615c6d5da7ad" + "to": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" } ] }, @@ -1321,14 +1320,14 @@ "ok": true, "summary": { "event_count": 1, - "adoption_sample_count": 0, - "activation_count": 0, - "accepted_count": 0, + "adoption_sample_count": 1, + "activation_count": 1, + "accepted_count": 1, "edited_count": 0, "rejected_count": 0, "missed_count": 0, "failed_count": 0, - "adoption_rate": 0, + "adoption_rate": 100.0, "missed_trigger_count": 0, "wrong_trigger_count": 0, "bad_output_count": 0, @@ -1337,7 +1336,7 @@ "review_overdue_count": 0, "risk_band": "low", "event_types": { - "review_event": 1 + "skill_activation": 1 }, "failure_types": {}, "source_types": { @@ -1604,7 +1603,7 @@ "status": "external_required", "category": "external", "owner": "Browser/Chrome/IDE/provider client integrator", - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1657,8 +1656,8 @@ "missing_submission_count": 4, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "submitted_but_pending_count": 0, "source_accepted_without_valid_submission_count": 0, "overclaim_guard_active": true, @@ -2008,7 +2007,7 @@ "status": "pending", "source_status": "external_required", "source_accepted": false, - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -2045,7 +2044,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 0, + "adoption_sample_count": 1, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -2066,8 +2065,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 0, - "status": "blocked", + "actual": 1, + "status": "pass", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -2083,8 +2082,8 @@ } ], "source_check_count": 3, - "source_pass_count": 1, - "source_blocked_count": 2, + "source_pass_count": 2, + "source_blocked_count": 1, "submission_state": { "status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json", diff --git a/reports/skill-overview.html b/reports/skill-overview.html index c6d5f21..c456d93 100644 --- a/reports/skill-overview.html +++ b/reports/skill-overview.html @@ -910,7 +910,7 @@ -

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks6 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
  • 采用样本Adoption sample
+

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks7 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
diff --git a/reports/skill-overview.json b/reports/skill-overview.json index 375ea30..70ad616 100644 --- a/reports/skill-overview.json +++ b/reports/skill-overview.json @@ -411,7 +411,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 6, + "source_pass_count": 7, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -470,8 +470,7 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events", - "Adoption sample" + "External events" ] } ] @@ -1000,8 +999,8 @@ "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "678c3839e1d41692b4fb991e542e7d6ef60fbca5248dab2e3f313e138aedde77", - "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", + "evidence_bundle_sha256": "c31f7210165c76ab9b288c6c7d5d228fd46c8c8627cd86ca095cc1033607e34f", + "source_contract_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1017,14 +1016,14 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, "public_claim_blocker_count": 5, "working_tree_dirty": true, - "changed_file_count": 18 + "changed_file_count": 9 }, - "commit": "3961b60341a893464c1ea96f9921f804809f37f9", + "commit": "7476c7ccbfaa094932809d405accf3aca9bc16ea", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1129,7 +1128,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 231, - "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" + "package_sha256": "9cb9a19e774540e721b143a240baf41b41c32b9f10df532eb4a7d1cf7a2fe67b" }, "skill_atlas": { "skill_count": 12, @@ -1167,7 +1166,7 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "a163b6e288c5342e683875b47fdfe62989d3f09241c0c8fd86ae615c6d5da7ad", + "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b" }, "compatibility": { @@ -1204,7 +1203,7 @@ "install_simulated": true, "install_simulation": "reports/install_simulation.json" }, - "generated_at": "2026-06-16" + "generated_at": "2026-06-17" }, "failures": [], "warnings": [] @@ -1227,7 +1226,7 @@ "ok": true, "summary": { "archive_present": true, - "archive_entry_count": 658, + "archive_entry_count": 677, "archive_extracted": true, "entrypoint_loaded": true, "manifest_loaded": true, @@ -1299,7 +1298,7 @@ { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "a163b6e288c5342e683875b47fdfe62989d3f09241c0c8fd86ae615c6d5da7ad" + "to": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" } ] }, @@ -1316,14 +1315,14 @@ "ok": true, "summary": { "event_count": 1, - "adoption_sample_count": 0, - "activation_count": 0, - "accepted_count": 0, + "adoption_sample_count": 1, + "activation_count": 1, + "accepted_count": 1, "edited_count": 0, "rejected_count": 0, "missed_count": 0, "failed_count": 0, - "adoption_rate": 0, + "adoption_rate": 100.0, "missed_trigger_count": 0, "wrong_trigger_count": 0, "bad_output_count": 0, @@ -1332,7 +1331,7 @@ "review_overdue_count": 0, "risk_band": "low", "event_types": { - "review_event": 1 + "skill_activation": 1 }, "failure_types": {}, "source_types": { @@ -1599,7 +1598,7 @@ "status": "external_required", "category": "external", "owner": "Browser/Chrome/IDE/provider client integrator", - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1652,8 +1651,8 @@ "missing_submission_count": 4, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "submitted_but_pending_count": 0, "source_accepted_without_valid_submission_count": 0, "overclaim_guard_active": true, @@ -2003,7 +2002,7 @@ "status": "pending", "source_status": "external_required", "source_accepted": false, - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -2040,7 +2039,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 0, + "adoption_sample_count": 1, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -2061,8 +2060,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 0, - "status": "blocked", + "actual": 1, + "status": "pass", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -2078,8 +2077,8 @@ } ], "source_check_count": 3, - "source_pass_count": 1, - "source_blocked_count": 2, + "source_pass_count": 2, + "source_blocked_count": 1, "submission_state": { "status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json",