diff --git a/registry/index.json b/registry/index.json index 6811f38..7399bf0 100644 --- a/registry/index.json +++ b/registry/index.json @@ -16,7 +16,7 @@ "vscode" ], "package_metadata": "registry/packages/yao-meta-skill.json", - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" } ] } diff --git a/registry/packages/yao-meta-skill.json b/registry/packages/yao-meta-skill.json index 82dc9a5..9c57cbe 100644 --- a/registry/packages/yao-meta-skill.json +++ b/registry/packages/yao-meta-skill.json @@ -16,8 +16,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d", - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, "compatibility": { "openai": "pass", @@ -48,7 +48,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" diff --git a/reports/adoption_drift_report.json b/reports/adoption_drift_report.json index dccfe43..eced7fe 100644 --- a/reports/adoption_drift_report.json +++ b/reports/adoption_drift_report.json @@ -1,7 +1,7 @@ { "ok": true, "schema_version": "2.0", - "generated_at": "2026-06-14T00:00:00Z", + "generated_at": "2026-06-13T17:32:44Z", "skill_dir": ".", "privacy_contract": { "storage": "local-first", diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json index fa7e84d..bd2e165 100644 --- a/reports/benchmark_reproducibility.json +++ b/reports/benchmark_reproducibility.json @@ -3,24 +3,24 @@ "ok": true, "generated_at": "2026-06-14", "skill_dir": ".", - "commit": "ee4879f296d6e6b5dde76afc37b81504806afc82", + "commit": "4737358f3c7acb6650d5463793e5f0e8f913aada", "git_status": { "available": true, "dirty": true, - "changed_file_count": 44, + "changed_file_count": 30, "sample": [ - " M Makefile", - " M README.md", " M registry/index.json", " M registry/packages/yao-meta-skill.json", + " M reports/adoption_drift_report.json", " M reports/benchmark_reproducibility.json", " M reports/benchmark_reproducibility.md", " M reports/compiled_targets.json", " M reports/context_budget.json", - " M reports/install_simulation.json", " M reports/output_execution_runs.json", " M reports/output_execution_runs.md", - " M reports/package_verification.json" + " M reports/package_verification.json", + " M reports/package_verification.md", + " M reports/registry_audit.json" ] }, "summary": { @@ -42,7 +42,7 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "working_tree_dirty": true, - "changed_file_count": 44 + "changed_file_count": 30 }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -115,8 +115,8 @@ "label": "output_execution", "path": "reports/output_execution_runs.json", "exists": true, - "bytes": 7966, - "sha256": "0340b973c3e967c0e46b36109cc6c0419446be1658a9e02d706aec2fd946acea" + "bytes": 7967, + "sha256": "cd6de0ee193725d03c3e16b3b116eda21ee2db71d830505de50da4c3624c8f3e" }, { "label": "blind_review", @@ -151,7 +151,7 @@ "path": "reports/security_trust_report.json", "exists": true, "bytes": 87272, - "sha256": "0eed3f527cd23ee75ddc2f5bf067366571f43eb99708f3180f2a399810d5d48f" + "sha256": "3cbe50361df9b509477ee70f27da9fc4e95462d588b1e7b71ba31da8e9f75d7b" }, { "label": "python_compatibility", @@ -165,14 +165,14 @@ "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, - "sha256": "dee17f5e1f35cd167f1237e55609b4422e9ff21caf9d87e00a06c1d5c657f319" + "sha256": "d293ac8472d4c16f32f77cecb735d7c32f5a0bf9a409f43006c74b7d826de34b" }, { "label": "package_verification", "path": "reports/package_verification.json", "exists": true, "bytes": 19325, - "sha256": "b247e849e011158e40128cda8a40ca4b6f6093b67f7aaa4b26a845fa2a981b9e" + "sha256": "bef665206e2f6ca0e998caa941fc6ffe75cf782036945595d743f83f156fc04f" }, { "label": "install_simulation", diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md index 9806c3c..6c44ec7 100644 --- a/reports/benchmark_reproducibility.md +++ b/reports/benchmark_reproducibility.md @@ -1,7 +1,7 @@ # Benchmark Reproducibility Generated at: `2026-06-14` -Commit: `ee4879f296d6e6b5dde76afc37b81504806afc82` +Commit: `4737358f3c7acb6650d5463793e5f0e8f913aada` Working tree dirty at generation: `true` ## Summary @@ -16,7 +16,7 @@ Working tree dirty at generation: `true` - provider evidence complete: `false` - human review complete: `false` - world-class ready: `false` -- changed files at generation: `44` +- changed files at generation: `30` This report proves local benchmark reproducibility only. It keeps external provider and human-review gaps visible instead of counting them as complete. @@ -40,15 +40,15 @@ This report proves local benchmark reproducibility only. It keeps external provi | output_cases | `evals/output/cases.jsonl` | present | `a6ae96857116` | | output_schema | `evals/output/schema.json` | present | `8ee340c95064` | | output_scorecard | `reports/output_quality_scorecard.json` | present | `0806258a8e08` | -| output_execution | `reports/output_execution_runs.json` | present | `0340b973c3e9` | +| output_execution | `reports/output_execution_runs.json` | present | `cd6de0ee1937` | | blind_review | `reports/output_blind_review_pack.json` | present | `bbe2db8ec277` | | review_adjudication | `reports/output_review_adjudication.json` | present | `ddd9af90d42e` | | trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` | | runtime_conformance | `reports/conformance_matrix.json` | present | `8251329e663d` | -| trust_report | `reports/security_trust_report.json` | present | `0eed3f527cd2` | +| trust_report | `reports/security_trust_report.json` | present | `3cbe50361df9` | | python_compatibility | `reports/python_compatibility.json` | present | `435dabcd3ce0` | -| registry_audit | `reports/registry_audit.json` | present | `dee17f5e1f35` | -| package_verification | `reports/package_verification.json` | present | `b247e849e011` | +| registry_audit | `reports/registry_audit.json` | present | `d293ac8472d4` | +| package_verification | `reports/package_verification.json` | present | `bef665206e2f` | | install_simulation | `reports/install_simulation.json` | present | `2bb990fc2198` | | skill_os2_audit | `reports/skill_os2_audit.json` | present | `ef085ddc8a1e` | | world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `ed1274b7c18c` | diff --git a/reports/compiled_targets.json b/reports/compiled_targets.json index 7eeb01f..7d5f1f9 100644 --- a/reports/compiled_targets.json +++ b/reports/compiled_targets.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-14", "skill_dir": ".", "summary": { "target_count": 5, diff --git a/reports/context_budget.json b/reports/context_budget.json index 44b1823..7bfbc4c 100644 --- a/reports/context_budget.json +++ b/reports/context_budget.json @@ -6,9 +6,9 @@ "context_budget_tier": "production", "context_budget_limit": 1000, "skill_body_tokens": 751, - "other_text_tokens": 1067478, + "other_text_tokens": 1073524, "estimated_initial_load_tokens": 944, - "estimated_total_text_tokens": 1068229, + "estimated_total_text_tokens": 1074275, "relevant_file_count": 461, "unused_resource_dirs": [], "quality_signal_points": 130, diff --git a/reports/output_execution_runs.json b/reports/output_execution_runs.json index 0af2703..4911b40 100644 --- a/reports/output_execution_runs.json +++ b/reports/output_execution_runs.json @@ -34,7 +34,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.61, + "duration_ms": 25.92, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -62,7 +62,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.26, + "duration_ms": 25.31, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -85,7 +85,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.3, + "duration_ms": 25.31, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -113,7 +113,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.97, + "duration_ms": 25.65, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -136,7 +136,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.09, + "duration_ms": 25.46, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -164,7 +164,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.12, + "duration_ms": 25.46, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -187,7 +187,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.99, + "duration_ms": 25.63, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -214,7 +214,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.69, + "duration_ms": 25.51, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -237,7 +237,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.08, + "duration_ms": 25.72, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -266,7 +266,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.82, + "duration_ms": 25.64, "provider": "local-output-eval-runner", "model": "", "usage": { diff --git a/reports/output_execution_runs.md b/reports/output_execution_runs.md index 6f05043..ce7e819 100644 --- a/reports/output_execution_runs.md +++ b/reports/output_execution_runs.md @@ -23,16 +23,16 @@ Command runner evidence is present. This proves the eval harness executed an ext | Case | Variant | Mode | Model | Duration ms | Tokens | Score | Status | | --- | --- | --- | --- | ---: | ---: | ---: | --- | -| skill-package-contract | baseline | command | local-output-eval-runner | 28.61 | 33 | 0.0 | pass | -| skill-package-contract | with_skill | command | local-output-eval-runner | 28.26 | 73 | 100.0 | pass | -| output-eval-expectation | baseline | command | local-output-eval-runner | 28.3 | 36 | 0.0 | pass | -| output-eval-expectation | with_skill | command | local-output-eval-runner | 28.97 | 80 | 100.0 | pass | -| ir-before-packaging | baseline | command | local-output-eval-runner | 28.09 | 33 | 0.0 | pass | -| ir-before-packaging | with_skill | command | local-output-eval-runner | 28.12 | 80 | 100.0 | pass | -| near-neighbor-boundary | baseline | command | local-output-eval-runner | 27.99 | 36 | 0.0 | pass | -| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 27.69 | 65 | 100.0 | pass | -| file-backed-governed-package | baseline | command | local-output-eval-runner | 28.08 | 37 | 0.0 | pass | -| file-backed-governed-package | with_skill | command | local-output-eval-runner | 27.82 | 98 | 100.0 | pass | +| skill-package-contract | baseline | command | local-output-eval-runner | 25.92 | 33 | 0.0 | pass | +| skill-package-contract | with_skill | command | local-output-eval-runner | 25.31 | 73 | 100.0 | pass | +| output-eval-expectation | baseline | command | local-output-eval-runner | 25.31 | 36 | 0.0 | pass | +| output-eval-expectation | with_skill | command | local-output-eval-runner | 25.65 | 80 | 100.0 | pass | +| ir-before-packaging | baseline | command | local-output-eval-runner | 25.46 | 33 | 0.0 | pass | +| ir-before-packaging | with_skill | command | local-output-eval-runner | 25.46 | 80 | 100.0 | pass | +| near-neighbor-boundary | baseline | command | local-output-eval-runner | 25.63 | 36 | 0.0 | pass | +| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 25.51 | 65 | 100.0 | pass | +| file-backed-governed-package | baseline | command | local-output-eval-runner | 25.72 | 37 | 0.0 | pass | +| file-backed-governed-package | with_skill | command | local-output-eval-runner | 25.64 | 98 | 100.0 | pass | ## Next Fixes diff --git a/reports/package_verification.json b/reports/package_verification.json index f953535..1722488 100644 --- a/reports/package_verification.json +++ b/reports/package_verification.json @@ -8,7 +8,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "archive_entry_count": 547, "failure_count": 0, "warning_count": 0 diff --git a/reports/package_verification.md b/reports/package_verification.md index 150612c..66f5746 100644 --- a/reports/package_verification.md +++ b/reports/package_verification.md @@ -4,7 +4,7 @@ - Package directory: `dist` - Targets: `4 / 4` adapters present - Archive present: `True` -- Archive SHA256: `13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05` +- Archive SHA256: `d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2` - Failures: `0` - Warnings: `0` diff --git a/reports/registry_audit.json b/reports/registry_audit.json index 2624968..1e71d6f 100644 --- a/reports/registry_audit.json +++ b/reports/registry_audit.json @@ -21,8 +21,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d", - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, "compatibility": { "openai": "pass", @@ -53,7 +53,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -78,7 +78,7 @@ "vscode" ], "package_metadata": "registry/packages/yao-meta-skill.json", - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" } ] }, diff --git a/reports/registry_audit.md b/reports/registry_audit.md index 192a2e4..dcd9d41 100644 --- a/reports/registry_audit.md +++ b/reports/registry_audit.md @@ -6,8 +6,8 @@ - Maturity: `governed` - Owner: `Yao Team` - License: `MIT` -- Package SHA256: `5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d` -- Archive SHA256: `13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05` +- Package SHA256: `e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723` +- Archive SHA256: `d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2` - Install simulated: `True` ## Compatibility diff --git a/reports/review-studio.html b/reports/review-studio.html index 8d99597..04f68bf 100644 --- a/reports/review-studio.html +++ b/reports/review-studio.html @@ -297,7 +297,7 @@

审查闸门

-
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 944/1000; quality density 137.7

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 82 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low

reports/adoption_drift_report.json 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

reports/promotion_decisions.json + reports/upgrade_check.json + docs/migration-v2.md 证据
+
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 944/1000; quality density 137.7

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 82 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

Python 兼容

Python 3.11; 140 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards

reports/python_compatibility.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low

reports/adoption_drift_report.json 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

@@ -363,7 +363,7 @@
-

信任报告

Secret
0
脚本数
82
网络脚本
3
Help 失败
0
包体哈希
5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d
+

信任报告

Secret
0
脚本数
82
网络脚本
3
Help 失败
0
包体哈希
e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723

安全边界

高风险 secret、远程 inline execution、缺失依赖策略或无法解释的脚本接口应阻断 governed release。

@@ -425,12 +425,12 @@

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

-

包体元数据

名称
yao-meta-skill
版本
1.1.0
Maturity
governed
Owner
Yao Team
License
MIT
信任级别
local
目标平台
openai, claude, generic, agent-skills-compatible, vscode
兼容通过
6/6
归档哈希
13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05
+

包体元数据

名称
yao-meta-skill
版本
1.1.0
Maturity
governed
Owner
Yao Team
License
MIT
信任级别
local
目标平台
openai, claude, generic, agent-skills-compatible, vscode
兼容通过
6/6
归档哈希
d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

-

包体验证

目标数
4
Adapter
4
归档存在
Zip 条目
547
失败数
0
警告数
0
归档哈希
13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05
+

包体验证

目标数
4
Adapter
4
归档存在
Zip 条目
547
失败数
0
警告数
0
归档哈希
d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2
diff --git a/reports/review-studio.json b/reports/review-studio.json index 8af4548..942cc04 100644 --- a/reports/review-studio.json +++ b/reports/review-studio.json @@ -5,7 +5,7 @@ "summary": { "decision": "review", "world_class_score": 90, - "gate_count": 14, + "gate_count": 15, "blocker_count": 0, "warning_count": 3, "action_count": 3, @@ -63,6 +63,14 @@ "evidence": "reports/security_trust_report.json", "link": "security_trust_report.md" }, + { + "key": "python-compat", + "label": "Python 兼容", + "status": "pass", + "detail": "Python 3.11; 140 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards", + "evidence": "reports/python_compatibility.json", + "link": "python_compatibility.md" + }, { "key": "permission-gates", "label": "权限批准", @@ -1289,9 +1297,9 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "working_tree_dirty": true, - "changed_file_count": 44 + "changed_file_count": 35 }, - "commit": "ee4879f296d6e6b5dde76afc37b81504806afc82", + "commit": "4737358f3c7acb6650d5463793e5f0e8f913aada", "missing_artifacts": [], "limitations": [ "Local command-runner evidence is reproducible but does not replace provider-backed model holdout evidence.", @@ -1387,7 +1395,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 166, - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" }, "skill_atlas": { "skill_count": 12, @@ -1425,8 +1433,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d", - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, "compatibility": { "openai": "pass", @@ -1457,7 +1465,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -1473,7 +1481,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "archive_entry_count": 547, "failure_count": 0, "warning_count": 0 @@ -1552,12 +1560,12 @@ { "field": "archive_sha256", "from": "", - "to": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05" + "to": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "to": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" } ] }, @@ -3674,7 +3682,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.61, + "duration_ms": 25.92, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -3702,7 +3710,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.26, + "duration_ms": 25.31, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -3725,7 +3733,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.3, + "duration_ms": 25.31, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -3753,7 +3761,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.97, + "duration_ms": 25.65, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -3776,7 +3784,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.09, + "duration_ms": 25.46, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -3804,7 +3812,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.12, + "duration_ms": 25.46, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -3827,7 +3835,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.99, + "duration_ms": 25.63, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -3854,7 +3862,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.69, + "duration_ms": 25.51, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -3877,7 +3885,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 28.08, + "duration_ms": 25.72, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -3906,7 +3914,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.82, + "duration_ms": 25.64, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4452,7 +4460,7 @@ "label": "Review Studio", "status": "pass", "objective": "One HTML page supports first-pass production review across trigger, output, runtime, trust, release, and evidence actions.", - "current": "14 gates; decision review; warnings 3", + "current": "15 gates; decision review; warnings 3", "command": "python3 scripts/yao.py review-studio .", "test": "python3 tests/verify_review_studio.py", "evidence": [ @@ -4770,7 +4778,7 @@ "label": "Review Studio 2.0", "status": "pass", "objective": "Recommended Skill OS 2.0 implementation PR from the upgrade plan.", - "current": "14 review gates", + "current": "15 review gates", "command": "make ci-test", "test": "tests/verify_review_studio.py", "evidence": [ @@ -4832,7 +4840,7 @@ "compiled_targets": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-14", "skill_dir": ".", "summary": { "target_count": 5, @@ -9585,7 +9593,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 166, - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" }, "failures": [], "warnings": [], @@ -12607,9 +12615,9 @@ "context_budget_tier": "production", "context_budget_limit": 1000, "skill_body_tokens": 751, - "other_text_tokens": 1067478, + "other_text_tokens": 1073524, "estimated_initial_load_tokens": 944, - "estimated_total_text_tokens": 1068229, + "estimated_total_text_tokens": 1074275, "relevant_file_count": 461, "unused_resource_dirs": [], "quality_signal_points": 130, @@ -13454,7 +13462,7 @@ }, "catalog": { "workspace_root": ".", - "generated_at": "2026-06-14", + "generated_at": "2026-06-13", "skills": [ { "name": "yao-meta-skill", @@ -13619,8 +13627,8 @@ "report": "reports/adoption_drift_report.json", "risk_band": "low", "event_count": 1, - "adoption_sample_count": 0, - "adoption_rate": 0, + "adoption_sample_count": 1, + "adoption_rate": 100.0, "candidate_count": 0 } }, @@ -14283,7 +14291,7 @@ "name": "incident-command-governor", "path": "examples/governed-incident-command/generated-skill", "reason": "review overdue by cadence monthly", - "age_days": 75, + "age_days": 74, "allowed_days": 31, "actionable": false, "scope": "example" @@ -14451,7 +14459,7 @@ "adoption_drift": { "ok": true, "schema_version": "2.0", - "generated_at": "2026-06-14T00:00:00Z", + "generated_at": "2026-06-13T17:32:44Z", "skill_dir": ".", "privacy_contract": { "storage": "local-first", @@ -14536,7 +14544,7 @@ "schema_version": "1.0", "ok": true, "skill_dir": ".", - "generated_at": "2026-06-13", + "generated_at": "2026-06-14", "summary": { "waiver_count": 0, "active_count": 0, @@ -15221,8 +15229,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d", - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, "compatibility": { "openai": "pass", @@ -15253,7 +15261,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -15278,7 +15286,7 @@ "vscode" ], "package_metadata": "registry/packages/yao-meta-skill.json", - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" } ] }, @@ -15301,7 +15309,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "archive_entry_count": 547, "failure_count": 0, "warning_count": 0 @@ -16309,12 +16317,12 @@ { "field": "archive_sha256", "from": "", - "to": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05" + "to": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "to": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" } ] }, diff --git a/reports/review-viewer.json b/reports/review-viewer.json index a30c9ad..fa87d2b 100644 --- a/reports/review-viewer.json +++ b/reports/review-viewer.json @@ -19,6 +19,7 @@ "reports/output_blind_review_pack.md", "reports/output_blind_answer_key.json", "reports/output_review_adjudication.md", + "reports/benchmark_reproducibility.md", "reports/conformance_matrix.md", "reports/security_trust_report.md", "reports/runtime_permission_probes.md", @@ -29,6 +30,8 @@ "reports/upgrade_check.md", "reports/adoption_drift_report.md", "reports/review_waivers.md", + "reports/world_class_evidence_plan.md", + "reports/world_class_evidence_ledger.md", "reports/review_annotations.md", "reports/review-studio.html", "reports/skill-overview.html" @@ -434,7 +437,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 74 + "file_count": 82 }, { "path": "evals", @@ -446,10 +449,10 @@ "path": "reports", "label": "Generated evidence and overview artifacts", "kind": "folder", - "file_count": 173 + "file_count": 193 } ], - "file_count": 311, + "file_count": 339, "folder_count": 4, "distribution": [ { @@ -474,7 +477,7 @@ }, { "label": "scripts", - "value": 74 + "value": 82 }, { "label": "evals", @@ -482,7 +485,7 @@ }, { "label": "reports", - "value": 173 + "value": 193 } ] }, @@ -603,7 +606,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 74 + "file_count": 82 }, { "path": "evals", @@ -615,7 +618,7 @@ "path": "reports", "label": "Generated evidence and overview artifacts", "kind": "folder", - "file_count": 173 + "file_count": 193 } ], "strengths": [ @@ -882,6 +885,37 @@ "reviewed_at": "", "failures": [] }, + "benchmark_reproducibility": { + "ok": true, + "summary": { + "reproducibility_ready": true, + "methodology_complete": true, + "required_artifact_count": 20, + "missing_artifact_count": 0, + "output_case_count": 5, + "failure_disclosure_count": 3, + "command_count": 19, + "command_executed_count": 10, + "timing_observed_count": 10, + "model_executed_count": 0, + "token_observed_count": 0, + "human_review_complete": false, + "provider_evidence_complete": false, + "world_class_ready": false, + "world_class_open_gap_count": 4, + "world_class_task_count": 4, + "world_class_ledger_pending_count": 4, + "working_tree_dirty": true, + "changed_file_count": 35 + }, + "commit": "4737358f3c7acb6650d5463793e5f0e8f913aada", + "missing_artifacts": [], + "limitations": [ + "Local command-runner evidence is reproducible but does not replace provider-backed model holdout evidence.", + "Pending blind-review decisions are visible but do not count as human adjudication.", + "World-class readiness remains false until external and human evidence gaps close." + ] + }, "runtime_conformance": { "target_count": 5, "pass_count": 5, @@ -949,8 +983,8 @@ "failures": [] }, "trust_security": { - "scanned_files": 158, - "script_count": 74, + "scanned_files": 166, + "script_count": 82, "internal_module_count": 10, "secret_findings": 0, "dependency_files": [ @@ -959,18 +993,18 @@ "network_script_count": 3, "network_policy_covered_count": 3, "network_policy_missing_count": 0, - "file_write_script_count": 51, + "file_write_script_count": 59, "permission_required_count": 3, "permission_approved_count": 3, "permission_missing_count": 0, "permission_invalid_count": 0, "permission_expired_count": 0, - "help_smoke_checked_count": 64, + "help_smoke_checked_count": 72, "help_smoke_failed_count": 0, "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", - "package_hash_file_count": 158, - "package_sha256": "e42a9c4cf0f22cd6fc047dff3da1a380f5cc9e17157a48d8a1ac2071f7c84fbc" + "package_hash_file_count": 166, + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" }, "skill_atlas": { "skill_count": 12, @@ -1008,8 +1042,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "e42a9c4cf0f22cd6fc047dff3da1a380f5cc9e17157a48d8a1ac2071f7c84fbc", - "archive_sha256": "7494bf9b978d58b663cc0e8b6ffe2531a84602385f6f517774320197d30170d3" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, "compatibility": { "openai": "pass", @@ -1040,12 +1074,12 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "7494bf9b978d58b663cc0e8b6ffe2531a84602385f6f517774320197d30170d3", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" }, - "generated_at": "2026-06-13" + "generated_at": "2026-06-14" }, "failures": [], "warnings": [] @@ -1056,8 +1090,8 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "7494bf9b978d58b663cc0e8b6ffe2531a84602385f6f517774320197d30170d3", - "archive_entry_count": 505, + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", + "archive_entry_count": 547, "failure_count": 0, "warning_count": 0 }, @@ -1068,7 +1102,7 @@ "ok": true, "summary": { "archive_present": true, - "archive_entry_count": 505, + "archive_entry_count": 547, "archive_extracted": true, "entrypoint_loaded": true, "manifest_loaded": true, @@ -1078,7 +1112,7 @@ "installer_permission_failure_count": 0, "permission_target_count": 4, "permission_capability_count": 3, - "install_root_is_temp": false, + "install_root_is_temp": true, "failure_count": 0, "warning_count": 0 }, @@ -1135,12 +1169,12 @@ { "field": "archive_sha256", "from": "", - "to": "7494bf9b978d58b663cc0e8b6ffe2531a84602385f6f517774320197d30170d3" + "to": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "e42a9c4cf0f22cd6fc047dff3da1a380f5cc9e17157a48d8a1ac2071f7c84fbc" + "to": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" } ] }, @@ -1252,6 +1286,389 @@ "annotations": [], "failures": [] }, + "world_class_evidence_plan": { + "ok": true, + "summary": { + "audit_decision": "continue-iteration", + "world_class_ready": false, + "ready_to_claim_world_class": false, + "task_count": 4, + "human_task_count": 1, + "external_task_count": 3, + "review_task_count": 0, + "decision": "collect-external-evidence" + }, + "tasks": [ + { + "key": "provider-holdout", + "label": "Provider Holdout", + "status": "external_required", + "category": "external", + "owner": "operator with provider credentials", + "current": "model-executed 0; token-observed 0", + "objective": "Collect at least one provider-backed output-eval holdout run with model, timing, and token metadata.", + "runbook": [ + "YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY= python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60", + "python3 scripts/yao.py skill-os2-audit . --generated-at ", + "Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields.", + "python3 scripts/yao.py world-class-intake ." + ], + "success_checks": [ + "reports/output_execution_runs.json summary.model_executed_count > 0", + "reports/output_execution_runs.json summary.timing_observed_count > 0", + "reports/output_execution_runs.json summary.token_observed_count > 0", + "reports/skill_os2_audit.json item provider-holdout status becomes pass" + ], + "evidence_artifacts": [ + "reports/output_execution_runs.json", + "reports/output_execution_runs.md", + "reports/skill_os2_audit.json", + "evidence/world_class/intake.schema.json", + "evidence/world_class/templates/provider-holdout.intake.json", + "reports/world_class_evidence_intake.json", + "reports/world_class_evidence_intake.md" + ], + "privacy_contract": [ + "Do not commit provider credentials or environment dumps.", + "The output execution report records output hashes and aggregate run metadata, not raw provider prompts." + ], + "audit_next_action": "Run provider-backed holdout cases with real credentials and commit only aggregate evidence." + }, + { + "key": "human-adjudication", + "label": "Human Adjudication", + "status": "human_required", + "category": "human", + "owner": "human reviewer", + "current": "0/5 decisions; pending 5", + "objective": "Record real blind A/B reviewer decisions before claiming human output review completion.", + "runbook": [ + "python3 scripts/adjudicate_output_review.py --write-template", + "Open reports/output_blind_review_pack.md and choose A or B for each pair without opening the answer key.", + "Edit reports/output_review_decisions.json with winner_variant values and reviewer metadata.", + "python3 scripts/yao.py output-review", + "python3 scripts/yao.py skill-os2-audit . --generated-at ", + "Copy evidence/world_class/templates/human-adjudication.intake.json to evidence/world_class/submissions/human-adjudication.json and fill only real evidence fields.", + "python3 scripts/yao.py world-class-intake ." + ], + "success_checks": [ + "reports/output_review_adjudication.json summary.pending_count == 0", + "reports/output_review_adjudication.json summary.judgment_count == summary.pair_count", + "reports/output_review_adjudication.json summary.invalid_decision_count == 0", + "reports/skill_os2_audit.json item human-adjudication status becomes pass" + ], + "evidence_artifacts": [ + "reports/output_blind_review_pack.md", + "reports/output_review_decisions.json", + "reports/output_review_adjudication.json", + "reports/output_review_adjudication.md", + "evidence/world_class/intake.schema.json", + "evidence/world_class/templates/human-adjudication.intake.json", + "reports/world_class_evidence_intake.json", + "reports/world_class_evidence_intake.md" + ], + "privacy_contract": [ + "Reviewer decisions should not include raw user data or private customer detail.", + "Keep the answer key separate until after decisions are recorded." + ], + "audit_next_action": "Record real A/B choices in the decision template, then regenerate adjudication." + }, + { + "key": "native-permission-enforcement", + "label": "Native Permission Enforcement", + "status": "external_required", + "category": "external", + "owner": "target client or installer integrator", + "current": "native-enforced targets 0", + "objective": "Prove at least one target or installer enforces approved high-permission capabilities at runtime.", + "runbook": [ + "Implement or connect a real target client/installer guard that blocks undeclared network, file_write, or subprocess capabilities.", + "Update the generated target adapter only when the guard is actually enforced by that target.", + "python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip", + "python3 scripts/yao.py runtime-permissions . --package-dir dist", + "python3 scripts/yao.py skill-os2-audit . --generated-at ", + "Copy evidence/world_class/templates/native-permission-enforcement.intake.json to evidence/world_class/submissions/native-permission-enforcement.json and fill only real evidence fields.", + "python3 scripts/yao.py world-class-intake ." + ], + "success_checks": [ + "reports/runtime_permission_probes.json summary.native_enforcement_count > 0", + "reports/runtime_permission_probes.json summary.failure_count == 0", + "reports/skill_os2_audit.json item native-permission-enforcement status becomes pass" + ], + "evidence_artifacts": [ + "dist/targets/*/adapter.json", + "reports/runtime_permission_probes.json", + "reports/runtime_permission_probes.md", + "security/permission_policy.json", + "evidence/world_class/intake.schema.json", + "evidence/world_class/templates/native-permission-enforcement.intake.json", + "reports/world_class_evidence_intake.json", + "reports/world_class_evidence_intake.md" + ], + "privacy_contract": [ + "Do not mark native_enforcement true for metadata-only fallbacks.", + "Keep residual risks visible for targets that still rely on operator enforcement." + ], + "audit_next_action": "Integrate a real client or installer runtime guard before claiming native permission enforcement." + }, + { + "key": "native-client-telemetry", + "label": "Native Client Telemetry", + "status": "external_required", + "category": "external", + "owner": "Browser/Chrome/IDE/provider client integrator", + "current": "external source events 0; adoption samples 0", + "objective": "Import production metadata-only events from a real external client into the local drift loop.", + "runbook": [ + "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", + "Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event.", + "python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl", + "python3 scripts/yao.py skill-atlas --workspace-root .", + "python3 scripts/yao.py skill-os2-audit . --generated-at ", + "Copy evidence/world_class/templates/native-client-telemetry.intake.json to evidence/world_class/submissions/native-client-telemetry.json and fill only real evidence fields.", + "python3 scripts/yao.py world-class-intake ." + ], + "success_checks": [ + "reports/adoption_drift_report.json summary.source_types.external > 0", + "reports/adoption_drift_report.json summary.adoption_sample_count > 0", + "reports/skill_os2_audit.json item native-client-telemetry status becomes pass" + ], + "evidence_artifacts": [ + "reports/adoption_drift_report.json", + "reports/adoption_drift_report.md", + "reports/telemetry_hook_recipes.json", + "scripts/telemetry_native_host.py", + "evidence/world_class/intake.schema.json", + "evidence/world_class/templates/native-client-telemetry.intake.json", + "reports/world_class_evidence_intake.json", + "reports/world_class_evidence_intake.md" + ], + "privacy_contract": [ + "Telemetry must remain metadata-only and local-first.", + "Do not package reports/telemetry_events.jsonl or any raw prompt, output, transcript, note, or message field." + ], + "audit_next_action": "Install a real client against the native host and import production metadata-only events." + } + ], + "source_audit": { + "json": "reports/skill_os2_audit.json", + "markdown": "reports/skill_os2_audit.md", + "open_gap_count": 4 + } + }, + "world_class_evidence_ledger": { + "ok": true, + "summary": { + "ledger_entry_count": 4, + "accepted_count": 0, + "pending_count": 4, + "human_pending_count": 1, + "external_pending_count": 3, + "overclaim_guard_active": true, + "ready_to_claim_world_class": false, + "decision": "evidence-pending" + }, + "entries": [ + { + "key": "provider-holdout", + "label": "Provider Holdout", + "category": "external", + "owner": "operator with provider credentials", + "status": "pending", + "source_status": "external_required", + "current": "model-executed 0; token-observed 0", + "objective": "Collect at least one provider-backed output-eval holdout run with model, timing, and token metadata.", + "provenance_requirements": [ + "provider-backed model run", + "observed timing", + "observed token metadata" + ], + "success_checks": [ + "reports/output_execution_runs.json summary.model_executed_count > 0", + "reports/output_execution_runs.json summary.timing_observed_count > 0", + "reports/output_execution_runs.json summary.token_observed_count > 0", + "reports/skill_os2_audit.json item provider-holdout status becomes pass" + ], + "evidence_artifacts": [ + "reports/output_execution_runs.json", + "reports/output_execution_runs.md", + "reports/skill_os2_audit.json", + "evidence/world_class/intake.schema.json", + "evidence/world_class/templates/provider-holdout.intake.json", + "reports/world_class_evidence_intake.json", + "reports/world_class_evidence_intake.md" + ], + "privacy_contract": [ + "Do not commit provider credentials or environment dumps.", + "The output execution report records output hashes and aggregate run metadata, not raw provider prompts." + ], + "observed_state": { + "model_executed_count": 0, + "timing_observed_count": 10, + "token_observed_count": 0, + "accepted": false + }, + "anti_overclaim": { + "planned_work_counts_as_evidence": false, + "metadata_fallback_counts_as_native_enforcement": false, + "pending_review_counts_as_human_decision": false, + "local_command_runner_counts_as_provider_model": false + }, + "next_action": "Run provider-backed holdout cases with real credentials and commit only aggregate evidence." + }, + { + "key": "human-adjudication", + "label": "Human Adjudication", + "category": "human", + "owner": "human reviewer", + "status": "pending", + "source_status": "human_required", + "current": "0/5 decisions; pending 5", + "objective": "Record real blind A/B reviewer decisions before claiming human output review completion.", + "provenance_requirements": [ + "real reviewer identity", + "blind A/B decisions", + "answer key unopened until decisions exist" + ], + "success_checks": [ + "reports/output_review_adjudication.json summary.pending_count == 0", + "reports/output_review_adjudication.json summary.judgment_count == summary.pair_count", + "reports/output_review_adjudication.json summary.invalid_decision_count == 0", + "reports/skill_os2_audit.json item human-adjudication status becomes pass" + ], + "evidence_artifacts": [ + "reports/output_blind_review_pack.md", + "reports/output_review_decisions.json", + "reports/output_review_adjudication.json", + "reports/output_review_adjudication.md", + "evidence/world_class/intake.schema.json", + "evidence/world_class/templates/human-adjudication.intake.json", + "reports/world_class_evidence_intake.json", + "reports/world_class_evidence_intake.md" + ], + "privacy_contract": [ + "Reviewer decisions should not include raw user data or private customer detail.", + "Keep the answer key separate until after decisions are recorded." + ], + "observed_state": { + "pair_count": 5, + "judgment_count": 0, + "pending_count": 5, + "invalid_decision_count": 0, + "answer_revealed_count": 0, + "accepted": false + }, + "anti_overclaim": { + "planned_work_counts_as_evidence": false, + "metadata_fallback_counts_as_native_enforcement": false, + "pending_review_counts_as_human_decision": false, + "local_command_runner_counts_as_provider_model": false + }, + "next_action": "Record real A/B choices in the decision template, then regenerate adjudication." + }, + { + "key": "native-permission-enforcement", + "label": "Native Permission Enforcement", + "category": "external", + "owner": "target client or installer integrator", + "status": "pending", + "source_status": "external_required", + "current": "native-enforced targets 0", + "objective": "Prove at least one target or installer enforces approved high-permission capabilities at runtime.", + "provenance_requirements": [ + "real target or installer guard", + "native enforcement flag", + "residual risk retained for fallback targets" + ], + "success_checks": [ + "reports/runtime_permission_probes.json summary.native_enforcement_count > 0", + "reports/runtime_permission_probes.json summary.failure_count == 0", + "reports/skill_os2_audit.json item native-permission-enforcement status becomes pass" + ], + "evidence_artifacts": [ + "dist/targets/*/adapter.json", + "reports/runtime_permission_probes.json", + "reports/runtime_permission_probes.md", + "security/permission_policy.json", + "evidence/world_class/intake.schema.json", + "evidence/world_class/templates/native-permission-enforcement.intake.json", + "reports/world_class_evidence_intake.json", + "reports/world_class_evidence_intake.md" + ], + "privacy_contract": [ + "Do not mark native_enforcement true for metadata-only fallbacks.", + "Keep residual risks visible for targets that still rely on operator enforcement." + ], + "observed_state": { + "native_enforcement_count": 0, + "metadata_fallback_count": 4, + "residual_risk_count": 4, + "failure_count": 0, + "accepted": false + }, + "anti_overclaim": { + "planned_work_counts_as_evidence": false, + "metadata_fallback_counts_as_native_enforcement": false, + "pending_review_counts_as_human_decision": false, + "local_command_runner_counts_as_provider_model": false + }, + "next_action": "Integrate a real client or installer runtime guard before claiming native permission enforcement." + }, + { + "key": "native-client-telemetry", + "label": "Native Client Telemetry", + "category": "external", + "owner": "Browser/Chrome/IDE/provider client integrator", + "status": "pending", + "source_status": "external_required", + "current": "external source events 0; adoption samples 0", + "objective": "Import production metadata-only events from a real external client into the local drift loop.", + "provenance_requirements": [ + "real external client source", + "metadata-only event", + "local-first import path" + ], + "success_checks": [ + "reports/adoption_drift_report.json summary.source_types.external > 0", + "reports/adoption_drift_report.json summary.adoption_sample_count > 0", + "reports/skill_os2_audit.json item native-client-telemetry status becomes pass" + ], + "evidence_artifacts": [ + "reports/adoption_drift_report.json", + "reports/adoption_drift_report.md", + "reports/telemetry_hook_recipes.json", + "scripts/telemetry_native_host.py", + "evidence/world_class/intake.schema.json", + "evidence/world_class/templates/native-client-telemetry.intake.json", + "reports/world_class_evidence_intake.json", + "reports/world_class_evidence_intake.md" + ], + "privacy_contract": [ + "Telemetry must remain metadata-only and local-first.", + "Do not package reports/telemetry_events.jsonl or any raw prompt, output, transcript, note, or message field." + ], + "observed_state": { + "external_source_events": 0, + "adoption_sample_count": 0, + "raw_content_allowed": false, + "risk_band": "low", + "accepted": false + }, + "anti_overclaim": { + "planned_work_counts_as_evidence": false, + "metadata_fallback_counts_as_native_enforcement": false, + "pending_review_counts_as_human_decision": false, + "local_command_runner_counts_as_provider_model": false + }, + "next_action": "Install a real client against the native host and import production metadata-only events." + } + ], + "source_plan": { + "json": "reports/world_class_evidence_plan.json", + "markdown": "reports/world_class_evidence_plan.md", + "task_count": 4 + } + }, "synthesis_highlights": [ "Borrow progressive disclosure: keep the entrypoint lean and move depth into references or scripts.", "Borrow a review checkpoint wherever trust matters more than raw speed.", diff --git a/reports/review_waivers.json b/reports/review_waivers.json index c3a9d45..6588ad1 100644 --- a/reports/review_waivers.json +++ b/reports/review_waivers.json @@ -2,7 +2,7 @@ "schema_version": "1.0", "ok": true, "skill_dir": ".", - "generated_at": "2026-06-13", + "generated_at": "2026-06-14", "summary": { "waiver_count": 0, "active_count": 0, diff --git a/reports/security_trust_report.json b/reports/security_trust_report.json index aa46206..cbe99cd 100644 --- a/reports/security_trust_report.json +++ b/reports/security_trust_report.json @@ -23,7 +23,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 166, - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" }, "failures": [], "warnings": [], diff --git a/reports/security_trust_report.md b/reports/security_trust_report.md index acd13fe..a2c3152 100644 --- a/reports/security_trust_report.md +++ b/reports/security_trust_report.md @@ -16,7 +16,7 @@ - Interactive scripts: `0` - Package hash scope: `source-contract-without-generated-reports` - Package hash files: `166` -- Package SHA256: `5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d` +- Package SHA256: `e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723` ## Failures diff --git a/reports/skill-overview.json b/reports/skill-overview.json index c4701bb..286ad47 100644 --- a/reports/skill-overview.json +++ b/reports/skill-overview.json @@ -905,9 +905,9 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "working_tree_dirty": true, - "changed_file_count": 44 + "changed_file_count": 35 }, - "commit": "ee4879f296d6e6b5dde76afc37b81504806afc82", + "commit": "4737358f3c7acb6650d5463793e5f0e8f913aada", "missing_artifacts": [], "limitations": [ "Local command-runner evidence is reproducible but does not replace provider-backed model holdout evidence.", @@ -1003,7 +1003,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 166, - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" }, "skill_atlas": { "skill_count": 12, @@ -1041,8 +1041,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d", - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05" + "package_sha256": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, "compatibility": { "openai": "pass", @@ -1073,7 +1073,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -1089,7 +1089,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05", + "archive_sha256": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2", "archive_entry_count": 547, "failure_count": 0, "warning_count": 0 @@ -1168,12 +1168,12 @@ { "field": "archive_sha256", "from": "", - "to": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05" + "to": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "to": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" } ] }, diff --git a/reports/skill_atlas.json b/reports/skill_atlas.json index c316735..33bcd0f 100644 --- a/reports/skill_atlas.json +++ b/reports/skill_atlas.json @@ -44,7 +44,7 @@ }, "catalog": { "workspace_root": ".", - "generated_at": "2026-06-14", + "generated_at": "2026-06-13", "skills": [ { "name": "yao-meta-skill", @@ -209,8 +209,8 @@ "report": "reports/adoption_drift_report.json", "risk_band": "low", "event_count": 1, - "adoption_sample_count": 0, - "adoption_rate": 0, + "adoption_sample_count": 1, + "adoption_rate": 100.0, "candidate_count": 0 } }, @@ -873,7 +873,7 @@ "name": "incident-command-governor", "path": "examples/governed-incident-command/generated-skill", "reason": "review overdue by cadence monthly", - "age_days": 75, + "age_days": 74, "allowed_days": 31, "actionable": false, "scope": "example" diff --git a/reports/skill_os2_coverage.json b/reports/skill_os2_coverage.json index d13eb9d..d7ca42a 100644 --- a/reports/skill_os2_coverage.json +++ b/reports/skill_os2_coverage.json @@ -237,7 +237,7 @@ "label": "Review Studio", "status": "pass", "objective": "One HTML page supports first-pass production review across trigger, output, runtime, trust, release, and evidence actions.", - "current": "14 gates; decision review; warnings 3", + "current": "15 gates; decision review; warnings 3", "command": "python3 scripts/yao.py review-studio .", "test": "python3 tests/verify_review_studio.py", "evidence": [ @@ -555,7 +555,7 @@ "label": "Review Studio 2.0", "status": "pass", "objective": "Recommended Skill OS 2.0 implementation PR from the upgrade plan.", - "current": "14 review gates", + "current": "15 review gates", "command": "make ci-test", "test": "tests/verify_review_studio.py", "evidence": [ diff --git a/reports/skill_os2_coverage.md b/reports/skill_os2_coverage.md index 092c9df..a322992 100644 --- a/reports/skill_os2_coverage.md +++ b/reports/skill_os2_coverage.md @@ -24,7 +24,7 @@ This report maps the Skill OS 2.0 upgrade blueprint to concrete local artifacts, | Trust Security | `pass` | 82 scripts; secrets 0; help failures 0 | `python3 scripts/yao.py trust .` | `python3 tests/verify_trust_check.py` | | Skill Atlas | `pass` | 12 scanned skills; actionable collisions 0 | `python3 scripts/yao.py skill-atlas --workspace-root .` | `python3 tests/verify_skill_atlas.py` | | Registry Distribution | `pass` | archive entries 547; install failures 0 | `python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip && python3 scripts/yao.py registry-audit .` | `python3 tests/verify_registry_audit.py` | -| Review Studio | `pass` | 14 gates; decision review; warnings 3 | `python3 scripts/yao.py review-studio .` | `python3 tests/verify_review_studio.py` | +| Review Studio | `pass` | 15 gates; decision review; warnings 3 | `python3 scripts/yao.py review-studio .` | `python3 tests/verify_review_studio.py` | | Telemetry Drift | `pass` | events 1; recipes 5; risk low | `python3 scripts/yao.py telemetry-hooks . && python3 scripts/yao.py adoption-drift .` | `python3 tests/verify_telemetry_hooks.py` | ## Recommended PR Coverage @@ -41,7 +41,7 @@ This report maps the Skill OS 2.0 upgrade blueprint to concrete local artifacts, | Trust Check | `pass` | secret findings 0 | `make ci-test` | `tests/verify_trust_check.py` | | Skill Atlas Generator | `pass` | 12 scanned skills | `make ci-test` | `tests/verify_skill_atlas.py` | | Registry Package Format | `pass` | registry ok True | `make ci-test` | `tests/verify_registry_audit.py` | -| Review Studio 2.0 | `pass` | 14 review gates | `make ci-test` | `tests/verify_review_studio.py` | +| Review Studio 2.0 | `pass` | 15 review gates | `make ci-test` | `tests/verify_review_studio.py` | | Migration V2 Docs | `pass` | migration guide present | `make ci-test` | `docs review` | ## Next Highest-Leverage Moves diff --git a/reports/upgrade_check.json b/reports/upgrade_check.json index a21f3fd..5353014 100644 --- a/reports/upgrade_check.json +++ b/reports/upgrade_check.json @@ -70,12 +70,12 @@ { "field": "archive_sha256", "from": "", - "to": "13b8b221acd80b5267c0ba63341560fa24ab99c0ee51cd55b2c763ffa458bc05" + "to": "d879a6c6c16f91854dae70c114fa6cea05aeb423d8d11918e2a3f87930ed1bf2" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "5466073395d5a54812f5b45061e745bc4e1ad67c446eeb5a41f6e206b916405d" + "to": "e377e8c99c14f87933224d1abebbe9d5d971fd852e1f3fe9b2a4a80861e90723" } ] }, diff --git a/scripts/render_review_studio.py b/scripts/render_review_studio.py index a6123b8..f906558 100644 --- a/scripts/render_review_studio.py +++ b/scripts/render_review_studio.py @@ -440,6 +440,17 @@ ACTION_GUIDANCE: dict[str, dict[str, str]] = { ], "verification": "python3 scripts/trust_check.py .", }, + "python-compat": { + "summary": "修复 Python 3.11 语法兼容问题,尤其是 f-string 表达式内的反斜杠转义。", + "why": "目标运行环境可能仍停留在 Python 3.11,语法漂移会让 CLI、报告生成和 CI 在发布后直接失败。", + "source_fix": "reports/python_compatibility.md + scripts/*.py + tests/*.py", + "source_paths": [ + {"path": "reports/python_compatibility.md", "label": "Python compatibility", "kind": "report", "patterns": ["# Python"]}, + {"path": "scripts/python_compat_check.py", "label": "compatibility checker", "kind": "source", "patterns": ["SCRIPT_INTERFACE"]}, + {"path": ".github/workflows/test.yml", "label": "CI test workflow", "kind": "ci", "patterns": ["python"]}, + ], + "verification": "python3 scripts/yao.py python-compat .", + }, "permission-gates": { "summary": "补齐高权限能力的 reviewer、scope、reason、expires_at 和目标端 enforcement 说明。", "why": "权限契约只有在批准人、有效期和目标端处置方式明确时,才能支撑 governed release。", diff --git a/scripts/review_studio_gates.py b/scripts/review_studio_gates.py index 22aaad7..23e9c0e 100644 --- a/scripts/review_studio_gates.py +++ b/scripts/review_studio_gates.py @@ -259,6 +259,38 @@ def build_gates(skill_dir: Path, output_html: Path, data: dict[str, dict[str, An ) ) + python_compat = data["python_compatibility"] + python_compat_summary = python_compat.get("summary", {}) + if not python_compat: + python_compat_status = "warn" + python_compat_detail = "Python compatibility report is missing" + else: + issue_count = int(python_compat_summary.get("issue_count", 0) or 0) + syntax_error_count = int(python_compat_summary.get("syntax_error_count", 0) or 0) + fstring_violation_count = int(python_compat_summary.get("fstring_311_violation_count", 0) or 0) + python_compat_status = ( + "block" + if not python_compat.get("ok", True) or issue_count or syntax_error_count or fstring_violation_count + else "pass" + ) + python_compat_detail = ( + f"Python {python_compat_summary.get('target_python', '3.11')}; " + f"{python_compat_summary.get('file_count', 0)} files; " + f"{issue_count} compatibility issues; " + f"{syntax_error_count} syntax; " + f"{fstring_violation_count} f-string 3.11 hazards" + ) + gates.append( + gate( + "python-compat", + "Python 兼容", + python_compat_status, + python_compat_detail, + "reports/python_compatibility.json", + _report_link(output_html, skill_dir, "reports/python_compatibility.md"), + ) + ) + permission_governance = trust.get("permission_governance", {}) if isinstance(trust.get("permission_governance", {}), dict) else {} if trust and not permission_governance: permission_governance = fallback_permission_governance(skill_dir) @@ -533,6 +565,7 @@ def weighted_score(gates: list[dict[str, str]]) -> int: "context-budget": 10, "runtime-matrix": 10, "trust-report": 10, + "python-compat": 10, "permission-gates": 10, "permission-runtime": 10, "skill-atlas": 10, diff --git a/skill_atlas/catalog.json b/skill_atlas/catalog.json index b4ff3c9..89055cf 100644 --- a/skill_atlas/catalog.json +++ b/skill_atlas/catalog.json @@ -1,6 +1,6 @@ { "workspace_root": ".", - "generated_at": "2026-06-14", + "generated_at": "2026-06-13", "skills": [ { "name": "yao-meta-skill", @@ -165,8 +165,8 @@ "report": "reports/adoption_drift_report.json", "risk_band": "low", "event_count": 1, - "adoption_sample_count": 0, - "adoption_rate": 0, + "adoption_sample_count": 1, + "adoption_rate": 100.0, "candidate_count": 0 } }, diff --git a/skill_atlas/stale_skills.json b/skill_atlas/stale_skills.json index a391983..6602e82 100644 --- a/skill_atlas/stale_skills.json +++ b/skill_atlas/stale_skills.json @@ -24,7 +24,7 @@ "name": "incident-command-governor", "path": "examples/governed-incident-command/generated-skill", "reason": "review overdue by cadence monthly", - "age_days": 75, + "age_days": 74, "allowed_days": 31, "actionable": false, "scope": "example" diff --git a/tests/verify_review_studio.py b/tests/verify_review_studio.py index 3aaa85e..7fa7d7d 100644 --- a/tests/verify_review_studio.py +++ b/tests/verify_review_studio.py @@ -262,7 +262,7 @@ def main() -> None: assert payload["ok"], payload assert payload["schema_version"] == "2.0", payload assert payload["summary"]["decision"] == "review", payload - assert payload["summary"]["gate_count"] == 14, payload + assert payload["summary"]["gate_count"] == 15, payload assert payload["summary"]["world_class_score"] == 90, payload assert payload["summary"]["warning_count"] == 3, payload assert payload["summary"]["blocker_count"] == 0, payload @@ -272,7 +272,7 @@ def main() -> None: assert payload["summary"]["action_count"] == payload["summary"]["warning_count"] + payload["summary"]["blocker_count"], payload assert {item["gate_key"] for item in payload["review_actions"]} == {"output-lab", "review-waivers", "world-class-evidence"}, payload gate_keys = {item["key"] for item in payload["gates"]} - assert {"intent-canvas", "trigger-lab", "output-lab", "runtime-matrix", "trust-report", "permission-gates", "permission-runtime", "skill-atlas", "operations-loop", "review-waivers", "world-class-evidence", "registry-audit", "release-notes"} <= gate_keys, payload + assert {"intent-canvas", "trigger-lab", "output-lab", "runtime-matrix", "trust-report", "python-compat", "permission-gates", "permission-runtime", "skill-atlas", "operations-loop", "review-waivers", "world-class-evidence", "registry-audit", "release-notes"} <= gate_keys, payload output_gate = next(item for item in payload["gates"] if item["key"] == "output-lab") assert output_gate["status"] == "warn", output_gate assert "5/5 cases" in output_gate["detail"], output_gate @@ -293,6 +293,12 @@ def main() -> None: assert trust_gate["status"] == "pass", trust_gate assert "3 network-capable scripts" in trust_gate["detail"], trust_gate assert "0 help smoke failures" in trust_gate["detail"], trust_gate + python_compat_gate = next(item for item in payload["gates"] if item["key"] == "python-compat") + assert python_compat_gate["status"] == "pass", python_compat_gate + assert "Python 3.11" in python_compat_gate["detail"], python_compat_gate + assert "0 compatibility issues" in python_compat_gate["detail"], python_compat_gate + assert "0 f-string 3.11 hazards" in python_compat_gate["detail"], python_compat_gate + assert python_compat_gate["evidence"] == "reports/python_compatibility.json", python_compat_gate permission_gate = next(item for item in payload["gates"] if item["key"] == "permission-gates") assert permission_gate["status"] == "pass", permission_gate assert "permissions approved" in permission_gate["detail"], permission_gate diff --git a/tests/verify_yao_cli.py b/tests/verify_yao_cli.py index 834747a..953281a 100644 --- a/tests/verify_yao_cli.py +++ b/tests/verify_yao_cli.py @@ -382,7 +382,7 @@ def main() -> None: review_studio_result = run("review-studio", str(created)) assert review_studio_result["ok"], review_studio_result assert review_studio_result["payload"]["artifacts"]["html"].endswith("reports/review-studio.html"), review_studio_result - assert review_studio_result["payload"]["summary"]["gate_count"] == 14, review_studio_result + assert review_studio_result["payload"]["summary"]["gate_count"] == 15, review_studio_result created_world_class_gate = next(item for item in review_studio_result["payload"]["gates"] if item["key"] == "world-class-evidence") assert created_world_class_gate["status"] == "pass", created_world_class_gate assert "optional" in created_world_class_gate["detail"], created_world_class_gate