diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json
index 33adf36..fb98c44 100644
--- a/reports/benchmark_reproducibility.json
+++ b/reports/benchmark_reproducibility.json
@@ -3,7 +3,7 @@
"ok": true,
"generated_at": "2026-06-15",
"skill_dir": ".",
- "commit": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e",
+ "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"git_status": {
"available": true,
"dirty": false,
@@ -17,9 +17,9 @@
"methodology_complete": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "evidence_bundle_sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc",
- "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"output_case_count": 5,
"failure_disclosure_count": 3,
"command_count": 22,
@@ -54,7 +54,7 @@
},
"release_lock": {
"ready": true,
- "commit": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e",
+ "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"status_scope": "generation-time status before this report is written",
"reason": "clean generation-time HEAD"
},
@@ -64,7 +64,7 @@
"existing_count": 24,
"missing_count": 0,
"missing_paths": [],
- "sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc"
+ "sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3"
},
"methodology": {
"path": "reports/benchmark_methodology.md",
@@ -138,7 +138,7 @@
"path": "reports/output_execution_runs.json",
"exists": true,
"bytes": 7967,
- "sha256": "cec7afbfb8daa00e5fcfb3c79dac10163e312e307431896cc0f62b0f47416659"
+ "sha256": "a4c8cacaab5d2f5178513e20cfbb58787052e82607b9ec4fe3eaeae58d7509c8"
},
{
"label": "blind_review",
@@ -165,50 +165,50 @@
"label": "runtime_conformance",
"path": "reports/conformance_matrix.json",
"exists": true,
- "bytes": 10313,
- "sha256": "8251329e663dda51472f29b7721e73d72ccbec9760d96fda022f6218a5a6e347"
+ "bytes": 10342,
+ "sha256": "97f9ba949c23a60b00e9ba2ff279ca03ba517845cc7c62aa8a42645d58006c7e"
},
{
"label": "trust_report",
"path": "reports/security_trust_report.json",
"exists": true,
- "bytes": 106773,
- "sha256": "6409321f1c0d2a7b9208595e05360de8f3dbb56602c8cee61226fb8c31d61f32"
+ "bytes": 107994,
+ "sha256": "d7dae65ba9cfdc52001b73242ac88c84cf302bc6091e0df8805fdb79ac2594ce"
},
{
"label": "python_compatibility",
"path": "reports/python_compatibility.json",
"exists": true,
- "bytes": 22510,
- "sha256": "73c6c2a81af9980cc1d8e2d1bb5dfb30bd8daccac48125fe71ec359dbde824dd"
+ "bytes": 22768,
+ "sha256": "cdd46eb42010f406fb2732063bde224866a4493ae4be1dc0c835417550a45d18"
},
{
"label": "registry_audit",
"path": "reports/registry_audit.json",
"exists": true,
"bytes": 3183,
- "sha256": "76a55d6dfc15afe8f461e37faa1a3661650a09eb3b2c84a8d7c3b04abae4623e"
+ "sha256": "9ca263ecce9ace6b3915b2485eb7807edaf375d1aeccc689bc8b2d91417d8d2e"
},
{
"label": "package_verification",
"path": "reports/package_verification.json",
"exists": true,
"bytes": 19338,
- "sha256": "2476ae8ec9c491a3f7f31ec3f0c6c78cceead8e80700dcb696838eec226c675b"
+ "sha256": "1abff9930aa56d337b47dbc421f293eaa9eb3620c187d1b45a26941594b392ef"
},
{
"label": "install_simulation",
"path": "reports/install_simulation.json",
"exists": true,
- "bytes": 8758,
- "sha256": "490e1f665580f5d9794b05bd0fcc57342dc66971776577037176f181a5c2c875"
+ "bytes": 8557,
+ "sha256": "958586fa59a71aeea10962c6047083876fbcd4b809f24fbeaec4ece8aa499dfc"
},
{
"label": "skill_os2_audit",
"path": "reports/skill_os2_audit.json",
"exists": true,
"bytes": 14310,
- "sha256": "a4cf40478f3ad9404a9cb3ccff9f152481703b8936c33841ae898fd589b78fa0"
+ "sha256": "55526d80bfa78b574f990442165a0159b0f177feca00523826657b00f7855d29"
},
{
"label": "world_class_evidence_plan",
diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md
index b4adb9f..87ea8de 100644
--- a/reports/benchmark_reproducibility.md
+++ b/reports/benchmark_reproducibility.md
@@ -1,9 +1,9 @@
# Benchmark Reproducibility
Generated at: `2026-06-15`
-Commit: `2df5c9acd093f45e27318bbe5ff984fae3a5fd8e`
+Commit: `0e6327ca9399c6995b192fdf710fa4c99699c2c8`
Working tree dirty at generation: `false`
-Evidence bundle SHA256: `4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc`
+Evidence bundle SHA256: `885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3`
## Summary
@@ -12,8 +12,8 @@ Evidence bundle SHA256: `4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f
- methodology complete: `true`
- required artifacts: `24`
- missing artifacts: `0`
-- source contract sha256: `4660a11db949`
-- archive sha256: `6852cf91a74d`
+- source contract sha256: `76edc3f9936d`
+- archive sha256: `798153ff8e15`
- output cases: `5`
- disclosed failure cases: `3`
- reproduction commands: `22`
@@ -50,7 +50,7 @@ This report proves local benchmark reproducibility only. It keeps external provi
- algorithm: `sha256(path,label,exists,artifact_sha256)`
- artifacts: `24` / `24`
-- sha256: `4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc`
+- sha256: `885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3`
## Methodology Sections
@@ -72,17 +72,17 @@ This report proves local benchmark reproducibility only. It keeps external provi
| output_cases | `evals/output/cases.jsonl` | present | `a6ae96857116` |
| output_schema | `evals/output/schema.json` | present | `8ee340c95064` |
| output_scorecard | `reports/output_quality_scorecard.json` | present | `0806258a8e08` |
-| output_execution | `reports/output_execution_runs.json` | present | `cec7afbfb8da` |
+| output_execution | `reports/output_execution_runs.json` | present | `a4c8cacaab5d` |
| blind_review | `reports/output_blind_review_pack.json` | present | `bbe2db8ec277` |
| review_adjudication | `reports/output_review_adjudication.json` | present | `240485a721af` |
| trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` |
-| runtime_conformance | `reports/conformance_matrix.json` | present | `8251329e663d` |
-| trust_report | `reports/security_trust_report.json` | present | `6409321f1c0d` |
-| python_compatibility | `reports/python_compatibility.json` | present | `73c6c2a81af9` |
-| registry_audit | `reports/registry_audit.json` | present | `76a55d6dfc15` |
-| package_verification | `reports/package_verification.json` | present | `2476ae8ec9c4` |
-| install_simulation | `reports/install_simulation.json` | present | `490e1f665580` |
-| skill_os2_audit | `reports/skill_os2_audit.json` | present | `a4cf40478f3a` |
+| runtime_conformance | `reports/conformance_matrix.json` | present | `97f9ba949c23` |
+| trust_report | `reports/security_trust_report.json` | present | `d7dae65ba9cf` |
+| python_compatibility | `reports/python_compatibility.json` | present | `cdd46eb42010` |
+| registry_audit | `reports/registry_audit.json` | present | `9ca263ecce9a` |
+| package_verification | `reports/package_verification.json` | present | `1abff9930aa5` |
+| install_simulation | `reports/install_simulation.json` | present | `958586fa59a7` |
+| skill_os2_audit | `reports/skill_os2_audit.json` | present | `55526d80bfa7` |
| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `933cdb002181` |
| world_class_evidence_ledger | `reports/world_class_evidence_ledger.json` | present | `5407409841eb` |
| world_class_evidence_intake | `reports/world_class_evidence_intake.json` | present | `b10e1ce0a5a1` |
diff --git a/reports/evidence_consistency.json b/reports/evidence_consistency.json
index a4241c8..eda8400 100644
--- a/reports/evidence_consistency.json
+++ b/reports/evidence_consistency.json
@@ -48,8 +48,8 @@
"key": "overview-benchmark-commit",
"label": "overview embeds the benchmark commit",
"status": "pass",
- "expected": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e",
- "actual": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e",
+ "expected": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
+ "actual": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"paths": [
"reports/benchmark_reproducibility.json",
"reports/skill-overview.json"
@@ -64,8 +64,8 @@
"release_lock_ready": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"world_class_ledger_pending_count": 4,
"world_class_source_check_count": 13,
"world_class_source_pass_count": 6,
@@ -77,8 +77,8 @@
"release_lock_ready": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"world_class_ledger_pending_count": 4,
"world_class_source_check_count": 13,
"world_class_source_pass_count": 6,
@@ -194,8 +194,8 @@
"key": "interpretation-benchmark-commit",
"label": "interpretation embeds the benchmark commit",
"status": "pass",
- "expected": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e",
- "actual": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e",
+ "expected": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
+ "actual": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"paths": [
"reports/benchmark_reproducibility.json",
"reports/skill-interpretation.json"
@@ -210,8 +210,8 @@
"release_lock_ready": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"world_class_ledger_pending_count": 4,
"world_class_source_check_count": 13,
"world_class_source_pass_count": 6,
@@ -223,8 +223,8 @@
"release_lock_ready": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"world_class_ledger_pending_count": 4,
"world_class_source_check_count": 13,
"world_class_source_pass_count": 6,
@@ -1312,7 +1312,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 107
+ "file_count": 109
},
{
"path": "evals",
@@ -1327,7 +1327,7 @@
"file_count": 215
}
],
- "file_count": 389,
+ "file_count": 391,
"folder_count": 4,
"distribution": [
{
@@ -1352,7 +1352,7 @@
},
{
"label": "scripts",
- "value": 107
+ "value": 109
},
{
"label": "evals",
@@ -1400,7 +1400,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 107
+ "file_count": 109
},
{
"path": "evals",
@@ -1415,7 +1415,7 @@
"file_count": 215
}
],
- "file_count": 389,
+ "file_count": 391,
"folder_count": 4,
"distribution": [
{
@@ -1440,7 +1440,7 @@
},
{
"label": "scripts",
- "value": 107
+ "value": 109
},
{
"label": "evals",
diff --git a/reports/review-studio.html b/reports/review-studio.html
index 10a710a..f298284 100644
--- a/reports/review-studio.html
+++ b/reports/review-studio.html
@@ -676,22 +676,22 @@
核心指标
- Skill IR 2.0.0 5 targets in platform-neutral contract
Compiler 5/5 target contracts compiled from Skill IR
Output Delta 100.0 5 cases; 1 file-backed
Exec Runs 10 command 10; model 0; recorded 0
Blind A/B 5 review pairs hide baseline vs with-skill labels
Review Kit 0/5 pending 5; answer key hidden
Review A/B 0/5 adjudication decisions; pending 5
Public Claim blocked 5 blockers; local reproducible true
Blueprint 20/20 2.0 coverage; extensions partial 1, planned 0; evidence pending 4
Runtime 5/5 target conformance pass rate
Perm Probe 4/4 0 native; 4 installer-enforced
Trust 0 106 scripts scanned; secrets found
Py Compat 0 171 files scanned for Python 3.11
Arch Debt 0 886 largest lines; 34 CLI handlers
Atlas 5 12 scanned skills; route collisions
Drift low 1 metadata events; 0 missed triggers
Waivers 0 0 gates covered; human risk decisions
Intake 4/4 0 valid submissions; 0 invalid
Claim Guard 0 76 public surfaces scanned
Notes 0/0 0 open blocker annotations
Registry 1.1.0 5 targets; MIT license
Archive pass 586 zip entries; package verification
Install pass 4 adapters; 12 permissions enforced; 0 permission failures
Upgrade minor declared minor; 0 breaking changes
+ Skill IR 2.0.0 5 targets in platform-neutral contract
Compiler 5/5 target contracts compiled from Skill IR
Output Delta 100.0 5 cases; 1 file-backed
Exec Runs 10 command 10; model 0; recorded 0
Blind A/B 5 review pairs hide baseline vs with-skill labels
Review Kit 0/5 pending 5; answer key hidden
Review A/B 0/5 adjudication decisions; pending 5
Public Claim blocked 4 blockers; local reproducible true
Blueprint 21/21 2.0 coverage; extensions partial 1, planned 0; evidence pending 4
Runtime 5/5 target conformance pass rate
Perm Probe 4/4 0 native; 4 installer-enforced
Trust 0 109 scripts scanned; secrets found
Py Compat 0 175 files scanned for Python 3.11
Arch Debt 0 897 largest lines; 63 CLI handlers; 18 in entrypoint
Atlas 5 12 scanned skills; route collisions
Drift low 1 metadata events; 0 missed triggers
Waivers 0 0 gates covered; human risk decisions
Intake 4/4 0 valid submissions; 0 invalid
Claim Guard 0 77 public surfaces scanned
Notes 0/0 0 open blocker annotations
Registry 1.1.0 5 targets; MIT license
Archive pass 609 zip entries; package verification
Install pass 4 adapters; 12 permissions enforced; 0 permission failures
Upgrade minor declared minor; 0 breaking changes
审查闸门
- 通过
意图画布 intent confidence 100/100; Intent is clear enough to package the first routeable version.
reports/intent-confidence.json 证据 通过
触发实验 13 trigger cases; 0 misroutes; 0 ambiguous
reports/route_scorecard.json 证据 关注
输出实验 5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5
reports/output_quality_scorecard.json 证据 通过
上下文 initial load 960/1000; deferred 416415/120000; top deferred scripts 367825; resource governance governed; quality density 135.4
reports/context_budget.json 证据 通过
运行矩阵 5 / 5 targets pass
reports/conformance_matrix.json 证据 通过
信任报告 0 secrets; 106 scripts; 3 network-capable scripts; 0 help smoke failures
reports/security_trust_report.json 证据 通过
Python 兼容 Python 3.11; 171 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards
reports/python_compatibility.json 证据 通过
架构维护 168 Python files; 0 hotspots; 0 blockers; largest 886 lines; 34 CLI handlers
reports/architecture_maintainability.json 证据 通过
权限批准 3/3 permissions approved; gaps 0; required file_write, network, subprocess
reports/security_trust_report.json + security/permission_policy.json 证据 通过
权限探针 4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4
reports/runtime_permission_probes.json 证据 通过
组合治理 12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues
reports/skill_atlas.json 证据 通过
运营回路 1 metadata events; adoption 100.0; missed 0; bad-output 0; risk low
reports/adoption_drift_report.json 证据 关注
人工批准 0 active waivers; 1 warning gates still need reviewer decision
reports/review_waivers.json 证据 关注
世界证据 4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 7/13 pass; 6 blocked; overclaim guard true
reports/world_class_evidence_ledger.json 证据 通过
注册审计 yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures
reports/registry_audit.json + reports/install_simulation.json 证据 通过
发布路线 0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended
reports/promotion_decisions.json + reports/upgrade_check.json + docs/migration-v2.md 证据
+ 通过
意图画布 intent confidence 100/100; Intent is clear enough to package the first routeable version.
reports/intent-confidence.json 证据 通过
触发实验 13 trigger cases; 0 misroutes; 0 ambiguous
reports/route_scorecard.json 证据 关注
输出实验 5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5
reports/output_quality_scorecard.json 证据 通过
上下文 initial load 960/1000; deferred 421932/120000; top deferred scripts 373342; resource governance governed; quality density 135.4
reports/context_budget.json 证据 通过
运行矩阵 5 / 5 targets pass
reports/conformance_matrix.json 证据 通过
信任报告 0 secrets; 109 scripts; 3 network-capable scripts; 0 help smoke failures
reports/security_trust_report.json 证据 通过
Python 兼容 Python 3.11; 175 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards
reports/python_compatibility.json 证据 通过
架构维护 172 Python files; 0 hotspots; 0 blockers; largest 897 lines; 63 CLI handlers; 18 in entrypoint
reports/architecture_maintainability.json 证据 通过
权限批准 3/3 permissions approved; gaps 0; required file_write, network, subprocess
reports/security_trust_report.json + security/permission_policy.json 证据 通过
权限探针 4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4
reports/runtime_permission_probes.json 证据 通过
组合治理 12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues
reports/skill_atlas.json 证据 通过
运营回路 1 metadata events; adoption 0; missed 0; bad-output 0; risk low
reports/adoption_drift_report.json 证据 关注
人工批准 0 active waivers; 1 warning gates still need reviewer decision
reports/review_waivers.json 证据 关注
世界证据 4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true
reports/world_class_evidence_ledger.json 证据 通过
注册审计 yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures
reports/registry_audit.json + reports/install_simulation.json 证据 通过
发布路线 0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended
reports/promotion_decisions.json + reports/upgrade_check.json + docs/migration-v2.md 证据
-
关注事项 输出实验 5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5 人工批准 0 active waivers; 1 warning gates still need reviewer decision 世界证据 4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 7/13 pass; 6 blocked; overclaim guard true
+
关注事项 输出实验 5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5 人工批准 0 active waivers; 1 warning gates still need reviewer decision 世界证据 4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true
修复动作
- 关注
输出实验 补足 output eval 覆盖、execution evidence、blind A/B 和 reviewer adjudication。
没有输出质量和人工盲评证据时,Skill 只能证明会触发,不能证明输出真的更好且经得起审查。 修复位置 evals/output/cases.jsonl + reports/output_quality_scorecard.md + reports/output_review_kit.html + reports/output_review_adjudication.md 验证命令 python3 scripts/adjudicate_output_review.py --write-template && python3 scripts/yao.py output-reviewreports/output_quality_scorecard.json 打开证据 关注
人工批准 对保留的 warning 写入 reviewer、理由、范围和到期时间,或修掉 warning。
warning 可以被接受,但必须可审计、会过期,并且不能掩盖 blocker。 修复位置 reports/review_waivers.md 验证命令 python3 scripts/render_review_waivers.py .reports/review_waivers.json 打开证据 关注
世界证据 补齐 provider、真人盲评、原生权限执行和真实客户端遥测证据,或明确本次发布不声明 world-class 完成。
世界级结论必须来自已接受的外部/人工证据;计划、metadata fallback、待评审和本地命令都不能替代完成证据。 修复位置 reports/world_class_operator_runbook.html + reports/world_class_evidence_ledger.md + reports/world_class_evidence_intake.md + reports/world_class_submission_review.md 验证命令 python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py review-studio .证据采集 以下条目仍需真实外部或人工证据;提交文件、校验命令和阻断检查必须同时闭环。
pending · external
Provider Holdout model-executed 0; token-observed 0
提交 evidence/world_class/submissions/provider-holdout.json模板 evidence/world_class/templates/provider-holdout.intake.json阻断 2 blocked / 1 pass 下一步 Run provider-backed holdout cases with real credentials and commit only aggregate evidence. 阻断检查 Provider model run model_executed_count: 0 / >0Run provider-backed output-exec with real credentials. Token usage observed token_observed_count: 0 / >0Provider execution should return non-estimated token usage. 操作命令 准备提交 python3 scripts/yao.py world-class-submission-kit . --evidence-key provider-holdout --output-dir evidence/world_class/submissions校验入口 python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions审查提交 python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions刷新台账 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions首要步骤 YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY=<redacted> python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60 python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields. pending · human
Human Adjudication 0/5 decisions; pending 5
提交 evidence/world_class/submissions/human-adjudication.json模板 evidence/world_class/templates/human-adjudication.intake.json阻断 2 blocked / 2 pass 下一步 Record real A/B choices in the decision template, then regenerate adjudication. 阻断检查 No pending decisions pending_count: 5 / ==0Record a reviewer choice for every pair. Judgments complete judgment_count: 0 / ==pair_countEvery pair needs one valid human judgment. 操作命令 准备提交 python3 scripts/yao.py world-class-submission-kit . --evidence-key human-adjudication --output-dir evidence/world_class/submissions校验入口 python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions审查提交 python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions刷新台账 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions首要步骤 python3 scripts/yao.py output-review-kit --write-template Open reports/output_review_kit.md and choose A or B for each pair without opening the answer key. python3 scripts/adjudicate_output_review.py --write-template pending · external
Native Permission Enforcement native-enforced targets 0; installer-enforced targets 4
提交 evidence/world_class/submissions/native-permission-enforcement.json模板 evidence/world_class/templates/native-permission-enforcement.intake.json阻断 1 blocked / 2 pass 下一步 Integrate a real target-client or external installer runtime guard before claiming native permission enforcement. 阻断检查 Native enforcement native_enforcement_count: 0 / >0Collect real target-client or external runtime guard proof. 操作命令 准备提交 python3 scripts/yao.py world-class-submission-kit . --evidence-key native-permission-enforcement --output-dir evidence/world_class/submissions校验入口 python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions审查提交 python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions刷新台账 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions首要步骤 Implement or connect a real target client or external installer runtime guard that blocks undeclared network, file_write, or subprocess capabilities. Update the generated target adapter only when the guard is actually enforced by that target. python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip pending · external
Native Client Telemetry external source events 0; adoption samples 1
提交 evidence/world_class/submissions/native-client-telemetry.json模板 evidence/world_class/templates/native-client-telemetry.intake.json阻断 1 blocked / 2 pass 下一步 Install a real client against the native host and import production metadata-only events. 阻断检查 External events external_source_events: 0 / >0Import at least one metadata-only event from a real client. 操作命令 准备提交 python3 scripts/yao.py world-class-submission-kit . --evidence-key native-client-telemetry --output-dir evidence/world_class/submissions校验入口 python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions审查提交 python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions刷新台账 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions首要步骤 python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension://<extension-id>/ Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event. python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl reports/world_class_evidence_ledger.json 打开证据
+ 关注
输出实验 补足 output eval 覆盖、execution evidence、blind A/B 和 reviewer adjudication。
没有输出质量和人工盲评证据时,Skill 只能证明会触发,不能证明输出真的更好且经得起审查。 修复位置 evals/output/cases.jsonl + reports/output_quality_scorecard.md + reports/output_review_kit.html + reports/output_review_adjudication.md 验证命令 python3 scripts/adjudicate_output_review.py --write-template && python3 scripts/yao.py output-reviewreports/output_quality_scorecard.json 打开证据 关注
人工批准 对保留的 warning 写入 reviewer、理由、范围和到期时间,或修掉 warning。
warning 可以被接受,但必须可审计、会过期,并且不能掩盖 blocker。 修复位置 reports/review_waivers.md 验证命令 python3 scripts/render_review_waivers.py .reports/review_waivers.json 打开证据 关注
世界证据 补齐 provider、真人盲评、原生权限执行和真实客户端遥测证据,或明确本次发布不声明 world-class 完成。
世界级结论必须来自已接受的外部/人工证据;计划、metadata fallback、待评审和本地命令都不能替代完成证据。 修复位置 reports/world_class_operator_runbook.html + reports/world_class_evidence_ledger.md + reports/world_class_evidence_intake.md + reports/world_class_submission_review.md 验证命令 python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py review-studio .证据采集 以下条目仍需真实外部或人工证据;提交文件、校验命令和阻断检查必须同时闭环。
pending · external
Provider Holdout model-executed 0; token-observed 0
提交 evidence/world_class/submissions/provider-holdout.json模板 evidence/world_class/templates/provider-holdout.intake.json阻断 2 blocked / 1 pass 下一步 Run provider-backed holdout cases with real credentials and commit only aggregate evidence. 阻断检查 Provider model run model_executed_count: 0 / >0Run provider-backed output-exec with real credentials. Token usage observed token_observed_count: 0 / >0Provider execution should return non-estimated token usage. 操作命令 准备提交 python3 scripts/yao.py world-class-submission-kit . --evidence-key provider-holdout --output-dir evidence/world_class/submissions校验入口 python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions审查提交 python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions刷新台账 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions首要步骤 YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY=<redacted> python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60 python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields. pending · human
Human Adjudication 0/5 decisions; pending 5
提交 evidence/world_class/submissions/human-adjudication.json模板 evidence/world_class/templates/human-adjudication.intake.json阻断 2 blocked / 2 pass 下一步 Record real A/B choices in the decision template, then regenerate adjudication. 阻断检查 No pending decisions pending_count: 5 / ==0Record a reviewer choice for every pair. Judgments complete judgment_count: 0 / ==pair_countEvery pair needs one valid human judgment. 操作命令 准备提交 python3 scripts/yao.py world-class-submission-kit . --evidence-key human-adjudication --output-dir evidence/world_class/submissions校验入口 python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions审查提交 python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions刷新台账 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions首要步骤 python3 scripts/yao.py output-review-kit --write-template Open reports/output_review_kit.md and choose A or B for each pair without opening the answer key. python3 scripts/adjudicate_output_review.py --write-template pending · external
Native Permission Enforcement native-enforced targets 0; installer-enforced targets 4
提交 evidence/world_class/submissions/native-permission-enforcement.json模板 evidence/world_class/templates/native-permission-enforcement.intake.json阻断 1 blocked / 2 pass 下一步 Integrate a real target-client or external installer runtime guard before claiming native permission enforcement. 阻断检查 Native enforcement native_enforcement_count: 0 / >0Collect real target-client or external runtime guard proof. 操作命令 准备提交 python3 scripts/yao.py world-class-submission-kit . --evidence-key native-permission-enforcement --output-dir evidence/world_class/submissions校验入口 python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions审查提交 python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions刷新台账 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions首要步骤 Implement or connect a real target client or external installer runtime guard that blocks undeclared network, file_write, or subprocess capabilities. Update the generated target adapter only when the guard is actually enforced by that target. python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip pending · external
Native Client Telemetry external source events 0; adoption samples 0
提交 evidence/world_class/submissions/native-client-telemetry.json模板 evidence/world_class/templates/native-client-telemetry.intake.json阻断 2 blocked / 1 pass 下一步 Install a real client against the native host and import production metadata-only events. 阻断检查 External events external_source_events: 0 / >0Import at least one metadata-only event from a real client. Adoption sample adoption_sample_count: 0 / >0Telemetry must include adoption outcome evidence. 操作命令 准备提交 python3 scripts/yao.py world-class-submission-kit . --evidence-key native-client-telemetry --output-dir evidence/world_class/submissions校验入口 python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions审查提交 python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions刷新台账 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions首要步骤 python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension://<extension-id>/ Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event. python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl reports/world_class_evidence_ledger.json 打开证据
- 上下文 initial load 960/1000; deferred 416415/120000; top deferred scripts 367825; resource governance governed; quality density 135.4
+ 上下文 initial load 960/1000; deferred 421932/120000; top deferred scripts 373342; resource governance governed; quality density 135.4
编译证据 Review reports/compiled_targets.md before packaging to inspect target adapter modes, generated files, preserved semantics, warnings, and unsupported features.
- 信任报告
Secret 0
脚本数 106
网络脚本 3
Help 失败 0
包体哈希 28625ce8b3fcb3e214ef3b8ffc748d9cff551e4c5a1b78786c1a2b7d7005f679
+ 信任报告
Secret 0
脚本数 109
网络脚本 3
Help 失败 0
包体哈希 76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4
安全边界 高风险 secret、远程 inline execution、缺失依赖策略或无法解释的脚本接口应阻断 governed release。
- Python 兼容
目标 Python 3.11
文件数 171
问题数 0
语法错误 0
F-string 3.11 0
+ Python 兼容
目标 Python 3.11
文件数 175
问题数 0
语法错误 0
F-string 3.11 0
解释器边界 CI 和发布审查以 Python 3.11 兼容为底线;本地更高版本允许的新语法不能绕过兼容门禁。
@@ -773,8 +773,8 @@
- 运营回路 1 metadata events; adoption 100.0; missed 0; bad-output 0; risk low
- 漂移信号
事件数 1
采用率 100
漏触发 0
Bad Output Count 0
风险带 low
+ 运营回路 1 metadata events; adoption 0; missed 0; bad-output 0; risk low
+ 漂移信号
事件数 1
采用率 0
漏触发 0
Bad Output Count 0
风险带 low
@@ -785,13 +785,13 @@
批准台账
Waiver Count 0
Active Count 0
Expired Count 0
Invalid Count 0
覆盖 Gate 0
批准候选
- 可批准 · needs-reviewer-decision
Output Lab review pending 5; model-executed 0; output failures 0
Gate output-lab证据 reports/output_review_adjudication.md选项 accepted-risk, false-positive, temporary-exception 边界 Does not count as provider, human, or public world-class completion evidence. 审查条件 Reviewer confirms this release does not claim provider-backed or human-adjudicated output superiority. Reviewer names the release scope and expiry date. Reviewer links output_review_adjudication or output_execution evidence. 建议命令 python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer "<reviewer>" --reason "Output Lab has pending human/provider evidence; accepted only for this bounded review scope." --expires-at 2027-06-13 --evidence reports/output_review_adjudication.md
不可批准 · cannot-waive
World-Class Evidence 4 pending evidence entries; 1 human pending; 3 external pending
Gate world-class-evidence证据 reports/world_class_evidence_ledger.md选项 无 边界 Non-waivable completion boundary. 审查条件 Do not use a waiver to claim public world-class readiness. Either submit accepted ledger evidence or state that this release does not claim world-class completion. Keep claim guard active until ledger summary.ready_to_claim_world_class is true. 建议命令 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-claim-guard .
+ 可批准 · needs-reviewer-decision
Output Lab review pending 5; model-executed 0; output failures 0
Gate output-lab证据 reports/output_review_adjudication.md选项 accepted-risk, false-positive, temporary-exception 边界 Does not count as provider, human, or public world-class completion evidence. 审查条件 Reviewer confirms this release does not claim provider-backed or human-adjudicated output superiority. Reviewer names the release scope and expiry date. Reviewer links output_review_adjudication or output_execution evidence. 建议命令 python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer "<reviewer>" --reason "Output Lab has pending human/provider evidence; accepted only for this bounded review scope." --expires-at 2027-06-15 --evidence reports/output_review_adjudication.md
不可批准 · cannot-waive
World-Class Evidence 4 pending evidence entries; 1 human pending; 3 external pending
Gate world-class-evidence证据 reports/world_class_evidence_ledger.md选项 无 边界 Non-waivable completion boundary. 审查条件 Do not use a waiver to claim public world-class readiness. Either submit accepted ledger evidence or state that this release does not claim world-class completion. Keep claim guard active until ledger summary.ready_to_claim_world_class is true. 建议命令 python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-claim-guard .
世界证据
这里列出每个 world-class 证据项的当前状态、完成定义、证据来源、隐私约束和下一步;计划、metadata fallback、待评审和本地命令不会被当成完成证据。
- 待补证 · external
Provider Holdout Collect at least one provider-backed output-eval holdout run with model, timing, and token metadata.
负责人 operator with provider credentials 当前状态 model-executed 0; token-observed 0 下一步 Run provider-backed holdout cases with real credentials and commit only aggregate evidence. 观测值 model_executed_count: 0; timing_observed_count: 10; token_observed_count: 0; accepted: False 提交态 status: missing; path: evidence/world_class/submissions/provider-holdout.json; attested_real_evidence: False; privacy_contract_satisfied: False 执行步骤 YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY=<redacted> python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60 python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields. python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions 完成定义 reports/output_execution_runs.json summary.model_executed_count > 0 reports/output_execution_runs.json summary.timing_observed_count > 0 reports/output_execution_runs.json summary.token_observed_count > 0 reports/skill_os2_audit.json item provider-holdout status becomes pass 证据来源 reports/output_execution_runs.json reports/output_execution_runs.md reports/skill_os2_audit.json evidence/world_class/intake.schema.json evidence/world_class/templates/provider-holdout.intake.json reports/world_class_evidence_intake.json reports/world_class_evidence_intake.md 隐私约束 Do not commit provider credentials or environment dumps. The output execution report records output hashes and aggregate run metadata, not raw provider prompts. 源证据检查 Provider model run model_executed_count: 0 / >0Run provider-backed output-exec with real credentials. Timing observed timing_observed_count: 10 / >0Provider execution should record timing metadata. Token usage observed token_observed_count: 0 / >0Provider execution should return non-estimated token usage. 待补证 · human
Human Adjudication Record real blind A/B reviewer decisions before claiming human output review completion.
负责人 human reviewer 当前状态 0/5 decisions; pending 5 下一步 Record real A/B choices in the decision template, then regenerate adjudication. 观测值 pair_count: 5; judgment_count: 0; pending_count: 5; invalid_decision_count: 0; answer_revealed_count: 0; accepted: False 提交态 status: missing; path: evidence/world_class/submissions/human-adjudication.json; attested_real_evidence: False; privacy_contract_satisfied: False 执行步骤 python3 scripts/yao.py output-review-kit --write-template Open reports/output_review_kit.md and choose A or B for each pair without opening the answer key. python3 scripts/adjudicate_output_review.py --write-template Edit reports/output_review_decisions.json with winner_variant values and reviewer metadata. python3 scripts/yao.py output-review python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/human-adjudication.intake.json to evidence/world_class/submissions/human-adjudication.json and fill only real evidence fields. python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions 完成定义 reports/output_review_adjudication.json summary.pending_count == 0 reports/output_review_adjudication.json summary.judgment_count == summary.pair_count reports/output_review_adjudication.json summary.invalid_decision_count == 0 reports/skill_os2_audit.json item human-adjudication status becomes pass 证据来源 reports/output_blind_review_pack.md reports/output_review_kit.md reports/output_review_decisions.json reports/output_review_adjudication.json reports/output_review_adjudication.md evidence/world_class/intake.schema.json evidence/world_class/templates/human-adjudication.intake.json reports/world_class_evidence_intake.json reports/world_class_evidence_intake.md 隐私约束 Reviewer decisions should not include raw user data or private customer detail. Keep the answer key separate until after decisions are recorded. 源证据检查 Review pairs exist pair_count: 5 / >0Generate the blind A/B review pack. No pending decisions pending_count: 5 / ==0Record a reviewer choice for every pair. Judgments complete judgment_count: 0 / ==pair_countEvery pair needs one valid human judgment. No invalid decisions invalid_decision_count: 0 / ==0Fix malformed winner/confidence entries. 待补证 · external
Native Permission Enforcement Prove at least one real target client or external installer runtime guard enforces approved high-permission capabilities.
负责人 target client or installer integrator 当前状态 native-enforced targets 0; installer-enforced targets 4 下一步 Integrate a real target-client or external installer runtime guard before claiming native permission enforcement. 观测值 native_enforcement_count: 0; metadata_fallback_count: 4; installer_enforcement_pass_count: 4; installer_permission_failure_count: 0; installer_enforcement_ready: True; residual_risk_count: 4; failure_count: 0; accepted: False 提交态 status: missing; path: evidence/world_class/submissions/native-permission-enforcement.json; attested_real_evidence: False; privacy_contract_satisfied: False 执行步骤 Implement or connect a real target client or external installer runtime guard that blocks undeclared network, file_write, or subprocess capabilities. Update the generated target adapter only when the guard is actually enforced by that target. python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip python3 scripts/yao.py install-simulate . --package-dir dist --install-root dist/install-simulation python3 scripts/yao.py runtime-permissions . --package-dir dist python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/native-permission-enforcement.intake.json to evidence/world_class/submissions/native-permission-enforcement.json and fill only real evidence fields. python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions 完成定义 reports/runtime_permission_probes.json summary.native_enforcement_count > 0 reports/runtime_permission_probes.json summary.failure_count == 0 reports/runtime_permission_probes.json summary.installer_enforcement_pass_count records local installer enforcement but does not replace native evidence reports/skill_os2_audit.json item native-permission-enforcement status becomes pass 证据来源 dist/targets/*/adapter.json reports/runtime_permission_probes.json reports/runtime_permission_probes.md reports/install_simulation.json reports/install_simulation.md security/permission_policy.json evidence/world_class/intake.schema.json evidence/world_class/templates/native-permission-enforcement.intake.json reports/world_class_evidence_intake.json reports/world_class_evidence_intake.md 隐私约束 Do not mark native_enforcement true for metadata-only fallbacks. Keep residual risks visible for targets that still rely on operator enforcement. 源证据检查 Native enforcement native_enforcement_count: 0 / >0Collect real target-client or external runtime guard proof. Probe failures failure_count: 0 / ==0Runtime permission probes must stay clean. Installer support installer_enforcement_ready: True / trueInstaller enforcement is supporting evidence, not native proof. 待补证 · external
Native Client Telemetry Import production metadata-only events from a real external client into the local drift loop.
负责人 Browser/Chrome/IDE/provider client integrator 当前状态 external source events 0; adoption samples 1 下一步 Install a real client against the native host and import production metadata-only events. 观测值 external_source_events: 0; adoption_sample_count: 1; raw_content_allowed: False; risk_band: low; accepted: False 提交态 status: missing; path: evidence/world_class/submissions/native-client-telemetry.json; attested_real_evidence: False; privacy_contract_satisfied: False 执行步骤 python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension://<extension-id>/ Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event. python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl python3 scripts/yao.py skill-atlas --workspace-root . python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/native-client-telemetry.intake.json to evidence/world_class/submissions/native-client-telemetry.json and fill only real evidence fields. python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions 完成定义 reports/adoption_drift_report.json summary.source_types.external > 0 reports/adoption_drift_report.json summary.adoption_sample_count > 0 reports/skill_os2_audit.json item native-client-telemetry status becomes pass 证据来源 reports/adoption_drift_report.json reports/adoption_drift_report.md reports/telemetry_hook_recipes.json scripts/telemetry_native_host.py evidence/world_class/intake.schema.json evidence/world_class/templates/native-client-telemetry.intake.json reports/world_class_evidence_intake.json reports/world_class_evidence_intake.md 隐私约束 Telemetry must remain metadata-only and local-first. Do not package reports/telemetry_events.jsonl or any raw prompt, output, transcript, note, or message field. 源证据检查 External events external_source_events: 0 / >0Import at least one metadata-only event from a real client. Adoption sample adoption_sample_count: 1 / >0Telemetry must include adoption outcome evidence. Raw content blocked raw_content_allowed: False / falseTelemetry must stay metadata-only.
+ 待补证 · external
Provider Holdout Collect at least one provider-backed output-eval holdout run with model, timing, and token metadata.
负责人 operator with provider credentials 当前状态 model-executed 0; token-observed 0 下一步 Run provider-backed holdout cases with real credentials and commit only aggregate evidence. 观测值 model_executed_count: 0; timing_observed_count: 10; token_observed_count: 0; accepted: False 提交态 status: missing; path: evidence/world_class/submissions/provider-holdout.json; attested_real_evidence: False; privacy_contract_satisfied: False 执行步骤 YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY=<redacted> python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60 python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields. python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions 完成定义 reports/output_execution_runs.json summary.model_executed_count > 0 reports/output_execution_runs.json summary.timing_observed_count > 0 reports/output_execution_runs.json summary.token_observed_count > 0 reports/skill_os2_audit.json item provider-holdout status becomes pass 证据来源 reports/output_execution_runs.json reports/output_execution_runs.md reports/skill_os2_audit.json evidence/world_class/intake.schema.json evidence/world_class/templates/provider-holdout.intake.json reports/world_class_evidence_intake.json reports/world_class_evidence_intake.md 隐私约束 Do not commit provider credentials or environment dumps. The output execution report records output hashes and aggregate run metadata, not raw provider prompts. 源证据检查 Provider model run model_executed_count: 0 / >0Run provider-backed output-exec with real credentials. Timing observed timing_observed_count: 10 / >0Provider execution should record timing metadata. Token usage observed token_observed_count: 0 / >0Provider execution should return non-estimated token usage. 待补证 · human
Human Adjudication Record real blind A/B reviewer decisions before claiming human output review completion.
负责人 human reviewer 当前状态 0/5 decisions; pending 5 下一步 Record real A/B choices in the decision template, then regenerate adjudication. 观测值 pair_count: 5; judgment_count: 0; pending_count: 5; invalid_decision_count: 0; answer_revealed_count: 0; accepted: False 提交态 status: missing; path: evidence/world_class/submissions/human-adjudication.json; attested_real_evidence: False; privacy_contract_satisfied: False 执行步骤 python3 scripts/yao.py output-review-kit --write-template Open reports/output_review_kit.md and choose A or B for each pair without opening the answer key. python3 scripts/adjudicate_output_review.py --write-template Edit reports/output_review_decisions.json with winner_variant values and reviewer metadata. python3 scripts/yao.py output-review python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/human-adjudication.intake.json to evidence/world_class/submissions/human-adjudication.json and fill only real evidence fields. python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions 完成定义 reports/output_review_adjudication.json summary.pending_count == 0 reports/output_review_adjudication.json summary.judgment_count == summary.pair_count reports/output_review_adjudication.json summary.invalid_decision_count == 0 reports/skill_os2_audit.json item human-adjudication status becomes pass 证据来源 reports/output_blind_review_pack.md reports/output_review_kit.md reports/output_review_decisions.json reports/output_review_adjudication.json reports/output_review_adjudication.md evidence/world_class/intake.schema.json evidence/world_class/templates/human-adjudication.intake.json reports/world_class_evidence_intake.json reports/world_class_evidence_intake.md 隐私约束 Reviewer decisions should not include raw user data or private customer detail. Keep the answer key separate until after decisions are recorded. 源证据检查 Review pairs exist pair_count: 5 / >0Generate the blind A/B review pack. No pending decisions pending_count: 5 / ==0Record a reviewer choice for every pair. Judgments complete judgment_count: 0 / ==pair_countEvery pair needs one valid human judgment. No invalid decisions invalid_decision_count: 0 / ==0Fix malformed winner/confidence entries. 待补证 · external
Native Permission Enforcement Prove at least one real target client or external installer runtime guard enforces approved high-permission capabilities.
负责人 target client or installer integrator 当前状态 native-enforced targets 0; installer-enforced targets 4 下一步 Integrate a real target-client or external installer runtime guard before claiming native permission enforcement. 观测值 native_enforcement_count: 0; metadata_fallback_count: 4; installer_enforcement_pass_count: 4; installer_permission_failure_count: 0; installer_enforcement_ready: True; residual_risk_count: 4; failure_count: 0; accepted: False 提交态 status: missing; path: evidence/world_class/submissions/native-permission-enforcement.json; attested_real_evidence: False; privacy_contract_satisfied: False 执行步骤 Implement or connect a real target client or external installer runtime guard that blocks undeclared network, file_write, or subprocess capabilities. Update the generated target adapter only when the guard is actually enforced by that target. python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip python3 scripts/yao.py install-simulate . --package-dir dist --install-root dist/install-simulation python3 scripts/yao.py runtime-permissions . --package-dir dist python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/native-permission-enforcement.intake.json to evidence/world_class/submissions/native-permission-enforcement.json and fill only real evidence fields. python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions 完成定义 reports/runtime_permission_probes.json summary.native_enforcement_count > 0 reports/runtime_permission_probes.json summary.failure_count == 0 reports/runtime_permission_probes.json summary.installer_enforcement_pass_count records local installer enforcement but does not replace native evidence reports/skill_os2_audit.json item native-permission-enforcement status becomes pass 证据来源 dist/targets/*/adapter.json reports/runtime_permission_probes.json reports/runtime_permission_probes.md reports/install_simulation.json reports/install_simulation.md security/permission_policy.json evidence/world_class/intake.schema.json evidence/world_class/templates/native-permission-enforcement.intake.json reports/world_class_evidence_intake.json reports/world_class_evidence_intake.md 隐私约束 Do not mark native_enforcement true for metadata-only fallbacks. Keep residual risks visible for targets that still rely on operator enforcement. 源证据检查 Native enforcement native_enforcement_count: 0 / >0Collect real target-client or external runtime guard proof. Probe failures failure_count: 0 / ==0Runtime permission probes must stay clean. Installer support installer_enforcement_ready: True / trueInstaller enforcement is supporting evidence, not native proof. 待补证 · external
Native Client Telemetry Import production metadata-only events from a real external client into the local drift loop.
负责人 Browser/Chrome/IDE/provider client integrator 当前状态 external source events 0; adoption samples 0 下一步 Install a real client against the native host and import production metadata-only events. 观测值 external_source_events: 0; adoption_sample_count: 0; raw_content_allowed: False; risk_band: low; accepted: False 提交态 status: missing; path: evidence/world_class/submissions/native-client-telemetry.json; attested_real_evidence: False; privacy_contract_satisfied: False 执行步骤 python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension://<extension-id>/ Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event. python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl python3 scripts/yao.py skill-atlas --workspace-root . python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD> Copy evidence/world_class/templates/native-client-telemetry.intake.json to evidence/world_class/submissions/native-client-telemetry.json and fill only real evidence fields. python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions 完成定义 reports/adoption_drift_report.json summary.source_types.external > 0 reports/adoption_drift_report.json summary.adoption_sample_count > 0 reports/skill_os2_audit.json item native-client-telemetry status becomes pass 证据来源 reports/adoption_drift_report.json reports/adoption_drift_report.md reports/telemetry_hook_recipes.json scripts/telemetry_native_host.py evidence/world_class/intake.schema.json evidence/world_class/templates/native-client-telemetry.intake.json reports/world_class_evidence_intake.json reports/world_class_evidence_intake.md 隐私约束 Telemetry must remain metadata-only and local-first. Do not package reports/telemetry_events.jsonl or any raw prompt, output, transcript, note, or message field. 源证据检查 External events external_source_events: 0 / >0Import at least one metadata-only event from a real client. Adoption sample adoption_sample_count: 0 / >0Telemetry must include adoption outcome evidence. Raw content blocked raw_content_allowed: False / falseTelemetry must stay metadata-only.
- 蓝图覆盖
项目数 20
模块数 8
建议 PR 12
通过数 20
Warn Count 0
缺失数 0
Extension Track Count 2
Extension Covered Count 1
Extension Partial Count 1
Extension Planned Count 0
Adaptive Extension Ready 否
本地蓝图 是
世界级 否
待补证据 4
+ 蓝图覆盖
项目数 21
模块数 8
建议 PR 13
通过数 21
Warn Count 0
缺失数 0
Extension Track Count 2
Extension Covered Count 1
Extension Partial Count 1
Extension Planned Count 0
Adaptive Extension Ready 否
本地蓝图 是
世界级 否
待补证据 4
覆盖边界 蓝图覆盖只证明 2.0 模块、建议 PR、脚本、报告和测试在本地闭环;public world-class 仍以 world-class evidence ledger 的真人和外部证据为准。
- 公开声明
本地复现 是
发布锁 否
可公开声明 否
声明阻断 5
Provider 证据 否
人审完成 否
世界级就绪 否
- 声明阻断 阻断 release lock is not clean or commit is unavailable 阻断 provider-backed model holdout evidence is incomplete 阻断 human blind-review adjudication is incomplete 阻断 world-class evidence is not accepted yet (4 open gaps, 4 ledger pending) 阻断 world-class source checks are not all accepted (6/13 pass, 7 blocked)
+ 公开声明
本地复现 是
发布锁 是
可公开声明 否
声明阻断 4
Provider 证据 否
人审完成 否
世界级就绪 否
+ 声明阻断 阻断 provider-backed model holdout evidence is incomplete 阻断 human blind-review adjudication is incomplete 阻断 world-class evidence is not accepted yet (4 open gaps, 4 ledger pending) 阻断 world-class source checks are not all accepted (6/13 pass, 7 blocked)
- 世界证据 4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 7/13 pass; 6 blocked; overclaim guard true
+ 世界证据 4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true
证据台账
Ledger Entry Count 4
Accepted Count 0
待审 4
Human Pending Count 1
External Pending Count 3
Overclaim Guard Active 是
Ready To Claim World Class 否
@@ -821,18 +821,18 @@
- 声明守卫
台账可声明 否
台账待补 4
声明面 76
违规数 0
Overclaim Guard Active 是
+ 声明守卫
台账可声明 否
台账待补 4
声明面 77
违规数 0
Overclaim Guard Active 是
声明边界 claim guard 扫描 README、docs 和 reports 中的完成态表述;ledger 未 ready 时,任何英文完成断言、true 状态声明或中文完成态都会阻断发布审查。
注册审计 yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures
- 包体元数据
名称 yao-meta-skill
版本 1.1.0
Maturity governed
Owner Yao Team
License MIT
信任级别 local
目标平台 openai, claude, generic, agent-skills-compatible, vscode
兼容通过 6/6
归档哈希 6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc
+ 包体元数据
名称 yao-meta-skill
版本 1.1.0
Maturity governed
Owner Yao Team
License MIT
信任级别 local
目标平台 openai, claude, generic, agent-skills-compatible, vscode
兼容通过 6/6
归档哈希 798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8
发布路线 0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended
- 包体验证
目标数 4
Adapter 4
归档存在 是
Zip 条目 586
失败数 0
警告数 0
归档哈希 6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc
+ 包体验证
目标数 4
Adapter 4
归档存在 是
Zip 条目 609
失败数 0
警告数 0
归档哈希 798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8
diff --git a/reports/review-studio.json b/reports/review-studio.json
index 9acc523..02faaa6 100644
--- a/reports/review-studio.json
+++ b/reports/review-studio.json
@@ -43,7 +43,7 @@
"key": "context-budget",
"label": "上下文",
"status": "pass",
- "detail": "initial load 960/1000; deferred 416415/120000; top deferred scripts 367825; resource governance governed; quality density 135.4",
+ "detail": "initial load 960/1000; deferred 421932/120000; top deferred scripts 373342; resource governance governed; quality density 135.4",
"evidence": "reports/context_budget.json",
"link": "context_budget.md"
},
@@ -59,7 +59,7 @@
"key": "trust-report",
"label": "信任报告",
"status": "pass",
- "detail": "0 secrets; 106 scripts; 3 network-capable scripts; 0 help smoke failures",
+ "detail": "0 secrets; 109 scripts; 3 network-capable scripts; 0 help smoke failures",
"evidence": "reports/security_trust_report.json",
"link": "security_trust_report.md"
},
@@ -67,7 +67,7 @@
"key": "python-compat",
"label": "Python 兼容",
"status": "pass",
- "detail": "Python 3.11; 171 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards",
+ "detail": "Python 3.11; 175 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards",
"evidence": "reports/python_compatibility.json",
"link": "python_compatibility.md"
},
@@ -75,7 +75,7 @@
"key": "architecture-maintainability",
"label": "架构维护",
"status": "pass",
- "detail": "168 Python files; 0 hotspots; 0 blockers; largest 886 lines; 34 CLI handlers",
+ "detail": "172 Python files; 0 hotspots; 0 blockers; largest 897 lines; 63 CLI handlers; 18 in entrypoint",
"evidence": "reports/architecture_maintainability.json",
"link": "architecture_maintainability.md"
},
@@ -107,7 +107,7 @@
"key": "operations-loop",
"label": "运营回路",
"status": "pass",
- "detail": "1 metadata events; adoption 100.0; missed 0; bad-output 0; risk low",
+ "detail": "1 metadata events; adoption 0; missed 0; bad-output 0; risk low",
"evidence": "reports/adoption_drift_report.json",
"link": "adoption_drift_report.md"
},
@@ -123,7 +123,7 @@
"key": "world-class-evidence",
"label": "世界证据",
"status": "warn",
- "detail": "4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 7/13 pass; 6 blocked; overclaim guard true",
+ "detail": "4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true",
"evidence": "reports/world_class_evidence_ledger.json",
"link": "world_class_evidence_ledger.md"
},
@@ -166,7 +166,7 @@
"key": "world-class-evidence",
"label": "世界证据",
"status": "warn",
- "detail": "4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 7/13 pass; 6 blocked; overclaim guard true",
+ "detail": "4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true",
"evidence": "reports/world_class_evidence_ledger.json",
"link": "world_class_evidence_ledger.md"
}
@@ -579,12 +579,12 @@
"category": "external",
"status": "pending",
"readiness": "awaiting-submission",
- "current": "external source events 0; adoption samples 1",
+ "current": "external source events 0; adoption samples 0",
"next_action": "Install a real client against the native host and import production metadata-only events.",
"submission_path": "evidence/world_class/submissions/native-client-telemetry.json",
"template_path": "evidence/world_class/templates/native-client-telemetry.intake.json",
- "source_pass_count": 2,
- "source_blocked_count": 1,
+ "source_pass_count": 1,
+ "source_blocked_count": 2,
"blocked_checks": [
{
"label": "External events",
@@ -593,6 +593,14 @@
"expected": ">0",
"status": "blocked",
"next_action": "Import at least one metadata-only event from a real client."
+ },
+ {
+ "label": "Adoption sample",
+ "field": "adoption_sample_count",
+ "actual": 0,
+ "expected": ">0",
+ "status": "blocked",
+ "next_action": "Telemetry must include adoption outcome evidence."
}
],
"commands": [
@@ -704,6 +712,7 @@
"reports/world_class_evidence_ledger.md",
"reports/review_annotations.md",
"reports/review-studio.html",
+ "reports/skill-interpretation.html",
"reports/skill-overview.html"
],
"flow": [
@@ -1181,7 +1190,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 106
+ "file_count": 109
},
{
"path": "evals",
@@ -1193,10 +1202,10 @@
"path": "reports",
"label": "Generated evidence and overview artifacts",
"kind": "folder",
- "file_count": 211
+ "file_count": 215
}
],
- "file_count": 384,
+ "file_count": 391,
"folder_count": 4,
"distribution": [
{
@@ -1221,7 +1230,7 @@
},
{
"label": "scripts",
- "value": 106
+ "value": 109
},
{
"label": "evals",
@@ -1229,7 +1238,7 @@
},
{
"label": "reports",
- "value": 211
+ "value": 215
}
]
},
@@ -1351,7 +1360,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 106
+ "file_count": 109
},
{
"path": "evals",
@@ -1363,7 +1372,7 @@
"path": "reports",
"label": "Generated evidence and overview artifacts",
"kind": "folder",
- "file_count": 211
+ "file_count": 215
}
],
"strengths": [
@@ -1664,16 +1673,16 @@
"ok": true,
"summary": {
"reproducibility_ready": true,
- "release_lock_ready": false,
+ "release_lock_ready": true,
"methodology_complete": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "evidence_bundle_sha256": "119f209f1d4665b38a489cc372140ec9f01c663c4d2637048e82b924cfadb6c5",
- "source_contract_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"output_case_count": 5,
"failure_disclosure_count": 3,
- "command_count": 21,
+ "command_count": 22,
"command_executed_count": 10,
"timing_observed_count": 10,
"model_executed_count": 0,
@@ -1688,11 +1697,11 @@
"world_class_source_pass_count": 6,
"world_class_source_blocked_count": 7,
"public_claim_ready": false,
- "public_claim_blocker_count": 5,
- "working_tree_dirty": true,
- "changed_file_count": 61
+ "public_claim_blocker_count": 4,
+ "working_tree_dirty": false,
+ "changed_file_count": 0
},
- "commit": "5495734e7e33b19b84a932626112d783a3edf271",
+ "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"missing_artifacts": [],
"limitations": [
"The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.",
@@ -1776,9 +1785,9 @@
"failures": []
},
"trust_security": {
- "scanned_files": 193,
- "script_count": 106,
- "internal_module_count": 26,
+ "scanned_files": 196,
+ "script_count": 109,
+ "internal_module_count": 28,
"secret_findings": 0,
"dependency_files": [
"requirements-ci.txt"
@@ -1786,18 +1795,18 @@
"network_script_count": 3,
"network_policy_covered_count": 3,
"network_policy_missing_count": 0,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"permission_required_count": 3,
"permission_approved_count": 3,
"permission_missing_count": 0,
"permission_invalid_count": 0,
"permission_expired_count": 0,
- "help_smoke_checked_count": 80,
+ "help_smoke_checked_count": 81,
"help_smoke_failed_count": 0,
"interactive_script_count": 0,
"package_hash_scope": "source-contract-without-generated-reports",
- "package_hash_file_count": 193,
- "package_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6"
+ "package_hash_file_count": 196,
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4"
},
"skill_atlas": {
"skill_count": 12,
@@ -1835,8 +1844,8 @@
"trust_level": "local",
"license": "MIT",
"checksums": {
- "package_sha256": "37a1ec0f35323886bdae64523b4113f5dc4ba3153117ee4980b8e5cd54b49aba",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc"
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8"
},
"compatibility": {
"openai": "pass",
@@ -1867,12 +1876,12 @@
},
"distribution": {
"archive_verified": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"package_verification": "reports/package_verification.json",
"install_simulated": true,
"install_simulation": "reports/install_simulation.json"
},
- "generated_at": "2026-06-13"
+ "generated_at": "2026-06-15"
},
"failures": [],
"warnings": []
@@ -1883,8 +1892,8 @@
"target_count": 4,
"adapter_count": 4,
"archive_present": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
- "archive_entry_count": 586,
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
+ "archive_entry_count": 609,
"failure_count": 0,
"warning_count": 0
},
@@ -1895,7 +1904,7 @@
"ok": true,
"summary": {
"archive_present": true,
- "archive_entry_count": 601,
+ "archive_entry_count": 609,
"archive_extracted": true,
"entrypoint_loaded": true,
"manifest_loaded": true,
@@ -1905,7 +1914,7 @@
"installer_permission_failure_count": 0,
"permission_target_count": 4,
"permission_capability_count": 3,
- "install_root_is_temp": false,
+ "install_root_is_temp": true,
"failure_count": 0,
"warning_count": 0
},
@@ -1967,7 +1976,7 @@
{
"field": "package_sha256",
"from": "0000000000000000000000000000000000000000000000000000000000000000",
- "to": "37a1ec0f35323886bdae64523b4113f5dc4ba3153117ee4980b8e5cd54b49aba"
+ "to": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306"
}
]
},
@@ -4334,7 +4343,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 42.05,
+ "duration_ms": 35.01,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -4362,7 +4371,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 36.82,
+ "duration_ms": 34.89,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -4385,7 +4394,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 38.87,
+ "duration_ms": 36.42,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -4413,7 +4422,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 35.73,
+ "duration_ms": 39.48,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -4436,7 +4445,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 35.53,
+ "duration_ms": 36.04,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -4464,7 +4473,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 48.73,
+ "duration_ms": 36.06,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -4487,7 +4496,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 65.97,
+ "duration_ms": 34.85,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -4514,7 +4523,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 60.32,
+ "duration_ms": 40.88,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -4537,7 +4546,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 50.55,
+ "duration_ms": 42.57,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -4566,7 +4575,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 44.37,
+ "duration_ms": 42.98,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -5287,39 +5296,26 @@
"ok": true,
"generated_at": "2026-06-15",
"skill_dir": ".",
- "commit": "5495734e7e33b19b84a932626112d783a3edf271",
+ "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"git_status": {
"available": true,
- "dirty": true,
- "changed_file_count": 61,
- "sample": [
- " M Makefile",
- " M README.md",
- " M reports/adoption_drift_report.json",
- " M reports/adoption_drift_report.md",
- " M reports/architecture_maintainability.json",
- " M reports/architecture_maintainability.md",
- " M reports/benchmark_reproducibility.json",
- " M reports/benchmark_reproducibility.md",
- " M reports/compiled_targets.json",
- " M reports/context_budget.json",
- " M reports/context_budget.md",
- " M reports/context_budget_summary.json"
- ],
+ "dirty": false,
+ "changed_file_count": 0,
+ "sample": [],
"scope": "generation-time status before this report is written"
},
"summary": {
"reproducibility_ready": true,
- "release_lock_ready": false,
+ "release_lock_ready": true,
"methodology_complete": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "evidence_bundle_sha256": "119f209f1d4665b38a489cc372140ec9f01c663c4d2637048e82b924cfadb6c5",
- "source_contract_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"output_case_count": 5,
"failure_disclosure_count": 3,
- "command_count": 21,
+ "command_count": 22,
"command_executed_count": 10,
"timing_observed_count": 10,
"model_executed_count": 0,
@@ -5334,15 +5330,14 @@
"world_class_source_pass_count": 6,
"world_class_source_blocked_count": 7,
"public_claim_ready": false,
- "public_claim_blocker_count": 5,
- "working_tree_dirty": true,
- "changed_file_count": 61
+ "public_claim_blocker_count": 4,
+ "working_tree_dirty": false,
+ "changed_file_count": 0
},
"public_claim": {
"ready": false,
"scope": "public benchmark or world-class readiness claim",
"blockers": [
- "release lock is not clean or commit is unavailable",
"provider-backed model holdout evidence is incomplete",
"human blind-review adjudication is incomplete",
"world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)",
@@ -5351,10 +5346,10 @@
"policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks."
},
"release_lock": {
- "ready": false,
- "commit": "5495734e7e33b19b84a932626112d783a3edf271",
+ "ready": true,
+ "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"status_scope": "generation-time status before this report is written",
- "reason": "working tree was dirty at generation time"
+ "reason": "clean generation-time HEAD"
},
"evidence_bundle": {
"algorithm": "sha256(path,label,exists,artifact_sha256)",
@@ -5362,7 +5357,7 @@
"existing_count": 24,
"missing_count": 0,
"missing_paths": [],
- "sha256": "119f209f1d4665b38a489cc372140ec9f01c663c4d2637048e82b924cfadb6c5"
+ "sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3"
},
"methodology": {
"path": "reports/benchmark_methodology.md",
@@ -5435,8 +5430,8 @@
"label": "output_execution",
"path": "reports/output_execution_runs.json",
"exists": true,
- "bytes": 7965,
- "sha256": "e559e6acf9f2e709f65dc36ea5646199a367fb760d7b7e6014570c59d3f46a55"
+ "bytes": 7967,
+ "sha256": "a4c8cacaab5d2f5178513e20cfbb58787052e82607b9ec4fe3eaeae58d7509c8"
},
{
"label": "blind_review",
@@ -5463,50 +5458,50 @@
"label": "runtime_conformance",
"path": "reports/conformance_matrix.json",
"exists": true,
- "bytes": 10313,
- "sha256": "8251329e663dda51472f29b7721e73d72ccbec9760d96fda022f6218a5a6e347"
+ "bytes": 10342,
+ "sha256": "97f9ba949c23a60b00e9ba2ff279ca03ba517845cc7c62aa8a42645d58006c7e"
},
{
"label": "trust_report",
"path": "reports/security_trust_report.json",
"exists": true,
- "bytes": 105694,
- "sha256": "5c1cc74ef5306f2959c7bc4d01068dd2cc36a840500724922f2fb346e63ddbc5"
+ "bytes": 107994,
+ "sha256": "d7dae65ba9cfdc52001b73242ac88c84cf302bc6091e0df8805fdb79ac2594ce"
},
{
"label": "python_compatibility",
"path": "reports/python_compatibility.json",
"exists": true,
- "bytes": 22252,
- "sha256": "fcf288346cb5159e070b2df75a8e371cca2501945c971aead521e7ac3557a4bc"
+ "bytes": 22768,
+ "sha256": "cdd46eb42010f406fb2732063bde224866a4493ae4be1dc0c835417550a45d18"
},
{
"label": "registry_audit",
"path": "reports/registry_audit.json",
"exists": true,
"bytes": 3183,
- "sha256": "9b54f3fd07680c2a9dd18ec9845e1faa1e19e8fbc854cd13a2d5bc1b84360f0e"
+ "sha256": "9ca263ecce9ace6b3915b2485eb7807edaf375d1aeccc689bc8b2d91417d8d2e"
},
{
"label": "package_verification",
"path": "reports/package_verification.json",
"exists": true,
"bytes": 19338,
- "sha256": "2476ae8ec9c491a3f7f31ec3f0c6c78cceead8e80700dcb696838eec226c675b"
+ "sha256": "1abff9930aa56d337b47dbc421f293eaa9eb3620c187d1b45a26941594b392ef"
},
{
"label": "install_simulation",
"path": "reports/install_simulation.json",
"exists": true,
- "bytes": 8758,
- "sha256": "491fabea4624d41967ff04952b72e0b165882c1b3b276124047c921daea4f3b7"
+ "bytes": 8557,
+ "sha256": "958586fa59a71aeea10962c6047083876fbcd4b809f24fbeaec4ece8aa499dfc"
},
{
"label": "skill_os2_audit",
"path": "reports/skill_os2_audit.json",
"exists": true,
"bytes": 14310,
- "sha256": "8481f4d78f2ad4ff0e74936b58b643a91052a30b6d12bcbbf24e0b4291b8fb1f"
+ "sha256": "55526d80bfa78b574f990442165a0159b0f177feca00523826657b00f7855d29"
},
{
"label": "world_class_evidence_plan",
@@ -5561,8 +5556,8 @@
"label": "world_class_claim_guard",
"path": "reports/world_class_claim_guard.json",
"exists": true,
- "bytes": 8634,
- "sha256": "a2e6953e92e667cc0913c08e48b6954ea48420ad3a4056ed15eb19a183825e67"
+ "bytes": 8814,
+ "sha256": "7e5a2eac1020f4f58688b52d3c59b5df1c74d7bc04f4e599b48e5be4cc23d786"
}
],
"missing_artifacts": [],
@@ -5667,6 +5662,11 @@
"command": "python3 scripts/yao.py world-class-claim-guard .",
"evidence": "reports/world_class_claim_guard.json"
},
+ {
+ "label": "evidence consistency",
+ "command": "python3 scripts/yao.py evidence-consistency .",
+ "evidence": "reports/evidence_consistency.json"
+ },
{
"label": "full ci",
"command": "make ci-test",
@@ -5695,10 +5695,10 @@
"generated_at": "2026-06-15",
"skill_dir": ".",
"summary": {
- "item_count": 20,
+ "item_count": 21,
"module_count": 8,
- "recommended_pr_count": 12,
- "pass_count": 20,
+ "recommended_pr_count": 13,
+ "pass_count": 21,
"warn_count": 0,
"missing_count": 0,
"extension_track_count": 2,
@@ -5712,7 +5712,7 @@
"decision": "local-blueprint-covered-evidence-pending"
},
"status_counts": {
- "pass": 20,
+ "pass": 21,
"warn": 0,
"missing": 0
},
@@ -5822,7 +5822,7 @@
"label": "Trust Security",
"status": "pass",
"objective": "Scripts, dependencies, permissions, secrets, and package hash are reviewable for team distribution.",
- "current": "106 scripts; secrets 0; help failures 0",
+ "current": "109 scripts; secrets 0; help failures 0",
"command": "python3 scripts/yao.py trust .",
"test": "python3 tests/verify_trust_check.py",
"evidence": [
@@ -5896,7 +5896,7 @@
"label": "Registry Distribution",
"status": "pass",
"objective": "Skill packages are installable, versioned, checksumed, and upgrade-reviewable.",
- "current": "archive entries 586; install failures 0",
+ "current": "archive entries 609; install failures 0",
"command": "python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip && python3 scripts/yao.py registry-audit .",
"test": "python3 tests/verify_registry_audit.py",
"evidence": [
@@ -6298,6 +6298,31 @@
}
],
"next_action": "Keep this item covered as the implementation evolves."
+ },
+ {
+ "key": "evidence-consistency",
+ "category": "recommended-pr",
+ "label": "Evidence Consistency",
+ "status": "pass",
+ "objective": "Recommended Skill OS 2.0 implementation PR from the upgrade plan.",
+ "current": "26 consistency checks",
+ "command": "make ci-test",
+ "test": "tests/verify_evidence_consistency.py",
+ "evidence": [
+ {
+ "path": "scripts/render_evidence_consistency.py",
+ "exists": true
+ },
+ {
+ "path": "reports/evidence_consistency.json",
+ "exists": true
+ },
+ {
+ "path": "tests/verify_evidence_consistency.py",
+ "exists": true
+ }
+ ],
+ "next_action": "Keep this item covered as the implementation evolves."
}
],
"reference_extension_tracks": [
@@ -6409,7 +6434,7 @@
"source_blueprint": {
"title": "Skill Overview / Skill OS 2.0 upgrade plan",
"core_module_count": 8,
- "recommended_pr_count": 12,
+ "recommended_pr_count": 13,
"reference_extension_count": 2,
"reference_extensions": [
"Skill interpretation report",
@@ -6424,7 +6449,7 @@
"compiled_targets": {
"schema_version": "1.0",
"ok": true,
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"skill_dir": ".",
"summary": {
"target_count": 5,
@@ -6609,6 +6634,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -6623,6 +6649,7 @@
"scripts/render_review_studio.py",
"scripts/render_review_viewer.py",
"scripts/render_review_waivers.py",
+ "scripts/render_skill_interpretation.py",
"scripts/render_skill_os2_audit.py",
"scripts/render_skill_os2_coverage.py",
"scripts/render_skill_overview.py",
@@ -6648,9 +6675,7 @@
"scripts/review_viewer_data.py",
"scripts/run_conformance_suite.py",
"scripts/run_description_optimization_suite.py",
- "scripts/run_eval_suite.py",
- "scripts/run_output_eval.py",
- "scripts/run_output_execution.py"
+ "scripts/run_eval_suite.py"
],
"assets": [
"templates/basic_skill.md.j2",
@@ -6810,7 +6835,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -6841,6 +6866,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -6917,14 +6943,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -6946,7 +6972,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -7059,7 +7085,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -7090,6 +7116,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -7166,14 +7193,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -7193,7 +7220,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -7468,6 +7495,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -7482,6 +7510,7 @@
"scripts/render_review_studio.py",
"scripts/render_review_viewer.py",
"scripts/render_review_waivers.py",
+ "scripts/render_skill_interpretation.py",
"scripts/render_skill_os2_audit.py",
"scripts/render_skill_os2_coverage.py",
"scripts/render_skill_overview.py",
@@ -7507,9 +7536,7 @@
"scripts/review_viewer_data.py",
"scripts/run_conformance_suite.py",
"scripts/run_description_optimization_suite.py",
- "scripts/run_eval_suite.py",
- "scripts/run_output_eval.py",
- "scripts/run_output_execution.py"
+ "scripts/run_eval_suite.py"
],
"assets": [
"templates/basic_skill.md.j2",
@@ -7669,7 +7696,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -7700,6 +7727,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -7776,14 +7804,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -7805,7 +7833,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -7918,7 +7946,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -7949,6 +7977,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -8025,14 +8054,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -8052,7 +8081,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -8327,6 +8356,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -8341,6 +8371,7 @@
"scripts/render_review_studio.py",
"scripts/render_review_viewer.py",
"scripts/render_review_waivers.py",
+ "scripts/render_skill_interpretation.py",
"scripts/render_skill_os2_audit.py",
"scripts/render_skill_os2_coverage.py",
"scripts/render_skill_overview.py",
@@ -8366,9 +8397,7 @@
"scripts/review_viewer_data.py",
"scripts/run_conformance_suite.py",
"scripts/run_description_optimization_suite.py",
- "scripts/run_eval_suite.py",
- "scripts/run_output_eval.py",
- "scripts/run_output_execution.py"
+ "scripts/run_eval_suite.py"
],
"assets": [
"templates/basic_skill.md.j2",
@@ -8528,7 +8557,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -8559,6 +8588,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -8635,14 +8665,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -8664,7 +8694,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -8770,7 +8800,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -8801,6 +8831,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -8877,14 +8908,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -8904,7 +8935,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -9170,6 +9201,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -9184,6 +9216,7 @@
"scripts/render_review_studio.py",
"scripts/render_review_viewer.py",
"scripts/render_review_waivers.py",
+ "scripts/render_skill_interpretation.py",
"scripts/render_skill_os2_audit.py",
"scripts/render_skill_os2_coverage.py",
"scripts/render_skill_overview.py",
@@ -9209,9 +9242,7 @@
"scripts/review_viewer_data.py",
"scripts/run_conformance_suite.py",
"scripts/run_description_optimization_suite.py",
- "scripts/run_eval_suite.py",
- "scripts/run_output_eval.py",
- "scripts/run_output_execution.py"
+ "scripts/run_eval_suite.py"
],
"assets": [
"templates/basic_skill.md.j2",
@@ -9371,7 +9402,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -9402,6 +9433,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -9478,14 +9510,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -9507,7 +9539,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -9613,7 +9645,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -9644,6 +9676,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -9720,14 +9753,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -9747,7 +9780,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -10013,6 +10046,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -10027,6 +10061,7 @@
"scripts/render_review_studio.py",
"scripts/render_review_viewer.py",
"scripts/render_review_waivers.py",
+ "scripts/render_skill_interpretation.py",
"scripts/render_skill_os2_audit.py",
"scripts/render_skill_os2_coverage.py",
"scripts/render_skill_overview.py",
@@ -10052,9 +10087,7 @@
"scripts/review_viewer_data.py",
"scripts/run_conformance_suite.py",
"scripts/run_description_optimization_suite.py",
- "scripts/run_eval_suite.py",
- "scripts/run_output_eval.py",
- "scripts/run_output_execution.py"
+ "scripts/run_eval_suite.py"
],
"assets": [
"templates/basic_skill.md.j2",
@@ -10214,7 +10247,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -10245,6 +10278,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -10321,14 +10355,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -10350,7 +10384,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -10460,7 +10494,7 @@
},
"file_write": {
"required": true,
- "script_count": 67,
+ "script_count": 68,
"scripts": [
"scripts/adjudicate_output_review.py",
"scripts/build_confusion_matrix.py",
@@ -10491,6 +10525,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -10567,14 +10602,14 @@
},
"help_smoke": {
"enabled": true,
- "checked_count": 80,
+ "checked_count": 81,
"failed_count": 0,
"failed_scripts": []
},
"trust_summary": {
"secret_findings": 0,
"network_script_count": 3,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"subprocess_script_count": 9,
"interactive_script_count": 0,
"help_smoke_failed_count": 0
@@ -10594,7 +10629,7 @@
],
"capability_counts": {
"network": 3,
- "file_write": 67,
+ "file_write": 68,
"subprocess": 9,
"interactive": 0
},
@@ -10734,6 +10769,7 @@
"Skill IR description matches frontmatter",
"references resource resolves: references/artifact-design-doctrine.md",
"references resource resolves: references/authoring-discipline.md",
+ "references resource resolves: references/autonomous-adaptation.md",
"references resource resolves: references/distribution-registry-method.md",
"references resource resolves: references/eval-playbook.md",
"references resource resolves: references/gate-selection.md",
@@ -10742,8 +10778,7 @@
"references resource resolves: references/intent-dialogue.md",
"references resource resolves: references/iteration-philosophy.md",
"references resource resolves: references/non-skill-decision-tree.md",
- "references resource resolves: references/operating-modes.md",
- "references resource resolves: references/output-eval-method.md"
+ "references resource resolves: references/operating-modes.md"
],
"failures": [],
"warnings": []
@@ -10778,6 +10813,7 @@
"Skill IR description matches frontmatter",
"references resource resolves: references/artifact-design-doctrine.md",
"references resource resolves: references/authoring-discipline.md",
+ "references resource resolves: references/autonomous-adaptation.md",
"references resource resolves: references/distribution-registry-method.md",
"references resource resolves: references/eval-playbook.md",
"references resource resolves: references/gate-selection.md",
@@ -10786,8 +10822,7 @@
"references resource resolves: references/intent-dialogue.md",
"references resource resolves: references/iteration-philosophy.md",
"references resource resolves: references/non-skill-decision-tree.md",
- "references resource resolves: references/operating-modes.md",
- "references resource resolves: references/output-eval-method.md"
+ "references resource resolves: references/operating-modes.md"
],
"failures": [],
"warnings": []
@@ -10801,7 +10836,7 @@
"frontmatter description exists",
"description length <= 1024",
"name is runtime-safe",
- "directory name matches skill name",
+ "package identity derives from skill name",
"manifest.name exists",
"manifest.version exists",
"manifest.owner exists",
@@ -10822,6 +10857,7 @@
"Skill IR description matches frontmatter",
"references resource resolves: references/artifact-design-doctrine.md",
"references resource resolves: references/authoring-discipline.md",
+ "references resource resolves: references/autonomous-adaptation.md",
"references resource resolves: references/distribution-registry-method.md",
"references resource resolves: references/eval-playbook.md",
"references resource resolves: references/gate-selection.md",
@@ -10830,8 +10866,7 @@
"references resource resolves: references/intent-dialogue.md",
"references resource resolves: references/iteration-philosophy.md",
"references resource resolves: references/non-skill-decision-tree.md",
- "references resource resolves: references/operating-modes.md",
- "references resource resolves: references/output-eval-method.md"
+ "references resource resolves: references/operating-modes.md"
],
"failures": [],
"warnings": [
@@ -10847,7 +10882,7 @@
"frontmatter description exists",
"description length <= 1024",
"name is runtime-safe",
- "directory name matches skill name",
+ "package identity derives from skill name",
"manifest.name exists",
"manifest.version exists",
"manifest.owner exists",
@@ -10868,6 +10903,7 @@
"Skill IR description matches frontmatter",
"references resource resolves: references/artifact-design-doctrine.md",
"references resource resolves: references/authoring-discipline.md",
+ "references resource resolves: references/autonomous-adaptation.md",
"references resource resolves: references/distribution-registry-method.md",
"references resource resolves: references/eval-playbook.md",
"references resource resolves: references/gate-selection.md",
@@ -10876,8 +10912,7 @@
"references resource resolves: references/intent-dialogue.md",
"references resource resolves: references/iteration-philosophy.md",
"references resource resolves: references/non-skill-decision-tree.md",
- "references resource resolves: references/operating-modes.md",
- "references resource resolves: references/output-eval-method.md"
+ "references resource resolves: references/operating-modes.md"
],
"failures": [],
"warnings": [
@@ -10914,6 +10949,7 @@
"Skill IR description matches frontmatter",
"references resource resolves: references/artifact-design-doctrine.md",
"references resource resolves: references/authoring-discipline.md",
+ "references resource resolves: references/autonomous-adaptation.md",
"references resource resolves: references/distribution-registry-method.md",
"references resource resolves: references/eval-playbook.md",
"references resource resolves: references/gate-selection.md",
@@ -10922,8 +10958,7 @@
"references resource resolves: references/intent-dialogue.md",
"references resource resolves: references/iteration-philosophy.md",
"references resource resolves: references/non-skill-decision-tree.md",
- "references resource resolves: references/operating-modes.md",
- "references resource resolves: references/output-eval-method.md"
+ "references resource resolves: references/operating-modes.md"
],
"failures": [],
"warnings": []
@@ -10943,7 +10978,7 @@
"schema_version": "1.0",
"ok": true,
"skill_dir": ".",
- "package_dir": "tests/tmp_review_studio/dist",
+ "package_dir": "dist",
"expected_capabilities": [
"file_write",
"network",
@@ -10983,7 +11018,7 @@
{
"target": "openai",
"status": "pass",
- "adapter": "tests/tmp_review_studio/dist/targets/openai/adapter.json",
+ "adapter": "dist/targets/openai/adapter.json",
"permission_model": "metadata-only",
"native_enforcement": false,
"metadata_fallback_explicit": true,
@@ -11080,7 +11115,7 @@
{
"target": "claude",
"status": "pass",
- "adapter": "tests/tmp_review_studio/dist/targets/claude/adapter.json",
+ "adapter": "dist/targets/claude/adapter.json",
"permission_model": "neutral-source-plus-adapter",
"native_enforcement": false,
"metadata_fallback_explicit": true,
@@ -11162,7 +11197,7 @@
{
"target": "generic",
"status": "pass",
- "adapter": "tests/tmp_review_studio/dist/targets/generic/adapter.json",
+ "adapter": "dist/targets/generic/adapter.json",
"permission_model": "agent-skills-compatible-metadata",
"native_enforcement": false,
"metadata_fallback_explicit": true,
@@ -11244,7 +11279,7 @@
{
"target": "vscode",
"status": "pass",
- "adapter": "tests/tmp_review_studio/dist/targets/vscode/adapter.json",
+ "adapter": "dist/targets/vscode/adapter.json",
"permission_model": "vscode-workspace-trust-plus-metadata",
"native_enforcement": false,
"metadata_fallback_explicit": true,
@@ -11334,9 +11369,9 @@
"ok": true,
"skill_dir": ".",
"summary": {
- "scanned_files": 193,
- "script_count": 106,
- "internal_module_count": 26,
+ "scanned_files": 196,
+ "script_count": 109,
+ "internal_module_count": 28,
"secret_findings": 0,
"dependency_files": [
"requirements-ci.txt"
@@ -11344,18 +11379,18 @@
"network_script_count": 3,
"network_policy_covered_count": 3,
"network_policy_missing_count": 0,
- "file_write_script_count": 67,
+ "file_write_script_count": 68,
"permission_required_count": 3,
"permission_approved_count": 3,
"permission_missing_count": 0,
"permission_invalid_count": 0,
"permission_expired_count": 0,
- "help_smoke_checked_count": 80,
+ "help_smoke_checked_count": 81,
"help_smoke_failed_count": 0,
"interactive_script_count": 0,
"package_hash_scope": "source-contract-without-generated-reports",
- "package_hash_file_count": 193,
- "package_sha256": "28625ce8b3fcb3e214ef3b8ffc748d9cff551e4c5a1b78786c1a2b7d7005f679"
+ "package_hash_file_count": 196,
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4"
},
"failures": [],
"warnings": [],
@@ -11906,6 +11941,20 @@
"network_urls": [],
"network_hosts": []
},
+ {
+ "path": "scripts/render_evidence_consistency.py",
+ "interface": "cli",
+ "interface_declared": true,
+ "interface_reason": "Renders a cross-report evidence consistency gate for generated Skill OS reports.",
+ "has_argparse": true,
+ "has_main_guard": true,
+ "uses_input": false,
+ "uses_network": false,
+ "uses_file_write": true,
+ "uses_subprocess": false,
+ "network_urls": [],
+ "network_hosts": []
+ },
{
"path": "scripts/render_intent_confidence.py",
"interface": "cli",
@@ -12806,6 +12855,34 @@
"network_urls": [],
"network_hosts": []
},
+ {
+ "path": "scripts/yao_cli_distribution_commands.py",
+ "interface": "internal-module",
+ "interface_declared": true,
+ "interface_reason": "Imported by yao.py to keep distribution and runtime gate handlers outside the thin CLI orchestrator.",
+ "has_argparse": true,
+ "has_main_guard": false,
+ "uses_input": false,
+ "uses_network": false,
+ "uses_file_write": false,
+ "uses_subprocess": false,
+ "network_urls": [],
+ "network_hosts": []
+ },
+ {
+ "path": "scripts/yao_cli_output_commands.py",
+ "interface": "internal-module",
+ "interface_declared": true,
+ "interface_reason": "Imported by yao.py to keep output evaluation and review handlers outside the thin CLI orchestrator.",
+ "has_argparse": true,
+ "has_main_guard": false,
+ "uses_input": false,
+ "uses_network": false,
+ "uses_file_write": false,
+ "uses_subprocess": false,
+ "network_urls": [],
+ "network_hosts": []
+ },
{
"path": "scripts/yao_cli_parser.py",
"interface": "internal-module",
@@ -12893,11 +12970,11 @@
"help_smoke": {
"enabled": true,
"timeout_seconds": 5.0,
- "candidate_count": 80,
- "checked_count": 80,
- "passed_count": 80,
+ "candidate_count": 81,
+ "checked_count": 81,
+ "passed_count": 81,
"failed_count": 0,
- "skipped_count": 26,
+ "skipped_count": 28,
"failed_scripts": [],
"results": [
{
@@ -13270,6 +13347,16 @@
"stdout_excerpt": "usage: render_eval_dashboard.py [-h] [--eval-dir EVAL_DIR]\n [--description-file DESCRIPTION_FILE]\n [--baseline-description-file BASELINE_DESCRIPTION_FILE]\n ",
"stderr_excerpt": ""
},
+ {
+ "path": "scripts/render_evidence_consistency.py",
+ "command": "python3 scripts/render_evidence_consistency.py --help",
+ "returncode": 0,
+ "timed_out": false,
+ "passed": true,
+ "has_help_text": true,
+ "stdout_excerpt": "usage: render_evidence_consistency.py [-h] [--output-json OUTPUT_JSON]\n [--output-md OUTPUT_MD]\n [--generated-at GENERATED_AT]\n [",
+ "stderr_excerpt": ""
+ },
{
"path": "scripts/render_intent_confidence.py",
"command": "python3 scripts/render_intent_confidence.py --help",
@@ -13327,7 +13414,7 @@
"timed_out": false,
"passed": true,
"has_help_text": true,
- "stdout_excerpt": "usage: render_portability_report.py [-h] [--output-json OUTPUT_JSON]\n [--output-md OUTPUT_MD]\n\nRender a portability score from neutral metadata, contracts, and snapshots.\n\noptions:\n -h, --help ",
+ "stdout_excerpt": "usage: render_portability_report.py [-h] [--output-json OUTPUT_JSON]\n [--output-md OUTPUT_MD]\n [skill_dir]\n\nRender a portability score from neutral metadata, contracts, a",
"stderr_excerpt": ""
},
{
@@ -13790,6 +13877,14 @@
"path": "scripts/yao_cli_create_commands.py",
"reason": "internal module"
},
+ {
+ "path": "scripts/yao_cli_distribution_commands.py",
+ "reason": "internal module"
+ },
+ {
+ "path": "scripts/yao_cli_output_commands.py",
+ "reason": "internal module"
+ },
{
"path": "scripts/yao_cli_parser.py",
"reason": "internal module"
@@ -13889,6 +13984,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -13990,11 +14086,11 @@
"python_compatibility": {
"schema_version": "1.0",
"ok": true,
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"root": ".",
"summary": {
"target_python": "3.11",
- "file_count": 171,
+ "file_count": 175,
"issue_count": 0,
"syntax_error_count": 0,
"fstring_311_violation_count": 0,
@@ -14257,6 +14353,12 @@
"issue_count": 0,
"issues": []
},
+ {
+ "path": "scripts/render_evidence_consistency.py",
+ "ok": true,
+ "issue_count": 0,
+ "issues": []
+ },
{
"path": "scripts/render_intent_confidence.py",
"ok": true,
@@ -14641,6 +14743,18 @@
"issue_count": 0,
"issues": []
},
+ {
+ "path": "scripts/yao_cli_distribution_commands.py",
+ "ok": true,
+ "issue_count": 0,
+ "issues": []
+ },
+ {
+ "path": "scripts/yao_cli_output_commands.py",
+ "ok": true,
+ "issue_count": 0,
+ "issues": []
+ },
{
"path": "scripts/yao_cli_parser.py",
"ok": true,
@@ -14725,6 +14839,12 @@
"issue_count": 0,
"issues": []
},
+ {
+ "path": "tests/verify_evidence_consistency.py",
+ "ok": true,
+ "issue_count": 0,
+ "issues": []
+ },
{
"path": "tests/verify_failure_regressions.py",
"ok": true,
@@ -15047,33 +15167,28 @@
"architecture_maintainability": {
"schema_version": "1.0",
"ok": true,
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"skill_dir": ".",
"summary": {
- "python_file_count": 168,
- "script_file_count": 106,
- "test_file_count": 62,
- "internal_module_count": 29,
- "cli_script_count": 79,
- "command_handler_count": 34,
+ "python_file_count": 172,
+ "script_file_count": 109,
+ "test_file_count": 63,
+ "internal_module_count": 31,
+ "cli_script_count": 80,
+ "command_handler_count": 63,
+ "entrypoint_command_handler_count": 18,
+ "command_module_count": 6,
"warn_line_threshold": 900,
"block_line_threshold": 1500,
- "largest_file_lines": 886,
+ "largest_file_lines": 897,
"hotspot_count": 0,
"blocker_count": 0,
"decision": "pass"
},
"largest_files": [
- {
- "path": "scripts/yao.py",
- "lines": 886,
- "kind": "cli-script",
- "severity": "pass",
- "recommendation": "Split command handlers by domain while keeping scripts/yao.py as the thin CLI orchestrator."
- },
{
"path": "tests/verify_yao_cli.py",
- "lines": 886,
+ "lines": 897,
"kind": "test",
"severity": "pass",
"recommendation": "Break broad integration assertions into focused verifier helpers when the next behavior change lands."
@@ -15087,14 +15202,14 @@
},
{
"path": "scripts/skill_report_model.py",
- "lines": 800,
+ "lines": 801,
"kind": "internal-module",
"severity": "pass",
"recommendation": "Watch this file before adding new responsibilities; extract a helper module when one concern dominates."
},
{
"path": "scripts/yao_cli_parser.py",
- "lines": 774,
+ "lines": 784,
"kind": "internal-module",
"severity": "pass",
"recommendation": "Watch this file before adding new responsibilities; extract a helper module when one concern dominates."
@@ -15147,6 +15262,13 @@
"kind": "cli-script",
"severity": "pass",
"recommendation": "Split viewer data assembly from HTML section rendering."
+ },
+ {
+ "path": "scripts/cross_packager.py",
+ "lines": 650,
+ "kind": "cli-script",
+ "severity": "pass",
+ "recommendation": "Watch this file before adding new responsibilities; extract a helper module when one concern dominates."
}
],
"hotspots": [],
@@ -15164,16 +15286,16 @@
"context_budget_tier": "production",
"context_budget_limit": 1000,
"skill_body_tokens": 767,
- "other_text_tokens": 1307850,
+ "other_text_tokens": 1334868,
"estimated_initial_load_tokens": 960,
- "estimated_total_text_tokens": 1308617,
- "deferred_resource_tokens": 416415,
+ "estimated_total_text_tokens": 1335635,
+ "deferred_resource_tokens": 421932,
"deferred_resource_warn_threshold": 120000,
"deferred_resource_dirs": [
{
"path": "scripts",
- "estimated_tokens": 367825,
- "file_count": 106
+ "estimated_tokens": 373342,
+ "file_count": 109
},
{
"path": "references",
@@ -15189,8 +15311,8 @@
"large_deferred_resource_dirs": [
{
"path": "scripts",
- "estimated_tokens": 367825,
- "file_count": 106
+ "estimated_tokens": 373342,
+ "file_count": 109
}
],
"deferred_resource_governance": {
@@ -15212,14 +15334,14 @@
],
"missing": [],
"path": "scripts",
- "estimated_tokens": 367825,
- "file_count": 106,
+ "estimated_tokens": 373342,
+ "file_count": 109,
"rationale": "Script resources are deterministic deferred tools, not initial-load prompt context."
}
],
"summary": "Large deferred resources are indexed and backed by evidence."
},
- "relevant_file_count": 544,
+ "relevant_file_count": 554,
"unused_resource_dirs": [],
"quality_signal_points": 130,
"quality_density": 135.4
@@ -16121,6 +16243,7 @@
"scripts/render_context_reports.py",
"scripts/render_description_drift_history.py",
"scripts/render_eval_dashboard.py",
+ "scripts/render_evidence_consistency.py",
"scripts/render_intent_confidence.py",
"scripts/render_intent_dialogue.py",
"scripts/render_iteration_directions.py",
@@ -17086,7 +17209,7 @@
"adoption_drift": {
"ok": true,
"schema_version": "2.0",
- "generated_at": "2026-06-15T13:28:42Z",
+ "generated_at": "2026-06-15T14:35:15Z",
"skill_dir": ".",
"privacy_contract": {
"storage": "local-first",
@@ -17110,14 +17233,14 @@
},
"summary": {
"event_count": 1,
- "adoption_sample_count": 1,
- "activation_count": 1,
- "accepted_count": 1,
+ "adoption_sample_count": 0,
+ "activation_count": 0,
+ "accepted_count": 0,
"edited_count": 0,
"rejected_count": 0,
"missed_count": 0,
"failed_count": 0,
- "adoption_rate": 100.0,
+ "adoption_rate": 0,
"missed_trigger_count": 0,
"wrong_trigger_count": 0,
"bad_output_count": 0,
@@ -17126,7 +17249,7 @@
"review_overdue_count": 0,
"risk_band": "low",
"event_types": {
- "skill_activation": 1
+ "review_event": 1
},
"failure_types": {},
"source_types": {
@@ -17138,31 +17261,31 @@
{
"skill": "yao-meta-skill",
"events": 1,
- "adoption_events": 1,
- "accepted": 1,
+ "adoption_events": 0,
+ "accepted": 0,
"edited": 0,
"rejected": 0,
"missed": 0,
- "adoption_rate": 100.0
+ "adoption_rate": 0
}
],
"next_iteration_candidates": [],
"recent_events": [
{
"command": "unknown",
- "event": "skill_activation",
+ "event": "review_event",
"skill": "yao-meta-skill",
"source": "manual",
"version": "1.1.0",
- "activation_type": "explicit",
- "outcome": "accepted",
+ "activation_type": "manual",
+ "outcome": "reviewed",
"failure_type": "none",
- "timestamp": "2026-06-13T10:00:00Z"
+ "timestamp": "2026-06-13T12:00:00Z"
}
],
"failures": [],
"artifacts": {
- "events_jsonl": "tests/tmp_review_studio/telemetry_events.jsonl",
+ "events_jsonl": "reports/telemetry_events.jsonl",
"json": "reports/adoption_drift_report.json",
"markdown": "reports/adoption_drift_report.md"
}
@@ -17171,7 +17294,7 @@
"schema_version": "1.0",
"ok": true,
"skill_dir": ".",
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"summary": {
"waiver_count": 0,
"active_count": 0,
@@ -17221,7 +17344,7 @@
"Reviewer links output_review_adjudication or output_execution evidence."
],
"suggested_evidence": "reports/output_review_adjudication.md",
- "suggested_command": "python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer \"\" --reason \"Output Lab has pending human/provider evidence; accepted only for this bounded review scope.\" --expires-at 2027-06-13 --evidence reports/output_review_adjudication.md",
+ "suggested_command": "python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer \"\" --reason \"Output Lab has pending human/provider evidence; accepted only for this bounded review scope.\" --expires-at 2027-06-15 --evidence reports/output_review_adjudication.md",
"world_class_boundary": "Does not count as provider, human, or public world-class completion evidence."
},
{
@@ -17251,7 +17374,7 @@
"schema_version": "1.0",
"ok": true,
"skill_dir": ".",
- "source": "tests/tmp_review_studio/empty_review_annotations_input.json",
+ "source": "reports/review_annotations_input.json",
"summary": {
"annotation_count": 0,
"open_count": 0,
@@ -17274,7 +17397,7 @@
"world_class_evidence_ledger": {
"schema_version": "1.0",
"ok": true,
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"skill_dir": ".",
"summary": {
"ledger_entry_count": 4,
@@ -17287,8 +17410,8 @@
"missing_submission_count": 4,
"invalid_submission_count": 0,
"source_check_count": 13,
- "source_pass_count": 7,
- "source_blocked_count": 6,
+ "source_pass_count": 6,
+ "source_blocked_count": 7,
"submitted_but_pending_count": 0,
"source_accepted_without_valid_submission_count": 0,
"overclaim_guard_active": true,
@@ -17624,7 +17747,7 @@
"status": "pending",
"source_status": "external_required",
"source_accepted": false,
- "current": "external source events 0; adoption samples 1",
+ "current": "external source events 0; adoption samples 0",
"objective": "Import production metadata-only events from a real external client into the local drift loop.",
"runbook": [
"python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///",
@@ -17661,7 +17784,7 @@
],
"observed_state": {
"external_source_events": 0,
- "adoption_sample_count": 1,
+ "adoption_sample_count": 0,
"raw_content_allowed": false,
"risk_band": "low",
"accepted": false
@@ -17682,8 +17805,8 @@
"label": "Adoption sample",
"field": "adoption_sample_count",
"expected": ">0",
- "actual": 1,
- "status": "pass",
+ "actual": 0,
+ "status": "blocked",
"source_accepted": false,
"next_action": "Telemetry must include adoption outcome evidence."
},
@@ -17699,8 +17822,8 @@
}
],
"source_check_count": 3,
- "source_pass_count": 2,
- "source_blocked_count": 1,
+ "source_pass_count": 1,
+ "source_blocked_count": 2,
"submission_state": {
"status": "missing",
"path": "evidence/world_class/submissions/native-client-telemetry.json",
@@ -17738,7 +17861,7 @@
"world_class_evidence_intake": {
"schema_version": "1.0",
"ok": true,
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"skill_dir": ".",
"summary": {
"schema_present": true,
@@ -18044,7 +18167,7 @@
"source_accepted": false,
"observed_state": {
"external_source_events": 0,
- "adoption_sample_count": 1,
+ "adoption_sample_count": 0,
"raw_content_allowed": false,
"risk_band": "low",
"accepted": false
@@ -18115,7 +18238,7 @@
"world_class_submission_review": {
"schema_version": "1.0",
"ok": true,
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"skill_dir": ".",
"summary": {
"review_item_count": 4,
@@ -18127,8 +18250,8 @@
"unmatched_submission_count": 0,
"invalid_submission_count": 0,
"source_check_count": 13,
- "source_pass_count": 7,
- "source_blocked_count": 6,
+ "source_pass_count": 6,
+ "source_blocked_count": 7,
"ready_to_claim_world_class": false,
"review_counts_submission_as_completion": false,
"decision": "awaiting-submissions"
@@ -18379,7 +18502,7 @@
"intake_errors": [],
"observed_state": {
"external_source_events": 0,
- "adoption_sample_count": 1,
+ "adoption_sample_count": 0,
"raw_content_allowed": false,
"risk_band": "low",
"accepted": false
@@ -18400,8 +18523,8 @@
"label": "Adoption sample",
"field": "adoption_sample_count",
"expected": ">0",
- "actual": 1,
- "status": "pass",
+ "actual": 0,
+ "status": "blocked",
"source_accepted": false,
"next_action": "Telemetry must include adoption outcome evidence."
},
@@ -18417,8 +18540,8 @@
}
],
"source_check_count": 3,
- "source_pass_count": 2,
- "source_blocked_count": 1,
+ "source_pass_count": 1,
+ "source_blocked_count": 2,
"success_checks": [
"reports/adoption_drift_report.json summary.source_types.external > 0",
"reports/adoption_drift_report.json summary.adoption_sample_count > 0",
@@ -18444,7 +18567,7 @@
"world_class_operator_runbook": {
"schema_version": "1.0",
"ok": true,
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"skill_dir": ".",
"summary": {
"evidence_item_count": 4,
@@ -18455,8 +18578,8 @@
"valid_packet_source_incomplete_count": 0,
"invalid_submission_count": 0,
"source_check_count": 13,
- "source_pass_count": 7,
- "source_blocked_count": 6,
+ "source_pass_count": 6,
+ "source_blocked_count": 7,
"ready_to_claim_world_class": false,
"runbook_counts_as_completion": false,
"decision": "collect-evidence"
@@ -18837,7 +18960,7 @@
"review_state": "awaiting-submission",
"source_accepted": false,
"objective": "Import production metadata-only events from a real external client into the local drift loop.",
- "current": "external source events 0; adoption samples 1",
+ "current": "external source events 0; adoption samples 0",
"execution_runbook": [
"python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///",
"Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event.",
@@ -18885,7 +19008,7 @@
],
"observed_state": {
"external_source_events": 0,
- "adoption_sample_count": 1,
+ "adoption_sample_count": 0,
"raw_content_allowed": false,
"risk_band": "low",
"accepted": false
@@ -18906,8 +19029,8 @@
"label": "Adoption sample",
"field": "adoption_sample_count",
"expected": ">0",
- "actual": 1,
- "status": "pass",
+ "actual": 0,
+ "status": "blocked",
"source_accepted": false,
"next_action": "Telemetry must include adoption outcome evidence."
},
@@ -18922,9 +19045,10 @@
"next_action": "Telemetry must stay metadata-only."
}
],
- "blocked_source_check_count": 1,
+ "blocked_source_check_count": 2,
"next_source_actions": [
- "Import at least one metadata-only event from a real client."
+ "Import at least one metadata-only event from a real client.",
+ "Telemetry must include adoption outcome evidence."
],
"submission_state": {
"status": "missing",
@@ -18957,12 +19081,12 @@
"world_class_claim_guard": {
"schema_version": "1.0",
"ok": true,
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"skill_dir": ".",
"summary": {
"ledger_ready_to_claim_world_class": false,
"ledger_pending_count": 4,
- "claim_surface_count": 76,
+ "claim_surface_count": 77,
"violation_count": 0,
"overclaim_guard_active": true,
"decision": "claim-guard-pass-evidence-pending"
@@ -19079,6 +19203,10 @@
"path": "reports/description_optimization_suite.md",
"violation_count": 0
},
+ {
+ "path": "reports/evidence_consistency.md",
+ "violation_count": 0
+ },
{
"path": "reports/family_summary.md",
"violation_count": 0
@@ -19333,8 +19461,8 @@
"trust_level": "local",
"license": "MIT",
"checksums": {
- "package_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc"
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8"
},
"compatibility": {
"openai": "pass",
@@ -19365,16 +19493,16 @@
},
"distribution": {
"archive_verified": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"package_verification": "reports/package_verification.json",
"install_simulated": true,
"install_simulation": "reports/install_simulation.json"
},
- "generated_at": "2026-06-13"
+ "generated_at": "2026-06-15"
},
"index": {
"schema_version": "2.0",
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"package_count": 1,
"packages": [
{
@@ -19390,7 +19518,7 @@
"vscode"
],
"package_metadata": "registry/packages/yao-meta-skill.json",
- "package_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6"
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4"
}
]
},
@@ -19413,8 +19541,8 @@
"target_count": 4,
"adapter_count": 4,
"archive_present": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
- "archive_entry_count": 586,
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
+ "archive_entry_count": 609,
"failure_count": 0,
"warning_count": 0
},
@@ -20072,14 +20200,14 @@
"install_simulation": {
"ok": true,
"schema_version": "2.0",
- "generated_at": "2026-06-13",
+ "generated_at": "2026-06-15",
"skill_dir": ".",
- "package_dir": "tests/tmp_review_studio/dist",
- "install_root": "tests/tmp_review_studio/install-root/simulate-yao-meta-skill",
- "installed_skill_dir": "tests/tmp_review_studio/install-root/simulate-yao-meta-skill/yao-meta-skill",
+ "package_dir": "dist",
+ "install_root": "[temporary-install-root]",
+ "installed_skill_dir": "[temporary-install-root]/yao-meta-skill",
"summary": {
"archive_present": true,
- "archive_entry_count": 603,
+ "archive_entry_count": 609,
"archive_extracted": true,
"entrypoint_loaded": true,
"manifest_loaded": true,
@@ -20089,7 +20217,7 @@
"installer_permission_failure_count": 0,
"permission_target_count": 4,
"permission_capability_count": 3,
- "install_root_is_temp": false,
+ "install_root_is_temp": true,
"failure_count": 0,
"warning_count": 0
},
@@ -20097,7 +20225,7 @@
{
"id": "archive-present",
"status": "pass",
- "detail": "Package archive exists: tests/tmp_review_studio/dist/yao-meta-skill.zip"
+ "detail": "Package archive exists: dist/yao-meta-skill.zip"
},
{
"id": "archive-safe-paths",
@@ -20343,10 +20471,10 @@
"failures": [],
"warnings": [],
"artifacts": {
- "archive": "tests/tmp_review_studio/dist/yao-meta-skill.zip",
- "package_manifest": "tests/tmp_review_studio/dist/manifest.json",
- "json": "tests/tmp_review_studio/install_simulation.json",
- "markdown": "tests/tmp_review_studio/install_simulation.md"
+ "archive": "dist/yao-meta-skill.zip",
+ "package_manifest": "dist/manifest.json",
+ "json": "reports/install_simulation.json",
+ "markdown": "reports/install_simulation.md"
}
},
"upgrade_check": {
@@ -20426,7 +20554,7 @@
{
"field": "package_sha256",
"from": "0000000000000000000000000000000000000000000000000000000000000000",
- "to": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6"
+ "to": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306"
}
]
},
diff --git a/reports/review-viewer.json b/reports/review-viewer.json
index bae1199..7517a52 100644
--- a/reports/review-viewer.json
+++ b/reports/review-viewer.json
@@ -513,7 +513,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 107
+ "file_count": 109
},
{
"path": "evals",
@@ -528,7 +528,7 @@
"file_count": 215
}
],
- "file_count": 389,
+ "file_count": 391,
"folder_count": 4,
"distribution": [
{
@@ -553,7 +553,7 @@
},
{
"label": "scripts",
- "value": 107
+ "value": 109
},
{
"label": "evals",
@@ -683,7 +683,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 107
+ "file_count": 109
},
{
"path": "evals",
@@ -996,13 +996,13 @@
"ok": true,
"summary": {
"reproducibility_ready": true,
- "release_lock_ready": false,
+ "release_lock_ready": true,
"methodology_complete": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "evidence_bundle_sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc",
- "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"output_case_count": 5,
"failure_disclosure_count": 3,
"command_count": 22,
@@ -1020,11 +1020,11 @@
"world_class_source_pass_count": 6,
"world_class_source_blocked_count": 7,
"public_claim_ready": false,
- "public_claim_blocker_count": 5,
- "working_tree_dirty": true,
- "changed_file_count": 12
+ "public_claim_blocker_count": 4,
+ "working_tree_dirty": false,
+ "changed_file_count": 0
},
- "commit": "6e2dfd48d080901f09b4e67bdf35b9500ec3dcec",
+ "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"missing_artifacts": [],
"limitations": [
"The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.",
@@ -1108,9 +1108,9 @@
"failures": []
},
"trust_security": {
- "scanned_files": 194,
- "script_count": 107,
- "internal_module_count": 26,
+ "scanned_files": 196,
+ "script_count": 109,
+ "internal_module_count": 28,
"secret_findings": 0,
"dependency_files": [
"requirements-ci.txt"
@@ -1128,8 +1128,8 @@
"help_smoke_failed_count": 0,
"interactive_script_count": 0,
"package_hash_scope": "source-contract-without-generated-reports",
- "package_hash_file_count": 194,
- "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306"
+ "package_hash_file_count": 196,
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4"
},
"skill_atlas": {
"skill_count": 12,
@@ -1167,8 +1167,8 @@
"trust_level": "local",
"license": "MIT",
"checksums": {
- "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc"
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8"
},
"compatibility": {
"openai": "pass",
@@ -1199,12 +1199,12 @@
},
"distribution": {
"archive_verified": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"package_verification": "reports/package_verification.json",
"install_simulated": true,
"install_simulation": "reports/install_simulation.json"
},
- "generated_at": "2026-06-13"
+ "generated_at": "2026-06-15"
},
"failures": [],
"warnings": []
@@ -1215,8 +1215,8 @@
"target_count": 4,
"adapter_count": 4,
"archive_present": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
- "archive_entry_count": 586,
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
+ "archive_entry_count": 609,
"failure_count": 0,
"warning_count": 0
},
@@ -1227,7 +1227,7 @@
"ok": true,
"summary": {
"archive_present": true,
- "archive_entry_count": 607,
+ "archive_entry_count": 609,
"archive_extracted": true,
"entrypoint_loaded": true,
"manifest_loaded": true,
@@ -1237,7 +1237,7 @@
"installer_permission_failure_count": 0,
"permission_target_count": 4,
"permission_capability_count": 3,
- "install_root_is_temp": false,
+ "install_root_is_temp": true,
"failure_count": 0,
"warning_count": 0
},
diff --git a/reports/skill-interpretation.html b/reports/skill-interpretation.html
index 6e8bccf..c963e30 100644
--- a/reports/skill-interpretation.html
+++ b/reports/skill-interpretation.html
@@ -930,7 +930,7 @@
让 reviewer 快速确认关键文件、目录和资产分布。 Lets reviewers confirm key files, directories, and asset distribution quickly.
-
资产分布 Asset Distribution 389项 389 items SKILL.md SKILL.md README.md README.md agents/interface.yaml agents/interface.yaml manifest.json manifest.json references references scripts scripts 资产分布图展示当前包体的文件和目录重心。 The asset distribution chart shows where files and directories are concentrated.
+
资产分布 Asset Distribution 391项 391 items SKILL.md SKILL.md README.md README.md agents/interface.yaml agents/interface.yaml manifest.json manifest.json references references scripts scripts 资产分布图展示当前包体的文件和目录重心。 The asset distribution chart shows where files and directories are concentrated.
路径 Path 作用 Role 类型 Type
SKILL.md Skill 入口文件 Skill entrypoint 文件 file README.md 人类可读使用说明 Human-readable usage guide 文件 file agents/interface.yaml 跨平台接口元数据 Neutral interface metadata 文件 file manifest.json 生命周期与打包元数据 Lifecycle and portability metadata 文件 file references 扩展指导与复用资料 Extended guidance and reusable notes 目录 folder scripts 确定性脚本或本地工具 Deterministic helpers or local tooling 目录 folder evals 触发与质量检查 Trigger and quality checks 目录 folder reports 生成的证据与总结报告 Generated evidence and overview artifacts 目录 folder
diff --git a/reports/skill-interpretation.json b/reports/skill-interpretation.json
index a7c772d..80a56ea 100644
--- a/reports/skill-interpretation.json
+++ b/reports/skill-interpretation.json
@@ -513,7 +513,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 107
+ "file_count": 109
},
{
"path": "evals",
@@ -528,7 +528,7 @@
"file_count": 215
}
],
- "file_count": 389,
+ "file_count": 391,
"folder_count": 4,
"distribution": [
{
@@ -553,7 +553,7 @@
},
{
"label": "scripts",
- "value": 107
+ "value": 109
},
{
"label": "evals",
@@ -687,7 +687,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 107
+ "file_count": 109
},
{
"path": "evals",
@@ -1004,9 +1004,9 @@
"methodology_complete": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "evidence_bundle_sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc",
- "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"output_case_count": 5,
"failure_disclosure_count": 3,
"command_count": 22,
@@ -1028,7 +1028,7 @@
"working_tree_dirty": false,
"changed_file_count": 0
},
- "commit": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e",
+ "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"missing_artifacts": [],
"limitations": [
"The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.",
@@ -1112,9 +1112,9 @@
"failures": []
},
"trust_security": {
- "scanned_files": 194,
- "script_count": 107,
- "internal_module_count": 26,
+ "scanned_files": 196,
+ "script_count": 109,
+ "internal_module_count": 28,
"secret_findings": 0,
"dependency_files": [
"requirements-ci.txt"
@@ -1132,8 +1132,8 @@
"help_smoke_failed_count": 0,
"interactive_script_count": 0,
"package_hash_scope": "source-contract-without-generated-reports",
- "package_hash_file_count": 194,
- "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306"
+ "package_hash_file_count": 196,
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4"
},
"skill_atlas": {
"skill_count": 12,
@@ -1171,8 +1171,8 @@
"trust_level": "local",
"license": "MIT",
"checksums": {
- "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc"
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8"
},
"compatibility": {
"openai": "pass",
@@ -1203,12 +1203,12 @@
},
"distribution": {
"archive_verified": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"package_verification": "reports/package_verification.json",
"install_simulated": true,
"install_simulation": "reports/install_simulation.json"
},
- "generated_at": "2026-06-13"
+ "generated_at": "2026-06-15"
},
"failures": [],
"warnings": []
@@ -1219,8 +1219,8 @@
"target_count": 4,
"adapter_count": 4,
"archive_present": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
- "archive_entry_count": 586,
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
+ "archive_entry_count": 609,
"failure_count": 0,
"warning_count": 0
},
@@ -1231,7 +1231,7 @@
"ok": true,
"summary": {
"archive_present": true,
- "archive_entry_count": 607,
+ "archive_entry_count": 609,
"archive_extracted": true,
"entrypoint_loaded": true,
"manifest_loaded": true,
@@ -1241,7 +1241,7 @@
"installer_permission_failure_count": 0,
"permission_target_count": 4,
"permission_capability_count": 3,
- "install_root_is_temp": false,
+ "install_root_is_temp": true,
"failure_count": 0,
"warning_count": 0
},
diff --git a/reports/skill-overview.html b/reports/skill-overview.html
index 4936f8c..249c25d 100644
--- a/reports/skill-overview.html
+++ b/reports/skill-overview.html
@@ -930,7 +930,7 @@
让 reviewer 快速确认关键文件、目录和资产分布。 Lets reviewers confirm key files, directories, and asset distribution quickly.
-
资产分布 Asset Distribution 389项 389 items SKILL.md SKILL.md README.md README.md agents/interface.yaml agents/interface.yaml manifest.json manifest.json references references scripts scripts 资产分布图展示当前包体的文件和目录重心。 The asset distribution chart shows where files and directories are concentrated.
+
资产分布 Asset Distribution 391项 391 items SKILL.md SKILL.md README.md README.md agents/interface.yaml agents/interface.yaml manifest.json manifest.json references references scripts scripts 资产分布图展示当前包体的文件和目录重心。 The asset distribution chart shows where files and directories are concentrated.
路径 Path 作用 Role 类型 Type
SKILL.md Skill 入口文件 Skill entrypoint 文件 file README.md 人类可读使用说明 Human-readable usage guide 文件 file agents/interface.yaml 跨平台接口元数据 Neutral interface metadata 文件 file manifest.json 生命周期与打包元数据 Lifecycle and portability metadata 文件 file references 扩展指导与复用资料 Extended guidance and reusable notes 目录 folder scripts 确定性脚本或本地工具 Deterministic helpers or local tooling 目录 folder evals 触发与质量检查 Trigger and quality checks 目录 folder reports 生成的证据与总结报告 Generated evidence and overview artifacts 目录 folder
diff --git a/reports/skill-overview.json b/reports/skill-overview.json
index 25bb730..e82c79e 100644
--- a/reports/skill-overview.json
+++ b/reports/skill-overview.json
@@ -512,7 +512,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 107
+ "file_count": 109
},
{
"path": "evals",
@@ -527,7 +527,7 @@
"file_count": 215
}
],
- "file_count": 389,
+ "file_count": 391,
"folder_count": 4,
"distribution": [
{
@@ -552,7 +552,7 @@
},
{
"label": "scripts",
- "value": 107
+ "value": 109
},
{
"label": "evals",
@@ -682,7 +682,7 @@
"path": "scripts",
"label": "Deterministic helpers or local tooling",
"kind": "folder",
- "file_count": 107
+ "file_count": 109
},
{
"path": "evals",
@@ -999,9 +999,9 @@
"methodology_complete": true,
"required_artifact_count": 24,
"missing_artifact_count": 0,
- "evidence_bundle_sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc",
- "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3",
+ "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"output_case_count": 5,
"failure_disclosure_count": 3,
"command_count": 22,
@@ -1023,7 +1023,7 @@
"working_tree_dirty": false,
"changed_file_count": 0
},
- "commit": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e",
+ "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8",
"missing_artifacts": [],
"limitations": [
"The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.",
@@ -1107,9 +1107,9 @@
"failures": []
},
"trust_security": {
- "scanned_files": 194,
- "script_count": 107,
- "internal_module_count": 26,
+ "scanned_files": 196,
+ "script_count": 109,
+ "internal_module_count": 28,
"secret_findings": 0,
"dependency_files": [
"requirements-ci.txt"
@@ -1127,8 +1127,8 @@
"help_smoke_failed_count": 0,
"interactive_script_count": 0,
"package_hash_scope": "source-contract-without-generated-reports",
- "package_hash_file_count": 194,
- "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306"
+ "package_hash_file_count": 196,
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4"
},
"skill_atlas": {
"skill_count": 12,
@@ -1166,8 +1166,8 @@
"trust_level": "local",
"license": "MIT",
"checksums": {
- "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306",
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc"
+ "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8"
},
"compatibility": {
"openai": "pass",
@@ -1198,12 +1198,12 @@
},
"distribution": {
"archive_verified": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
"package_verification": "reports/package_verification.json",
"install_simulated": true,
"install_simulation": "reports/install_simulation.json"
},
- "generated_at": "2026-06-13"
+ "generated_at": "2026-06-15"
},
"failures": [],
"warnings": []
@@ -1214,8 +1214,8 @@
"target_count": 4,
"adapter_count": 4,
"archive_present": true,
- "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc",
- "archive_entry_count": 586,
+ "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8",
+ "archive_entry_count": 609,
"failure_count": 0,
"warning_count": 0
},
@@ -1226,7 +1226,7 @@
"ok": true,
"summary": {
"archive_present": true,
- "archive_entry_count": 607,
+ "archive_entry_count": 609,
"archive_extracted": true,
"entrypoint_loaded": true,
"manifest_loaded": true,
@@ -1236,7 +1236,7 @@
"installer_permission_failure_count": 0,
"permission_target_count": 4,
"permission_capability_count": 3,
- "install_root_is_temp": false,
+ "install_root_is_temp": true,
"failure_count": 0,
"warning_count": 0
},