diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json index 33adf36..fb98c44 100644 --- a/reports/benchmark_reproducibility.json +++ b/reports/benchmark_reproducibility.json @@ -3,7 +3,7 @@ "ok": true, "generated_at": "2026-06-15", "skill_dir": ".", - "commit": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e", + "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "git_status": { "available": true, "dirty": false, @@ -17,9 +17,9 @@ "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc", - "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 22, @@ -54,7 +54,7 @@ }, "release_lock": { "ready": true, - "commit": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e", + "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "status_scope": "generation-time status before this report is written", "reason": "clean generation-time HEAD" }, @@ -64,7 +64,7 @@ "existing_count": 24, "missing_count": 0, "missing_paths": [], - "sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc" + "sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -138,7 +138,7 @@ "path": "reports/output_execution_runs.json", "exists": true, "bytes": 7967, - "sha256": "cec7afbfb8daa00e5fcfb3c79dac10163e312e307431896cc0f62b0f47416659" + "sha256": "a4c8cacaab5d2f5178513e20cfbb58787052e82607b9ec4fe3eaeae58d7509c8" }, { "label": "blind_review", @@ -165,50 +165,50 @@ "label": "runtime_conformance", "path": "reports/conformance_matrix.json", "exists": true, - "bytes": 10313, - "sha256": "8251329e663dda51472f29b7721e73d72ccbec9760d96fda022f6218a5a6e347" + "bytes": 10342, + "sha256": "97f9ba949c23a60b00e9ba2ff279ca03ba517845cc7c62aa8a42645d58006c7e" }, { "label": "trust_report", "path": "reports/security_trust_report.json", "exists": true, - "bytes": 106773, - "sha256": "6409321f1c0d2a7b9208595e05360de8f3dbb56602c8cee61226fb8c31d61f32" + "bytes": 107994, + "sha256": "d7dae65ba9cfdc52001b73242ac88c84cf302bc6091e0df8805fdb79ac2594ce" }, { "label": "python_compatibility", "path": "reports/python_compatibility.json", "exists": true, - "bytes": 22510, - "sha256": "73c6c2a81af9980cc1d8e2d1bb5dfb30bd8daccac48125fe71ec359dbde824dd" + "bytes": 22768, + "sha256": "cdd46eb42010f406fb2732063bde224866a4493ae4be1dc0c835417550a45d18" }, { "label": "registry_audit", "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, - "sha256": "76a55d6dfc15afe8f461e37faa1a3661650a09eb3b2c84a8d7c3b04abae4623e" + "sha256": "9ca263ecce9ace6b3915b2485eb7807edaf375d1aeccc689bc8b2d91417d8d2e" }, { "label": "package_verification", "path": "reports/package_verification.json", "exists": true, "bytes": 19338, - "sha256": "2476ae8ec9c491a3f7f31ec3f0c6c78cceead8e80700dcb696838eec226c675b" + "sha256": "1abff9930aa56d337b47dbc421f293eaa9eb3620c187d1b45a26941594b392ef" }, { "label": "install_simulation", "path": "reports/install_simulation.json", "exists": true, - "bytes": 8758, - "sha256": "490e1f665580f5d9794b05bd0fcc57342dc66971776577037176f181a5c2c875" + "bytes": 8557, + "sha256": "958586fa59a71aeea10962c6047083876fbcd4b809f24fbeaec4ece8aa499dfc" }, { "label": "skill_os2_audit", "path": "reports/skill_os2_audit.json", "exists": true, "bytes": 14310, - "sha256": "a4cf40478f3ad9404a9cb3ccff9f152481703b8936c33841ae898fd589b78fa0" + "sha256": "55526d80bfa78b574f990442165a0159b0f177feca00523826657b00f7855d29" }, { "label": "world_class_evidence_plan", diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md index b4adb9f..87ea8de 100644 --- a/reports/benchmark_reproducibility.md +++ b/reports/benchmark_reproducibility.md @@ -1,9 +1,9 @@ # Benchmark Reproducibility Generated at: `2026-06-15` -Commit: `2df5c9acd093f45e27318bbe5ff984fae3a5fd8e` +Commit: `0e6327ca9399c6995b192fdf710fa4c99699c2c8` Working tree dirty at generation: `false` -Evidence bundle SHA256: `4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc` +Evidence bundle SHA256: `885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3` ## Summary @@ -12,8 +12,8 @@ Evidence bundle SHA256: `4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f - methodology complete: `true` - required artifacts: `24` - missing artifacts: `0` -- source contract sha256: `4660a11db949` -- archive sha256: `6852cf91a74d` +- source contract sha256: `76edc3f9936d` +- archive sha256: `798153ff8e15` - output cases: `5` - disclosed failure cases: `3` - reproduction commands: `22` @@ -50,7 +50,7 @@ This report proves local benchmark reproducibility only. It keeps external provi - algorithm: `sha256(path,label,exists,artifact_sha256)` - artifacts: `24` / `24` -- sha256: `4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc` +- sha256: `885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3` ## Methodology Sections @@ -72,17 +72,17 @@ This report proves local benchmark reproducibility only. It keeps external provi | output_cases | `evals/output/cases.jsonl` | present | `a6ae96857116` | | output_schema | `evals/output/schema.json` | present | `8ee340c95064` | | output_scorecard | `reports/output_quality_scorecard.json` | present | `0806258a8e08` | -| output_execution | `reports/output_execution_runs.json` | present | `cec7afbfb8da` | +| output_execution | `reports/output_execution_runs.json` | present | `a4c8cacaab5d` | | blind_review | `reports/output_blind_review_pack.json` | present | `bbe2db8ec277` | | review_adjudication | `reports/output_review_adjudication.json` | present | `240485a721af` | | trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` | -| runtime_conformance | `reports/conformance_matrix.json` | present | `8251329e663d` | -| trust_report | `reports/security_trust_report.json` | present | `6409321f1c0d` | -| python_compatibility | `reports/python_compatibility.json` | present | `73c6c2a81af9` | -| registry_audit | `reports/registry_audit.json` | present | `76a55d6dfc15` | -| package_verification | `reports/package_verification.json` | present | `2476ae8ec9c4` | -| install_simulation | `reports/install_simulation.json` | present | `490e1f665580` | -| skill_os2_audit | `reports/skill_os2_audit.json` | present | `a4cf40478f3a` | +| runtime_conformance | `reports/conformance_matrix.json` | present | `97f9ba949c23` | +| trust_report | `reports/security_trust_report.json` | present | `d7dae65ba9cf` | +| python_compatibility | `reports/python_compatibility.json` | present | `cdd46eb42010` | +| registry_audit | `reports/registry_audit.json` | present | `9ca263ecce9a` | +| package_verification | `reports/package_verification.json` | present | `1abff9930aa5` | +| install_simulation | `reports/install_simulation.json` | present | `958586fa59a7` | +| skill_os2_audit | `reports/skill_os2_audit.json` | present | `55526d80bfa7` | | world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `933cdb002181` | | world_class_evidence_ledger | `reports/world_class_evidence_ledger.json` | present | `5407409841eb` | | world_class_evidence_intake | `reports/world_class_evidence_intake.json` | present | `b10e1ce0a5a1` | diff --git a/reports/evidence_consistency.json b/reports/evidence_consistency.json index a4241c8..eda8400 100644 --- a/reports/evidence_consistency.json +++ b/reports/evidence_consistency.json @@ -48,8 +48,8 @@ "key": "overview-benchmark-commit", "label": "overview embeds the benchmark commit", "status": "pass", - "expected": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e", - "actual": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e", + "expected": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", + "actual": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-overview.json" @@ -64,8 +64,8 @@ "release_lock_ready": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, "world_class_source_pass_count": 6, @@ -77,8 +77,8 @@ "release_lock_ready": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, "world_class_source_pass_count": 6, @@ -194,8 +194,8 @@ "key": "interpretation-benchmark-commit", "label": "interpretation embeds the benchmark commit", "status": "pass", - "expected": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e", - "actual": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e", + "expected": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", + "actual": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-interpretation.json" @@ -210,8 +210,8 @@ "release_lock_ready": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, "world_class_source_pass_count": 6, @@ -223,8 +223,8 @@ "release_lock_ready": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, "world_class_source_pass_count": 6, @@ -1312,7 +1312,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 107 + "file_count": 109 }, { "path": "evals", @@ -1327,7 +1327,7 @@ "file_count": 215 } ], - "file_count": 389, + "file_count": 391, "folder_count": 4, "distribution": [ { @@ -1352,7 +1352,7 @@ }, { "label": "scripts", - "value": 107 + "value": 109 }, { "label": "evals", @@ -1400,7 +1400,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 107 + "file_count": 109 }, { "path": "evals", @@ -1415,7 +1415,7 @@ "file_count": 215 } ], - "file_count": 389, + "file_count": 391, "folder_count": 4, "distribution": [ { @@ -1440,7 +1440,7 @@ }, { "label": "scripts", - "value": 107 + "value": 109 }, { "label": "evals", diff --git a/reports/review-studio.html b/reports/review-studio.html index 10a710a..f298284 100644 --- a/reports/review-studio.html +++ b/reports/review-studio.html @@ -676,22 +676,22 @@

核心指标

-
Skill IR2.0.0

5 targets in platform-neutral contract

Compiler5/5

target contracts compiled from Skill IR

Output Delta100.0

5 cases; 1 file-backed

Exec Runs10

command 10; model 0; recorded 0

Blind A/B5

review pairs hide baseline vs with-skill labels

Review Kit0/5

pending 5; answer key hidden

Review A/B0/5

adjudication decisions; pending 5

Public Claimblocked

5 blockers; local reproducible true

Blueprint20/20

2.0 coverage; extensions partial 1, planned 0; evidence pending 4

Runtime5/5

target conformance pass rate

Perm Probe4/4

0 native; 4 installer-enforced

Trust0

106 scripts scanned; secrets found

Py Compat0

171 files scanned for Python 3.11

Arch Debt0

886 largest lines; 34 CLI handlers

Atlas5

12 scanned skills; route collisions

Driftlow

1 metadata events; 0 missed triggers

Waivers0

0 gates covered; human risk decisions

Intake4/4

0 valid submissions; 0 invalid

Claim Guard0

76 public surfaces scanned

Notes0/0

0 open blocker annotations

Registry1.1.0

5 targets; MIT license

Archivepass

586 zip entries; package verification

Installpass

4 adapters; 12 permissions enforced; 0 permission failures

Upgrademinor

declared minor; 0 breaking changes

+
Skill IR2.0.0

5 targets in platform-neutral contract

Compiler5/5

target contracts compiled from Skill IR

Output Delta100.0

5 cases; 1 file-backed

Exec Runs10

command 10; model 0; recorded 0

Blind A/B5

review pairs hide baseline vs with-skill labels

Review Kit0/5

pending 5; answer key hidden

Review A/B0/5

adjudication decisions; pending 5

Public Claimblocked

4 blockers; local reproducible true

Blueprint21/21

2.0 coverage; extensions partial 1, planned 0; evidence pending 4

Runtime5/5

target conformance pass rate

Perm Probe4/4

0 native; 4 installer-enforced

Trust0

109 scripts scanned; secrets found

Py Compat0

175 files scanned for Python 3.11

Arch Debt0

897 largest lines; 63 CLI handlers; 18 in entrypoint

Atlas5

12 scanned skills; route collisions

Driftlow

1 metadata events; 0 missed triggers

Waivers0

0 gates covered; human risk decisions

Intake4/4

0 valid submissions; 0 invalid

Claim Guard0

77 public surfaces scanned

Notes0/0

0 open blocker annotations

Registry1.1.0

5 targets; MIT license

Archivepass

609 zip entries; package verification

Installpass

4 adapters; 12 permissions enforced; 0 permission failures

Upgrademinor

declared minor; 0 breaking changes

审查闸门

-
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 960/1000; deferred 416415/120000; top deferred scripts 367825; resource governance governed; quality density 135.4

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 106 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

Python 兼容

Python 3.11; 171 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards

reports/python_compatibility.json 证据
通过

架构维护

168 Python files; 0 hotspots; 0 blockers; largest 886 lines; 34 CLI handlers

reports/architecture_maintainability.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 100.0; missed 0; bad-output 0; risk low

reports/adoption_drift_report.json 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 7/13 pass; 6 blocked; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

reports/promotion_decisions.json + reports/upgrade_check.json + docs/migration-v2.md 证据
+
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 960/1000; deferred 421932/120000; top deferred scripts 373342; resource governance governed; quality density 135.4

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 109 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

Python 兼容

Python 3.11; 175 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards

reports/python_compatibility.json 证据
通过

架构维护

172 Python files; 0 hotspots; 0 blockers; largest 897 lines; 63 CLI handlers; 18 in entrypoint

reports/architecture_maintainability.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low

reports/adoption_drift_report.json 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

阻断事项

无。

-

关注事项

+

关注事项

修复动作

-
关注

输出实验

补足 output eval 覆盖、execution evidence、blind A/B 和 reviewer adjudication。

没有输出质量和人工盲评证据时,Skill 只能证明会触发,不能证明输出真的更好且经得起审查。
修复位置
evals/output/cases.jsonl + reports/output_quality_scorecard.md + reports/output_review_kit.html + reports/output_review_adjudication.md
验证命令
python3 scripts/adjudicate_output_review.py --write-template && python3 scripts/yao.py output-review
关注

人工批准

对保留的 warning 写入 reviewer、理由、范围和到期时间,或修掉 warning。

warning 可以被接受,但必须可审计、会过期,并且不能掩盖 blocker。
修复位置
reports/review_waivers.md
验证命令
python3 scripts/render_review_waivers.py .
关注

世界证据

补齐 provider、真人盲评、原生权限执行和真实客户端遥测证据,或明确本次发布不声明 world-class 完成。

世界级结论必须来自已接受的外部/人工证据;计划、metadata fallback、待评审和本地命令都不能替代完成证据。
修复位置
reports/world_class_operator_runbook.html + reports/world_class_evidence_ledger.md + reports/world_class_evidence_intake.md + reports/world_class_submission_review.md
验证命令
python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py review-studio .

证据采集

以下条目仍需真实外部或人工证据;提交文件、校验命令和阻断检查必须同时闭环。

pending · external

Provider Holdout

model-executed 0; token-observed 0

提交
evidence/world_class/submissions/provider-holdout.json
模板
evidence/world_class/templates/provider-holdout.intake.json
阻断
2 blocked / 1 pass
下一步
Run provider-backed holdout cases with real credentials and commit only aggregate evidence.
阻断检查
  • Provider model runmodel_executed_count: 0 / >0Run provider-backed output-exec with real credentials.
  • Token usage observedtoken_observed_count: 0 / >0Provider execution should return non-estimated token usage.
操作命令
  • 准备提交python3 scripts/yao.py world-class-submission-kit . --evidence-key provider-holdout --output-dir evidence/world_class/submissions
  • 校验入口python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions
  • 审查提交python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions
  • 刷新台账python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions
首要步骤
  1. YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY=<redacted> python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60
  2. python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  3. Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields.
pending · human

Human Adjudication

0/5 decisions; pending 5

提交
evidence/world_class/submissions/human-adjudication.json
模板
evidence/world_class/templates/human-adjudication.intake.json
阻断
2 blocked / 2 pass
下一步
Record real A/B choices in the decision template, then regenerate adjudication.
阻断检查
  • No pending decisionspending_count: 5 / ==0Record a reviewer choice for every pair.
  • Judgments completejudgment_count: 0 / ==pair_countEvery pair needs one valid human judgment.
操作命令
  • 准备提交python3 scripts/yao.py world-class-submission-kit . --evidence-key human-adjudication --output-dir evidence/world_class/submissions
  • 校验入口python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions
  • 审查提交python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions
  • 刷新台账python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions
首要步骤
  1. python3 scripts/yao.py output-review-kit --write-template
  2. Open reports/output_review_kit.md and choose A or B for each pair without opening the answer key.
  3. python3 scripts/adjudicate_output_review.py --write-template
pending · external

Native Permission Enforcement

native-enforced targets 0; installer-enforced targets 4

提交
evidence/world_class/submissions/native-permission-enforcement.json
模板
evidence/world_class/templates/native-permission-enforcement.intake.json
阻断
1 blocked / 2 pass
下一步
Integrate a real target-client or external installer runtime guard before claiming native permission enforcement.
阻断检查
  • Native enforcementnative_enforcement_count: 0 / >0Collect real target-client or external runtime guard proof.
操作命令
  • 准备提交python3 scripts/yao.py world-class-submission-kit . --evidence-key native-permission-enforcement --output-dir evidence/world_class/submissions
  • 校验入口python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions
  • 审查提交python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions
  • 刷新台账python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions
首要步骤
  1. Implement or connect a real target client or external installer runtime guard that blocks undeclared network, file_write, or subprocess capabilities.
  2. Update the generated target adapter only when the guard is actually enforced by that target.
  3. python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip
pending · external

Native Client Telemetry

external source events 0; adoption samples 1

提交
evidence/world_class/submissions/native-client-telemetry.json
模板
evidence/world_class/templates/native-client-telemetry.intake.json
阻断
1 blocked / 2 pass
下一步
Install a real client against the native host and import production metadata-only events.
阻断检查
  • External eventsexternal_source_events: 0 / >0Import at least one metadata-only event from a real client.
操作命令
  • 准备提交python3 scripts/yao.py world-class-submission-kit . --evidence-key native-client-telemetry --output-dir evidence/world_class/submissions
  • 校验入口python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions
  • 审查提交python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions
  • 刷新台账python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions
首要步骤
  1. python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension://<extension-id>/
  2. Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event.
  3. python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl
+
关注

输出实验

补足 output eval 覆盖、execution evidence、blind A/B 和 reviewer adjudication。

没有输出质量和人工盲评证据时,Skill 只能证明会触发,不能证明输出真的更好且经得起审查。
修复位置
evals/output/cases.jsonl + reports/output_quality_scorecard.md + reports/output_review_kit.html + reports/output_review_adjudication.md
验证命令
python3 scripts/adjudicate_output_review.py --write-template && python3 scripts/yao.py output-review
关注

人工批准

对保留的 warning 写入 reviewer、理由、范围和到期时间,或修掉 warning。

warning 可以被接受,但必须可审计、会过期,并且不能掩盖 blocker。
修复位置
reports/review_waivers.md
验证命令
python3 scripts/render_review_waivers.py .
关注

世界证据

补齐 provider、真人盲评、原生权限执行和真实客户端遥测证据,或明确本次发布不声明 world-class 完成。

世界级结论必须来自已接受的外部/人工证据;计划、metadata fallback、待评审和本地命令都不能替代完成证据。
修复位置
reports/world_class_operator_runbook.html + reports/world_class_evidence_ledger.md + reports/world_class_evidence_intake.md + reports/world_class_submission_review.md
验证命令
python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py review-studio .

证据采集

以下条目仍需真实外部或人工证据;提交文件、校验命令和阻断检查必须同时闭环。

pending · external

Provider Holdout

model-executed 0; token-observed 0

提交
evidence/world_class/submissions/provider-holdout.json
模板
evidence/world_class/templates/provider-holdout.intake.json
阻断
2 blocked / 1 pass
下一步
Run provider-backed holdout cases with real credentials and commit only aggregate evidence.
阻断检查
  • Provider model runmodel_executed_count: 0 / >0Run provider-backed output-exec with real credentials.
  • Token usage observedtoken_observed_count: 0 / >0Provider execution should return non-estimated token usage.
操作命令
  • 准备提交python3 scripts/yao.py world-class-submission-kit . --evidence-key provider-holdout --output-dir evidence/world_class/submissions
  • 校验入口python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions
  • 审查提交python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions
  • 刷新台账python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions
首要步骤
  1. YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY=<redacted> python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60
  2. python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  3. Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields.
pending · human

Human Adjudication

0/5 decisions; pending 5

提交
evidence/world_class/submissions/human-adjudication.json
模板
evidence/world_class/templates/human-adjudication.intake.json
阻断
2 blocked / 2 pass
下一步
Record real A/B choices in the decision template, then regenerate adjudication.
阻断检查
  • No pending decisionspending_count: 5 / ==0Record a reviewer choice for every pair.
  • Judgments completejudgment_count: 0 / ==pair_countEvery pair needs one valid human judgment.
操作命令
  • 准备提交python3 scripts/yao.py world-class-submission-kit . --evidence-key human-adjudication --output-dir evidence/world_class/submissions
  • 校验入口python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions
  • 审查提交python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions
  • 刷新台账python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions
首要步骤
  1. python3 scripts/yao.py output-review-kit --write-template
  2. Open reports/output_review_kit.md and choose A or B for each pair without opening the answer key.
  3. python3 scripts/adjudicate_output_review.py --write-template
pending · external

Native Permission Enforcement

native-enforced targets 0; installer-enforced targets 4

提交
evidence/world_class/submissions/native-permission-enforcement.json
模板
evidence/world_class/templates/native-permission-enforcement.intake.json
阻断
1 blocked / 2 pass
下一步
Integrate a real target-client or external installer runtime guard before claiming native permission enforcement.
阻断检查
  • Native enforcementnative_enforcement_count: 0 / >0Collect real target-client or external runtime guard proof.
操作命令
  • 准备提交python3 scripts/yao.py world-class-submission-kit . --evidence-key native-permission-enforcement --output-dir evidence/world_class/submissions
  • 校验入口python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions
  • 审查提交python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions
  • 刷新台账python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions
首要步骤
  1. Implement or connect a real target client or external installer runtime guard that blocks undeclared network, file_write, or subprocess capabilities.
  2. Update the generated target adapter only when the guard is actually enforced by that target.
  3. python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip
pending · external

Native Client Telemetry

external source events 0; adoption samples 0

提交
evidence/world_class/submissions/native-client-telemetry.json
模板
evidence/world_class/templates/native-client-telemetry.intake.json
阻断
2 blocked / 1 pass
下一步
Install a real client against the native host and import production metadata-only events.
阻断检查
  • External eventsexternal_source_events: 0 / >0Import at least one metadata-only event from a real client.
  • Adoption sampleadoption_sample_count: 0 / >0Telemetry must include adoption outcome evidence.
操作命令
  • 准备提交python3 scripts/yao.py world-class-submission-kit . --evidence-key native-client-telemetry --output-dir evidence/world_class/submissions
  • 校验入口python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions
  • 审查提交python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions
  • 刷新台账python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions
首要步骤
  1. python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension://<extension-id>/
  2. Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event.
  3. python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl
@@ -743,17 +743,17 @@
-

上下文

initial load 960/1000; deferred 416415/120000; top deferred scripts 367825; resource governance governed; quality density 135.4

+

上下文

initial load 960/1000; deferred 421932/120000; top deferred scripts 373342; resource governance governed; quality density 135.4

编译证据

Review reports/compiled_targets.md before packaging to inspect target adapter modes, generated files, preserved semantics, warnings, and unsupported features.

-

信任报告

Secret
0
脚本数
106
网络脚本
3
Help 失败
0
包体哈希
28625ce8b3fcb3e214ef3b8ffc748d9cff551e4c5a1b78786c1a2b7d7005f679
+

信任报告

Secret
0
脚本数
109
网络脚本
3
Help 失败
0
包体哈希
76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4

安全边界

高风险 secret、远程 inline execution、缺失依赖策略或无法解释的脚本接口应阻断 governed release。

-

Python 兼容

目标 Python
3.11
文件数
171
问题数
0
语法错误
0
F-string 3.11
0
+

Python 兼容

目标 Python
3.11
文件数
175
问题数
0
语法错误
0
F-string 3.11
0

解释器边界

CI 和发布审查以 Python 3.11 兼容为底线;本地更高版本允许的新语法不能绕过兼容门禁。

@@ -773,8 +773,8 @@
-

运营回路

1 metadata events; adoption 100.0; missed 0; bad-output 0; risk low

-

漂移信号

事件数
1
采用率
100
漏触发
0
Bad Output Count
0
风险带
low
+

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low

+

漂移信号

事件数
1
采用率
0
漏触发
0
Bad Output Count
0
风险带
low
@@ -785,13 +785,13 @@

批准台账

Waiver Count
0
Active Count
0
Expired Count
0
Invalid Count
0
覆盖 Gate
0

批准候选

-
可批准 · needs-reviewer-decision

Output Lab

review pending 5; model-executed 0; output failures 0

Gate
output-lab
证据
reports/output_review_adjudication.md
选项
accepted-risk, false-positive, temporary-exception
边界
Does not count as provider, human, or public world-class completion evidence.

审查条件

  • Reviewer confirms this release does not claim provider-backed or human-adjudicated output superiority.
  • Reviewer names the release scope and expiry date.
  • Reviewer links output_review_adjudication or output_execution evidence.
建议命令python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer "<reviewer>" --reason "Output Lab has pending human/provider evidence; accepted only for this bounded review scope." --expires-at 2027-06-13 --evidence reports/output_review_adjudication.md
不可批准 · cannot-waive

World-Class Evidence

4 pending evidence entries; 1 human pending; 3 external pending

Gate
world-class-evidence
证据
reports/world_class_evidence_ledger.md
选项
边界
Non-waivable completion boundary.

审查条件

  • Do not use a waiver to claim public world-class readiness.
  • Either submit accepted ledger evidence or state that this release does not claim world-class completion.
  • Keep claim guard active until ledger summary.ready_to_claim_world_class is true.
建议命令python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-claim-guard .
+
可批准 · needs-reviewer-decision

Output Lab

review pending 5; model-executed 0; output failures 0

Gate
output-lab
证据
reports/output_review_adjudication.md
选项
accepted-risk, false-positive, temporary-exception
边界
Does not count as provider, human, or public world-class completion evidence.

审查条件

  • Reviewer confirms this release does not claim provider-backed or human-adjudicated output superiority.
  • Reviewer names the release scope and expiry date.
  • Reviewer links output_review_adjudication or output_execution evidence.
建议命令python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer "<reviewer>" --reason "Output Lab has pending human/provider evidence; accepted only for this bounded review scope." --expires-at 2027-06-15 --evidence reports/output_review_adjudication.md
不可批准 · cannot-waive

World-Class Evidence

4 pending evidence entries; 1 human pending; 3 external pending

Gate
world-class-evidence
证据
reports/world_class_evidence_ledger.md
选项
边界
Non-waivable completion boundary.

审查条件

  • Do not use a waiver to claim public world-class readiness.
  • Either submit accepted ledger evidence or state that this release does not claim world-class completion.
  • Keep claim guard active until ledger summary.ready_to_claim_world_class is true.
建议命令python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-claim-guard .

世界证据

这里列出每个 world-class 证据项的当前状态、完成定义、证据来源、隐私约束和下一步;计划、metadata fallback、待评审和本地命令不会被当成完成证据。

-
待补证 · external

Provider Holdout

Collect at least one provider-backed output-eval holdout run with model, timing, and token metadata.

负责人
operator with provider credentials
当前状态
model-executed 0; token-observed 0
下一步
Run provider-backed holdout cases with real credentials and commit only aggregate evidence.
观测值
model_executed_count: 0; timing_observed_count: 10; token_observed_count: 0; accepted: False
提交态
status: missing; path: evidence/world_class/submissions/provider-holdout.json; attested_real_evidence: False; privacy_contract_satisfied: False

执行步骤

  • YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY=<redacted> python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60
  • python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  • Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields.
  • python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions

完成定义

  • reports/output_execution_runs.json summary.model_executed_count > 0
  • reports/output_execution_runs.json summary.timing_observed_count > 0
  • reports/output_execution_runs.json summary.token_observed_count > 0
  • reports/skill_os2_audit.json item provider-holdout status becomes pass

证据来源

  • reports/output_execution_runs.json
  • reports/output_execution_runs.md
  • reports/skill_os2_audit.json
  • evidence/world_class/intake.schema.json
  • evidence/world_class/templates/provider-holdout.intake.json
  • reports/world_class_evidence_intake.json
  • reports/world_class_evidence_intake.md

隐私约束

  • Do not commit provider credentials or environment dumps.
  • The output execution report records output hashes and aggregate run metadata, not raw provider prompts.

源证据检查

  • Provider model runmodel_executed_count: 0 / >0Run provider-backed output-exec with real credentials.
  • Timing observedtiming_observed_count: 10 / >0Provider execution should record timing metadata.
  • Token usage observedtoken_observed_count: 0 / >0Provider execution should return non-estimated token usage.
待补证 · human

Human Adjudication

Record real blind A/B reviewer decisions before claiming human output review completion.

负责人
human reviewer
当前状态
0/5 decisions; pending 5
下一步
Record real A/B choices in the decision template, then regenerate adjudication.
观测值
pair_count: 5; judgment_count: 0; pending_count: 5; invalid_decision_count: 0; answer_revealed_count: 0; accepted: False
提交态
status: missing; path: evidence/world_class/submissions/human-adjudication.json; attested_real_evidence: False; privacy_contract_satisfied: False

执行步骤

  • python3 scripts/yao.py output-review-kit --write-template
  • Open reports/output_review_kit.md and choose A or B for each pair without opening the answer key.
  • python3 scripts/adjudicate_output_review.py --write-template
  • Edit reports/output_review_decisions.json with winner_variant values and reviewer metadata.
  • python3 scripts/yao.py output-review
  • python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  • Copy evidence/world_class/templates/human-adjudication.intake.json to evidence/world_class/submissions/human-adjudication.json and fill only real evidence fields.
  • python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions

完成定义

  • reports/output_review_adjudication.json summary.pending_count == 0
  • reports/output_review_adjudication.json summary.judgment_count == summary.pair_count
  • reports/output_review_adjudication.json summary.invalid_decision_count == 0
  • reports/skill_os2_audit.json item human-adjudication status becomes pass

证据来源

  • reports/output_blind_review_pack.md
  • reports/output_review_kit.md
  • reports/output_review_decisions.json
  • reports/output_review_adjudication.json
  • reports/output_review_adjudication.md
  • evidence/world_class/intake.schema.json
  • evidence/world_class/templates/human-adjudication.intake.json
  • reports/world_class_evidence_intake.json
  • reports/world_class_evidence_intake.md

隐私约束

  • Reviewer decisions should not include raw user data or private customer detail.
  • Keep the answer key separate until after decisions are recorded.

源证据检查

  • Review pairs existpair_count: 5 / >0Generate the blind A/B review pack.
  • No pending decisionspending_count: 5 / ==0Record a reviewer choice for every pair.
  • Judgments completejudgment_count: 0 / ==pair_countEvery pair needs one valid human judgment.
  • No invalid decisionsinvalid_decision_count: 0 / ==0Fix malformed winner/confidence entries.
待补证 · external

Native Permission Enforcement

Prove at least one real target client or external installer runtime guard enforces approved high-permission capabilities.

负责人
target client or installer integrator
当前状态
native-enforced targets 0; installer-enforced targets 4
下一步
Integrate a real target-client or external installer runtime guard before claiming native permission enforcement.
观测值
native_enforcement_count: 0; metadata_fallback_count: 4; installer_enforcement_pass_count: 4; installer_permission_failure_count: 0; installer_enforcement_ready: True; residual_risk_count: 4; failure_count: 0; accepted: False
提交态
status: missing; path: evidence/world_class/submissions/native-permission-enforcement.json; attested_real_evidence: False; privacy_contract_satisfied: False

执行步骤

  • Implement or connect a real target client or external installer runtime guard that blocks undeclared network, file_write, or subprocess capabilities.
  • Update the generated target adapter only when the guard is actually enforced by that target.
  • python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip
  • python3 scripts/yao.py install-simulate . --package-dir dist --install-root dist/install-simulation
  • python3 scripts/yao.py runtime-permissions . --package-dir dist
  • python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  • Copy evidence/world_class/templates/native-permission-enforcement.intake.json to evidence/world_class/submissions/native-permission-enforcement.json and fill only real evidence fields.
  • python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions

完成定义

  • reports/runtime_permission_probes.json summary.native_enforcement_count > 0
  • reports/runtime_permission_probes.json summary.failure_count == 0
  • reports/runtime_permission_probes.json summary.installer_enforcement_pass_count records local installer enforcement but does not replace native evidence
  • reports/skill_os2_audit.json item native-permission-enforcement status becomes pass

证据来源

  • dist/targets/*/adapter.json
  • reports/runtime_permission_probes.json
  • reports/runtime_permission_probes.md
  • reports/install_simulation.json
  • reports/install_simulation.md
  • security/permission_policy.json
  • evidence/world_class/intake.schema.json
  • evidence/world_class/templates/native-permission-enforcement.intake.json
  • reports/world_class_evidence_intake.json
  • reports/world_class_evidence_intake.md

隐私约束

  • Do not mark native_enforcement true for metadata-only fallbacks.
  • Keep residual risks visible for targets that still rely on operator enforcement.

源证据检查

  • Native enforcementnative_enforcement_count: 0 / >0Collect real target-client or external runtime guard proof.
  • Probe failuresfailure_count: 0 / ==0Runtime permission probes must stay clean.
  • Installer supportinstaller_enforcement_ready: True / trueInstaller enforcement is supporting evidence, not native proof.
待补证 · external

Native Client Telemetry

Import production metadata-only events from a real external client into the local drift loop.

负责人
Browser/Chrome/IDE/provider client integrator
当前状态
external source events 0; adoption samples 1
下一步
Install a real client against the native host and import production metadata-only events.
观测值
external_source_events: 0; adoption_sample_count: 1; raw_content_allowed: False; risk_band: low; accepted: False
提交态
status: missing; path: evidence/world_class/submissions/native-client-telemetry.json; attested_real_evidence: False; privacy_contract_satisfied: False

执行步骤

  • python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension://<extension-id>/
  • Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event.
  • python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl
  • python3 scripts/yao.py skill-atlas --workspace-root .
  • python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  • Copy evidence/world_class/templates/native-client-telemetry.intake.json to evidence/world_class/submissions/native-client-telemetry.json and fill only real evidence fields.
  • python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions

完成定义

  • reports/adoption_drift_report.json summary.source_types.external > 0
  • reports/adoption_drift_report.json summary.adoption_sample_count > 0
  • reports/skill_os2_audit.json item native-client-telemetry status becomes pass

证据来源

  • reports/adoption_drift_report.json
  • reports/adoption_drift_report.md
  • reports/telemetry_hook_recipes.json
  • scripts/telemetry_native_host.py
  • evidence/world_class/intake.schema.json
  • evidence/world_class/templates/native-client-telemetry.intake.json
  • reports/world_class_evidence_intake.json
  • reports/world_class_evidence_intake.md

隐私约束

  • Telemetry must remain metadata-only and local-first.
  • Do not package reports/telemetry_events.jsonl or any raw prompt, output, transcript, note, or message field.

源证据检查

  • External eventsexternal_source_events: 0 / >0Import at least one metadata-only event from a real client.
  • Adoption sampleadoption_sample_count: 1 / >0Telemetry must include adoption outcome evidence.
  • Raw content blockedraw_content_allowed: False / falseTelemetry must stay metadata-only.
+
待补证 · external

Provider Holdout

Collect at least one provider-backed output-eval holdout run with model, timing, and token metadata.

负责人
operator with provider credentials
当前状态
model-executed 0; token-observed 0
下一步
Run provider-backed holdout cases with real credentials and commit only aggregate evidence.
观测值
model_executed_count: 0; timing_observed_count: 10; token_observed_count: 0; accepted: False
提交态
status: missing; path: evidence/world_class/submissions/provider-holdout.json; attested_real_evidence: False; privacy_contract_satisfied: False

执行步骤

  • YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY=<redacted> python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60
  • python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  • Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields.
  • python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions

完成定义

  • reports/output_execution_runs.json summary.model_executed_count > 0
  • reports/output_execution_runs.json summary.timing_observed_count > 0
  • reports/output_execution_runs.json summary.token_observed_count > 0
  • reports/skill_os2_audit.json item provider-holdout status becomes pass

证据来源

  • reports/output_execution_runs.json
  • reports/output_execution_runs.md
  • reports/skill_os2_audit.json
  • evidence/world_class/intake.schema.json
  • evidence/world_class/templates/provider-holdout.intake.json
  • reports/world_class_evidence_intake.json
  • reports/world_class_evidence_intake.md

隐私约束

  • Do not commit provider credentials or environment dumps.
  • The output execution report records output hashes and aggregate run metadata, not raw provider prompts.

源证据检查

  • Provider model runmodel_executed_count: 0 / >0Run provider-backed output-exec with real credentials.
  • Timing observedtiming_observed_count: 10 / >0Provider execution should record timing metadata.
  • Token usage observedtoken_observed_count: 0 / >0Provider execution should return non-estimated token usage.
待补证 · human

Human Adjudication

Record real blind A/B reviewer decisions before claiming human output review completion.

负责人
human reviewer
当前状态
0/5 decisions; pending 5
下一步
Record real A/B choices in the decision template, then regenerate adjudication.
观测值
pair_count: 5; judgment_count: 0; pending_count: 5; invalid_decision_count: 0; answer_revealed_count: 0; accepted: False
提交态
status: missing; path: evidence/world_class/submissions/human-adjudication.json; attested_real_evidence: False; privacy_contract_satisfied: False

执行步骤

  • python3 scripts/yao.py output-review-kit --write-template
  • Open reports/output_review_kit.md and choose A or B for each pair without opening the answer key.
  • python3 scripts/adjudicate_output_review.py --write-template
  • Edit reports/output_review_decisions.json with winner_variant values and reviewer metadata.
  • python3 scripts/yao.py output-review
  • python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  • Copy evidence/world_class/templates/human-adjudication.intake.json to evidence/world_class/submissions/human-adjudication.json and fill only real evidence fields.
  • python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions

完成定义

  • reports/output_review_adjudication.json summary.pending_count == 0
  • reports/output_review_adjudication.json summary.judgment_count == summary.pair_count
  • reports/output_review_adjudication.json summary.invalid_decision_count == 0
  • reports/skill_os2_audit.json item human-adjudication status becomes pass

证据来源

  • reports/output_blind_review_pack.md
  • reports/output_review_kit.md
  • reports/output_review_decisions.json
  • reports/output_review_adjudication.json
  • reports/output_review_adjudication.md
  • evidence/world_class/intake.schema.json
  • evidence/world_class/templates/human-adjudication.intake.json
  • reports/world_class_evidence_intake.json
  • reports/world_class_evidence_intake.md

隐私约束

  • Reviewer decisions should not include raw user data or private customer detail.
  • Keep the answer key separate until after decisions are recorded.

源证据检查

  • Review pairs existpair_count: 5 / >0Generate the blind A/B review pack.
  • No pending decisionspending_count: 5 / ==0Record a reviewer choice for every pair.
  • Judgments completejudgment_count: 0 / ==pair_countEvery pair needs one valid human judgment.
  • No invalid decisionsinvalid_decision_count: 0 / ==0Fix malformed winner/confidence entries.
待补证 · external

Native Permission Enforcement

Prove at least one real target client or external installer runtime guard enforces approved high-permission capabilities.

负责人
target client or installer integrator
当前状态
native-enforced targets 0; installer-enforced targets 4
下一步
Integrate a real target-client or external installer runtime guard before claiming native permission enforcement.
观测值
native_enforcement_count: 0; metadata_fallback_count: 4; installer_enforcement_pass_count: 4; installer_permission_failure_count: 0; installer_enforcement_ready: True; residual_risk_count: 4; failure_count: 0; accepted: False
提交态
status: missing; path: evidence/world_class/submissions/native-permission-enforcement.json; attested_real_evidence: False; privacy_contract_satisfied: False

执行步骤

  • Implement or connect a real target client or external installer runtime guard that blocks undeclared network, file_write, or subprocess capabilities.
  • Update the generated target adapter only when the guard is actually enforced by that target.
  • python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip
  • python3 scripts/yao.py install-simulate . --package-dir dist --install-root dist/install-simulation
  • python3 scripts/yao.py runtime-permissions . --package-dir dist
  • python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  • Copy evidence/world_class/templates/native-permission-enforcement.intake.json to evidence/world_class/submissions/native-permission-enforcement.json and fill only real evidence fields.
  • python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions

完成定义

  • reports/runtime_permission_probes.json summary.native_enforcement_count > 0
  • reports/runtime_permission_probes.json summary.failure_count == 0
  • reports/runtime_permission_probes.json summary.installer_enforcement_pass_count records local installer enforcement but does not replace native evidence
  • reports/skill_os2_audit.json item native-permission-enforcement status becomes pass

证据来源

  • dist/targets/*/adapter.json
  • reports/runtime_permission_probes.json
  • reports/runtime_permission_probes.md
  • reports/install_simulation.json
  • reports/install_simulation.md
  • security/permission_policy.json
  • evidence/world_class/intake.schema.json
  • evidence/world_class/templates/native-permission-enforcement.intake.json
  • reports/world_class_evidence_intake.json
  • reports/world_class_evidence_intake.md

隐私约束

  • Do not mark native_enforcement true for metadata-only fallbacks.
  • Keep residual risks visible for targets that still rely on operator enforcement.

源证据检查

  • Native enforcementnative_enforcement_count: 0 / >0Collect real target-client or external runtime guard proof.
  • Probe failuresfailure_count: 0 / ==0Runtime permission probes must stay clean.
  • Installer supportinstaller_enforcement_ready: True / trueInstaller enforcement is supporting evidence, not native proof.
待补证 · external

Native Client Telemetry

Import production metadata-only events from a real external client into the local drift loop.

负责人
Browser/Chrome/IDE/provider client integrator
当前状态
external source events 0; adoption samples 0
下一步
Install a real client against the native host and import production metadata-only events.
观测值
external_source_events: 0; adoption_sample_count: 0; raw_content_allowed: False; risk_band: low; accepted: False
提交态
status: missing; path: evidence/world_class/submissions/native-client-telemetry.json; attested_real_evidence: False; privacy_contract_satisfied: False

执行步骤

  • python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension://<extension-id>/
  • Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event.
  • python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl
  • python3 scripts/yao.py skill-atlas --workspace-root .
  • python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  • Copy evidence/world_class/templates/native-client-telemetry.intake.json to evidence/world_class/submissions/native-client-telemetry.json and fill only real evidence fields.
  • python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions

完成定义

  • reports/adoption_drift_report.json summary.source_types.external > 0
  • reports/adoption_drift_report.json summary.adoption_sample_count > 0
  • reports/skill_os2_audit.json item native-client-telemetry status becomes pass

证据来源

  • reports/adoption_drift_report.json
  • reports/adoption_drift_report.md
  • reports/telemetry_hook_recipes.json
  • scripts/telemetry_native_host.py
  • evidence/world_class/intake.schema.json
  • evidence/world_class/templates/native-client-telemetry.intake.json
  • reports/world_class_evidence_intake.json
  • reports/world_class_evidence_intake.md

隐私约束

  • Telemetry must remain metadata-only and local-first.
  • Do not package reports/telemetry_events.jsonl or any raw prompt, output, transcript, note, or message field.

源证据检查

  • External eventsexternal_source_events: 0 / >0Import at least one metadata-only event from a real client.
  • Adoption sampleadoption_sample_count: 0 / >0Telemetry must include adoption outcome evidence.
  • Raw content blockedraw_content_allowed: False / falseTelemetry must stay metadata-only.
@@ -801,17 +801,17 @@
-

蓝图覆盖

项目数
20
模块数
8
建议 PR
12
通过数
20
Warn Count
0
缺失数
0
Extension Track Count
2
Extension Covered Count
1
Extension Partial Count
1
Extension Planned Count
0
Adaptive Extension Ready
本地蓝图
世界级
待补证据
4
+

蓝图覆盖

项目数
21
模块数
8
建议 PR
13
通过数
21
Warn Count
0
缺失数
0
Extension Track Count
2
Extension Covered Count
1
Extension Partial Count
1
Extension Planned Count
0
Adaptive Extension Ready
本地蓝图
世界级
待补证据
4

覆盖边界

蓝图覆盖只证明 2.0 模块、建议 PR、脚本、报告和测试在本地闭环;public world-class 仍以 world-class evidence ledger 的真人和外部证据为准。

-

公开声明

本地复现
发布锁
可公开声明
声明阻断
5
Provider 证据
人审完成
世界级就绪
-

声明阻断

+

公开声明

本地复现
发布锁
可公开声明
声明阻断
4
Provider 证据
人审完成
世界级就绪
+

声明阻断

-

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 7/13 pass; 6 blocked; overclaim guard true

+

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true

证据台账

Ledger Entry Count
4
Accepted Count
0
待审
4
Human Pending Count
1
External Pending Count
3
Overclaim Guard Active
Ready To Claim World Class
@@ -821,18 +821,18 @@
-

声明守卫

台账可声明
台账待补
4
声明面
76
违规数
0
Overclaim Guard Active
+

声明守卫

台账可声明
台账待补
4
声明面
77
违规数
0
Overclaim Guard Active

声明边界

claim guard 扫描 README、docs 和 reports 中的完成态表述;ledger 未 ready 时,任何英文完成断言、true 状态声明或中文完成态都会阻断发布审查。

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

-

包体元数据

名称
yao-meta-skill
版本
1.1.0
Maturity
governed
Owner
Yao Team
License
MIT
信任级别
local
目标平台
openai, claude, generic, agent-skills-compatible, vscode
兼容通过
6/6
归档哈希
6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc
+

包体元数据

名称
yao-meta-skill
版本
1.1.0
Maturity
governed
Owner
Yao Team
License
MIT
信任级别
local
目标平台
openai, claude, generic, agent-skills-compatible, vscode
兼容通过
6/6
归档哈希
798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

-

包体验证

目标数
4
Adapter
4
归档存在
Zip 条目
586
失败数
0
警告数
0
归档哈希
6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc
+

包体验证

目标数
4
Adapter
4
归档存在
Zip 条目
609
失败数
0
警告数
0
归档哈希
798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8
diff --git a/reports/review-studio.json b/reports/review-studio.json index 9acc523..02faaa6 100644 --- a/reports/review-studio.json +++ b/reports/review-studio.json @@ -43,7 +43,7 @@ "key": "context-budget", "label": "上下文", "status": "pass", - "detail": "initial load 960/1000; deferred 416415/120000; top deferred scripts 367825; resource governance governed; quality density 135.4", + "detail": "initial load 960/1000; deferred 421932/120000; top deferred scripts 373342; resource governance governed; quality density 135.4", "evidence": "reports/context_budget.json", "link": "context_budget.md" }, @@ -59,7 +59,7 @@ "key": "trust-report", "label": "信任报告", "status": "pass", - "detail": "0 secrets; 106 scripts; 3 network-capable scripts; 0 help smoke failures", + "detail": "0 secrets; 109 scripts; 3 network-capable scripts; 0 help smoke failures", "evidence": "reports/security_trust_report.json", "link": "security_trust_report.md" }, @@ -67,7 +67,7 @@ "key": "python-compat", "label": "Python 兼容", "status": "pass", - "detail": "Python 3.11; 171 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards", + "detail": "Python 3.11; 175 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards", "evidence": "reports/python_compatibility.json", "link": "python_compatibility.md" }, @@ -75,7 +75,7 @@ "key": "architecture-maintainability", "label": "架构维护", "status": "pass", - "detail": "168 Python files; 0 hotspots; 0 blockers; largest 886 lines; 34 CLI handlers", + "detail": "172 Python files; 0 hotspots; 0 blockers; largest 897 lines; 63 CLI handlers; 18 in entrypoint", "evidence": "reports/architecture_maintainability.json", "link": "architecture_maintainability.md" }, @@ -107,7 +107,7 @@ "key": "operations-loop", "label": "运营回路", "status": "pass", - "detail": "1 metadata events; adoption 100.0; missed 0; bad-output 0; risk low", + "detail": "1 metadata events; adoption 0; missed 0; bad-output 0; risk low", "evidence": "reports/adoption_drift_report.json", "link": "adoption_drift_report.md" }, @@ -123,7 +123,7 @@ "key": "world-class-evidence", "label": "世界证据", "status": "warn", - "detail": "4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 7/13 pass; 6 blocked; overclaim guard true", + "detail": "4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true", "evidence": "reports/world_class_evidence_ledger.json", "link": "world_class_evidence_ledger.md" }, @@ -166,7 +166,7 @@ "key": "world-class-evidence", "label": "世界证据", "status": "warn", - "detail": "4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 7/13 pass; 6 blocked; overclaim guard true", + "detail": "4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true", "evidence": "reports/world_class_evidence_ledger.json", "link": "world_class_evidence_ledger.md" } @@ -579,12 +579,12 @@ "category": "external", "status": "pending", "readiness": "awaiting-submission", - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "next_action": "Install a real client against the native host and import production metadata-only events.", "submission_path": "evidence/world_class/submissions/native-client-telemetry.json", "template_path": "evidence/world_class/templates/native-client-telemetry.intake.json", - "source_pass_count": 2, - "source_blocked_count": 1, + "source_pass_count": 1, + "source_blocked_count": 2, "blocked_checks": [ { "label": "External events", @@ -593,6 +593,14 @@ "expected": ">0", "status": "blocked", "next_action": "Import at least one metadata-only event from a real client." + }, + { + "label": "Adoption sample", + "field": "adoption_sample_count", + "actual": 0, + "expected": ">0", + "status": "blocked", + "next_action": "Telemetry must include adoption outcome evidence." } ], "commands": [ @@ -704,6 +712,7 @@ "reports/world_class_evidence_ledger.md", "reports/review_annotations.md", "reports/review-studio.html", + "reports/skill-interpretation.html", "reports/skill-overview.html" ], "flow": [ @@ -1181,7 +1190,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 106 + "file_count": 109 }, { "path": "evals", @@ -1193,10 +1202,10 @@ "path": "reports", "label": "Generated evidence and overview artifacts", "kind": "folder", - "file_count": 211 + "file_count": 215 } ], - "file_count": 384, + "file_count": 391, "folder_count": 4, "distribution": [ { @@ -1221,7 +1230,7 @@ }, { "label": "scripts", - "value": 106 + "value": 109 }, { "label": "evals", @@ -1229,7 +1238,7 @@ }, { "label": "reports", - "value": 211 + "value": 215 } ] }, @@ -1351,7 +1360,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 106 + "file_count": 109 }, { "path": "evals", @@ -1363,7 +1372,7 @@ "path": "reports", "label": "Generated evidence and overview artifacts", "kind": "folder", - "file_count": 211 + "file_count": 215 } ], "strengths": [ @@ -1664,16 +1673,16 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "119f209f1d4665b38a489cc372140ec9f01c663c4d2637048e82b924cfadb6c5", - "source_contract_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "output_case_count": 5, "failure_disclosure_count": 3, - "command_count": 21, + "command_count": 22, "command_executed_count": 10, "timing_observed_count": 10, "model_executed_count": 0, @@ -1688,11 +1697,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 61 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, - "commit": "5495734e7e33b19b84a932626112d783a3edf271", + "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1776,9 +1785,9 @@ "failures": [] }, "trust_security": { - "scanned_files": 193, - "script_count": 106, - "internal_module_count": 26, + "scanned_files": 196, + "script_count": 109, + "internal_module_count": 28, "secret_findings": 0, "dependency_files": [ "requirements-ci.txt" @@ -1786,18 +1795,18 @@ "network_script_count": 3, "network_policy_covered_count": 3, "network_policy_missing_count": 0, - "file_write_script_count": 67, + "file_write_script_count": 68, "permission_required_count": 3, "permission_approved_count": 3, "permission_missing_count": 0, "permission_invalid_count": 0, "permission_expired_count": 0, - "help_smoke_checked_count": 80, + "help_smoke_checked_count": 81, "help_smoke_failed_count": 0, "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", - "package_hash_file_count": 193, - "package_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6" + "package_hash_file_count": 196, + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4" }, "skill_atlas": { "skill_count": 12, @@ -1835,8 +1844,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "37a1ec0f35323886bdae64523b4113f5dc4ba3153117ee4980b8e5cd54b49aba", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc" + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8" }, "compatibility": { "openai": "pass", @@ -1867,12 +1876,12 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" }, - "generated_at": "2026-06-13" + "generated_at": "2026-06-15" }, "failures": [], "warnings": [] @@ -1883,8 +1892,8 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", - "archive_entry_count": 586, + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", + "archive_entry_count": 609, "failure_count": 0, "warning_count": 0 }, @@ -1895,7 +1904,7 @@ "ok": true, "summary": { "archive_present": true, - "archive_entry_count": 601, + "archive_entry_count": 609, "archive_extracted": true, "entrypoint_loaded": true, "manifest_loaded": true, @@ -1905,7 +1914,7 @@ "installer_permission_failure_count": 0, "permission_target_count": 4, "permission_capability_count": 3, - "install_root_is_temp": false, + "install_root_is_temp": true, "failure_count": 0, "warning_count": 0 }, @@ -1967,7 +1976,7 @@ { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "37a1ec0f35323886bdae64523b4113f5dc4ba3153117ee4980b8e5cd54b49aba" + "to": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306" } ] }, @@ -4334,7 +4343,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 42.05, + "duration_ms": 35.01, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4362,7 +4371,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 36.82, + "duration_ms": 34.89, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4385,7 +4394,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 38.87, + "duration_ms": 36.42, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4413,7 +4422,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 35.73, + "duration_ms": 39.48, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4436,7 +4445,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 35.53, + "duration_ms": 36.04, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4464,7 +4473,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 48.73, + "duration_ms": 36.06, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4487,7 +4496,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 65.97, + "duration_ms": 34.85, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4514,7 +4523,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 60.32, + "duration_ms": 40.88, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4537,7 +4546,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 50.55, + "duration_ms": 42.57, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4566,7 +4575,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 44.37, + "duration_ms": 42.98, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -5287,39 +5296,26 @@ "ok": true, "generated_at": "2026-06-15", "skill_dir": ".", - "commit": "5495734e7e33b19b84a932626112d783a3edf271", + "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "git_status": { "available": true, - "dirty": true, - "changed_file_count": 61, - "sample": [ - " M Makefile", - " M README.md", - " M reports/adoption_drift_report.json", - " M reports/adoption_drift_report.md", - " M reports/architecture_maintainability.json", - " M reports/architecture_maintainability.md", - " M reports/benchmark_reproducibility.json", - " M reports/benchmark_reproducibility.md", - " M reports/compiled_targets.json", - " M reports/context_budget.json", - " M reports/context_budget.md", - " M reports/context_budget_summary.json" - ], + "dirty": false, + "changed_file_count": 0, + "sample": [], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "119f209f1d4665b38a489cc372140ec9f01c663c4d2637048e82b924cfadb6c5", - "source_contract_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "output_case_count": 5, "failure_disclosure_count": 3, - "command_count": 21, + "command_count": 22, "command_executed_count": 10, "timing_observed_count": 10, "model_executed_count": 0, @@ -5334,15 +5330,14 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 61 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ - "release lock is not clean or commit is unavailable", "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", @@ -5351,10 +5346,10 @@ "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { - "ready": false, - "commit": "5495734e7e33b19b84a932626112d783a3edf271", + "ready": true, + "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "status_scope": "generation-time status before this report is written", - "reason": "working tree was dirty at generation time" + "reason": "clean generation-time HEAD" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", @@ -5362,7 +5357,7 @@ "existing_count": 24, "missing_count": 0, "missing_paths": [], - "sha256": "119f209f1d4665b38a489cc372140ec9f01c663c4d2637048e82b924cfadb6c5" + "sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -5435,8 +5430,8 @@ "label": "output_execution", "path": "reports/output_execution_runs.json", "exists": true, - "bytes": 7965, - "sha256": "e559e6acf9f2e709f65dc36ea5646199a367fb760d7b7e6014570c59d3f46a55" + "bytes": 7967, + "sha256": "a4c8cacaab5d2f5178513e20cfbb58787052e82607b9ec4fe3eaeae58d7509c8" }, { "label": "blind_review", @@ -5463,50 +5458,50 @@ "label": "runtime_conformance", "path": "reports/conformance_matrix.json", "exists": true, - "bytes": 10313, - "sha256": "8251329e663dda51472f29b7721e73d72ccbec9760d96fda022f6218a5a6e347" + "bytes": 10342, + "sha256": "97f9ba949c23a60b00e9ba2ff279ca03ba517845cc7c62aa8a42645d58006c7e" }, { "label": "trust_report", "path": "reports/security_trust_report.json", "exists": true, - "bytes": 105694, - "sha256": "5c1cc74ef5306f2959c7bc4d01068dd2cc36a840500724922f2fb346e63ddbc5" + "bytes": 107994, + "sha256": "d7dae65ba9cfdc52001b73242ac88c84cf302bc6091e0df8805fdb79ac2594ce" }, { "label": "python_compatibility", "path": "reports/python_compatibility.json", "exists": true, - "bytes": 22252, - "sha256": "fcf288346cb5159e070b2df75a8e371cca2501945c971aead521e7ac3557a4bc" + "bytes": 22768, + "sha256": "cdd46eb42010f406fb2732063bde224866a4493ae4be1dc0c835417550a45d18" }, { "label": "registry_audit", "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, - "sha256": "9b54f3fd07680c2a9dd18ec9845e1faa1e19e8fbc854cd13a2d5bc1b84360f0e" + "sha256": "9ca263ecce9ace6b3915b2485eb7807edaf375d1aeccc689bc8b2d91417d8d2e" }, { "label": "package_verification", "path": "reports/package_verification.json", "exists": true, "bytes": 19338, - "sha256": "2476ae8ec9c491a3f7f31ec3f0c6c78cceead8e80700dcb696838eec226c675b" + "sha256": "1abff9930aa56d337b47dbc421f293eaa9eb3620c187d1b45a26941594b392ef" }, { "label": "install_simulation", "path": "reports/install_simulation.json", "exists": true, - "bytes": 8758, - "sha256": "491fabea4624d41967ff04952b72e0b165882c1b3b276124047c921daea4f3b7" + "bytes": 8557, + "sha256": "958586fa59a71aeea10962c6047083876fbcd4b809f24fbeaec4ece8aa499dfc" }, { "label": "skill_os2_audit", "path": "reports/skill_os2_audit.json", "exists": true, "bytes": 14310, - "sha256": "8481f4d78f2ad4ff0e74936b58b643a91052a30b6d12bcbbf24e0b4291b8fb1f" + "sha256": "55526d80bfa78b574f990442165a0159b0f177feca00523826657b00f7855d29" }, { "label": "world_class_evidence_plan", @@ -5561,8 +5556,8 @@ "label": "world_class_claim_guard", "path": "reports/world_class_claim_guard.json", "exists": true, - "bytes": 8634, - "sha256": "a2e6953e92e667cc0913c08e48b6954ea48420ad3a4056ed15eb19a183825e67" + "bytes": 8814, + "sha256": "7e5a2eac1020f4f58688b52d3c59b5df1c74d7bc04f4e599b48e5be4cc23d786" } ], "missing_artifacts": [], @@ -5667,6 +5662,11 @@ "command": "python3 scripts/yao.py world-class-claim-guard .", "evidence": "reports/world_class_claim_guard.json" }, + { + "label": "evidence consistency", + "command": "python3 scripts/yao.py evidence-consistency .", + "evidence": "reports/evidence_consistency.json" + }, { "label": "full ci", "command": "make ci-test", @@ -5695,10 +5695,10 @@ "generated_at": "2026-06-15", "skill_dir": ".", "summary": { - "item_count": 20, + "item_count": 21, "module_count": 8, - "recommended_pr_count": 12, - "pass_count": 20, + "recommended_pr_count": 13, + "pass_count": 21, "warn_count": 0, "missing_count": 0, "extension_track_count": 2, @@ -5712,7 +5712,7 @@ "decision": "local-blueprint-covered-evidence-pending" }, "status_counts": { - "pass": 20, + "pass": 21, "warn": 0, "missing": 0 }, @@ -5822,7 +5822,7 @@ "label": "Trust Security", "status": "pass", "objective": "Scripts, dependencies, permissions, secrets, and package hash are reviewable for team distribution.", - "current": "106 scripts; secrets 0; help failures 0", + "current": "109 scripts; secrets 0; help failures 0", "command": "python3 scripts/yao.py trust .", "test": "python3 tests/verify_trust_check.py", "evidence": [ @@ -5896,7 +5896,7 @@ "label": "Registry Distribution", "status": "pass", "objective": "Skill packages are installable, versioned, checksumed, and upgrade-reviewable.", - "current": "archive entries 586; install failures 0", + "current": "archive entries 609; install failures 0", "command": "python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --output-dir dist --zip && python3 scripts/yao.py registry-audit .", "test": "python3 tests/verify_registry_audit.py", "evidence": [ @@ -6298,6 +6298,31 @@ } ], "next_action": "Keep this item covered as the implementation evolves." + }, + { + "key": "evidence-consistency", + "category": "recommended-pr", + "label": "Evidence Consistency", + "status": "pass", + "objective": "Recommended Skill OS 2.0 implementation PR from the upgrade plan.", + "current": "26 consistency checks", + "command": "make ci-test", + "test": "tests/verify_evidence_consistency.py", + "evidence": [ + { + "path": "scripts/render_evidence_consistency.py", + "exists": true + }, + { + "path": "reports/evidence_consistency.json", + "exists": true + }, + { + "path": "tests/verify_evidence_consistency.py", + "exists": true + } + ], + "next_action": "Keep this item covered as the implementation evolves." } ], "reference_extension_tracks": [ @@ -6409,7 +6434,7 @@ "source_blueprint": { "title": "Skill Overview / Skill OS 2.0 upgrade plan", "core_module_count": 8, - "recommended_pr_count": 12, + "recommended_pr_count": 13, "reference_extension_count": 2, "reference_extensions": [ "Skill interpretation report", @@ -6424,7 +6449,7 @@ "compiled_targets": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "target_count": 5, @@ -6609,6 +6634,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -6623,6 +6649,7 @@ "scripts/render_review_studio.py", "scripts/render_review_viewer.py", "scripts/render_review_waivers.py", + "scripts/render_skill_interpretation.py", "scripts/render_skill_os2_audit.py", "scripts/render_skill_os2_coverage.py", "scripts/render_skill_overview.py", @@ -6648,9 +6675,7 @@ "scripts/review_viewer_data.py", "scripts/run_conformance_suite.py", "scripts/run_description_optimization_suite.py", - "scripts/run_eval_suite.py", - "scripts/run_output_eval.py", - "scripts/run_output_execution.py" + "scripts/run_eval_suite.py" ], "assets": [ "templates/basic_skill.md.j2", @@ -6810,7 +6835,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -6841,6 +6866,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -6917,14 +6943,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -6946,7 +6972,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -7059,7 +7085,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -7090,6 +7116,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -7166,14 +7193,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -7193,7 +7220,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -7468,6 +7495,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -7482,6 +7510,7 @@ "scripts/render_review_studio.py", "scripts/render_review_viewer.py", "scripts/render_review_waivers.py", + "scripts/render_skill_interpretation.py", "scripts/render_skill_os2_audit.py", "scripts/render_skill_os2_coverage.py", "scripts/render_skill_overview.py", @@ -7507,9 +7536,7 @@ "scripts/review_viewer_data.py", "scripts/run_conformance_suite.py", "scripts/run_description_optimization_suite.py", - "scripts/run_eval_suite.py", - "scripts/run_output_eval.py", - "scripts/run_output_execution.py" + "scripts/run_eval_suite.py" ], "assets": [ "templates/basic_skill.md.j2", @@ -7669,7 +7696,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -7700,6 +7727,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -7776,14 +7804,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -7805,7 +7833,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -7918,7 +7946,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -7949,6 +7977,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -8025,14 +8054,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -8052,7 +8081,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -8327,6 +8356,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -8341,6 +8371,7 @@ "scripts/render_review_studio.py", "scripts/render_review_viewer.py", "scripts/render_review_waivers.py", + "scripts/render_skill_interpretation.py", "scripts/render_skill_os2_audit.py", "scripts/render_skill_os2_coverage.py", "scripts/render_skill_overview.py", @@ -8366,9 +8397,7 @@ "scripts/review_viewer_data.py", "scripts/run_conformance_suite.py", "scripts/run_description_optimization_suite.py", - "scripts/run_eval_suite.py", - "scripts/run_output_eval.py", - "scripts/run_output_execution.py" + "scripts/run_eval_suite.py" ], "assets": [ "templates/basic_skill.md.j2", @@ -8528,7 +8557,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -8559,6 +8588,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -8635,14 +8665,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -8664,7 +8694,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -8770,7 +8800,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -8801,6 +8831,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -8877,14 +8908,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -8904,7 +8935,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -9170,6 +9201,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -9184,6 +9216,7 @@ "scripts/render_review_studio.py", "scripts/render_review_viewer.py", "scripts/render_review_waivers.py", + "scripts/render_skill_interpretation.py", "scripts/render_skill_os2_audit.py", "scripts/render_skill_os2_coverage.py", "scripts/render_skill_overview.py", @@ -9209,9 +9242,7 @@ "scripts/review_viewer_data.py", "scripts/run_conformance_suite.py", "scripts/run_description_optimization_suite.py", - "scripts/run_eval_suite.py", - "scripts/run_output_eval.py", - "scripts/run_output_execution.py" + "scripts/run_eval_suite.py" ], "assets": [ "templates/basic_skill.md.j2", @@ -9371,7 +9402,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -9402,6 +9433,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -9478,14 +9510,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -9507,7 +9539,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -9613,7 +9645,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -9644,6 +9676,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -9720,14 +9753,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -9747,7 +9780,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -10013,6 +10046,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -10027,6 +10061,7 @@ "scripts/render_review_studio.py", "scripts/render_review_viewer.py", "scripts/render_review_waivers.py", + "scripts/render_skill_interpretation.py", "scripts/render_skill_os2_audit.py", "scripts/render_skill_os2_coverage.py", "scripts/render_skill_overview.py", @@ -10052,9 +10087,7 @@ "scripts/review_viewer_data.py", "scripts/run_conformance_suite.py", "scripts/run_description_optimization_suite.py", - "scripts/run_eval_suite.py", - "scripts/run_output_eval.py", - "scripts/run_output_execution.py" + "scripts/run_eval_suite.py" ], "assets": [ "templates/basic_skill.md.j2", @@ -10214,7 +10247,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -10245,6 +10278,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -10321,14 +10355,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -10350,7 +10384,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -10460,7 +10494,7 @@ }, "file_write": { "required": true, - "script_count": 67, + "script_count": 68, "scripts": [ "scripts/adjudicate_output_review.py", "scripts/build_confusion_matrix.py", @@ -10491,6 +10525,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -10567,14 +10602,14 @@ }, "help_smoke": { "enabled": true, - "checked_count": 80, + "checked_count": 81, "failed_count": 0, "failed_scripts": [] }, "trust_summary": { "secret_findings": 0, "network_script_count": 3, - "file_write_script_count": 67, + "file_write_script_count": 68, "subprocess_script_count": 9, "interactive_script_count": 0, "help_smoke_failed_count": 0 @@ -10594,7 +10629,7 @@ ], "capability_counts": { "network": 3, - "file_write": 67, + "file_write": 68, "subprocess": 9, "interactive": 0 }, @@ -10734,6 +10769,7 @@ "Skill IR description matches frontmatter", "references resource resolves: references/artifact-design-doctrine.md", "references resource resolves: references/authoring-discipline.md", + "references resource resolves: references/autonomous-adaptation.md", "references resource resolves: references/distribution-registry-method.md", "references resource resolves: references/eval-playbook.md", "references resource resolves: references/gate-selection.md", @@ -10742,8 +10778,7 @@ "references resource resolves: references/intent-dialogue.md", "references resource resolves: references/iteration-philosophy.md", "references resource resolves: references/non-skill-decision-tree.md", - "references resource resolves: references/operating-modes.md", - "references resource resolves: references/output-eval-method.md" + "references resource resolves: references/operating-modes.md" ], "failures": [], "warnings": [] @@ -10778,6 +10813,7 @@ "Skill IR description matches frontmatter", "references resource resolves: references/artifact-design-doctrine.md", "references resource resolves: references/authoring-discipline.md", + "references resource resolves: references/autonomous-adaptation.md", "references resource resolves: references/distribution-registry-method.md", "references resource resolves: references/eval-playbook.md", "references resource resolves: references/gate-selection.md", @@ -10786,8 +10822,7 @@ "references resource resolves: references/intent-dialogue.md", "references resource resolves: references/iteration-philosophy.md", "references resource resolves: references/non-skill-decision-tree.md", - "references resource resolves: references/operating-modes.md", - "references resource resolves: references/output-eval-method.md" + "references resource resolves: references/operating-modes.md" ], "failures": [], "warnings": [] @@ -10801,7 +10836,7 @@ "frontmatter description exists", "description length <= 1024", "name is runtime-safe", - "directory name matches skill name", + "package identity derives from skill name", "manifest.name exists", "manifest.version exists", "manifest.owner exists", @@ -10822,6 +10857,7 @@ "Skill IR description matches frontmatter", "references resource resolves: references/artifact-design-doctrine.md", "references resource resolves: references/authoring-discipline.md", + "references resource resolves: references/autonomous-adaptation.md", "references resource resolves: references/distribution-registry-method.md", "references resource resolves: references/eval-playbook.md", "references resource resolves: references/gate-selection.md", @@ -10830,8 +10866,7 @@ "references resource resolves: references/intent-dialogue.md", "references resource resolves: references/iteration-philosophy.md", "references resource resolves: references/non-skill-decision-tree.md", - "references resource resolves: references/operating-modes.md", - "references resource resolves: references/output-eval-method.md" + "references resource resolves: references/operating-modes.md" ], "failures": [], "warnings": [ @@ -10847,7 +10882,7 @@ "frontmatter description exists", "description length <= 1024", "name is runtime-safe", - "directory name matches skill name", + "package identity derives from skill name", "manifest.name exists", "manifest.version exists", "manifest.owner exists", @@ -10868,6 +10903,7 @@ "Skill IR description matches frontmatter", "references resource resolves: references/artifact-design-doctrine.md", "references resource resolves: references/authoring-discipline.md", + "references resource resolves: references/autonomous-adaptation.md", "references resource resolves: references/distribution-registry-method.md", "references resource resolves: references/eval-playbook.md", "references resource resolves: references/gate-selection.md", @@ -10876,8 +10912,7 @@ "references resource resolves: references/intent-dialogue.md", "references resource resolves: references/iteration-philosophy.md", "references resource resolves: references/non-skill-decision-tree.md", - "references resource resolves: references/operating-modes.md", - "references resource resolves: references/output-eval-method.md" + "references resource resolves: references/operating-modes.md" ], "failures": [], "warnings": [ @@ -10914,6 +10949,7 @@ "Skill IR description matches frontmatter", "references resource resolves: references/artifact-design-doctrine.md", "references resource resolves: references/authoring-discipline.md", + "references resource resolves: references/autonomous-adaptation.md", "references resource resolves: references/distribution-registry-method.md", "references resource resolves: references/eval-playbook.md", "references resource resolves: references/gate-selection.md", @@ -10922,8 +10958,7 @@ "references resource resolves: references/intent-dialogue.md", "references resource resolves: references/iteration-philosophy.md", "references resource resolves: references/non-skill-decision-tree.md", - "references resource resolves: references/operating-modes.md", - "references resource resolves: references/output-eval-method.md" + "references resource resolves: references/operating-modes.md" ], "failures": [], "warnings": [] @@ -10943,7 +10978,7 @@ "schema_version": "1.0", "ok": true, "skill_dir": ".", - "package_dir": "tests/tmp_review_studio/dist", + "package_dir": "dist", "expected_capabilities": [ "file_write", "network", @@ -10983,7 +11018,7 @@ { "target": "openai", "status": "pass", - "adapter": "tests/tmp_review_studio/dist/targets/openai/adapter.json", + "adapter": "dist/targets/openai/adapter.json", "permission_model": "metadata-only", "native_enforcement": false, "metadata_fallback_explicit": true, @@ -11080,7 +11115,7 @@ { "target": "claude", "status": "pass", - "adapter": "tests/tmp_review_studio/dist/targets/claude/adapter.json", + "adapter": "dist/targets/claude/adapter.json", "permission_model": "neutral-source-plus-adapter", "native_enforcement": false, "metadata_fallback_explicit": true, @@ -11162,7 +11197,7 @@ { "target": "generic", "status": "pass", - "adapter": "tests/tmp_review_studio/dist/targets/generic/adapter.json", + "adapter": "dist/targets/generic/adapter.json", "permission_model": "agent-skills-compatible-metadata", "native_enforcement": false, "metadata_fallback_explicit": true, @@ -11244,7 +11279,7 @@ { "target": "vscode", "status": "pass", - "adapter": "tests/tmp_review_studio/dist/targets/vscode/adapter.json", + "adapter": "dist/targets/vscode/adapter.json", "permission_model": "vscode-workspace-trust-plus-metadata", "native_enforcement": false, "metadata_fallback_explicit": true, @@ -11334,9 +11369,9 @@ "ok": true, "skill_dir": ".", "summary": { - "scanned_files": 193, - "script_count": 106, - "internal_module_count": 26, + "scanned_files": 196, + "script_count": 109, + "internal_module_count": 28, "secret_findings": 0, "dependency_files": [ "requirements-ci.txt" @@ -11344,18 +11379,18 @@ "network_script_count": 3, "network_policy_covered_count": 3, "network_policy_missing_count": 0, - "file_write_script_count": 67, + "file_write_script_count": 68, "permission_required_count": 3, "permission_approved_count": 3, "permission_missing_count": 0, "permission_invalid_count": 0, "permission_expired_count": 0, - "help_smoke_checked_count": 80, + "help_smoke_checked_count": 81, "help_smoke_failed_count": 0, "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", - "package_hash_file_count": 193, - "package_sha256": "28625ce8b3fcb3e214ef3b8ffc748d9cff551e4c5a1b78786c1a2b7d7005f679" + "package_hash_file_count": 196, + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4" }, "failures": [], "warnings": [], @@ -11906,6 +11941,20 @@ "network_urls": [], "network_hosts": [] }, + { + "path": "scripts/render_evidence_consistency.py", + "interface": "cli", + "interface_declared": true, + "interface_reason": "Renders a cross-report evidence consistency gate for generated Skill OS reports.", + "has_argparse": true, + "has_main_guard": true, + "uses_input": false, + "uses_network": false, + "uses_file_write": true, + "uses_subprocess": false, + "network_urls": [], + "network_hosts": [] + }, { "path": "scripts/render_intent_confidence.py", "interface": "cli", @@ -12806,6 +12855,34 @@ "network_urls": [], "network_hosts": [] }, + { + "path": "scripts/yao_cli_distribution_commands.py", + "interface": "internal-module", + "interface_declared": true, + "interface_reason": "Imported by yao.py to keep distribution and runtime gate handlers outside the thin CLI orchestrator.", + "has_argparse": true, + "has_main_guard": false, + "uses_input": false, + "uses_network": false, + "uses_file_write": false, + "uses_subprocess": false, + "network_urls": [], + "network_hosts": [] + }, + { + "path": "scripts/yao_cli_output_commands.py", + "interface": "internal-module", + "interface_declared": true, + "interface_reason": "Imported by yao.py to keep output evaluation and review handlers outside the thin CLI orchestrator.", + "has_argparse": true, + "has_main_guard": false, + "uses_input": false, + "uses_network": false, + "uses_file_write": false, + "uses_subprocess": false, + "network_urls": [], + "network_hosts": [] + }, { "path": "scripts/yao_cli_parser.py", "interface": "internal-module", @@ -12893,11 +12970,11 @@ "help_smoke": { "enabled": true, "timeout_seconds": 5.0, - "candidate_count": 80, - "checked_count": 80, - "passed_count": 80, + "candidate_count": 81, + "checked_count": 81, + "passed_count": 81, "failed_count": 0, - "skipped_count": 26, + "skipped_count": 28, "failed_scripts": [], "results": [ { @@ -13270,6 +13347,16 @@ "stdout_excerpt": "usage: render_eval_dashboard.py [-h] [--eval-dir EVAL_DIR]\n [--description-file DESCRIPTION_FILE]\n [--baseline-description-file BASELINE_DESCRIPTION_FILE]\n ", "stderr_excerpt": "" }, + { + "path": "scripts/render_evidence_consistency.py", + "command": "python3 scripts/render_evidence_consistency.py --help", + "returncode": 0, + "timed_out": false, + "passed": true, + "has_help_text": true, + "stdout_excerpt": "usage: render_evidence_consistency.py [-h] [--output-json OUTPUT_JSON]\n [--output-md OUTPUT_MD]\n [--generated-at GENERATED_AT]\n [", + "stderr_excerpt": "" + }, { "path": "scripts/render_intent_confidence.py", "command": "python3 scripts/render_intent_confidence.py --help", @@ -13327,7 +13414,7 @@ "timed_out": false, "passed": true, "has_help_text": true, - "stdout_excerpt": "usage: render_portability_report.py [-h] [--output-json OUTPUT_JSON]\n [--output-md OUTPUT_MD]\n\nRender a portability score from neutral metadata, contracts, and snapshots.\n\noptions:\n -h, --help ", + "stdout_excerpt": "usage: render_portability_report.py [-h] [--output-json OUTPUT_JSON]\n [--output-md OUTPUT_MD]\n [skill_dir]\n\nRender a portability score from neutral metadata, contracts, a", "stderr_excerpt": "" }, { @@ -13790,6 +13877,14 @@ "path": "scripts/yao_cli_create_commands.py", "reason": "internal module" }, + { + "path": "scripts/yao_cli_distribution_commands.py", + "reason": "internal module" + }, + { + "path": "scripts/yao_cli_output_commands.py", + "reason": "internal module" + }, { "path": "scripts/yao_cli_parser.py", "reason": "internal module" @@ -13889,6 +13984,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -13990,11 +14086,11 @@ "python_compatibility": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "root": ".", "summary": { "target_python": "3.11", - "file_count": 171, + "file_count": 175, "issue_count": 0, "syntax_error_count": 0, "fstring_311_violation_count": 0, @@ -14257,6 +14353,12 @@ "issue_count": 0, "issues": [] }, + { + "path": "scripts/render_evidence_consistency.py", + "ok": true, + "issue_count": 0, + "issues": [] + }, { "path": "scripts/render_intent_confidence.py", "ok": true, @@ -14641,6 +14743,18 @@ "issue_count": 0, "issues": [] }, + { + "path": "scripts/yao_cli_distribution_commands.py", + "ok": true, + "issue_count": 0, + "issues": [] + }, + { + "path": "scripts/yao_cli_output_commands.py", + "ok": true, + "issue_count": 0, + "issues": [] + }, { "path": "scripts/yao_cli_parser.py", "ok": true, @@ -14725,6 +14839,12 @@ "issue_count": 0, "issues": [] }, + { + "path": "tests/verify_evidence_consistency.py", + "ok": true, + "issue_count": 0, + "issues": [] + }, { "path": "tests/verify_failure_regressions.py", "ok": true, @@ -15047,33 +15167,28 @@ "architecture_maintainability": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { - "python_file_count": 168, - "script_file_count": 106, - "test_file_count": 62, - "internal_module_count": 29, - "cli_script_count": 79, - "command_handler_count": 34, + "python_file_count": 172, + "script_file_count": 109, + "test_file_count": 63, + "internal_module_count": 31, + "cli_script_count": 80, + "command_handler_count": 63, + "entrypoint_command_handler_count": 18, + "command_module_count": 6, "warn_line_threshold": 900, "block_line_threshold": 1500, - "largest_file_lines": 886, + "largest_file_lines": 897, "hotspot_count": 0, "blocker_count": 0, "decision": "pass" }, "largest_files": [ - { - "path": "scripts/yao.py", - "lines": 886, - "kind": "cli-script", - "severity": "pass", - "recommendation": "Split command handlers by domain while keeping scripts/yao.py as the thin CLI orchestrator." - }, { "path": "tests/verify_yao_cli.py", - "lines": 886, + "lines": 897, "kind": "test", "severity": "pass", "recommendation": "Break broad integration assertions into focused verifier helpers when the next behavior change lands." @@ -15087,14 +15202,14 @@ }, { "path": "scripts/skill_report_model.py", - "lines": 800, + "lines": 801, "kind": "internal-module", "severity": "pass", "recommendation": "Watch this file before adding new responsibilities; extract a helper module when one concern dominates." }, { "path": "scripts/yao_cli_parser.py", - "lines": 774, + "lines": 784, "kind": "internal-module", "severity": "pass", "recommendation": "Watch this file before adding new responsibilities; extract a helper module when one concern dominates." @@ -15147,6 +15262,13 @@ "kind": "cli-script", "severity": "pass", "recommendation": "Split viewer data assembly from HTML section rendering." + }, + { + "path": "scripts/cross_packager.py", + "lines": 650, + "kind": "cli-script", + "severity": "pass", + "recommendation": "Watch this file before adding new responsibilities; extract a helper module when one concern dominates." } ], "hotspots": [], @@ -15164,16 +15286,16 @@ "context_budget_tier": "production", "context_budget_limit": 1000, "skill_body_tokens": 767, - "other_text_tokens": 1307850, + "other_text_tokens": 1334868, "estimated_initial_load_tokens": 960, - "estimated_total_text_tokens": 1308617, - "deferred_resource_tokens": 416415, + "estimated_total_text_tokens": 1335635, + "deferred_resource_tokens": 421932, "deferred_resource_warn_threshold": 120000, "deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 367825, - "file_count": 106 + "estimated_tokens": 373342, + "file_count": 109 }, { "path": "references", @@ -15189,8 +15311,8 @@ "large_deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 367825, - "file_count": 106 + "estimated_tokens": 373342, + "file_count": 109 } ], "deferred_resource_governance": { @@ -15212,14 +15334,14 @@ ], "missing": [], "path": "scripts", - "estimated_tokens": 367825, - "file_count": 106, + "estimated_tokens": 373342, + "file_count": 109, "rationale": "Script resources are deterministic deferred tools, not initial-load prompt context." } ], "summary": "Large deferred resources are indexed and backed by evidence." }, - "relevant_file_count": 544, + "relevant_file_count": 554, "unused_resource_dirs": [], "quality_signal_points": 130, "quality_density": 135.4 @@ -16121,6 +16243,7 @@ "scripts/render_context_reports.py", "scripts/render_description_drift_history.py", "scripts/render_eval_dashboard.py", + "scripts/render_evidence_consistency.py", "scripts/render_intent_confidence.py", "scripts/render_intent_dialogue.py", "scripts/render_iteration_directions.py", @@ -17086,7 +17209,7 @@ "adoption_drift": { "ok": true, "schema_version": "2.0", - "generated_at": "2026-06-15T13:28:42Z", + "generated_at": "2026-06-15T14:35:15Z", "skill_dir": ".", "privacy_contract": { "storage": "local-first", @@ -17110,14 +17233,14 @@ }, "summary": { "event_count": 1, - "adoption_sample_count": 1, - "activation_count": 1, - "accepted_count": 1, + "adoption_sample_count": 0, + "activation_count": 0, + "accepted_count": 0, "edited_count": 0, "rejected_count": 0, "missed_count": 0, "failed_count": 0, - "adoption_rate": 100.0, + "adoption_rate": 0, "missed_trigger_count": 0, "wrong_trigger_count": 0, "bad_output_count": 0, @@ -17126,7 +17249,7 @@ "review_overdue_count": 0, "risk_band": "low", "event_types": { - "skill_activation": 1 + "review_event": 1 }, "failure_types": {}, "source_types": { @@ -17138,31 +17261,31 @@ { "skill": "yao-meta-skill", "events": 1, - "adoption_events": 1, - "accepted": 1, + "adoption_events": 0, + "accepted": 0, "edited": 0, "rejected": 0, "missed": 0, - "adoption_rate": 100.0 + "adoption_rate": 0 } ], "next_iteration_candidates": [], "recent_events": [ { "command": "unknown", - "event": "skill_activation", + "event": "review_event", "skill": "yao-meta-skill", "source": "manual", "version": "1.1.0", - "activation_type": "explicit", - "outcome": "accepted", + "activation_type": "manual", + "outcome": "reviewed", "failure_type": "none", - "timestamp": "2026-06-13T10:00:00Z" + "timestamp": "2026-06-13T12:00:00Z" } ], "failures": [], "artifacts": { - "events_jsonl": "tests/tmp_review_studio/telemetry_events.jsonl", + "events_jsonl": "reports/telemetry_events.jsonl", "json": "reports/adoption_drift_report.json", "markdown": "reports/adoption_drift_report.md" } @@ -17171,7 +17294,7 @@ "schema_version": "1.0", "ok": true, "skill_dir": ".", - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "summary": { "waiver_count": 0, "active_count": 0, @@ -17221,7 +17344,7 @@ "Reviewer links output_review_adjudication or output_execution evidence." ], "suggested_evidence": "reports/output_review_adjudication.md", - "suggested_command": "python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer \"\" --reason \"Output Lab has pending human/provider evidence; accepted only for this bounded review scope.\" --expires-at 2027-06-13 --evidence reports/output_review_adjudication.md", + "suggested_command": "python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer \"\" --reason \"Output Lab has pending human/provider evidence; accepted only for this bounded review scope.\" --expires-at 2027-06-15 --evidence reports/output_review_adjudication.md", "world_class_boundary": "Does not count as provider, human, or public world-class completion evidence." }, { @@ -17251,7 +17374,7 @@ "schema_version": "1.0", "ok": true, "skill_dir": ".", - "source": "tests/tmp_review_studio/empty_review_annotations_input.json", + "source": "reports/review_annotations_input.json", "summary": { "annotation_count": 0, "open_count": 0, @@ -17274,7 +17397,7 @@ "world_class_evidence_ledger": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "ledger_entry_count": 4, @@ -17287,8 +17410,8 @@ "missing_submission_count": 4, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "submitted_but_pending_count": 0, "source_accepted_without_valid_submission_count": 0, "overclaim_guard_active": true, @@ -17624,7 +17747,7 @@ "status": "pending", "source_status": "external_required", "source_accepted": false, - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -17661,7 +17784,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -17682,8 +17805,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 1, - "status": "pass", + "actual": 0, + "status": "blocked", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -17699,8 +17822,8 @@ } ], "source_check_count": 3, - "source_pass_count": 2, - "source_blocked_count": 1, + "source_pass_count": 1, + "source_blocked_count": 2, "submission_state": { "status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json", @@ -17738,7 +17861,7 @@ "world_class_evidence_intake": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "schema_present": true, @@ -18044,7 +18167,7 @@ "source_accepted": false, "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -18115,7 +18238,7 @@ "world_class_submission_review": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "review_item_count": 4, @@ -18127,8 +18250,8 @@ "unmatched_submission_count": 0, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "ready_to_claim_world_class": false, "review_counts_submission_as_completion": false, "decision": "awaiting-submissions" @@ -18379,7 +18502,7 @@ "intake_errors": [], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -18400,8 +18523,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 1, - "status": "pass", + "actual": 0, + "status": "blocked", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -18417,8 +18540,8 @@ } ], "source_check_count": 3, - "source_pass_count": 2, - "source_blocked_count": 1, + "source_pass_count": 1, + "source_blocked_count": 2, "success_checks": [ "reports/adoption_drift_report.json summary.source_types.external > 0", "reports/adoption_drift_report.json summary.adoption_sample_count > 0", @@ -18444,7 +18567,7 @@ "world_class_operator_runbook": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "evidence_item_count": 4, @@ -18455,8 +18578,8 @@ "valid_packet_source_incomplete_count": 0, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "ready_to_claim_world_class": false, "runbook_counts_as_completion": false, "decision": "collect-evidence" @@ -18837,7 +18960,7 @@ "review_state": "awaiting-submission", "source_accepted": false, "objective": "Import production metadata-only events from a real external client into the local drift loop.", - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "execution_runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", "Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event.", @@ -18885,7 +19008,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -18906,8 +19029,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 1, - "status": "pass", + "actual": 0, + "status": "blocked", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -18922,9 +19045,10 @@ "next_action": "Telemetry must stay metadata-only." } ], - "blocked_source_check_count": 1, + "blocked_source_check_count": 2, "next_source_actions": [ - "Import at least one metadata-only event from a real client." + "Import at least one metadata-only event from a real client.", + "Telemetry must include adoption outcome evidence." ], "submission_state": { "status": "missing", @@ -18957,12 +19081,12 @@ "world_class_claim_guard": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "ledger_ready_to_claim_world_class": false, "ledger_pending_count": 4, - "claim_surface_count": 76, + "claim_surface_count": 77, "violation_count": 0, "overclaim_guard_active": true, "decision": "claim-guard-pass-evidence-pending" @@ -19079,6 +19203,10 @@ "path": "reports/description_optimization_suite.md", "violation_count": 0 }, + { + "path": "reports/evidence_consistency.md", + "violation_count": 0 + }, { "path": "reports/family_summary.md", "violation_count": 0 @@ -19333,8 +19461,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc" + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8" }, "compatibility": { "openai": "pass", @@ -19365,16 +19493,16 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" }, - "generated_at": "2026-06-13" + "generated_at": "2026-06-15" }, "index": { "schema_version": "2.0", - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "package_count": 1, "packages": [ { @@ -19390,7 +19518,7 @@ "vscode" ], "package_metadata": "registry/packages/yao-meta-skill.json", - "package_sha256": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6" + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4" } ] }, @@ -19413,8 +19541,8 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", - "archive_entry_count": 586, + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", + "archive_entry_count": 609, "failure_count": 0, "warning_count": 0 }, @@ -20072,14 +20200,14 @@ "install_simulation": { "ok": true, "schema_version": "2.0", - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", - "package_dir": "tests/tmp_review_studio/dist", - "install_root": "tests/tmp_review_studio/install-root/simulate-yao-meta-skill", - "installed_skill_dir": "tests/tmp_review_studio/install-root/simulate-yao-meta-skill/yao-meta-skill", + "package_dir": "dist", + "install_root": "[temporary-install-root]", + "installed_skill_dir": "[temporary-install-root]/yao-meta-skill", "summary": { "archive_present": true, - "archive_entry_count": 603, + "archive_entry_count": 609, "archive_extracted": true, "entrypoint_loaded": true, "manifest_loaded": true, @@ -20089,7 +20217,7 @@ "installer_permission_failure_count": 0, "permission_target_count": 4, "permission_capability_count": 3, - "install_root_is_temp": false, + "install_root_is_temp": true, "failure_count": 0, "warning_count": 0 }, @@ -20097,7 +20225,7 @@ { "id": "archive-present", "status": "pass", - "detail": "Package archive exists: tests/tmp_review_studio/dist/yao-meta-skill.zip" + "detail": "Package archive exists: dist/yao-meta-skill.zip" }, { "id": "archive-safe-paths", @@ -20343,10 +20471,10 @@ "failures": [], "warnings": [], "artifacts": { - "archive": "tests/tmp_review_studio/dist/yao-meta-skill.zip", - "package_manifest": "tests/tmp_review_studio/dist/manifest.json", - "json": "tests/tmp_review_studio/install_simulation.json", - "markdown": "tests/tmp_review_studio/install_simulation.md" + "archive": "dist/yao-meta-skill.zip", + "package_manifest": "dist/manifest.json", + "json": "reports/install_simulation.json", + "markdown": "reports/install_simulation.md" } }, "upgrade_check": { @@ -20426,7 +20554,7 @@ { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "ecb6c742c79077b26d1f9e61ede1b1c4c078b485fc57467b969d9c1744bc91b6" + "to": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306" } ] }, diff --git a/reports/review-viewer.json b/reports/review-viewer.json index bae1199..7517a52 100644 --- a/reports/review-viewer.json +++ b/reports/review-viewer.json @@ -513,7 +513,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 107 + "file_count": 109 }, { "path": "evals", @@ -528,7 +528,7 @@ "file_count": 215 } ], - "file_count": 389, + "file_count": 391, "folder_count": 4, "distribution": [ { @@ -553,7 +553,7 @@ }, { "label": "scripts", - "value": 107 + "value": 109 }, { "label": "evals", @@ -683,7 +683,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 107 + "file_count": 109 }, { "path": "evals", @@ -996,13 +996,13 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc", - "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 22, @@ -1020,11 +1020,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 12 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, - "commit": "6e2dfd48d080901f09b4e67bdf35b9500ec3dcec", + "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1108,9 +1108,9 @@ "failures": [] }, "trust_security": { - "scanned_files": 194, - "script_count": 107, - "internal_module_count": 26, + "scanned_files": 196, + "script_count": 109, + "internal_module_count": 28, "secret_findings": 0, "dependency_files": [ "requirements-ci.txt" @@ -1128,8 +1128,8 @@ "help_smoke_failed_count": 0, "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", - "package_hash_file_count": 194, - "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306" + "package_hash_file_count": 196, + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4" }, "skill_atlas": { "skill_count": 12, @@ -1167,8 +1167,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc" + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8" }, "compatibility": { "openai": "pass", @@ -1199,12 +1199,12 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" }, - "generated_at": "2026-06-13" + "generated_at": "2026-06-15" }, "failures": [], "warnings": [] @@ -1215,8 +1215,8 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", - "archive_entry_count": 586, + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", + "archive_entry_count": 609, "failure_count": 0, "warning_count": 0 }, @@ -1227,7 +1227,7 @@ "ok": true, "summary": { "archive_present": true, - "archive_entry_count": 607, + "archive_entry_count": 609, "archive_extracted": true, "entrypoint_loaded": true, "manifest_loaded": true, @@ -1237,7 +1237,7 @@ "installer_permission_failure_count": 0, "permission_target_count": 4, "permission_capability_count": 3, - "install_root_is_temp": false, + "install_root_is_temp": true, "failure_count": 0, "warning_count": 0 }, diff --git a/reports/skill-interpretation.html b/reports/skill-interpretation.html index 6e8bccf..c963e30 100644 --- a/reports/skill-interpretation.html +++ b/reports/skill-interpretation.html @@ -930,7 +930,7 @@

让 reviewer 快速确认关键文件、目录和资产分布。Lets reviewers confirm key files, directories, and asset distribution quickly.

-
资产分布Asset Distribution389项389 itemsSKILL.mdSKILL.mdREADME.mdREADME.mdagents/interface.yamlagents/interface.yamlmanifest.jsonmanifest.jsonreferencesreferencesscriptsscripts
资产分布图展示当前包体的文件和目录重心。The asset distribution chart shows where files and directories are concentrated.
+
资产分布Asset Distribution391项391 itemsSKILL.mdSKILL.mdREADME.mdREADME.mdagents/interface.yamlagents/interface.yamlmanifest.jsonmanifest.jsonreferencesreferencesscriptsscripts
资产分布图展示当前包体的文件和目录重心。The asset distribution chart shows where files and directories are concentrated.
diff --git a/reports/skill-interpretation.json b/reports/skill-interpretation.json index a7c772d..80a56ea 100644 --- a/reports/skill-interpretation.json +++ b/reports/skill-interpretation.json @@ -513,7 +513,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 107 + "file_count": 109 }, { "path": "evals", @@ -528,7 +528,7 @@ "file_count": 215 } ], - "file_count": 389, + "file_count": 391, "folder_count": 4, "distribution": [ { @@ -553,7 +553,7 @@ }, { "label": "scripts", - "value": 107 + "value": 109 }, { "label": "evals", @@ -687,7 +687,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 107 + "file_count": 109 }, { "path": "evals", @@ -1004,9 +1004,9 @@ "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc", - "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 22, @@ -1028,7 +1028,7 @@ "working_tree_dirty": false, "changed_file_count": 0 }, - "commit": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e", + "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1112,9 +1112,9 @@ "failures": [] }, "trust_security": { - "scanned_files": 194, - "script_count": 107, - "internal_module_count": 26, + "scanned_files": 196, + "script_count": 109, + "internal_module_count": 28, "secret_findings": 0, "dependency_files": [ "requirements-ci.txt" @@ -1132,8 +1132,8 @@ "help_smoke_failed_count": 0, "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", - "package_hash_file_count": 194, - "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306" + "package_hash_file_count": 196, + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4" }, "skill_atlas": { "skill_count": 12, @@ -1171,8 +1171,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc" + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8" }, "compatibility": { "openai": "pass", @@ -1203,12 +1203,12 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" }, - "generated_at": "2026-06-13" + "generated_at": "2026-06-15" }, "failures": [], "warnings": [] @@ -1219,8 +1219,8 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", - "archive_entry_count": 586, + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", + "archive_entry_count": 609, "failure_count": 0, "warning_count": 0 }, @@ -1231,7 +1231,7 @@ "ok": true, "summary": { "archive_present": true, - "archive_entry_count": 607, + "archive_entry_count": 609, "archive_extracted": true, "entrypoint_loaded": true, "manifest_loaded": true, @@ -1241,7 +1241,7 @@ "installer_permission_failure_count": 0, "permission_target_count": 4, "permission_capability_count": 3, - "install_root_is_temp": false, + "install_root_is_temp": true, "failure_count": 0, "warning_count": 0 }, diff --git a/reports/skill-overview.html b/reports/skill-overview.html index 4936f8c..249c25d 100644 --- a/reports/skill-overview.html +++ b/reports/skill-overview.html @@ -930,7 +930,7 @@

让 reviewer 快速确认关键文件、目录和资产分布。Lets reviewers confirm key files, directories, and asset distribution quickly.

-
资产分布Asset Distribution389项389 itemsSKILL.mdSKILL.mdREADME.mdREADME.mdagents/interface.yamlagents/interface.yamlmanifest.jsonmanifest.jsonreferencesreferencesscriptsscripts
资产分布图展示当前包体的文件和目录重心。The asset distribution chart shows where files and directories are concentrated.
+
资产分布Asset Distribution391项391 itemsSKILL.mdSKILL.mdREADME.mdREADME.mdagents/interface.yamlagents/interface.yamlmanifest.jsonmanifest.jsonreferencesreferencesscriptsscripts
资产分布图展示当前包体的文件和目录重心。The asset distribution chart shows where files and directories are concentrated.
路径Path作用Role类型Type
SKILL.mdSkill 入口文件Skill entrypoint文件file
README.md人类可读使用说明Human-readable usage guide文件file
agents/interface.yaml跨平台接口元数据Neutral interface metadata文件file
manifest.json生命周期与打包元数据Lifecycle and portability metadata文件file
references扩展指导与复用资料Extended guidance and reusable notes目录folder
scripts确定性脚本或本地工具Deterministic helpers or local tooling目录folder
evals触发与质量检查Trigger and quality checks目录folder
reports生成的证据与总结报告Generated evidence and overview artifacts目录folder
diff --git a/reports/skill-overview.json b/reports/skill-overview.json index 25bb730..e82c79e 100644 --- a/reports/skill-overview.json +++ b/reports/skill-overview.json @@ -512,7 +512,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 107 + "file_count": 109 }, { "path": "evals", @@ -527,7 +527,7 @@ "file_count": 215 } ], - "file_count": 389, + "file_count": 391, "folder_count": 4, "distribution": [ { @@ -552,7 +552,7 @@ }, { "label": "scripts", - "value": 107 + "value": 109 }, { "label": "evals", @@ -682,7 +682,7 @@ "path": "scripts", "label": "Deterministic helpers or local tooling", "kind": "folder", - "file_count": 107 + "file_count": 109 }, { "path": "evals", @@ -999,9 +999,9 @@ "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "4b8a09b7d324ac39b25a0cc9af8ad04c85e6c94c3d1280c6243146f0f4048bcc", - "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "evidence_bundle_sha256": "885429fb91fff38c5cae2dc22b26ca918e22bf01aa2fcc6b42d702d79ede39d3", + "source_contract_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 22, @@ -1023,7 +1023,7 @@ "working_tree_dirty": false, "changed_file_count": 0 }, - "commit": "2df5c9acd093f45e27318bbe5ff984fae3a5fd8e", + "commit": "0e6327ca9399c6995b192fdf710fa4c99699c2c8", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1107,9 +1107,9 @@ "failures": [] }, "trust_security": { - "scanned_files": 194, - "script_count": 107, - "internal_module_count": 26, + "scanned_files": 196, + "script_count": 109, + "internal_module_count": 28, "secret_findings": 0, "dependency_files": [ "requirements-ci.txt" @@ -1127,8 +1127,8 @@ "help_smoke_failed_count": 0, "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", - "package_hash_file_count": 194, - "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306" + "package_hash_file_count": 196, + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4" }, "skill_atlas": { "skill_count": 12, @@ -1166,8 +1166,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc" + "package_sha256": "76edc3f9936da8966eff0e0f689568c7d6d787691a45773a3e86bb38be6b0cc4", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8" }, "compatibility": { "openai": "pass", @@ -1198,12 +1198,12 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" }, - "generated_at": "2026-06-13" + "generated_at": "2026-06-15" }, "failures": [], "warnings": [] @@ -1214,8 +1214,8 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", - "archive_entry_count": 586, + "archive_sha256": "798153ff8e1569408e8f658be5aea1f144ee44bfe4edf658c71d6a2f0275c1c8", + "archive_entry_count": 609, "failure_count": 0, "warning_count": 0 }, @@ -1226,7 +1226,7 @@ "ok": true, "summary": { "archive_present": true, - "archive_entry_count": 607, + "archive_entry_count": 609, "archive_extracted": true, "entrypoint_loaded": true, "manifest_loaded": true, @@ -1236,7 +1236,7 @@ "installer_permission_failure_count": 0, "permission_target_count": 4, "permission_capability_count": 3, - "install_root_is_temp": false, + "install_root_is_temp": true, "failure_count": 0, "warning_count": 0 },
路径Path作用Role类型Type
SKILL.mdSkill 入口文件Skill entrypoint文件file
README.md人类可读使用说明Human-readable usage guide文件file
agents/interface.yaml跨平台接口元数据Neutral interface metadata文件file
manifest.json生命周期与打包元数据Lifecycle and portability metadata文件file
references扩展指导与复用资料Extended guidance and reusable notes目录folder
scripts确定性脚本或本地工具Deterministic helpers or local tooling目录folder
evals触发与质量检查Trigger and quality checks目录folder
reports生成的证据与总结报告Generated evidence and overview artifacts目录folder