diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json index 44c9f63..781915a 100644 --- a/reports/benchmark_reproducibility.json +++ b/reports/benchmark_reproducibility.json @@ -3,7 +3,7 @@ "ok": true, "generated_at": "2026-06-14", "skill_dir": ".", - "commit": "07274b4c50068d4635583631c55e5a6e3fc00a7a", + "commit": "328be185f6199ff1d7041378c97657b42e7e3c42", "git_status": { "available": true, "dirty": false, @@ -17,9 +17,9 @@ "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "69dfb6c9b80bc88d8043270714359b8f8074bfc60f5fb811d8a917a0da087aef", - "source_contract_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5", - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "evidence_bundle_sha256": "1e0bb8068a01ef836225760c31bdd6fa04dbf4014643f5b4532d5241357c4554", + "source_contract_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 21, @@ -50,7 +50,7 @@ }, "release_lock": { "ready": true, - "commit": "07274b4c50068d4635583631c55e5a6e3fc00a7a", + "commit": "328be185f6199ff1d7041378c97657b42e7e3c42", "status_scope": "generation-time status before this report is written", "reason": "clean generation-time HEAD" }, @@ -60,7 +60,7 @@ "existing_count": 24, "missing_count": 0, "missing_paths": [], - "sha256": "69dfb6c9b80bc88d8043270714359b8f8074bfc60f5fb811d8a917a0da087aef" + "sha256": "1e0bb8068a01ef836225760c31bdd6fa04dbf4014643f5b4532d5241357c4554" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -169,7 +169,7 @@ "path": "reports/security_trust_report.json", "exists": true, "bytes": 98830, - "sha256": "871eb9e4f88b0b91c211af94a66660f1c4e799907466362c15ebe63a9128bab1" + "sha256": "39ddc420a6f8d6e0b991b54311c61ce2f7be8de8e6c7241eda8d0ffe9434bb39" }, { "label": "python_compatibility", @@ -183,14 +183,14 @@ "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, - "sha256": "1b2feee7ce87257d042d6a2b9cb75bd2e495f4676ff20c21a7b3508c87aa6248" + "sha256": "d02dbd2d31655fd27613765d3395212d53ca07c8ef028deeff64732ca5059dbc" }, { "label": "package_verification", "path": "reports/package_verification.json", "exists": true, "bytes": 19325, - "sha256": "6e885151878efa803cb5e85d0636b493015729cbf598e22cafffb30b8df2936a" + "sha256": "e101808a5f8f97126c02fedac757d545fef3e65a7f01d9d1da2774d314af0840" }, { "label": "install_simulation", @@ -210,8 +210,8 @@ "label": "world_class_evidence_plan", "path": "reports/world_class_evidence_plan.json", "exists": true, - "bytes": 19532, - "sha256": "a9bd392cd45f3accabe9b5befb9090456f383b1edbc323cab7d20930179dc229" + "bytes": 19940, + "sha256": "164803bb1cea795c8b962ec82cb1b3a4a58bbc51e10a4edfce12de00d046905a" }, { "label": "world_class_evidence_ledger", @@ -342,22 +342,22 @@ }, { "label": "world-class evidence ledger", - "command": "python3 scripts/yao.py world-class-ledger .", + "command": "python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_evidence_ledger.json" }, { "label": "world-class evidence intake", - "command": "python3 scripts/yao.py world-class-intake .", + "command": "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_evidence_intake.json" }, { "label": "world-class submission review", - "command": "python3 scripts/yao.py world-class-submission-review .", + "command": "python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_submission_review.json" }, { "label": "world-class operator runbook", - "command": "python3 scripts/yao.py world-class-runbook .", + "command": "python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_operator_runbook.json" }, { diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md index 3bf92df..543430e 100644 --- a/reports/benchmark_reproducibility.md +++ b/reports/benchmark_reproducibility.md @@ -1,9 +1,9 @@ # Benchmark Reproducibility Generated at: `2026-06-14` -Commit: `07274b4c50068d4635583631c55e5a6e3fc00a7a` +Commit: `328be185f6199ff1d7041378c97657b42e7e3c42` Working tree dirty at generation: `false` -Evidence bundle SHA256: `69dfb6c9b80bc88d8043270714359b8f8074bfc60f5fb811d8a917a0da087aef` +Evidence bundle SHA256: `1e0bb8068a01ef836225760c31bdd6fa04dbf4014643f5b4532d5241357c4554` ## Summary @@ -12,8 +12,8 @@ Evidence bundle SHA256: `69dfb6c9b80bc88d8043270714359b8f8074bfc60f5fb811d8a917a - methodology complete: `true` - required artifacts: `24` - missing artifacts: `0` -- source contract sha256: `f6672acad15c` -- archive sha256: `7b0683ec4b2e` +- source contract sha256: `7bd945074bb9` +- archive sha256: `0488c5b84d90` - output cases: `5` - disclosed failure cases: `3` - reproduction commands: `21` @@ -48,7 +48,7 @@ This report proves local benchmark reproducibility only. It keeps external provi - algorithm: `sha256(path,label,exists,artifact_sha256)` - artifacts: `24` / `24` -- sha256: `69dfb6c9b80bc88d8043270714359b8f8074bfc60f5fb811d8a917a0da087aef` +- sha256: `1e0bb8068a01ef836225760c31bdd6fa04dbf4014643f5b4532d5241357c4554` ## Methodology Sections @@ -75,13 +75,13 @@ This report proves local benchmark reproducibility only. It keeps external provi | review_adjudication | `reports/output_review_adjudication.json` | present | `240485a721af` | | trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` | | runtime_conformance | `reports/conformance_matrix.json` | present | `8251329e663d` | -| trust_report | `reports/security_trust_report.json` | present | `871eb9e4f88b` | +| trust_report | `reports/security_trust_report.json` | present | `39ddc420a6f8` | | python_compatibility | `reports/python_compatibility.json` | present | `ae16e17266e4` | -| registry_audit | `reports/registry_audit.json` | present | `1b2feee7ce87` | -| package_verification | `reports/package_verification.json` | present | `6e885151878e` | +| registry_audit | `reports/registry_audit.json` | present | `d02dbd2d3165` | +| package_verification | `reports/package_verification.json` | present | `e101808a5f8f` | | install_simulation | `reports/install_simulation.json` | present | `8f987e805c92` | | skill_os2_audit | `reports/skill_os2_audit.json` | present | `6bb2dcb0e1e5` | -| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `a9bd392cd45f` | +| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `164803bb1cea` | | world_class_evidence_ledger | `reports/world_class_evidence_ledger.json` | present | `0bff14542475` | | world_class_evidence_intake | `reports/world_class_evidence_intake.json` | present | `72af5fa06f5b` | | world_class_submission_review | `reports/world_class_submission_review.json` | present | `4f03edb2ef28` | @@ -122,13 +122,13 @@ This report proves local benchmark reproducibility only. It keeps external provi - evidence: `reports/skill_os2_audit.json` - `python3 scripts/yao.py world-class-evidence .` - evidence: `reports/world_class_evidence_plan.json` -- `python3 scripts/yao.py world-class-ledger .` +- `python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions` - evidence: `reports/world_class_evidence_ledger.json` -- `python3 scripts/yao.py world-class-intake .` +- `python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions` - evidence: `reports/world_class_evidence_intake.json` -- `python3 scripts/yao.py world-class-submission-review .` +- `python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions` - evidence: `reports/world_class_submission_review.json` -- `python3 scripts/yao.py world-class-runbook .` +- `python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions` - evidence: `reports/world_class_operator_runbook.json` - `python3 scripts/yao.py world-class-claim-guard .` - evidence: `reports/world_class_claim_guard.json` diff --git a/reports/review-studio.html b/reports/review-studio.html index 1f7836d..37e0302 100644 --- a/reports/review-studio.html +++ b/reports/review-studio.html @@ -510,7 +510,7 @@

审查闸门

-
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 960/1000; deferred 388115/120000; top deferred scripts 340379; resource governance governed; quality density 135.4

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 97 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

Python 兼容

Python 3.11; 159 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards

reports/python_compatibility.json 证据
通过

架构维护

156 Python files; 0 hotspots; 0 blockers; largest 888 lines; 34 CLI handlers

reports/architecture_maintainability.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low

reports/adoption_drift_report.json 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

reports/promotion_decisions.json + reports/upgrade_check.json + docs/migration-v2.md 证据
+
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 960/1000; deferred 388240/120000; top deferred scripts 340504; resource governance governed; quality density 135.4

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 97 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

Python 兼容

Python 3.11; 159 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards

reports/python_compatibility.json 证据
通过

架构维护

156 Python files; 0 hotspots; 0 blockers; largest 888 lines; 34 CLI handlers

reports/architecture_maintainability.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low

reports/adoption_drift_report.json 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

@@ -520,7 +520,7 @@

修复动作

-
关注

输出实验

补足 output eval 覆盖、execution evidence、blind A/B 和 reviewer adjudication。

没有输出质量和人工盲评证据时,Skill 只能证明会触发,不能证明输出真的更好且经得起审查。
修复位置
evals/output/cases.jsonl + reports/output_quality_scorecard.md + reports/output_review_kit.html + reports/output_review_adjudication.md
验证命令
python3 scripts/adjudicate_output_review.py --write-template && python3 scripts/yao.py output-review
关注

人工批准

对保留的 warning 写入 reviewer、理由、范围和到期时间,或修掉 warning。

warning 可以被接受,但必须可审计、会过期,并且不能掩盖 blocker。
修复位置
reports/review_waivers.md
验证命令
python3 scripts/render_review_waivers.py .
关注

世界证据

补齐 provider、真人盲评、原生权限执行和真实客户端遥测证据,或明确本次发布不声明 world-class 完成。

世界级结论必须来自已接受的外部/人工证据;计划、metadata fallback、待评审和本地命令都不能替代完成证据。
修复位置
reports/world_class_operator_runbook.html + reports/world_class_evidence_ledger.md + reports/world_class_evidence_intake.md + reports/world_class_submission_review.md
验证命令
python3 scripts/yao.py world-class-runbook . && python3 scripts/yao.py world-class-ledger . && python3 scripts/yao.py review-studio .
+
关注

输出实验

补足 output eval 覆盖、execution evidence、blind A/B 和 reviewer adjudication。

没有输出质量和人工盲评证据时,Skill 只能证明会触发,不能证明输出真的更好且经得起审查。
修复位置
evals/output/cases.jsonl + reports/output_quality_scorecard.md + reports/output_review_kit.html + reports/output_review_adjudication.md
验证命令
python3 scripts/adjudicate_output_review.py --write-template && python3 scripts/yao.py output-review
关注

人工批准

对保留的 warning 写入 reviewer、理由、范围和到期时间,或修掉 warning。

warning 可以被接受,但必须可审计、会过期,并且不能掩盖 blocker。
修复位置
reports/review_waivers.md
验证命令
python3 scripts/render_review_waivers.py .
关注

世界证据

补齐 provider、真人盲评、原生权限执行和真实客户端遥测证据,或明确本次发布不声明 world-class 完成。

世界级结论必须来自已接受的外部/人工证据;计划、metadata fallback、待评审和本地命令都不能替代完成证据。
修复位置
reports/world_class_operator_runbook.html + reports/world_class_evidence_ledger.md + reports/world_class_evidence_intake.md + reports/world_class_submission_review.md
验证命令
python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py review-studio .
@@ -572,12 +572,12 @@
-

上下文

initial load 960/1000; deferred 388115/120000; top deferred scripts 340379; resource governance governed; quality density 135.4

+

上下文

initial load 960/1000; deferred 388240/120000; top deferred scripts 340504; resource governance governed; quality density 135.4

编译证据

Review reports/compiled_targets.md before packaging to inspect target adapter modes, generated files, preserved semantics, warnings, and unsupported features.

-

信任报告

Secret
0
脚本数
97
网络脚本
3
Help 失败
0
包体哈希
f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5
+

信任报告

Secret
0
脚本数
97
网络脚本
3
Help 失败
0
包体哈希
7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397

安全边界

高风险 secret、远程 inline execution、缺失依赖策略或无法解释的脚本接口应阻断 governed release。

@@ -614,7 +614,7 @@

批准台账

Waiver Count
0
Active Count
0
Expired Count
0
Invalid Count
0
覆盖 Gate
0

批准候选

-
可批准 · needs-reviewer-decision

Output Lab

review pending 5; model-executed 0; output failures 0

Gate
output-lab
证据
reports/output_review_adjudication.md
选项
accepted-risk, false-positive, temporary-exception
边界
Does not count as provider, human, or public world-class completion evidence.

审查条件

  • Reviewer confirms this release does not claim provider-backed or human-adjudicated output superiority.
  • Reviewer names the release scope and expiry date.
  • Reviewer links output_review_adjudication or output_execution evidence.
建议命令python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer "<reviewer>" --reason "Output Lab has pending human/provider evidence; accepted only for this bounded review scope." --expires-at 2027-06-13 --evidence reports/output_review_adjudication.md
不可批准 · cannot-waive

World-Class Evidence

4 pending evidence entries; 1 human pending; 3 external pending

Gate
world-class-evidence
证据
reports/world_class_evidence_ledger.md
选项
边界
Non-waivable completion boundary.

审查条件

  • Do not use a waiver to claim public world-class readiness.
  • Either submit accepted ledger evidence or state that this release does not claim world-class completion.
  • Keep claim guard active until ledger summary.ready_to_claim_world_class is true.
建议命令python3 scripts/yao.py world-class-ledger . && python3 scripts/yao.py world-class-claim-guard .
+
可批准 · needs-reviewer-decision

Output Lab

review pending 5; model-executed 0; output failures 0

Gate
output-lab
证据
reports/output_review_adjudication.md
选项
accepted-risk, false-positive, temporary-exception
边界
Does not count as provider, human, or public world-class completion evidence.

审查条件

  • Reviewer confirms this release does not claim provider-backed or human-adjudicated output superiority.
  • Reviewer names the release scope and expiry date.
  • Reviewer links output_review_adjudication or output_execution evidence.
建议命令python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer "<reviewer>" --reason "Output Lab has pending human/provider evidence; accepted only for this bounded review scope." --expires-at 2027-06-14 --evidence reports/output_review_adjudication.md
不可批准 · cannot-waive

World-Class Evidence

4 pending evidence entries; 1 human pending; 3 external pending

Gate
world-class-evidence
证据
reports/world_class_evidence_ledger.md
选项
边界
Non-waivable completion boundary.

审查条件

  • Do not use a waiver to claim public world-class readiness.
  • Either submit accepted ledger evidence or state that this release does not claim world-class completion.
  • Keep claim guard active until ledger summary.ready_to_claim_world_class is true.
建议命令python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-claim-guard .
@@ -656,12 +656,12 @@

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

-

包体元数据

名称
yao-meta-skill
版本
1.1.0
Maturity
governed
Owner
Yao Team
License
MIT
信任级别
local
目标平台
openai, claude, generic, agent-skills-compatible, vscode
兼容通过
6/6
归档哈希
7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f
+

包体元数据

名称
yao-meta-skill
版本
1.1.0
Maturity
governed
Owner
Yao Team
License
MIT
信任级别
local
目标平台
openai, claude, generic, agent-skills-compatible, vscode
兼容通过
6/6
归档哈希
0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

-

包体验证

目标数
4
Adapter
4
归档存在
Zip 条目
576
失败数
0
警告数
0
归档哈希
7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f
+

包体验证

目标数
4
Adapter
4
归档存在
Zip 条目
576
失败数
0
警告数
0
归档哈希
0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798
diff --git a/reports/review-studio.json b/reports/review-studio.json index ebe7acd..d89d799 100644 --- a/reports/review-studio.json +++ b/reports/review-studio.json @@ -43,7 +43,7 @@ "key": "context-budget", "label": "上下文", "status": "pass", - "detail": "initial load 960/1000; deferred 388115/120000; top deferred scripts 340379; resource governance governed; quality density 135.4", + "detail": "initial load 960/1000; deferred 388240/120000; top deferred scripts 340504; resource governance governed; quality density 135.4", "evidence": "reports/context_budget.json", "link": "context_budget.md" }, @@ -372,7 +372,7 @@ ], "evidence": "reports/world_class_evidence_ledger.json", "evidence_link": "world_class_evidence_ledger.md", - "verification_command": "python3 scripts/yao.py world-class-runbook . && python3 scripts/yao.py world-class-ledger . && python3 scripts/yao.py review-studio ." + "verification_command": "python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py review-studio ." } ], "evidence_paths": { @@ -1336,9 +1336,9 @@ "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "69dfb6c9b80bc88d8043270714359b8f8074bfc60f5fb811d8a917a0da087aef", - "source_contract_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5", - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "evidence_bundle_sha256": "1e0bb8068a01ef836225760c31bdd6fa04dbf4014643f5b4532d5241357c4554", + "source_contract_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 21, @@ -1357,7 +1357,7 @@ "working_tree_dirty": false, "changed_file_count": 0 }, - "commit": "07274b4c50068d4635583631c55e5a6e3fc00a7a", + "commit": "328be185f6199ff1d7041378c97657b42e7e3c42", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1462,7 +1462,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 181, - "package_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5" + "package_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397" }, "skill_atlas": { "skill_count": 12, @@ -1500,8 +1500,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5", - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f" + "package_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798" }, "compatibility": { "openai": "pass", @@ -1532,7 +1532,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -1548,7 +1548,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "archive_entry_count": 576, "failure_count": 0, "warning_count": 0 @@ -1627,12 +1627,12 @@ { "field": "archive_sha256", "from": "", - "to": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f" + "to": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5" + "to": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397" } ] }, @@ -1775,7 +1775,7 @@ "YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY= python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60", "python3 scripts/yao.py skill-os2-audit . --generated-at ", "Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields.", - "python3 scripts/yao.py world-class-intake ." + "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions" ], "success_checks": [ "reports/output_execution_runs.json summary.model_executed_count > 0", @@ -1814,7 +1814,7 @@ "python3 scripts/yao.py output-review", "python3 scripts/yao.py skill-os2-audit . --generated-at ", "Copy evidence/world_class/templates/human-adjudication.intake.json to evidence/world_class/submissions/human-adjudication.json and fill only real evidence fields.", - "python3 scripts/yao.py world-class-intake ." + "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions" ], "success_checks": [ "reports/output_review_adjudication.json summary.pending_count == 0", @@ -1855,7 +1855,7 @@ "python3 scripts/yao.py runtime-permissions . --package-dir dist", "python3 scripts/yao.py skill-os2-audit . --generated-at ", "Copy evidence/world_class/templates/native-permission-enforcement.intake.json to evidence/world_class/submissions/native-permission-enforcement.json and fill only real evidence fields.", - "python3 scripts/yao.py world-class-intake ." + "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions" ], "success_checks": [ "reports/runtime_permission_probes.json summary.native_enforcement_count > 0", @@ -1896,7 +1896,7 @@ "python3 scripts/yao.py skill-atlas --workspace-root .", "python3 scripts/yao.py skill-os2-audit . --generated-at ", "Copy evidence/world_class/templates/native-client-telemetry.intake.json to evidence/world_class/submissions/native-client-telemetry.json and fill only real evidence fields.", - "python3 scripts/yao.py world-class-intake ." + "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions" ], "success_checks": [ "reports/adoption_drift_report.json summary.source_types.external > 0", @@ -4764,7 +4764,7 @@ "ok": true, "generated_at": "2026-06-14", "skill_dir": ".", - "commit": "07274b4c50068d4635583631c55e5a6e3fc00a7a", + "commit": "328be185f6199ff1d7041378c97657b42e7e3c42", "git_status": { "available": true, "dirty": false, @@ -4778,9 +4778,9 @@ "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "69dfb6c9b80bc88d8043270714359b8f8074bfc60f5fb811d8a917a0da087aef", - "source_contract_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5", - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "evidence_bundle_sha256": "1e0bb8068a01ef836225760c31bdd6fa04dbf4014643f5b4532d5241357c4554", + "source_contract_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 21, @@ -4811,7 +4811,7 @@ }, "release_lock": { "ready": true, - "commit": "07274b4c50068d4635583631c55e5a6e3fc00a7a", + "commit": "328be185f6199ff1d7041378c97657b42e7e3c42", "status_scope": "generation-time status before this report is written", "reason": "clean generation-time HEAD" }, @@ -4821,7 +4821,7 @@ "existing_count": 24, "missing_count": 0, "missing_paths": [], - "sha256": "69dfb6c9b80bc88d8043270714359b8f8074bfc60f5fb811d8a917a0da087aef" + "sha256": "1e0bb8068a01ef836225760c31bdd6fa04dbf4014643f5b4532d5241357c4554" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -4930,7 +4930,7 @@ "path": "reports/security_trust_report.json", "exists": true, "bytes": 98830, - "sha256": "871eb9e4f88b0b91c211af94a66660f1c4e799907466362c15ebe63a9128bab1" + "sha256": "39ddc420a6f8d6e0b991b54311c61ce2f7be8de8e6c7241eda8d0ffe9434bb39" }, { "label": "python_compatibility", @@ -4944,14 +4944,14 @@ "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, - "sha256": "1b2feee7ce87257d042d6a2b9cb75bd2e495f4676ff20c21a7b3508c87aa6248" + "sha256": "d02dbd2d31655fd27613765d3395212d53ca07c8ef028deeff64732ca5059dbc" }, { "label": "package_verification", "path": "reports/package_verification.json", "exists": true, "bytes": 19325, - "sha256": "6e885151878efa803cb5e85d0636b493015729cbf598e22cafffb30b8df2936a" + "sha256": "e101808a5f8f97126c02fedac757d545fef3e65a7f01d9d1da2774d314af0840" }, { "label": "install_simulation", @@ -4971,8 +4971,8 @@ "label": "world_class_evidence_plan", "path": "reports/world_class_evidence_plan.json", "exists": true, - "bytes": 19532, - "sha256": "a9bd392cd45f3accabe9b5befb9090456f383b1edbc323cab7d20930179dc229" + "bytes": 19940, + "sha256": "164803bb1cea795c8b962ec82cb1b3a4a58bbc51e10a4edfce12de00d046905a" }, { "label": "world_class_evidence_ledger", @@ -5103,22 +5103,22 @@ }, { "label": "world-class evidence ledger", - "command": "python3 scripts/yao.py world-class-ledger .", + "command": "python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_evidence_ledger.json" }, { "label": "world-class evidence intake", - "command": "python3 scripts/yao.py world-class-intake .", + "command": "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_evidence_intake.json" }, { "label": "world-class submission review", - "command": "python3 scripts/yao.py world-class-submission-review .", + "command": "python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_submission_review.json" }, { "label": "world-class operator runbook", - "command": "python3 scripts/yao.py world-class-runbook .", + "command": "python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions", "evidence": "reports/world_class_operator_runbook.json" }, { @@ -10653,7 +10653,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 181, - "package_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5" + "package_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397" }, "failures": [], "warnings": [], @@ -14172,7 +14172,7 @@ }, { "path": "tests/verify_review_studio.py", - "lines": 679, + "lines": 680, "kind": "test", "severity": "pass", "recommendation": "Break broad integration assertions into focused verifier helpers when the next behavior change lands." @@ -14207,15 +14207,15 @@ "context_budget_tier": "production", "context_budget_limit": 1000, "skill_body_tokens": 767, - "other_text_tokens": 1203641, + "other_text_tokens": 1204831, "estimated_initial_load_tokens": 960, - "estimated_total_text_tokens": 1204408, - "deferred_resource_tokens": 388115, + "estimated_total_text_tokens": 1205598, + "deferred_resource_tokens": 388240, "deferred_resource_warn_threshold": 120000, "deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 340379, + "estimated_tokens": 340504, "file_count": 97 }, { @@ -14232,7 +14232,7 @@ "large_deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 340379, + "estimated_tokens": 340504, "file_count": 97 } ], @@ -14255,14 +14255,14 @@ ], "missing": [], "path": "scripts", - "estimated_tokens": 340379, + "estimated_tokens": 340504, "file_count": 97, "rationale": "Script resources are deterministic deferred tools, not initial-load prompt context." } ], "summary": "Large deferred resources are indexed and backed by evidence." }, - "relevant_file_count": 509, + "relevant_file_count": 510, "unused_resource_dirs": [], "quality_signal_points": 130, "quality_density": 135.4 @@ -16203,7 +16203,7 @@ "schema_version": "1.0", "ok": true, "skill_dir": ".", - "generated_at": "2026-06-13", + "generated_at": "2026-06-14", "summary": { "waiver_count": 0, "active_count": 0, @@ -16253,7 +16253,7 @@ "Reviewer links output_review_adjudication or output_execution evidence." ], "suggested_evidence": "reports/output_review_adjudication.md", - "suggested_command": "python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer \"\" --reason \"Output Lab has pending human/provider evidence; accepted only for this bounded review scope.\" --expires-at 2027-06-13 --evidence reports/output_review_adjudication.md", + "suggested_command": "python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer \"\" --reason \"Output Lab has pending human/provider evidence; accepted only for this bounded review scope.\" --expires-at 2027-06-14 --evidence reports/output_review_adjudication.md", "world_class_boundary": "Does not count as provider, human, or public world-class completion evidence." }, { @@ -16268,7 +16268,7 @@ "Keep claim guard active until ledger summary.ready_to_claim_world_class is true." ], "suggested_evidence": "reports/world_class_evidence_ledger.md", - "suggested_command": "python3 scripts/yao.py world-class-ledger . && python3 scripts/yao.py world-class-claim-guard .", + "suggested_command": "python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions && python3 scripts/yao.py world-class-claim-guard .", "world_class_boundary": "Non-waivable completion boundary." } ], @@ -17748,8 +17748,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5", - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f" + "package_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798" }, "compatibility": { "openai": "pass", @@ -17780,7 +17780,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -17805,7 +17805,7 @@ "vscode" ], "package_metadata": "registry/packages/yao-meta-skill.json", - "package_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5" + "package_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397" } ] }, @@ -17828,7 +17828,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "archive_entry_count": 576, "failure_count": 0, "warning_count": 0 @@ -18836,12 +18836,12 @@ { "field": "archive_sha256", "from": "", - "to": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f" + "to": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5" + "to": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397" } ] }, diff --git a/reports/skill-overview.json b/reports/skill-overview.json index e0534b2..b732291 100644 --- a/reports/skill-overview.json +++ b/reports/skill-overview.json @@ -922,9 +922,9 @@ "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "69dfb6c9b80bc88d8043270714359b8f8074bfc60f5fb811d8a917a0da087aef", - "source_contract_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5", - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "evidence_bundle_sha256": "1e0bb8068a01ef836225760c31bdd6fa04dbf4014643f5b4532d5241357c4554", + "source_contract_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 21, @@ -943,7 +943,7 @@ "working_tree_dirty": false, "changed_file_count": 0 }, - "commit": "07274b4c50068d4635583631c55e5a6e3fc00a7a", + "commit": "328be185f6199ff1d7041378c97657b42e7e3c42", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1048,7 +1048,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 181, - "package_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5" + "package_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397" }, "skill_atlas": { "skill_count": 12, @@ -1086,8 +1086,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5", - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f" + "package_sha256": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798" }, "compatibility": { "openai": "pass", @@ -1118,7 +1118,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -1134,7 +1134,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f", + "archive_sha256": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798", "archive_entry_count": 576, "failure_count": 0, "warning_count": 0 @@ -1213,12 +1213,12 @@ { "field": "archive_sha256", "from": "", - "to": "7b0683ec4b2e6cf991a023ca137f5734612c64410a8a0b5fc2adae97595b470f" + "to": "0488c5b84d906354c41570de908244ed8af5d55b55f3f01f94bdfdff99851798" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "f6672acad15cd131632032e808c09ee71ea1da1b4714cfedcf8dd35b0808fcf5" + "to": "7bd945074bb9066661e3f4f1c2b45a6147d0bdd467cadc3e96af1ce129dc3397" } ] }, @@ -1361,7 +1361,7 @@ "YAO_OUTPUT_EVAL_MODEL=gpt-4.1-mini OPENAI_API_KEY= python3 scripts/yao.py output-exec --provider-runner openai --timeout-seconds 60", "python3 scripts/yao.py skill-os2-audit . --generated-at ", "Copy evidence/world_class/templates/provider-holdout.intake.json to evidence/world_class/submissions/provider-holdout.json and fill only real evidence fields.", - "python3 scripts/yao.py world-class-intake ." + "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions" ], "success_checks": [ "reports/output_execution_runs.json summary.model_executed_count > 0", @@ -1400,7 +1400,7 @@ "python3 scripts/yao.py output-review", "python3 scripts/yao.py skill-os2-audit . --generated-at ", "Copy evidence/world_class/templates/human-adjudication.intake.json to evidence/world_class/submissions/human-adjudication.json and fill only real evidence fields.", - "python3 scripts/yao.py world-class-intake ." + "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions" ], "success_checks": [ "reports/output_review_adjudication.json summary.pending_count == 0", @@ -1441,7 +1441,7 @@ "python3 scripts/yao.py runtime-permissions . --package-dir dist", "python3 scripts/yao.py skill-os2-audit . --generated-at ", "Copy evidence/world_class/templates/native-permission-enforcement.intake.json to evidence/world_class/submissions/native-permission-enforcement.json and fill only real evidence fields.", - "python3 scripts/yao.py world-class-intake ." + "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions" ], "success_checks": [ "reports/runtime_permission_probes.json summary.native_enforcement_count > 0", @@ -1482,7 +1482,7 @@ "python3 scripts/yao.py skill-atlas --workspace-root .", "python3 scripts/yao.py skill-os2-audit . --generated-at ", "Copy evidence/world_class/templates/native-client-telemetry.intake.json to evidence/world_class/submissions/native-client-telemetry.json and fill only real evidence fields.", - "python3 scripts/yao.py world-class-intake ." + "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions" ], "success_checks": [ "reports/adoption_drift_report.json summary.source_types.external > 0",