diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json index 5510679..19e2d32 100644 --- a/reports/benchmark_reproducibility.json +++ b/reports/benchmark_reproducibility.json @@ -3,30 +3,17 @@ "ok": true, "generated_at": "2026-06-16", "skill_dir": ".", - "commit": "c9a7842991900287ce259e1b624b7f36e3750265", + "commit": "09518a4d8637c430311d50a4249913187dbc5057", "git_status": { "available": true, - "dirty": true, - "changed_file_count": 30, - "sample": [ - " M registry/index.json", - " M registry/packages/yao-meta-skill.json", - " M reports/adoption_drift_report.json", - " M reports/architecture_maintainability.json", - " M reports/architecture_maintainability.md", - " M reports/benchmark_reproducibility.json", - " M reports/benchmark_reproducibility.md", - " M reports/context_budget.json", - " M reports/context_budget.md", - " M reports/context_budget_summary.json", - " M reports/evidence_consistency.json", - " M reports/evidence_consistency.md" - ], + "dirty": false, + "changed_file_count": 0, + "sample": [], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, @@ -50,15 +37,14 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 30 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ - "release lock is not clean or commit is unavailable", "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", @@ -67,10 +53,10 @@ "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { - "ready": false, - "commit": "c9a7842991900287ce259e1b624b7f36e3750265", + "ready": true, + "commit": "09518a4d8637c430311d50a4249913187dbc5057", "status_scope": "generation-time status before this report is written", - "reason": "working tree was dirty at generation time" + "reason": "clean generation-time HEAD" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md index 44f9dd3..0445502 100644 --- a/reports/benchmark_reproducibility.md +++ b/reports/benchmark_reproducibility.md @@ -1,14 +1,14 @@ # Benchmark Reproducibility Generated at: `2026-06-16` -Commit: `c9a7842991900287ce259e1b624b7f36e3750265` -Working tree dirty at generation: `true` +Commit: `09518a4d8637c430311d50a4249913187dbc5057` +Working tree dirty at generation: `false` Evidence bundle SHA256: `9bc8af4e6aa04e1f61bc42b432335e0f3500515d3e3d59a6c430ce8c1177fbbe` ## Summary - reproducibility ready: `true` -- release lock ready: `false` +- release lock ready: `true` - methodology complete: `true` - required artifacts: `24` - missing artifacts: `0` @@ -22,8 +22,8 @@ Evidence bundle SHA256: `9bc8af4e6aa04e1f61bc42b432335e0f3500515d3e3d59a6c430ce8 - world-class ready: `false` - world-class source checks: `6` pass / `13` total; `7` blocked - public claim ready: `false` -- public claim blockers: `5` -- changed files at generation: `30` +- public claim blockers: `4` +- changed files at generation: `0` This report proves local benchmark reproducibility only. It keeps external provider and human-review gaps visible instead of counting them as complete. The git commit is generation-time context; the evidence bundle SHA is the durable anchor for the artifacts listed below. @@ -35,7 +35,6 @@ This report proves local benchmark reproducibility only. It keeps external provi | Blocker | | --- | -| release lock is not clean or commit is unavailable | | provider-backed model holdout evidence is incomplete | | human blind-review adjudication is incomplete | | world-class evidence is not accepted yet (4 open gaps, 4 ledger pending) | @@ -43,8 +42,8 @@ This report proves local benchmark reproducibility only. It keeps external provi ## Release Lock -- ready: `false` -- reason: working tree was dirty at generation time +- ready: `true` +- reason: clean generation-time HEAD - status scope: generation-time status before this report is written ## Evidence Bundle diff --git a/reports/evidence_consistency.json b/reports/evidence_consistency.json index 15ab4bc..548c9a3 100644 --- a/reports/evidence_consistency.json +++ b/reports/evidence_consistency.json @@ -47,8 +47,8 @@ "key": "benchmark-release-lock-self-consistency", "label": "Benchmark release lock matches git dirty state", "status": "pass", - "expected": false, - "actual": false, + "expected": true, + "actual": true, "paths": [ "reports/benchmark_reproducibility.json" ], @@ -58,8 +58,8 @@ "key": "overview-benchmark-commit", "label": "overview embeds the benchmark commit", "status": "pass", - "expected": "c9a7842991900287ce259e1b624b7f36e3750265", - "actual": "c9a7842991900287ce259e1b624b7f36e3750265", + "expected": "09518a4d8637c430311d50a4249913187dbc5057", + "actual": "09518a4d8637c430311d50a4249913187dbc5057", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-overview.json" @@ -71,7 +71,7 @@ "label": "overview embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": false, + "release_lock_ready": true, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "f48a71e36bdbcaa353bd4e6bfc499bda41f047aeffe3d8eae2f99509ec693086", @@ -81,10 +81,10 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5 + "public_claim_blocker_count": 4 }, "actual": { - "release_lock_ready": false, + "release_lock_ready": true, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "f48a71e36bdbcaa353bd4e6bfc499bda41f047aeffe3d8eae2f99509ec693086", @@ -94,7 +94,7 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5 + "public_claim_blocker_count": 4 }, "paths": [ "reports/benchmark_reproducibility.json", @@ -204,8 +204,8 @@ "key": "interpretation-benchmark-commit", "label": "interpretation embeds the benchmark commit", "status": "pass", - "expected": "c9a7842991900287ce259e1b624b7f36e3750265", - "actual": "c9a7842991900287ce259e1b624b7f36e3750265", + "expected": "09518a4d8637c430311d50a4249913187dbc5057", + "actual": "09518a4d8637c430311d50a4249913187dbc5057", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-interpretation.json" @@ -217,7 +217,7 @@ "label": "interpretation embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": false, + "release_lock_ready": true, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "f48a71e36bdbcaa353bd4e6bfc499bda41f047aeffe3d8eae2f99509ec693086", @@ -227,10 +227,10 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5 + "public_claim_blocker_count": 4 }, "actual": { - "release_lock_ready": false, + "release_lock_ready": true, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "f48a71e36bdbcaa353bd4e6bfc499bda41f047aeffe3d8eae2f99509ec693086", @@ -240,7 +240,7 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5 + "public_claim_blocker_count": 4 }, "paths": [ "reports/benchmark_reproducibility.json", diff --git a/reports/review-studio.html b/reports/review-studio.html index 0a0fc48..3b3c644 100644 --- a/reports/review-studio.html +++ b/reports/review-studio.html @@ -676,12 +676,12 @@

核心指标

-
Skill IR2.0.0

5 targets in platform-neutral contract

Compiler5/5

target contracts compiled from Skill IR

Output Delta100.0

5 cases; 1 file-backed

Exec Runs10

command 10; model 0; recorded 0

Blind A/B5

review pairs hide baseline vs with-skill labels

Review Kit0/5

pending 5; answer key hidden

Review A/B0/5

adjudication decisions; pending 5

Public Claimblocked

5 blockers; local reproducible true

Blueprint21/21

2.0 coverage; extensions partial 0, planned 0; evidence pending 4

Runtime5/5

target conformance pass rate

Perm Probe4/4

0 native; 4 installer-enforced

Trust0

110 scripts scanned; secrets found

Py Compat0

176 files scanned for Python 3.11

Arch Debt0

899 largest lines; 8 watchlist; 64 CLI handlers; 18 in entrypoint

Atlas5

12 scanned skills; route collisions

Driftlow

1 metadata events; 0 missed triggers

Waivers0

0 gates covered; human risk decisions

Intake4/4

0 valid submissions; 0 invalid

Claim Guard0

173 public surfaces scanned

Notes0/0

0 open blocker annotations

Registry1.1.0

5 targets; MIT license

Archivepass

617 zip entries; package verification

Installpass

4 adapters; 12 permissions enforced; 0 permission failures

Upgrademinor

declared minor; 0 breaking changes

+
Skill IR2.0.0

5 targets in platform-neutral contract

Compiler5/5

target contracts compiled from Skill IR

Output Delta100.0

5 cases; 1 file-backed

Exec Runs10

command 10; model 0; recorded 0

Blind A/B5

review pairs hide baseline vs with-skill labels

Review Kit0/5

pending 5; answer key hidden

Review A/B0/5

adjudication decisions; pending 5

Public Claimblocked

4 blockers; local reproducible true

Blueprint21/21

2.0 coverage; extensions partial 0, planned 0; evidence pending 4

Runtime5/5

target conformance pass rate

Perm Probe4/4

0 native; 4 installer-enforced

Trust0

110 scripts scanned; secrets found

Py Compat0

176 files scanned for Python 3.11

Arch Debt0

899 largest lines; 9 watchlist; 64 CLI handlers; 18 in entrypoint

Atlas5

12 scanned skills; route collisions

Driftlow

1 metadata events; 0 missed triggers

Waivers0

0 gates covered; human risk decisions

Intake4/4

0 valid submissions; 0 invalid

Claim Guard0

173 public surfaces scanned

Notes0/0

0 open blocker annotations

Registry1.1.0

5 targets; MIT license

Archivepass

617 zip entries; package verification

Installpass

4 adapters; 12 permissions enforced; 0 permission failures

Upgrademinor

declared minor; 0 breaking changes

审查闸门

-
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 960/1000; deferred 434418/120000; top deferred scripts 385207; resource governance governed; quality density 135.4

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 110 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

Python 兼容

Python 3.11; 176 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards

reports/python_compatibility.json 证据
通过

架构维护

173 Python files; 0 hotspots; 8 watchlist files; 0 blockers; largest 899 lines; 64 CLI handlers; 18 in entrypoint

reports/architecture_maintainability.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low

reports/adoption_drift_report.json 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

reports/promotion_decisions.json + reports/upgrade_check.json + docs/migration-v2.md 证据
+
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 960/1000; deferred 436849/120000; top deferred scripts 387638; resource governance governed; quality density 135.4

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 110 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

Python 兼容

Python 3.11; 176 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards

reports/python_compatibility.json 证据
通过

架构维护

173 Python files; 0 hotspots; 9 watchlist files; 0 blockers; largest 899 lines; 64 CLI handlers; 18 in entrypoint

reports/architecture_maintainability.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low

reports/adoption_drift_report.json 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

@@ -743,7 +743,7 @@
-

上下文

initial load 960/1000; deferred 434418/120000; top deferred scripts 385207; resource governance governed; quality density 135.4

+

上下文

initial load 960/1000; deferred 436849/120000; top deferred scripts 387638; resource governance governed; quality density 135.4

编译证据

Review reports/compiled_targets.md before packaging to inspect target adapter modes, generated files, preserved semantics, warnings, and unsupported features.

@@ -806,8 +806,8 @@
-

公开声明

本地复现
发布锁
可公开声明
声明阻断
5
Provider 证据
人审完成
世界级就绪
-

声明阻断

  • 阻断release lock is not clean or commit is unavailable
  • 阻断provider-backed model holdout evidence is incomplete
  • 阻断human blind-review adjudication is incomplete
  • 阻断world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)
  • 阻断world-class source checks are not all accepted (6/13 pass, 7 blocked)
+

公开声明

本地复现
发布锁
可公开声明
声明阻断
4
Provider 证据
人审完成
世界级就绪
+

声明阻断

  • 阻断provider-backed model holdout evidence is incomplete
  • 阻断human blind-review adjudication is incomplete
  • 阻断world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)
  • 阻断world-class source checks are not all accepted (6/13 pass, 7 blocked)
diff --git a/reports/review-studio.json b/reports/review-studio.json index 595e0ea..5a20c2f 100644 --- a/reports/review-studio.json +++ b/reports/review-studio.json @@ -43,7 +43,7 @@ "key": "context-budget", "label": "上下文", "status": "pass", - "detail": "initial load 960/1000; deferred 434418/120000; top deferred scripts 385207; resource governance governed; quality density 135.4", + "detail": "initial load 960/1000; deferred 436849/120000; top deferred scripts 387638; resource governance governed; quality density 135.4", "evidence": "reports/context_budget.json", "link": "context_budget.md" }, @@ -75,7 +75,7 @@ "key": "architecture-maintainability", "label": "架构维护", "status": "pass", - "detail": "173 Python files; 0 hotspots; 8 watchlist files; 0 blockers; largest 899 lines; 64 CLI handlers; 18 in entrypoint", + "detail": "173 Python files; 0 hotspots; 9 watchlist files; 0 blockers; largest 899 lines; 64 CLI handlers; 18 in entrypoint", "evidence": "reports/architecture_maintainability.json", "link": "architecture_maintainability.md" }, @@ -1676,13 +1676,13 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "a76544029263cb74d648e507a64bc91f1222fadeed84f128dc62ed3faafd4b4c", - "source_contract_sha256": "caa815571f7c86fa95447cd7407875eb452cac255c45360db63b017699425921", - "archive_sha256": "6721f9af9fc9784b0491f8e73dd92db8a38b3629fc08170ddd3269f2a3c9b48f", + "evidence_bundle_sha256": "9bc8af4e6aa04e1f61bc42b432335e0f3500515d3e3d59a6c430ce8c1177fbbe", + "source_contract_sha256": "f48a71e36bdbcaa353bd4e6bfc499bda41f047aeffe3d8eae2f99509ec693086", + "archive_sha256": "00151114ce7fee627882964137dd5259583f5521c75c6570789345e874454137", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 22, @@ -1700,11 +1700,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 2 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, - "commit": "c9a7842991900287ce259e1b624b7f36e3750265", + "commit": "09518a4d8637c430311d50a4249913187dbc5057", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -4346,7 +4346,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.63, + "duration_ms": 26.28, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4374,7 +4374,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.13, + "duration_ms": 26.19, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4397,7 +4397,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.15, + "duration_ms": 26.08, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4425,7 +4425,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 29.26, + "duration_ms": 26.33, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4448,7 +4448,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 29.51, + "duration_ms": 25.89, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4476,7 +4476,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.2, + "duration_ms": 26.56, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4499,7 +4499,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 25.82, + "duration_ms": 26.12, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4526,7 +4526,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.22, + "duration_ms": 26.4, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4549,7 +4549,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.09, + "duration_ms": 26.2, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4578,7 +4578,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.0, + "duration_ms": 26.1, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -5299,26 +5299,23 @@ "ok": true, "generated_at": "2026-06-16", "skill_dir": ".", - "commit": "c9a7842991900287ce259e1b624b7f36e3750265", + "commit": "09518a4d8637c430311d50a4249913187dbc5057", "git_status": { "available": true, - "dirty": true, - "changed_file_count": 2, - "sample": [ - " M scripts/render_evidence_consistency.py", - " M tests/verify_evidence_consistency.py" - ], + "dirty": false, + "changed_file_count": 0, + "sample": [], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "a76544029263cb74d648e507a64bc91f1222fadeed84f128dc62ed3faafd4b4c", - "source_contract_sha256": "caa815571f7c86fa95447cd7407875eb452cac255c45360db63b017699425921", - "archive_sha256": "6721f9af9fc9784b0491f8e73dd92db8a38b3629fc08170ddd3269f2a3c9b48f", + "evidence_bundle_sha256": "9bc8af4e6aa04e1f61bc42b432335e0f3500515d3e3d59a6c430ce8c1177fbbe", + "source_contract_sha256": "f48a71e36bdbcaa353bd4e6bfc499bda41f047aeffe3d8eae2f99509ec693086", + "archive_sha256": "00151114ce7fee627882964137dd5259583f5521c75c6570789345e874454137", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 22, @@ -5336,15 +5333,14 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 2 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ - "release lock is not clean or commit is unavailable", "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", @@ -5353,10 +5349,10 @@ "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { - "ready": false, - "commit": "c9a7842991900287ce259e1b624b7f36e3750265", + "ready": true, + "commit": "09518a4d8637c430311d50a4249913187dbc5057", "status_scope": "generation-time status before this report is written", - "reason": "working tree was dirty at generation time" + "reason": "clean generation-time HEAD" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", @@ -5364,7 +5360,7 @@ "existing_count": 24, "missing_count": 0, "missing_paths": [], - "sha256": "a76544029263cb74d648e507a64bc91f1222fadeed84f128dc62ed3faafd4b4c" + "sha256": "9bc8af4e6aa04e1f61bc42b432335e0f3500515d3e3d59a6c430ce8c1177fbbe" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -5437,8 +5433,8 @@ "label": "output_execution", "path": "reports/output_execution_runs.json", "exists": true, - "bytes": 7967, - "sha256": "0e4a474685833f04094d42f56dd84af5722185ece221fa6b63ec0996e84921d0" + "bytes": 7964, + "sha256": "30b43454ef2a4802b13d554100789b1c9c7a1baec9bebcd6fc8cf0ea9830ba41" }, { "label": "blind_review", @@ -5473,7 +5469,7 @@ "path": "reports/security_trust_report.json", "exists": true, "bytes": 109099, - "sha256": "4fd98fe60dcdd1a0b8c6d0e6569bd21b5a07d695a0987a2f771d4e7b25fdec46" + "sha256": "491354ae881f6682687a62a98c9a4988bbeef931b24565fdf43e6f3a206fd449" }, { "label": "python_compatibility", @@ -5487,14 +5483,14 @@ "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, - "sha256": "7c901c1abbfa4ef1e8216f5ff4426fd41f929034722914d35d5d9995c6b215e7" + "sha256": "870c455a5cd209ff8d9e8793861da074e668083d265230d04e38e8ee0ff1323a" }, { "label": "package_verification", "path": "reports/package_verification.json", "exists": true, "bytes": 19338, - "sha256": "0ffbc5dcb848ab2965a6e01708d7c39d243c98b8fa387ecef5280d04f112da6e" + "sha256": "c4acb824482d992e0e4339b7eeeb8d23d49da12d3414606222f4916c3ed09df6" }, { "label": "install_simulation", @@ -6311,7 +6307,7 @@ "label": "Evidence Consistency", "status": "pass", "objective": "Recommended Skill OS 2.0 implementation PR from the upgrade plan.", - "current": "28 consistency checks", + "current": "29 consistency checks", "command": "make ci-test", "test": "tests/verify_evidence_consistency.py", "evidence": [ @@ -15240,7 +15236,7 @@ "watch_line_threshold": 720, "block_line_threshold": 1500, "largest_file_lines": 899, - "watchlist_count": 8, + "watchlist_count": 9, "hotspot_count": 0, "blocker_count": 0, "decision": "pass" @@ -15253,6 +15249,13 @@ "severity": "pass", "recommendation": "Break broad integration assertions into focused verifier helpers when the next behavior change lands." }, + { + "path": "scripts/render_evidence_consistency.py", + "lines": 859, + "kind": "cli-script", + "severity": "pass", + "recommendation": "Watch this file before adding new responsibilities; extract a helper module when one concern dominates." + }, { "path": "scripts/yao_cli_parser.py", "lines": 810, @@ -15322,13 +15325,6 @@ "kind": "internal-module", "severity": "pass", "recommendation": "Watch this file before adding new responsibilities; extract a helper module when one concern dominates." - }, - { - "path": "scripts/render_review_viewer.py", - "lines": 685, - "kind": "cli-script", - "severity": "pass", - "recommendation": "Split viewer data assembly from HTML section rendering." } ], "watchlist": [ @@ -15339,6 +15335,13 @@ "severity": "pass", "recommendation": "Break broad integration assertions into focused verifier helpers when the next behavior change lands." }, + { + "path": "scripts/render_evidence_consistency.py", + "lines": 859, + "kind": "cli-script", + "severity": "pass", + "recommendation": "Watch this file before adding new responsibilities; extract a helper module when one concern dominates." + }, { "path": "scripts/yao_cli_parser.py", "lines": 810, @@ -15404,15 +15407,15 @@ "context_budget_tier": "production", "context_budget_limit": 1000, "skill_body_tokens": 767, - "other_text_tokens": 1370410, + "other_text_tokens": 1376201, "estimated_initial_load_tokens": 960, - "estimated_total_text_tokens": 1371177, - "deferred_resource_tokens": 434418, + "estimated_total_text_tokens": 1376968, + "deferred_resource_tokens": 436849, "deferred_resource_warn_threshold": 120000, "deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 385207, + "estimated_tokens": 387638, "file_count": 110 }, { @@ -15429,7 +15432,7 @@ "large_deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 385207, + "estimated_tokens": 387638, "file_count": 110 } ], @@ -15452,7 +15455,7 @@ ], "missing": [], "path": "scripts", - "estimated_tokens": 385207, + "estimated_tokens": 387638, "file_count": 110, "rationale": "Script resources are deterministic deferred tools, not initial-load prompt context." } @@ -17327,7 +17330,7 @@ "adoption_drift": { "ok": true, "schema_version": "2.0", - "generated_at": "2026-06-15T16:51:01Z", + "generated_at": "2026-06-15T16:57:40Z", "skill_dir": ".", "privacy_contract": { "storage": "local-first", diff --git a/reports/review-viewer.json b/reports/review-viewer.json index 8161618..44cc057 100644 --- a/reports/review-viewer.json +++ b/reports/review-viewer.json @@ -996,7 +996,7 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, @@ -1020,11 +1020,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 30 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, - "commit": "c9a7842991900287ce259e1b624b7f36e3750265", + "commit": "09518a4d8637c430311d50a4249913187dbc5057", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", diff --git a/reports/skill-interpretation.json b/reports/skill-interpretation.json index ec3303f..4bf0465 100644 --- a/reports/skill-interpretation.json +++ b/reports/skill-interpretation.json @@ -1000,7 +1000,7 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, @@ -1024,11 +1024,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 30 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, - "commit": "c9a7842991900287ce259e1b624b7f36e3750265", + "commit": "09518a4d8637c430311d50a4249913187dbc5057", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", diff --git a/reports/skill-overview.json b/reports/skill-overview.json index 77babfa..dc56723 100644 --- a/reports/skill-overview.json +++ b/reports/skill-overview.json @@ -995,7 +995,7 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, @@ -1019,11 +1019,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 30 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, - "commit": "c9a7842991900287ce259e1b624b7f36e3750265", + "commit": "09518a4d8637c430311d50a4249913187dbc5057", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.",