5 targets in platform-neutral contract
diff --git a/registry/index.json b/registry/index.json
index 611cd1f..e82fd8f 100644
--- a/registry/index.json
+++ b/registry/index.json
@@ -16,7 +16,7 @@
"vscode"
],
"package_metadata": "registry/packages/yao-meta-skill.json",
- "package_sha256": "1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642"
+ "package_sha256": "363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453"
}
]
}
diff --git a/registry/packages/yao-meta-skill.json b/registry/packages/yao-meta-skill.json
index aac73e6..ca8b6f6 100644
--- a/registry/packages/yao-meta-skill.json
+++ b/registry/packages/yao-meta-skill.json
@@ -16,8 +16,8 @@
"trust_level": "local",
"license": "MIT",
"checksums": {
- "package_sha256": "1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642",
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288"
+ "package_sha256": "363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482"
},
"compatibility": {
"openai": "pass",
@@ -48,7 +48,7 @@
},
"distribution": {
"archive_verified": true,
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482",
"package_verification": "reports/package_verification.json",
"install_simulated": true,
"install_simulation": "reports/install_simulation.json"
diff --git a/reports/architecture_maintainability.json b/reports/architecture_maintainability.json
index 4267177..469f18b 100644
--- a/reports/architecture_maintainability.json
+++ b/reports/architecture_maintainability.json
@@ -15,7 +15,7 @@
"warn_line_threshold": 900,
"watch_line_threshold": 720,
"block_line_threshold": 1500,
- "largest_file_lines": 836,
+ "largest_file_lines": 837,
"watchlist_count": 8,
"hotspot_count": 0,
"blocker_count": 0,
@@ -24,7 +24,7 @@
"largest_files": [
{
"path": "tests/verify_review_studio.py",
- "lines": 836,
+ "lines": 837,
"kind": "test",
"severity": "pass",
"recommendation": "Break broad integration assertions into focused verifier helpers when the next behavior change lands."
@@ -110,7 +110,7 @@
"watchlist": [
{
"path": "tests/verify_review_studio.py",
- "lines": 836,
+ "lines": 837,
"kind": "test",
"severity": "pass",
"recommendation": "Break broad integration assertions into focused verifier helpers when the next behavior change lands."
diff --git a/reports/architecture_maintainability.md b/reports/architecture_maintainability.md
index 6fabd89..df9f363 100644
--- a/reports/architecture_maintainability.md
+++ b/reports/architecture_maintainability.md
@@ -13,7 +13,7 @@ Generated at: `2026-06-16`
- Yao CLI command handlers: `68`
- entrypoint command handlers: `18`
- command modules: `6`
-- largest file lines: `836`
+- largest file lines: `837`
- watch threshold lines: `720`
- watchlist: `8`
- hotspots: `0`
@@ -29,7 +29,7 @@ No file-size hotspots found.
| File | Lines | Kind | Recommended next split |
| --- | ---: | --- | --- |
-| `tests/verify_review_studio.py` | `836` | `test` | Break broad integration assertions into focused verifier helpers when the next behavior change lands. |
+| `tests/verify_review_studio.py` | `837` | `test` | Break broad integration assertions into focused verifier helpers when the next behavior change lands. |
| `scripts/skill_report_model.py` | `800` | `internal-module` | Watch this file before adding new responsibilities; extract a helper module when one concern dominates. |
| `tests/verify_yao_cli.py` | `785` | `test` | Break broad integration assertions into focused verifier helpers when the next behavior change lands. |
| `scripts/render_evidence_consistency.py` | `766` | `cli-script` | Watch this file before adding new responsibilities; extract a helper module when one concern dominates. |
@@ -42,7 +42,7 @@ No file-size hotspots found.
| File | Lines | Kind | Severity |
| --- | ---: | --- | --- |
-| `tests/verify_review_studio.py` | `836` | `test` | `pass` |
+| `tests/verify_review_studio.py` | `837` | `test` | `pass` |
| `scripts/skill_report_model.py` | `800` | `internal-module` | `pass` |
| `tests/verify_yao_cli.py` | `785` | `test` | `pass` |
| `scripts/render_evidence_consistency.py` | `766` | `cli-script` | `pass` |
diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json
index 88431af..9590cab 100644
--- a/reports/benchmark_reproducibility.json
+++ b/reports/benchmark_reproducibility.json
@@ -3,23 +3,36 @@
"ok": true,
"generated_at": "2026-06-16",
"skill_dir": ".",
- "commit": "499214b052869193d96cdd7e73f537a57eb7bb8d",
+ "commit": "ca94d734ee43bd514d4bf83f5c9a7cb7bdc161da",
"git_status": {
"available": true,
- "dirty": false,
- "changed_file_count": 0,
- "sample": [],
+ "dirty": true,
+ "changed_file_count": 32,
+ "sample": [
+ " M registry/index.json",
+ " M registry/packages/yao-meta-skill.json",
+ " M reports/architecture_maintainability.json",
+ " M reports/architecture_maintainability.md",
+ " M reports/benchmark_reproducibility.json",
+ " M reports/benchmark_reproducibility.md",
+ " M reports/context_budget.json",
+ " M reports/context_budget.md",
+ " M reports/context_budget_summary.json",
+ " M reports/evidence_consistency.json",
+ " M reports/output_execution_runs.json",
+ " M reports/output_execution_runs.md"
+ ],
"scope": "generation-time status before this report is written"
},
"summary": {
"reproducibility_ready": true,
- "release_lock_ready": true,
+ "release_lock_ready": false,
"methodology_complete": true,
"required_artifact_count": 25,
"missing_artifact_count": 0,
- "evidence_bundle_sha256": "7540215c8aa2db458cd10433f2cac56bc2931c58df057232309fdae3a5f8e0e8",
- "source_contract_sha256": "1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642",
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288",
+ "evidence_bundle_sha256": "813517ab4bc2b504a720544f29c80e1bf7b09bbd1938ffdbb840b33aa5abb76a",
+ "source_contract_sha256": "363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482",
"output_case_count": 5,
"failure_disclosure_count": 3,
"command_count": 23,
@@ -37,14 +50,15 @@
"world_class_source_pass_count": 6,
"world_class_source_blocked_count": 7,
"public_claim_ready": false,
- "public_claim_blocker_count": 4,
- "working_tree_dirty": false,
- "changed_file_count": 0
+ "public_claim_blocker_count": 5,
+ "working_tree_dirty": true,
+ "changed_file_count": 32
},
"public_claim": {
"ready": false,
"scope": "public benchmark or world-class readiness claim",
"blockers": [
+ "release lock is not clean or commit is unavailable",
"provider-backed model holdout evidence is incomplete",
"human blind-review adjudication is incomplete",
"world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)",
@@ -53,10 +67,10 @@
"policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks."
},
"release_lock": {
- "ready": true,
- "commit": "499214b052869193d96cdd7e73f537a57eb7bb8d",
+ "ready": false,
+ "commit": "ca94d734ee43bd514d4bf83f5c9a7cb7bdc161da",
"status_scope": "generation-time status before this report is written",
- "reason": "clean generation-time HEAD"
+ "reason": "working tree was dirty at generation time"
},
"evidence_bundle": {
"algorithm": "sha256(path,label,exists,artifact_sha256)",
@@ -64,7 +78,7 @@
"existing_count": 25,
"missing_count": 0,
"missing_paths": [],
- "sha256": "7540215c8aa2db458cd10433f2cac56bc2931c58df057232309fdae3a5f8e0e8"
+ "sha256": "813517ab4bc2b504a720544f29c80e1bf7b09bbd1938ffdbb840b33aa5abb76a"
},
"methodology": {
"path": "reports/benchmark_methodology.md",
@@ -138,7 +152,7 @@
"path": "reports/output_execution_runs.json",
"exists": true,
"bytes": 7966,
- "sha256": "fd065f84dd529c38c2a7078d99f185fb23d08e3734fbdf7faab1c897f8129818"
+ "sha256": "93a5679326b29eb9d79c3ada70679f0070be6a8aed6e50d0452c854b48efa475"
},
{
"label": "blind_review",
@@ -173,7 +187,7 @@
"path": "reports/security_trust_report.json",
"exists": true,
"bytes": 120363,
- "sha256": "e6353c1589c371823d9f4944a9be8fa46f154d004668d678c3529c92143a8436"
+ "sha256": "bea31eb462ca6e8021e68181ce3c4d34eaa0122dd34476b697ba59d38391cf36"
},
{
"label": "python_compatibility",
@@ -187,14 +201,14 @@
"path": "reports/registry_audit.json",
"exists": true,
"bytes": 3183,
- "sha256": "879a11cef5138fedb93365e40253d650137c7bae0f810803242640a9c6d273c1"
+ "sha256": "c6d5cf3c3ec415480b7cc72623b1977c5a98e64707a7dbd13ba239dfd44159fe"
},
{
"label": "package_verification",
"path": "reports/package_verification.json",
"exists": true,
"bytes": 19338,
- "sha256": "45b51edb26bc22e7ab349bf8b050830d219a6971156a9c003538a99e91e47f55"
+ "sha256": "dd90aae6ad68c21bd0db241b033aa00038278fd6396e75fc42f86064911ce51b"
},
{
"label": "install_simulation",
diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md
index 07d9247..16dcac6 100644
--- a/reports/benchmark_reproducibility.md
+++ b/reports/benchmark_reproducibility.md
@@ -1,19 +1,19 @@
# Benchmark Reproducibility
Generated at: `2026-06-16`
-Commit: `499214b052869193d96cdd7e73f537a57eb7bb8d`
-Working tree dirty at generation: `false`
-Evidence bundle SHA256: `7540215c8aa2db458cd10433f2cac56bc2931c58df057232309fdae3a5f8e0e8`
+Commit: `ca94d734ee43bd514d4bf83f5c9a7cb7bdc161da`
+Working tree dirty at generation: `true`
+Evidence bundle SHA256: `813517ab4bc2b504a720544f29c80e1bf7b09bbd1938ffdbb840b33aa5abb76a`
## Summary
- reproducibility ready: `true`
-- release lock ready: `true`
+- release lock ready: `false`
- methodology complete: `true`
- required artifacts: `25`
- missing artifacts: `0`
-- source contract sha256: `1e466c4508d6`
-- archive sha256: `88dc8d392047`
+- source contract sha256: `363354808bdd`
+- archive sha256: `a5a62aeb079b`
- output cases: `5`
- disclosed failure cases: `3`
- reproduction commands: `23`
@@ -22,8 +22,8 @@ Evidence bundle SHA256: `7540215c8aa2db458cd10433f2cac56bc2931c58df057232309fdae
- world-class ready: `false`
- world-class source checks: `6` pass / `13` total; `7` blocked
- public claim ready: `false`
-- public claim blockers: `4`
-- changed files at generation: `0`
+- public claim blockers: `5`
+- changed files at generation: `32`
This report proves local benchmark reproducibility only. It keeps external provider and human-review gaps visible instead of counting them as complete. The git commit is generation-time context; the evidence bundle SHA is the durable anchor for the artifacts listed below.
@@ -35,6 +35,7 @@ This report proves local benchmark reproducibility only. It keeps external provi
| Blocker |
| --- |
+| release lock is not clean or commit is unavailable |
| provider-backed model holdout evidence is incomplete |
| human blind-review adjudication is incomplete |
| world-class evidence is not accepted yet (4 open gaps, 4 ledger pending) |
@@ -42,15 +43,15 @@ This report proves local benchmark reproducibility only. It keeps external provi
## Release Lock
-- ready: `true`
-- reason: clean generation-time HEAD
+- ready: `false`
+- reason: working tree was dirty at generation time
- status scope: generation-time status before this report is written
## Evidence Bundle
- algorithm: `sha256(path,label,exists,artifact_sha256)`
- artifacts: `25` / `25`
-- sha256: `7540215c8aa2db458cd10433f2cac56bc2931c58df057232309fdae3a5f8e0e8`
+- sha256: `813517ab4bc2b504a720544f29c80e1bf7b09bbd1938ffdbb840b33aa5abb76a`
## Methodology Sections
@@ -72,15 +73,15 @@ This report proves local benchmark reproducibility only. It keeps external provi
| output_cases | `evals/output/cases.jsonl` | present | `a6ae96857116` |
| output_schema | `evals/output/schema.json` | present | `8ee340c95064` |
| output_scorecard | `reports/output_quality_scorecard.json` | present | `0806258a8e08` |
-| output_execution | `reports/output_execution_runs.json` | present | `fd065f84dd52` |
+| output_execution | `reports/output_execution_runs.json` | present | `93a5679326b2` |
| blind_review | `reports/output_blind_review_pack.json` | present | `bbe2db8ec277` |
| review_adjudication | `reports/output_review_adjudication.json` | present | `bb8c72a9291e` |
| trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` |
| runtime_conformance | `reports/conformance_matrix.json` | present | `97f9ba949c23` |
-| trust_report | `reports/security_trust_report.json` | present | `e6353c1589c3` |
+| trust_report | `reports/security_trust_report.json` | present | `bea31eb462ca` |
| python_compatibility | `reports/python_compatibility.json` | present | `6843f5283fd7` |
-| registry_audit | `reports/registry_audit.json` | present | `879a11cef513` |
-| package_verification | `reports/package_verification.json` | present | `45b51edb26bc` |
+| registry_audit | `reports/registry_audit.json` | present | `c6d5cf3c3ec4` |
+| package_verification | `reports/package_verification.json` | present | `dd90aae6ad68` |
| install_simulation | `reports/install_simulation.json` | present | `1c3fdea61d50` |
| skill_os2_audit | `reports/skill_os2_audit.json` | present | `c0e499f9f051` |
| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `130161495dc4` |
diff --git a/reports/context_budget.json b/reports/context_budget.json
index 76b3031..80e62ef 100644
--- a/reports/context_budget.json
+++ b/reports/context_budget.json
@@ -6,15 +6,15 @@
"context_budget_tier": "production",
"context_budget_limit": 1000,
"skill_body_tokens": 797,
- "other_text_tokens": 1050165,
+ "other_text_tokens": 1050110,
"estimated_initial_load_tokens": 990,
- "estimated_total_text_tokens": 1050962,
- "deferred_resource_tokens": 486314,
+ "estimated_total_text_tokens": 1050907,
+ "deferred_resource_tokens": 486227,
"deferred_resource_warn_threshold": 120000,
"deferred_resource_dirs": [
{
"path": "scripts",
- "estimated_tokens": 426187,
+ "estimated_tokens": 426100,
"file_count": 125
},
{
@@ -36,7 +36,7 @@
"large_deferred_resource_dirs": [
{
"path": "scripts",
- "estimated_tokens": 426187,
+ "estimated_tokens": 426100,
"file_count": 125
}
],
@@ -59,7 +59,7 @@
],
"missing": [],
"path": "scripts",
- "estimated_tokens": 426187,
+ "estimated_tokens": 426100,
"file_count": 125,
"rationale": "Script resources are deterministic deferred tools, not initial-load prompt context."
}
diff --git a/reports/context_budget.md b/reports/context_budget.md
index 8d72382..71aaa7b 100644
--- a/reports/context_budget.md
+++ b/reports/context_budget.md
@@ -2,7 +2,7 @@
| Target | Path | Tier | Limit | Initial | SKILL | Deferred | Resource Governance | Large Deferred Dirs | Quality Density | Unused Dirs | Status |
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- | ---: | --- | --- |
-| root | `.` | `production` | 1000 | 990 | 797 | 486314 | `governed` | scripts:426187 | 131.3 | - | ok |
+| root | `.` | `production` | 1000 | 990 | 797 | 486227 | `governed` | scripts:426100 | 131.3 | - | ok |
| complex-release-orchestrator | `examples/complex-release-orchestrator/generated-skill` | `production` | 1000 | 790 | 718 | 1657 | `not-required` | - | 164.6 | - | ok |
| governed-incident-command | `examples/governed-incident-command/generated-skill` | `production` | 1000 | 760 | 658 | 1030 | `not-required` | - | 171.1 | - | ok |
diff --git a/reports/context_budget_summary.json b/reports/context_budget_summary.json
index bcef6ce..488d0e9 100644
--- a/reports/context_budget_summary.json
+++ b/reports/context_budget_summary.json
@@ -8,11 +8,11 @@
"budget_limit": 1000,
"initial_tokens": 990,
"skill_body_tokens": 797,
- "deferred_resource_tokens": 486314,
+ "deferred_resource_tokens": 486227,
"large_deferred_resource_dirs": [
{
"path": "scripts",
- "estimated_tokens": 426187,
+ "estimated_tokens": 426100,
"file_count": 125
}
],
@@ -35,7 +35,7 @@
],
"missing": [],
"path": "scripts",
- "estimated_tokens": 426187,
+ "estimated_tokens": 426100,
"file_count": 125,
"rationale": "Script resources are deterministic deferred tools, not initial-load prompt context."
}
diff --git a/reports/evidence_consistency.json b/reports/evidence_consistency.json
index d21cd09..7cd7aac 100644
--- a/reports/evidence_consistency.json
+++ b/reports/evidence_consistency.json
@@ -189,12 +189,12 @@
"status": "pass",
"expected": {
"status": "pass",
- "detail": "initial load 990/1000; deferred 486314/120000; top deferred scripts 426187; resource governance governed; quality density 131.3",
+ "detail": "initial load 990/1000; deferred 486227/120000; top deferred scripts 426100; resource governance governed; quality density 131.3",
"evidence": "reports/context_budget.json"
},
"actual": {
"status": "pass",
- "detail": "initial load 990/1000; deferred 486314/120000; top deferred scripts 426187; resource governance governed; quality density 131.3",
+ "detail": "initial load 990/1000; deferred 486227/120000; top deferred scripts 426100; resource governance governed; quality density 131.3",
"evidence": "reports/context_budget.json"
},
"paths": [
@@ -207,8 +207,8 @@
"key": "benchmark-release-lock-self-consistency",
"label": "Benchmark release lock matches git dirty state",
"status": "pass",
- "expected": true,
- "actual": true,
+ "expected": false,
+ "actual": false,
"paths": [
"reports/benchmark_reproducibility.json"
],
@@ -248,8 +248,8 @@
"key": "overview-benchmark-commit",
"label": "overview embeds the benchmark commit",
"status": "pass",
- "expected": "499214b052869193d96cdd7e73f537a57eb7bb8d",
- "actual": "499214b052869193d96cdd7e73f537a57eb7bb8d",
+ "expected": "ca94d734ee43bd514d4bf83f5c9a7cb7bdc161da",
+ "actual": "ca94d734ee43bd514d4bf83f5c9a7cb7bdc161da",
"paths": [
"reports/benchmark_reproducibility.json",
"reports/skill-overview.json"
@@ -261,30 +261,30 @@
"label": "overview embeds benchmark summary fields",
"status": "pass",
"expected": {
- "release_lock_ready": true,
+ "release_lock_ready": false,
"required_artifact_count": 25,
"missing_artifact_count": 0,
- "source_contract_sha256": "1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642",
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288",
+ "source_contract_sha256": "363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482",
"world_class_ledger_pending_count": 4,
"world_class_source_check_count": 13,
"world_class_source_pass_count": 6,
"world_class_source_blocked_count": 7,
"public_claim_ready": false,
- "public_claim_blocker_count": 4
+ "public_claim_blocker_count": 5
},
"actual": {
- "release_lock_ready": true,
+ "release_lock_ready": false,
"required_artifact_count": 25,
"missing_artifact_count": 0,
- "source_contract_sha256": "1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642",
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288",
+ "source_contract_sha256": "363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482",
"world_class_ledger_pending_count": 4,
"world_class_source_check_count": 13,
"world_class_source_pass_count": 6,
"world_class_source_blocked_count": 7,
"public_claim_ready": false,
- "public_claim_blocker_count": 4
+ "public_claim_blocker_count": 5
},
"paths": [
"reports/benchmark_reproducibility.json",
@@ -394,8 +394,8 @@
"key": "interpretation-benchmark-commit",
"label": "interpretation embeds the benchmark commit",
"status": "pass",
- "expected": "499214b052869193d96cdd7e73f537a57eb7bb8d",
- "actual": "499214b052869193d96cdd7e73f537a57eb7bb8d",
+ "expected": "ca94d734ee43bd514d4bf83f5c9a7cb7bdc161da",
+ "actual": "ca94d734ee43bd514d4bf83f5c9a7cb7bdc161da",
"paths": [
"reports/benchmark_reproducibility.json",
"reports/skill-interpretation.json"
@@ -407,30 +407,30 @@
"label": "interpretation embeds benchmark summary fields",
"status": "pass",
"expected": {
- "release_lock_ready": true,
+ "release_lock_ready": false,
"required_artifact_count": 25,
"missing_artifact_count": 0,
- "source_contract_sha256": "1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642",
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288",
+ "source_contract_sha256": "363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482",
"world_class_ledger_pending_count": 4,
"world_class_source_check_count": 13,
"world_class_source_pass_count": 6,
"world_class_source_blocked_count": 7,
"public_claim_ready": false,
- "public_claim_blocker_count": 4
+ "public_claim_blocker_count": 5
},
"actual": {
- "release_lock_ready": true,
+ "release_lock_ready": false,
"required_artifact_count": 25,
"missing_artifact_count": 0,
- "source_contract_sha256": "1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642",
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288",
+ "source_contract_sha256": "363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482",
"world_class_ledger_pending_count": 4,
"world_class_source_check_count": 13,
"world_class_source_pass_count": 6,
"world_class_source_blocked_count": 7,
"public_claim_ready": false,
- "public_claim_blocker_count": 4
+ "public_claim_blocker_count": 5
},
"paths": [
"reports/benchmark_reproducibility.json",
diff --git a/reports/output_execution_runs.json b/reports/output_execution_runs.json
index 2b2c026..3a619d8 100644
--- a/reports/output_execution_runs.json
+++ b/reports/output_execution_runs.json
@@ -34,7 +34,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 32.31,
+ "duration_ms": 31.35,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -62,7 +62,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 35.69,
+ "duration_ms": 31.06,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -85,7 +85,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 37.25,
+ "duration_ms": 31.69,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -113,7 +113,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 34.3,
+ "duration_ms": 33.39,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -136,7 +136,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 32.89,
+ "duration_ms": 31.94,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -164,7 +164,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 33.18,
+ "duration_ms": 32.09,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -187,7 +187,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 33.42,
+ "duration_ms": 31.3,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -214,7 +214,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 33.85,
+ "duration_ms": 31.15,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -237,7 +237,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 34.38,
+ "duration_ms": 31.77,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
@@ -266,7 +266,7 @@
"execution_mode": "command",
"model_executed": false,
"command_executed": true,
- "duration_ms": 34.62,
+ "duration_ms": 33.75,
"provider": "local-output-eval-runner",
"model": "",
"usage": {
diff --git a/reports/output_execution_runs.md b/reports/output_execution_runs.md
index 975c9fd..e319057 100644
--- a/reports/output_execution_runs.md
+++ b/reports/output_execution_runs.md
@@ -23,16 +23,16 @@ Command runner evidence is present. This proves the eval harness executed an ext
| Case | Variant | Mode | Model | Duration ms | Tokens | Score | Status |
| --- | --- | --- | --- | ---: | ---: | ---: | --- |
-| skill-package-contract | baseline | command | local-output-eval-runner | 32.31 | 33 | 0.0 | pass |
-| skill-package-contract | with_skill | command | local-output-eval-runner | 35.69 | 73 | 100.0 | pass |
-| output-eval-expectation | baseline | command | local-output-eval-runner | 37.25 | 36 | 0.0 | pass |
-| output-eval-expectation | with_skill | command | local-output-eval-runner | 34.3 | 80 | 100.0 | pass |
-| ir-before-packaging | baseline | command | local-output-eval-runner | 32.89 | 33 | 0.0 | pass |
-| ir-before-packaging | with_skill | command | local-output-eval-runner | 33.18 | 80 | 100.0 | pass |
-| near-neighbor-boundary | baseline | command | local-output-eval-runner | 33.42 | 36 | 0.0 | pass |
-| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 33.85 | 65 | 100.0 | pass |
-| file-backed-governed-package | baseline | command | local-output-eval-runner | 34.38 | 37 | 0.0 | pass |
-| file-backed-governed-package | with_skill | command | local-output-eval-runner | 34.62 | 98 | 100.0 | pass |
+| skill-package-contract | baseline | command | local-output-eval-runner | 31.35 | 33 | 0.0 | pass |
+| skill-package-contract | with_skill | command | local-output-eval-runner | 31.06 | 73 | 100.0 | pass |
+| output-eval-expectation | baseline | command | local-output-eval-runner | 31.69 | 36 | 0.0 | pass |
+| output-eval-expectation | with_skill | command | local-output-eval-runner | 33.39 | 80 | 100.0 | pass |
+| ir-before-packaging | baseline | command | local-output-eval-runner | 31.94 | 33 | 0.0 | pass |
+| ir-before-packaging | with_skill | command | local-output-eval-runner | 32.09 | 80 | 100.0 | pass |
+| near-neighbor-boundary | baseline | command | local-output-eval-runner | 31.3 | 36 | 0.0 | pass |
+| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 31.15 | 65 | 100.0 | pass |
+| file-backed-governed-package | baseline | command | local-output-eval-runner | 31.77 | 37 | 0.0 | pass |
+| file-backed-governed-package | with_skill | command | local-output-eval-runner | 33.75 | 98 | 100.0 | pass |
## Next Fixes
diff --git a/reports/package_verification.json b/reports/package_verification.json
index 43df8ed..2ce0684 100644
--- a/reports/package_verification.json
+++ b/reports/package_verification.json
@@ -8,7 +8,7 @@
"target_count": 4,
"adapter_count": 4,
"archive_present": true,
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482",
"archive_entry_count": 652,
"failure_count": 0,
"warning_count": 0
diff --git a/reports/package_verification.md b/reports/package_verification.md
index e590c59..20f3414 100644
--- a/reports/package_verification.md
+++ b/reports/package_verification.md
@@ -4,7 +4,7 @@
- Package directory: `dist`
- Targets: `4 / 4` adapters present
- Archive present: `True`
-- Archive SHA256: `88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288`
+- Archive SHA256: `a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482`
- Failures: `0`
- Warnings: `0`
diff --git a/reports/registry_audit.json b/reports/registry_audit.json
index 595c43d..503534b 100644
--- a/reports/registry_audit.json
+++ b/reports/registry_audit.json
@@ -21,8 +21,8 @@
"trust_level": "local",
"license": "MIT",
"checksums": {
- "package_sha256": "1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642",
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288"
+ "package_sha256": "363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482"
},
"compatibility": {
"openai": "pass",
@@ -53,7 +53,7 @@
},
"distribution": {
"archive_verified": true,
- "archive_sha256": "88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288",
+ "archive_sha256": "a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482",
"package_verification": "reports/package_verification.json",
"install_simulated": true,
"install_simulation": "reports/install_simulation.json"
@@ -78,7 +78,7 @@
"vscode"
],
"package_metadata": "registry/packages/yao-meta-skill.json",
- "package_sha256": "1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642"
+ "package_sha256": "363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453"
}
]
},
diff --git a/reports/registry_audit.md b/reports/registry_audit.md
index 234da30..87ab404 100644
--- a/reports/registry_audit.md
+++ b/reports/registry_audit.md
@@ -6,8 +6,8 @@
- Maturity: `governed`
- Owner: `Yao Team`
- License: `MIT`
-- Package SHA256: `1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642`
-- Archive SHA256: `88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288`
+- Package SHA256: `363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453`
+- Archive SHA256: `a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482`
- Install simulated: `True`
## Compatibility
diff --git a/reports/review-studio.html b/reports/review-studio.html
index fd0a6f1..db25bdd 100644
--- a/reports/review-studio.html
+++ b/reports/review-studio.html
@@ -740,12 +740,12 @@
5 targets in platform-neutral contract target contracts compiled from Skill IR 5 cases; 1 file-backed command 10; model 0; recorded 0 review pairs hide baseline vs with-skill labels pending 5; answer key hidden adjudication decisions; pending 5 4 blockers; local reproducible true 2.0 coverage; extensions partial 0, planned 0; evidence pending 4 target conformance pass rate 0 native; 4 installer-enforced 125 scripts scanned; secrets found 199 files scanned for Python 3.11 836 largest lines; 8 watchlist; 68 CLI handlers; 18 in entrypoint 12 scanned skills; route collisions 1 metadata events; 0 missed triggers proposal-review; approval 0; release lock true curator-review; ready 1; top score 88 0 gates covered; human risk decisions 0 valid submissions; 0 invalid 180 public surfaces scanned 0 open blocker annotations 5 targets; MIT license 652 zip entries; package verification 4 adapters; 12 permissions enforced; 0 permission failures declared minor; 0 breaking changes 5 targets in platform-neutral contract target contracts compiled from Skill IR 5 cases; 1 file-backed command 10; model 0; recorded 0 review pairs hide baseline vs with-skill labels pending 5; answer key hidden adjudication decisions; pending 5 5 blockers; local reproducible true 2.0 coverage; extensions partial 0, planned 0; evidence pending 4 target conformance pass rate 0 native; 4 installer-enforced 125 scripts scanned; secrets found 199 files scanned for Python 3.11 837 largest lines; 8 watchlist; 68 CLI handlers; 18 in entrypoint 12 scanned skills; route collisions 1 metadata events; 0 missed triggers proposal-review; approval 0; release lock false curator-review; ready 1; top score 88 0 gates covered; human risk decisions 0 valid submissions; 0 invalid 180 public surfaces scanned 0 open blocker annotations 5 targets; MIT license 652 zip entries; package verification 4 adapters; 12 permissions enforced; 0 permission failures declared minor; 0 breaking changes intent confidence 100/100; Intent is clear enough to package the first routeable version. 13 trigger cases; 0 misroutes; 0 ambiguous 5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5 initial load 990/1000; deferred 486314/120000; top deferred scripts 426187; resource governance governed; quality density 131.3 5 / 5 targets pass 0 secrets; 125 scripts; 3 network-capable scripts; 0 help smoke failures Python 3.11; 199 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards 196 Python files; 0 hotspots; 8 watchlist files; 0 blockers; largest 836 lines; 68 CLI handlers; 18 in entrypoint 3/3 permissions approved; gaps 0; required file_write, network, subprocess 4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4 12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues 1 metadata events; adoption 0; missed 0; bad-output 0; risk low; daily proposals 5; daily decision proposal-review; daily release lock true; weekly queue 5 unique; weekly ready 1; weekly top 88; weekly release lock true 0 active waivers; 1 warning gates still need reviewer decision 4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures 0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended intent confidence 100/100; Intent is clear enough to package the first routeable version. 13 trigger cases; 0 misroutes; 0 ambiguous 5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5 initial load 990/1000; deferred 486227/120000; top deferred scripts 426100; resource governance governed; quality density 131.3 5 / 5 targets pass 0 secrets; 125 scripts; 3 network-capable scripts; 0 help smoke failures Python 3.11; 199 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards 196 Python files; 0 hotspots; 8 watchlist files; 0 blockers; largest 837 lines; 68 CLI handlers; 18 in entrypoint 3/3 permissions approved; gaps 0; required file_write, network, subprocess 4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4 12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues 1 metadata events; adoption 0; missed 0; bad-output 0; risk low; daily proposals 5; daily decision proposal-review; daily release lock false; weekly queue 5 unique; weekly ready 1; weekly top 88; weekly release lock false 0 active waivers; 1 warning gates still need reviewer decision 4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures 0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended核心指标
- 审查闸门
- 意图画布
触发实验
输出实验
上下文
运行矩阵
信任报告
Python 兼容
架构维护
权限批准
权限探针
组合治理
运营回路
人工批准
世界证据
注册审计
发布路线
意图画布
触发实验
输出实验
上下文
运行矩阵
信任报告
Python 兼容
架构维护
权限批准
权限探针
组合治理
运营回路
人工批准
世界证据
注册审计
发布路线
initial load 990/1000; deferred 486314/120000; top deferred scripts 426187; resource governance governed; quality density 131.3
initial load 990/1000; deferred 486227/120000; top deferred scripts 426100; resource governance governed; quality density 131.3
Review reports/compiled_targets.md before packaging to inspect target adapter modes, generated files, preserved semantics, warnings, and unsupported features.
1e466c4508d6ebb845923331fe534197df4fddf9e3b778c789ce8d349a555642363354808bdd7c9517352e7a709ea13cd2a3878e526fb1f7121f71de5e746453高风险 secret、远程 inline execution、缺失依赖策略或无法解释的脚本接口应阻断 governed release。
1 metadata events; adoption 0; missed 0; bad-output 0; risk low; daily proposals 5; daily decision proposal-review; daily release lock true; weekly queue 5 unique; weekly ready 1; weekly top 88; weekly release lock true
1 metadata events; adoption 0; missed 0; bad-output 0; risk low; daily proposals 5; daily decision proposal-review; daily release lock false; weekly queue 5 unique; weekly ready 1; weekly top 88; weekly release lock false
yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures
88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c4820 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended
88dc8d39204730f91d5738858e9000ac9b657ba09ab02d6d4804dac1b2546288a5a62aeb079b925fe3e52446a607508209ab40a9fa9b2be2335dee35b182c482