diff --git a/registry/index.json b/registry/index.json index a2efa71..8898328 100644 --- a/registry/index.json +++ b/registry/index.json @@ -16,7 +16,7 @@ "vscode" ], "package_metadata": "registry/packages/yao-meta-skill.json", - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" } ] } diff --git a/registry/packages/yao-meta-skill.json b/registry/packages/yao-meta-skill.json index 2e66a0f..77273e0 100644 --- a/registry/packages/yao-meta-skill.json +++ b/registry/packages/yao-meta-skill.json @@ -16,8 +16,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, "compatibility": { "openai": "pass", @@ -48,7 +48,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json index 1ba9d08..7a1a9de 100644 --- a/reports/benchmark_reproducibility.json +++ b/reports/benchmark_reproducibility.json @@ -3,23 +3,36 @@ "ok": true, "generated_at": "2026-06-16", "skill_dir": ".", - "commit": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "commit": "cd933046c840affafa98edff697a2d0f68afd71e", "git_status": { "available": true, - "dirty": false, - "changed_file_count": 0, - "sample": [], + "dirty": true, + "changed_file_count": 22, + "sample": [ + " M registry/index.json", + " M registry/packages/yao-meta-skill.json", + " M reports/benchmark_reproducibility.json", + " M reports/benchmark_reproducibility.md", + " M reports/evidence_consistency.json", + " M reports/evidence_consistency.md", + " M reports/output_execution_runs.json", + " M reports/output_execution_runs.md", + " M reports/package_verification.json", + " M reports/package_verification.md", + " M reports/registry_audit.json", + " M reports/registry_audit.md" + ], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "bab9f7cd9fc387621005ec17e41186d232af6be1fb154824493e5c5b5cac6027", - "source_contract_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "evidence_bundle_sha256": "8e612a6cb7fbac70b9538d8c1f77713e1ce1806aad777502654b5489b54ce334", + "source_contract_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 22, @@ -37,14 +50,15 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 22 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ + "release lock is not clean or commit is unavailable", "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", @@ -53,10 +67,10 @@ "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { - "ready": true, - "commit": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "ready": false, + "commit": "cd933046c840affafa98edff697a2d0f68afd71e", "status_scope": "generation-time status before this report is written", - "reason": "clean generation-time HEAD" + "reason": "working tree was dirty at generation time" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", @@ -64,7 +78,7 @@ "existing_count": 24, "missing_count": 0, "missing_paths": [], - "sha256": "bab9f7cd9fc387621005ec17e41186d232af6be1fb154824493e5c5b5cac6027" + "sha256": "8e612a6cb7fbac70b9538d8c1f77713e1ce1806aad777502654b5489b54ce334" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -137,8 +151,8 @@ "label": "output_execution", "path": "reports/output_execution_runs.json", "exists": true, - "bytes": 7967, - "sha256": "ddb48310b50baee9dec234264e789cf75024016a6c1073ea1e391ca0b0381c53" + "bytes": 7966, + "sha256": "3d79a1bf2d0c9ce23b36a1cf59c94e4943cd68ef4a7a70a8b43b66f25190a551" }, { "label": "blind_review", @@ -173,7 +187,7 @@ "path": "reports/security_trust_report.json", "exists": true, "bytes": 109099, - "sha256": "154c7e8534de6ecf170bb28536f9ee3dc060ae2821855d1358f0ad0e83e90794" + "sha256": "0b11eb520b7275c36de490ba81e7c96e40ba9dd6baafd74fb4b1abd396042e33" }, { "label": "python_compatibility", @@ -187,14 +201,14 @@ "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, - "sha256": "afb07b53145703e057484fc6ca8c9efa3fb348d593d9449853d8e5daa7900f30" + "sha256": "944ae2224089a684a7495706685005d148b1ce8eea2efd1c18f318f1ffc59032" }, { "label": "package_verification", "path": "reports/package_verification.json", "exists": true, "bytes": 19338, - "sha256": "affc7184918121ab719b916c87b5c8c058902a2d4747356bf3c34a60a32e769f" + "sha256": "c7a048b6bb2f6b13e88f473137c6b469fe7d25e99c67ca4018b9cf0f7308f312" }, { "label": "install_simulation", diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md index 52f832e..6b8bc06 100644 --- a/reports/benchmark_reproducibility.md +++ b/reports/benchmark_reproducibility.md @@ -1,19 +1,19 @@ # Benchmark Reproducibility Generated at: `2026-06-16` -Commit: `ae5ce88ea7f4d002670ce9d3cb2bb339017902d7` -Working tree dirty at generation: `false` -Evidence bundle SHA256: `bab9f7cd9fc387621005ec17e41186d232af6be1fb154824493e5c5b5cac6027` +Commit: `cd933046c840affafa98edff697a2d0f68afd71e` +Working tree dirty at generation: `true` +Evidence bundle SHA256: `8e612a6cb7fbac70b9538d8c1f77713e1ce1806aad777502654b5489b54ce334` ## Summary - reproducibility ready: `true` -- release lock ready: `true` +- release lock ready: `false` - methodology complete: `true` - required artifacts: `24` - missing artifacts: `0` -- source contract sha256: `30502ca01a3d` -- archive sha256: `f1c499c597e8` +- source contract sha256: `002cb9e43ee2` +- archive sha256: `eddf1d422b1a` - output cases: `5` - disclosed failure cases: `3` - reproduction commands: `22` @@ -22,8 +22,8 @@ Evidence bundle SHA256: `bab9f7cd9fc387621005ec17e41186d232af6be1fb154824493e5c5 - world-class ready: `false` - world-class source checks: `6` pass / `13` total; `7` blocked - public claim ready: `false` -- public claim blockers: `4` -- changed files at generation: `0` +- public claim blockers: `5` +- changed files at generation: `22` This report proves local benchmark reproducibility only. It keeps external provider and human-review gaps visible instead of counting them as complete. The git commit is generation-time context; the evidence bundle SHA is the durable anchor for the artifacts listed below. @@ -35,6 +35,7 @@ This report proves local benchmark reproducibility only. It keeps external provi | Blocker | | --- | +| release lock is not clean or commit is unavailable | | provider-backed model holdout evidence is incomplete | | human blind-review adjudication is incomplete | | world-class evidence is not accepted yet (4 open gaps, 4 ledger pending) | @@ -42,15 +43,15 @@ This report proves local benchmark reproducibility only. It keeps external provi ## Release Lock -- ready: `true` -- reason: clean generation-time HEAD +- ready: `false` +- reason: working tree was dirty at generation time - status scope: generation-time status before this report is written ## Evidence Bundle - algorithm: `sha256(path,label,exists,artifact_sha256)` - artifacts: `24` / `24` -- sha256: `bab9f7cd9fc387621005ec17e41186d232af6be1fb154824493e5c5b5cac6027` +- sha256: `8e612a6cb7fbac70b9538d8c1f77713e1ce1806aad777502654b5489b54ce334` ## Methodology Sections @@ -72,15 +73,15 @@ This report proves local benchmark reproducibility only. It keeps external provi | output_cases | `evals/output/cases.jsonl` | present | `a6ae96857116` | | output_schema | `evals/output/schema.json` | present | `8ee340c95064` | | output_scorecard | `reports/output_quality_scorecard.json` | present | `0806258a8e08` | -| output_execution | `reports/output_execution_runs.json` | present | `ddb48310b50b` | +| output_execution | `reports/output_execution_runs.json` | present | `3d79a1bf2d0c` | | blind_review | `reports/output_blind_review_pack.json` | present | `bbe2db8ec277` | | review_adjudication | `reports/output_review_adjudication.json` | present | `240485a721af` | | trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` | | runtime_conformance | `reports/conformance_matrix.json` | present | `97f9ba949c23` | -| trust_report | `reports/security_trust_report.json` | present | `154c7e8534de` | +| trust_report | `reports/security_trust_report.json` | present | `0b11eb520b72` | | python_compatibility | `reports/python_compatibility.json` | present | `471c481ff9f9` | -| registry_audit | `reports/registry_audit.json` | present | `afb07b531457` | -| package_verification | `reports/package_verification.json` | present | `affc71849181` | +| registry_audit | `reports/registry_audit.json` | present | `944ae2224089` | +| package_verification | `reports/package_verification.json` | present | `c7a048b6bb2f` | | install_simulation | `reports/install_simulation.json` | present | `28ceb014e202` | | skill_os2_audit | `reports/skill_os2_audit.json` | present | `57536bc67370` | | world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `76a3f8e2b12b` | diff --git a/reports/evidence_consistency.json b/reports/evidence_consistency.json index 27faaf3..95df508 100644 --- a/reports/evidence_consistency.json +++ b/reports/evidence_consistency.json @@ -4,14 +4,14 @@ "generated_at": "2026-06-16", "skill_dir": ".", "summary": { - "check_count": 27, - "pass_count": 27, + "check_count": 28, + "pass_count": 28, "warn_count": 0, "fail_count": 0, "decision": "consistent" }, "status_counts": { - "pass": 27, + "pass": 28, "warn": 0, "fail": 0 }, @@ -34,6 +34,7 @@ "reports/install_simulation.json", "reports/security_trust_report.json", "reports/context_budget.json", + "reports/world_class_claim_guard.json", "reports/skill-os-2-review.md" ], "detail": "The consistency gate can only be trusted when every source JSON report parses and every source Markdown report is readable." @@ -42,8 +43,8 @@ "key": "benchmark-release-lock-self-consistency", "label": "Benchmark release lock matches git dirty state", "status": "pass", - "expected": true, - "actual": true, + "expected": false, + "actual": false, "paths": [ "reports/benchmark_reproducibility.json" ], @@ -53,8 +54,8 @@ "key": "overview-benchmark-commit", "label": "overview embeds the benchmark commit", "status": "pass", - "expected": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", - "actual": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "expected": "cd933046c840affafa98edff697a2d0f68afd71e", + "actual": "cd933046c840affafa98edff697a2d0f68afd71e", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-overview.json" @@ -66,7 +67,7 @@ "label": "overview embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", @@ -76,10 +77,10 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "actual": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", @@ -89,7 +90,7 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "paths": [ "reports/benchmark_reproducibility.json", @@ -199,8 +200,8 @@ "key": "interpretation-benchmark-commit", "label": "interpretation embeds the benchmark commit", "status": "pass", - "expected": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", - "actual": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "expected": "cd933046c840affafa98edff697a2d0f68afd71e", + "actual": "cd933046c840affafa98edff697a2d0f68afd71e", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-interpretation.json" @@ -212,7 +213,7 @@ "label": "interpretation embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", @@ -222,10 +223,10 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "actual": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", @@ -235,7 +236,7 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "paths": [ "reports/benchmark_reproducibility.json", @@ -1645,6 +1646,64 @@ ], "detail": "When world-class evidence is pending, Review Studio must stay in a review or warning posture." }, + { + "key": "claim-guard-package-runtime-surface", + "label": "Claim guard covers package and runtime claim surfaces", + "status": "pass", + "expected": { + "overclaim_guard_active": true, + "violation_count": 0, + "ledger_ready_to_claim_world_class": false, + "ledger_pending_count": 4, + "metadata_covers_json": true, + "package_surface_minimum": true, + "claim_surface_covers_package": true, + "required_surfaces": { + "README.md": true, + "SKILL.md": true, + "manifest.json": true, + "agents/interface.yaml": true, + "dist/manifest.json": true, + "dist/targets/openai/adapter.json": true, + "evidence/world_class/README.md": true, + "security/permission_policy.json": true, + "reports/world_class_evidence_ledger.json": true + }, + "prohibited_surfaces": [] + }, + "actual": { + "overclaim_guard_active": true, + "violation_count": 0, + "ledger_ready_to_claim_world_class": false, + "ledger_pending_count": 4, + "metadata_covers_json": true, + "package_surface_minimum": true, + "claim_surface_covers_package": true, + "required_surfaces": { + "README.md": true, + "SKILL.md": true, + "manifest.json": true, + "agents/interface.yaml": true, + "dist/manifest.json": true, + "dist/targets/openai/adapter.json": true, + "evidence/world_class/README.md": true, + "security/permission_policy.json": true, + "reports/world_class_evidence_ledger.json": true + }, + "prohibited_surfaces": [] + }, + "paths": [ + "reports/world_class_claim_guard.json", + "manifest.json", + "agents/interface.yaml", + "dist/manifest.json", + "dist/targets/openai/adapter.json", + "evidence/world_class/README.md", + "security/permission_policy.json", + "reports/world_class_evidence_ledger.json" + ], + "detail": "The overclaim guard must scan package manifests, adapter metadata, security policy, and ledger surfaces before public readiness can be trusted." + }, { "key": "skill-os-2-review-current-evidence", "label": "Skill OS 2.0 review summary mirrors current evidence", diff --git a/reports/evidence_consistency.md b/reports/evidence_consistency.md index 20cb6ec..1b8596a 100644 --- a/reports/evidence_consistency.md +++ b/reports/evidence_consistency.md @@ -5,8 +5,8 @@ Generated at: `2026-06-16` ## Summary - decision: `consistent` -- checks: `27` -- pass: `27` +- checks: `28` +- pass: `28` - warn: `0` - fail: `0` @@ -16,7 +16,7 @@ This gate compares generated evidence reports against each other. It does not cr | Check | Status | Detail | Paths | | --- | --- | --- | --- | -| Required report artifacts are readable | `pass` | The consistency gate can only be trusted when every source JSON report parses and every source Markdown report is readable. | `reports/benchmark_reproducibility.json`, `reports/skill-overview.json`, `reports/skill-interpretation.json`, `reports/adoption_drift_report.json`, `reports/world_class_evidence_ledger.json`, `reports/skill_os2_coverage.json`, `reports/review-studio.json`, `reports/package_verification.json`, `reports/install_simulation.json`, `reports/security_trust_report.json`, `reports/context_budget.json`, `reports/skill-os-2-review.md` | +| Required report artifacts are readable | `pass` | The consistency gate can only be trusted when every source JSON report parses and every source Markdown report is readable. | `reports/benchmark_reproducibility.json`, `reports/skill-overview.json`, `reports/skill-interpretation.json`, `reports/adoption_drift_report.json`, `reports/world_class_evidence_ledger.json`, `reports/skill_os2_coverage.json`, `reports/review-studio.json`, `reports/package_verification.json`, `reports/install_simulation.json`, `reports/security_trust_report.json`, `reports/context_budget.json`, `reports/world_class_claim_guard.json`, `reports/skill-os-2-review.md` | | Benchmark release lock matches git dirty state | `pass` | The benchmark release lock must reflect the generation-time git dirty flag. | `reports/benchmark_reproducibility.json` | | overview embeds the benchmark commit | `pass` | Human-facing reports must point to the same benchmark release-lock commit. | `reports/benchmark_reproducibility.json`, `reports/skill-overview.json` | | overview embeds benchmark summary fields | `pass` | Selected summary fields must match exactly across generated reports. | `reports/benchmark_reproducibility.json`, `reports/skill-overview.json` | @@ -42,4 +42,5 @@ This gate compares generated evidence reports against each other. It does not cr | Coverage report mirrors world-class evidence boundary | `pass` | Blueprint coverage can be locally complete while public world-class evidence remains pending. | `reports/world_class_evidence_ledger.json`, `reports/skill_os2_coverage.json` | | Benchmark report mirrors world-class evidence boundary | `pass` | Benchmark reproducibility must not overstate public claim readiness. | `reports/world_class_evidence_ledger.json`, `reports/benchmark_reproducibility.json` | | Review Studio does not overclaim pending world-class evidence | `pass` | When world-class evidence is pending, Review Studio must stay in a review or warning posture. | `reports/world_class_evidence_ledger.json`, `reports/review-studio.json` | +| Claim guard covers package and runtime claim surfaces | `pass` | The overclaim guard must scan package manifests, adapter metadata, security policy, and ledger surfaces before public readiness can be trusted. | `reports/world_class_claim_guard.json`, `manifest.json`, `agents/interface.yaml`, `dist/manifest.json`, `dist/targets/openai/adapter.json`, `evidence/world_class/README.md`, `security/permission_policy.json`, `reports/world_class_evidence_ledger.json` | | Skill OS 2.0 review summary mirrors current evidence | `pass` | Manual 2.0 review summaries must not drift from generated gate, package, trust, context, benchmark, or CI evidence. | `reports/skill-os-2-review.md`, `reports/review-studio.json`, `reports/package_verification.json`, `reports/install_simulation.json`, `reports/security_trust_report.json`, `reports/context_budget.json`, `reports/benchmark_reproducibility.json`, `scripts/ci_test.py` | diff --git a/reports/output_execution_runs.json b/reports/output_execution_runs.json index 4b27ba4..7a6ce3e 100644 --- a/reports/output_execution_runs.json +++ b/reports/output_execution_runs.json @@ -34,7 +34,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.58, + "duration_ms": 27.02, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -62,7 +62,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.04, + "duration_ms": 25.98, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -85,7 +85,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.21, + "duration_ms": 25.78, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -113,7 +113,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 29.32, + "duration_ms": 29.05, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -136,7 +136,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 29.89, + "duration_ms": 29.11, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -164,7 +164,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.45, + "duration_ms": 27.39, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -187,7 +187,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.34, + "duration_ms": 26.01, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -214,7 +214,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.11, + "duration_ms": 26.1, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -237,7 +237,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.19, + "duration_ms": 26.37, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -266,7 +266,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.05, + "duration_ms": 26.29, "provider": "local-output-eval-runner", "model": "", "usage": { diff --git a/reports/output_execution_runs.md b/reports/output_execution_runs.md index f4cb87b..cc0e918 100644 --- a/reports/output_execution_runs.md +++ b/reports/output_execution_runs.md @@ -23,16 +23,16 @@ Command runner evidence is present. This proves the eval harness executed an ext | Case | Variant | Mode | Model | Duration ms | Tokens | Score | Status | | --- | --- | --- | --- | ---: | ---: | ---: | --- | -| skill-package-contract | baseline | command | local-output-eval-runner | 27.58 | 33 | 0.0 | pass | -| skill-package-contract | with_skill | command | local-output-eval-runner | 26.04 | 73 | 100.0 | pass | -| output-eval-expectation | baseline | command | local-output-eval-runner | 26.21 | 36 | 0.0 | pass | -| output-eval-expectation | with_skill | command | local-output-eval-runner | 29.32 | 80 | 100.0 | pass | -| ir-before-packaging | baseline | command | local-output-eval-runner | 29.89 | 33 | 0.0 | pass | -| ir-before-packaging | with_skill | command | local-output-eval-runner | 27.45 | 80 | 100.0 | pass | -| near-neighbor-boundary | baseline | command | local-output-eval-runner | 26.34 | 36 | 0.0 | pass | -| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 26.11 | 65 | 100.0 | pass | -| file-backed-governed-package | baseline | command | local-output-eval-runner | 26.19 | 37 | 0.0 | pass | -| file-backed-governed-package | with_skill | command | local-output-eval-runner | 26.05 | 98 | 100.0 | pass | +| skill-package-contract | baseline | command | local-output-eval-runner | 27.02 | 33 | 0.0 | pass | +| skill-package-contract | with_skill | command | local-output-eval-runner | 25.98 | 73 | 100.0 | pass | +| output-eval-expectation | baseline | command | local-output-eval-runner | 25.78 | 36 | 0.0 | pass | +| output-eval-expectation | with_skill | command | local-output-eval-runner | 29.05 | 80 | 100.0 | pass | +| ir-before-packaging | baseline | command | local-output-eval-runner | 29.11 | 33 | 0.0 | pass | +| ir-before-packaging | with_skill | command | local-output-eval-runner | 27.39 | 80 | 100.0 | pass | +| near-neighbor-boundary | baseline | command | local-output-eval-runner | 26.01 | 36 | 0.0 | pass | +| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 26.1 | 65 | 100.0 | pass | +| file-backed-governed-package | baseline | command | local-output-eval-runner | 26.37 | 37 | 0.0 | pass | +| file-backed-governed-package | with_skill | command | local-output-eval-runner | 26.29 | 98 | 100.0 | pass | ## Next Fixes diff --git a/reports/package_verification.json b/reports/package_verification.json index d3d5a6c..dafd06b 100644 --- a/reports/package_verification.json +++ b/reports/package_verification.json @@ -8,7 +8,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "archive_entry_count": 617, "failure_count": 0, "warning_count": 0 diff --git a/reports/package_verification.md b/reports/package_verification.md index 05a87e9..3b2fed6 100644 --- a/reports/package_verification.md +++ b/reports/package_verification.md @@ -4,7 +4,7 @@ - Package directory: `dist` - Targets: `4 / 4` adapters present - Archive present: `True` -- Archive SHA256: `f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874` +- Archive SHA256: `eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8` - Failures: `0` - Warnings: `0` diff --git a/reports/registry_audit.json b/reports/registry_audit.json index 95613df..1ef84c1 100644 --- a/reports/registry_audit.json +++ b/reports/registry_audit.json @@ -21,8 +21,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, "compatibility": { "openai": "pass", @@ -53,7 +53,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -78,7 +78,7 @@ "vscode" ], "package_metadata": "registry/packages/yao-meta-skill.json", - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" } ] }, diff --git a/reports/registry_audit.md b/reports/registry_audit.md index aff052b..137a1d6 100644 --- a/reports/registry_audit.md +++ b/reports/registry_audit.md @@ -6,8 +6,8 @@ - Maturity: `governed` - Owner: `Yao Team` - License: `MIT` -- Package SHA256: `30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e` -- Archive SHA256: `f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874` +- Package SHA256: `002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8` +- Archive SHA256: `eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8` - Install simulated: `True` ## Compatibility diff --git a/reports/review-studio.html b/reports/review-studio.html index 7d9fa83..d207a56 100644 --- a/reports/review-studio.html +++ b/reports/review-studio.html @@ -676,7 +676,7 @@

核心指标

-
Skill IR2.0.0

5 targets in platform-neutral contract

Compiler5/5

target contracts compiled from Skill IR

Output Delta100.0

5 cases; 1 file-backed

Exec Runs10

command 10; model 0; recorded 0

Blind A/B5

review pairs hide baseline vs with-skill labels

Review Kit0/5

pending 5; answer key hidden

Review A/B0/5

adjudication decisions; pending 5

Public Claimblocked

4 blockers; local reproducible true

Blueprint21/21

2.0 coverage; extensions partial 0, planned 0; evidence pending 4

Runtime5/5

target conformance pass rate

Perm Probe4/4

0 native; 4 installer-enforced

Trust0

110 scripts scanned; secrets found

Py Compat0

176 files scanned for Python 3.11

Arch Debt0

899 largest lines; 8 watchlist; 64 CLI handlers; 18 in entrypoint

Atlas5

12 scanned skills; route collisions

Driftlow

1 metadata events; 0 missed triggers

Waivers0

0 gates covered; human risk decisions

Intake4/4

0 valid submissions; 0 invalid

Claim Guard0

173 public surfaces scanned

Notes0/0

0 open blocker annotations

Registry1.1.0

5 targets; MIT license

Archivepass

617 zip entries; package verification

Installpass

4 adapters; 12 permissions enforced; 0 permission failures

Upgrademinor

declared minor; 0 breaking changes

+
Skill IR2.0.0

5 targets in platform-neutral contract

Compiler5/5

target contracts compiled from Skill IR

Output Delta100.0

5 cases; 1 file-backed

Exec Runs10

command 10; model 0; recorded 0

Blind A/B5

review pairs hide baseline vs with-skill labels

Review Kit0/5

pending 5; answer key hidden

Review A/B0/5

adjudication decisions; pending 5

Public Claimblocked

5 blockers; local reproducible true

Blueprint21/21

2.0 coverage; extensions partial 0, planned 0; evidence pending 4

Runtime5/5

target conformance pass rate

Perm Probe4/4

0 native; 4 installer-enforced

Trust0

110 scripts scanned; secrets found

Py Compat0

176 files scanned for Python 3.11

Arch Debt0

899 largest lines; 8 watchlist; 64 CLI handlers; 18 in entrypoint

Atlas5

12 scanned skills; route collisions

Driftlow

1 metadata events; 0 missed triggers

Waivers0

0 gates covered; human risk decisions

Intake4/4

0 valid submissions; 0 invalid

Claim Guard0

173 public surfaces scanned

Notes0/0

0 open blocker annotations

Registry1.1.0

5 targets; MIT license

Archivepass

617 zip entries; package verification

Installpass

4 adapters; 12 permissions enforced; 0 permission failures

Upgrademinor

declared minor; 0 breaking changes

@@ -748,7 +748,7 @@
-

信任报告

Secret
0
脚本数
110
网络脚本
3
Help 失败
0
包体哈希
30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e
+

信任报告

Secret
0
脚本数
110
网络脚本
3
Help 失败
0
包体哈希
002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8

安全边界

高风险 secret、远程 inline execution、缺失依赖策略或无法解释的脚本接口应阻断 governed release。

@@ -806,8 +806,8 @@
-

公开声明

本地复现
发布锁
可公开声明
声明阻断
4
Provider 证据
人审完成
世界级就绪
-

声明阻断

+

公开声明

本地复现
发布锁
可公开声明
声明阻断
5
Provider 证据
人审完成
世界级就绪
+

声明阻断

@@ -827,12 +827,12 @@

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

-

包体元数据

名称
yao-meta-skill
版本
1.1.0
Maturity
governed
Owner
Yao Team
License
MIT
信任级别
local
目标平台
openai, claude, generic, agent-skills-compatible, vscode
兼容通过
6/6
归档哈希
f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874
+

包体元数据

名称
yao-meta-skill
版本
1.1.0
Maturity
governed
Owner
Yao Team
License
MIT
信任级别
local
目标平台
openai, claude, generic, agent-skills-compatible, vscode
兼容通过
6/6
归档哈希
eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

-

包体验证

目标数
4
Adapter
4
归档存在
Zip 条目
617
失败数
0
警告数
0
归档哈希
f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874
+

包体验证

目标数
4
Adapter
4
归档存在
Zip 条目
617
失败数
0
警告数
0
归档哈希
eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8
diff --git a/reports/review-studio.json b/reports/review-studio.json index bf09887..5ea29e1 100644 --- a/reports/review-studio.json +++ b/reports/review-studio.json @@ -1676,7 +1676,7 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, @@ -1700,11 +1700,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 2 }, - "commit": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "commit": "cd933046c840affafa98edff697a2d0f68afd71e", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1809,7 +1809,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 197, - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" }, "skill_atlas": { "skill_count": 12, @@ -1847,8 +1847,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, "compatibility": { "openai": "pass", @@ -1879,7 +1879,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -1895,7 +1895,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "archive_entry_count": 617, "failure_count": 0, "warning_count": 0 @@ -1974,12 +1974,12 @@ { "field": "archive_sha256", "from": "", - "to": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "to": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "to": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" } ] }, @@ -4346,7 +4346,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.58, + "duration_ms": 27.02, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4374,7 +4374,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.04, + "duration_ms": 25.98, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4397,7 +4397,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.21, + "duration_ms": 25.78, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4425,7 +4425,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 29.32, + "duration_ms": 29.05, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4448,7 +4448,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 29.89, + "duration_ms": 29.11, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4476,7 +4476,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.45, + "duration_ms": 27.39, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4499,7 +4499,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.34, + "duration_ms": 26.01, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4526,7 +4526,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.11, + "duration_ms": 26.1, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4549,7 +4549,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.19, + "duration_ms": 26.37, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4578,7 +4578,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.05, + "duration_ms": 26.29, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -5299,17 +5299,20 @@ "ok": true, "generated_at": "2026-06-16", "skill_dir": ".", - "commit": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "commit": "cd933046c840affafa98edff697a2d0f68afd71e", "git_status": { "available": true, - "dirty": false, - "changed_file_count": 0, - "sample": [], + "dirty": true, + "changed_file_count": 2, + "sample": [ + " M scripts/render_evidence_consistency.py", + " M tests/verify_evidence_consistency.py" + ], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, @@ -5333,14 +5336,15 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 2 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ + "release lock is not clean or commit is unavailable", "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", @@ -5349,10 +5353,10 @@ "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { - "ready": true, - "commit": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "ready": false, + "commit": "cd933046c840affafa98edff697a2d0f68afd71e", "status_scope": "generation-time status before this report is written", - "reason": "clean generation-time HEAD" + "reason": "working tree was dirty at generation time" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", @@ -11412,7 +11416,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 197, - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" }, "failures": [], "warnings": [], @@ -20438,8 +20442,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, "compatibility": { "openai": "pass", @@ -20470,7 +20474,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -20495,7 +20499,7 @@ "vscode" ], "package_metadata": "registry/packages/yao-meta-skill.json", - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" } ] }, @@ -20518,7 +20522,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "archive_entry_count": 617, "failure_count": 0, "warning_count": 0 @@ -21526,12 +21530,12 @@ { "field": "archive_sha256", "from": "", - "to": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "to": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "to": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" } ] }, diff --git a/reports/review-viewer.json b/reports/review-viewer.json index 2b2926c..1c87fac 100644 --- a/reports/review-viewer.json +++ b/reports/review-viewer.json @@ -996,7 +996,7 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, @@ -1020,11 +1020,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 2 }, - "commit": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "commit": "cd933046c840affafa98edff697a2d0f68afd71e", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1129,7 +1129,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 197, - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" }, "skill_atlas": { "skill_count": 12, @@ -1167,8 +1167,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, "compatibility": { "openai": "pass", @@ -1199,7 +1199,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -1215,7 +1215,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "archive_entry_count": 617, "failure_count": 0, "warning_count": 0 @@ -1294,12 +1294,12 @@ { "field": "archive_sha256", "from": "", - "to": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "to": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "to": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" } ] }, diff --git a/reports/security_trust_report.json b/reports/security_trust_report.json index f2d70b2..9200a15 100644 --- a/reports/security_trust_report.json +++ b/reports/security_trust_report.json @@ -23,7 +23,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 197, - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" }, "failures": [], "warnings": [], diff --git a/reports/security_trust_report.md b/reports/security_trust_report.md index 42016f4..df74b5a 100644 --- a/reports/security_trust_report.md +++ b/reports/security_trust_report.md @@ -16,7 +16,7 @@ - Interactive scripts: `0` - Package hash scope: `source-contract-without-generated-reports` - Package hash files: `197` -- Package SHA256: `30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e` +- Package SHA256: `002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8` ## Failures diff --git a/reports/skill-interpretation.json b/reports/skill-interpretation.json index e49f565..c9682d1 100644 --- a/reports/skill-interpretation.json +++ b/reports/skill-interpretation.json @@ -1000,13 +1000,13 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "bab9f7cd9fc387621005ec17e41186d232af6be1fb154824493e5c5b5cac6027", - "source_contract_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "evidence_bundle_sha256": "8e612a6cb7fbac70b9538d8c1f77713e1ce1806aad777502654b5489b54ce334", + "source_contract_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 22, @@ -1024,11 +1024,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 22 }, - "commit": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "commit": "cd933046c840affafa98edff697a2d0f68afd71e", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1133,7 +1133,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 197, - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" }, "skill_atlas": { "skill_count": 12, @@ -1171,8 +1171,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, "compatibility": { "openai": "pass", @@ -1203,7 +1203,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -1219,7 +1219,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "archive_entry_count": 617, "failure_count": 0, "warning_count": 0 @@ -1298,12 +1298,12 @@ { "field": "archive_sha256", "from": "", - "to": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "to": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "to": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" } ] }, diff --git a/reports/skill-overview.json b/reports/skill-overview.json index 8166e34..92b8ee9 100644 --- a/reports/skill-overview.json +++ b/reports/skill-overview.json @@ -995,13 +995,13 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "bab9f7cd9fc387621005ec17e41186d232af6be1fb154824493e5c5b5cac6027", - "source_contract_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "evidence_bundle_sha256": "8e612a6cb7fbac70b9538d8c1f77713e1ce1806aad777502654b5489b54ce334", + "source_contract_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 22, @@ -1019,11 +1019,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 22 }, - "commit": "ae5ce88ea7f4d002670ce9d3cb2bb339017902d7", + "commit": "cd933046c840affafa98edff697a2d0f68afd71e", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1128,7 +1128,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 197, - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" }, "skill_atlas": { "skill_count": 12, @@ -1166,8 +1166,8 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e", - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "package_sha256": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, "compatibility": { "openai": "pass", @@ -1198,7 +1198,7 @@ }, "distribution": { "archive_verified": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "package_verification": "reports/package_verification.json", "install_simulated": true, "install_simulation": "reports/install_simulation.json" @@ -1214,7 +1214,7 @@ "target_count": 4, "adapter_count": 4, "archive_present": true, - "archive_sha256": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874", + "archive_sha256": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8", "archive_entry_count": 617, "failure_count": 0, "warning_count": 0 @@ -1293,12 +1293,12 @@ { "field": "archive_sha256", "from": "", - "to": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "to": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "to": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" } ] }, diff --git a/reports/skill_os2_coverage.json b/reports/skill_os2_coverage.json index 77dcd43..47f92db 100644 --- a/reports/skill_os2_coverage.json +++ b/reports/skill_os2_coverage.json @@ -613,7 +613,7 @@ "label": "Evidence Consistency", "status": "pass", "objective": "Recommended Skill OS 2.0 implementation PR from the upgrade plan.", - "current": "27 consistency checks", + "current": "28 consistency checks", "command": "make ci-test", "test": "tests/verify_evidence_consistency.py", "evidence": [ diff --git a/reports/skill_os2_coverage.md b/reports/skill_os2_coverage.md index 2534e6a..0802af4 100644 --- a/reports/skill_os2_coverage.md +++ b/reports/skill_os2_coverage.md @@ -48,7 +48,7 @@ This report maps the Skill OS 2.0 upgrade blueprint to concrete local artifacts, | Registry Package Format | `pass` | registry ok True | `make ci-test` | `tests/verify_registry_audit.py` | | Review Studio 2.0 | `pass` | 16 review gates | `make ci-test` | `tests/verify_review_studio.py` | | Migration V2 Docs | `pass` | migration guide present | `make ci-test` | `docs review` | -| Evidence Consistency | `pass` | 27 consistency checks | `make ci-test` | `tests/verify_evidence_consistency.py` | +| Evidence Consistency | `pass` | 28 consistency checks | `make ci-test` | `tests/verify_evidence_consistency.py` | ## Reference Extension Tracks diff --git a/reports/upgrade_check.json b/reports/upgrade_check.json index 83e1b49..b3d5255 100644 --- a/reports/upgrade_check.json +++ b/reports/upgrade_check.json @@ -70,12 +70,12 @@ { "field": "archive_sha256", "from": "", - "to": "f1c499c597e88a4e1121d3464f5bcedfff2530128edf7a81f95602dda53e1874" + "to": "eddf1d422b1ad8d50dc0b49f0525c33010e4fe7db31a4c605dcd68c33adf56e8" }, { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "30502ca01a3d74140f6ad23d9d5fd499a8c80dcd4b8428ad1c9a812d2e3f9b5e" + "to": "002cb9e43ee22c7ba2a59422e4e123fb801242506a1a189b9a8bc8ad7d71f7f8" } ] }, diff --git a/scripts/render_evidence_consistency.py b/scripts/render_evidence_consistency.py index fcb9bb1..2f9d711 100644 --- a/scripts/render_evidence_consistency.py +++ b/scripts/render_evidence_consistency.py @@ -22,6 +22,7 @@ REQUIRED_REPORTS = { "install_simulation": "reports/install_simulation.json", "trust": "reports/security_trust_report.json", "context_budget": "reports/context_budget.json", + "world_class_claim_guard": "reports/world_class_claim_guard.json", } REQUIRED_TEXT_REPORTS = { "skill_os2_review": "reports/skill-os-2-review.md", @@ -122,6 +123,28 @@ def nested(payload: dict[str, Any], path: list[str], default: Any = None) -> Any return current +def scanned_surface_paths(payload: dict[str, Any]) -> set[str]: + surfaces = payload.get("scanned_surfaces") + if not isinstance(surfaces, list): + return set() + paths: set[str] = set() + for item in surfaces: + if isinstance(item, dict) and isinstance(item.get("path"), str): + paths.add(item["path"]) + elif isinstance(item, str): + paths.add(item) + return paths + + +def as_int(value: Any) -> int | None: + if isinstance(value, bool): + return None + try: + return int(value) + except (TypeError, ValueError): + return None + + def add_check( checks: list[dict[str, Any]], *, @@ -233,6 +256,7 @@ def build_report(skill_dir: Path, generated_at: str) -> dict[str, Any]: install_simulation = reports["install_simulation"] trust = reports["trust"] context_budget = reports["context_budget"] + claim_guard = reports["world_class_claim_guard"] benchmark_summary = nested(benchmark, ["summary"], {}) adoption_summary = nested(adoption, ["summary"], {}) @@ -243,6 +267,7 @@ def build_report(skill_dir: Path, generated_at: str) -> dict[str, Any]: install_summary = nested(install_simulation, ["summary"], {}) trust_summary = nested(trust, ["summary"], {}) context_stats = nested(context_budget, ["stats"], {}) + claim_guard_summary = nested(claim_guard, ["summary"], {}) if isinstance(benchmark_summary, dict): compare_values( checks, @@ -406,6 +431,80 @@ def build_report(skill_dir: Path, generated_at: str) -> dict[str, Any]: paths=[REQUIRED_REPORTS["world_class_ledger"], REQUIRED_REPORTS["review_studio"]], detail="When world-class evidence is pending, Review Studio must stay in a review or warning posture.", ) + claim_surface_paths = scanned_surface_paths(claim_guard) + required_claim_surfaces = [ + "README.md", + "SKILL.md", + "manifest.json", + "agents/interface.yaml", + "dist/manifest.json", + "dist/targets/openai/adapter.json", + "evidence/world_class/README.md", + "security/permission_policy.json", + "reports/world_class_evidence_ledger.json", + ] + prohibited_claim_surface_prefixes = [ + "dist/install-simulation/", + "evidence/world_class/submissions/", + ] + json_claim_surface_count = as_int(claim_guard_summary.get("json_claim_surface_count")) + metadata_claim_surface_count = as_int(claim_guard_summary.get("metadata_claim_surface_count")) + package_claim_surface_count = as_int(claim_guard_summary.get("package_claim_surface_count")) + claim_surface_count = as_int(claim_guard_summary.get("claim_surface_count")) + expected_claim_guard_surface = { + "overclaim_guard_active": True, + "violation_count": 0, + "ledger_ready_to_claim_world_class": ledger_summary.get("ready_to_claim_world_class") + if isinstance(ledger_summary, dict) + else None, + "ledger_pending_count": ledger_summary.get("pending_count") if isinstance(ledger_summary, dict) else None, + "metadata_covers_json": True, + "package_surface_minimum": True, + "claim_surface_covers_package": True, + "required_surfaces": {path: True for path in required_claim_surfaces}, + "prohibited_surfaces": [], + } + actual_claim_guard_surface = { + "overclaim_guard_active": claim_guard_summary.get("overclaim_guard_active"), + "violation_count": claim_guard_summary.get("violation_count"), + "ledger_ready_to_claim_world_class": claim_guard_summary.get("ledger_ready_to_claim_world_class"), + "ledger_pending_count": claim_guard_summary.get("ledger_pending_count"), + "metadata_covers_json": ( + metadata_claim_surface_count is not None + and json_claim_surface_count is not None + and metadata_claim_surface_count >= json_claim_surface_count + ), + "package_surface_minimum": package_claim_surface_count is not None and package_claim_surface_count >= 5, + "claim_surface_covers_package": ( + claim_surface_count is not None + and package_claim_surface_count is not None + and claim_surface_count >= package_claim_surface_count + ), + "required_surfaces": {path: path in claim_surface_paths for path in required_claim_surfaces}, + "prohibited_surfaces": sorted( + path + for path in claim_surface_paths + if any(path.startswith(prefix) for prefix in prohibited_claim_surface_prefixes) + ), + } + compare_values( + checks, + key="claim-guard-package-runtime-surface", + label="Claim guard covers package and runtime claim surfaces", + expected=expected_claim_guard_surface, + actual=actual_claim_guard_surface, + paths=[ + REQUIRED_REPORTS["world_class_claim_guard"], + "manifest.json", + "agents/interface.yaml", + "dist/manifest.json", + "dist/targets/openai/adapter.json", + "evidence/world_class/README.md", + "security/permission_policy.json", + REQUIRED_REPORTS["world_class_ledger"], + ], + detail="The overclaim guard must scan package manifests, adapter metadata, security policy, and ledger surfaces before public readiness can be trusted.", + ) skill_os2_review = text_reports.get("skill_os2_review", "") ci_target_count = ci_default_target_count(skill_dir / "scripts" / "ci_test.py") expected_review_snippets = [ diff --git a/tests/verify_evidence_consistency.py b/tests/verify_evidence_consistency.py index dabc19b..e648678 100644 --- a/tests/verify_evidence_consistency.py +++ b/tests/verify_evidence_consistency.py @@ -23,6 +23,7 @@ REPORT_FILES = [ "reports/install_simulation.json", "reports/security_trust_report.json", "reports/context_budget.json", + "reports/world_class_claim_guard.json", "reports/skill-os-2-review.md", "scripts/ci_test.py", ] @@ -44,6 +45,7 @@ def refresh_embedded_reports() -> None: script_names = [ "render_benchmark_reproducibility.py", "render_skill_os2_coverage.py", + "render_world_class_claim_guard.py", "render_skill_overview.py", "render_skill_interpretation.py", ] @@ -110,12 +112,15 @@ def main() -> None: assert payload["ok"] is True, payload assert payload["summary"]["decision"] == "consistent", payload assert payload["summary"]["fail_count"] == 0, payload - assert payload["summary"]["check_count"] >= 26, payload + assert payload["summary"]["check_count"] >= 28, payload checks = {item["key"]: item for item in payload["checks"]} assert checks["overview-benchmark-summary"]["status"] == "pass", checks["overview-benchmark-summary"] assert checks["interpretation-adoption-summary"]["status"] == "pass", checks["interpretation-adoption-summary"] assert checks["coverage-world-class-boundary"]["status"] == "pass", checks["coverage-world-class-boundary"] assert checks["review-studio-no-overclaim"]["status"] == "pass", checks["review-studio-no-overclaim"] + assert checks["claim-guard-package-runtime-surface"]["status"] == "pass", checks[ + "claim-guard-package-runtime-surface" + ] assert checks["skill-os-2-review-current-evidence"]["status"] == "pass", checks[ "skill-os-2-review-current-evidence" ] @@ -150,6 +155,35 @@ def main() -> None: assert drift_payload["summary"]["decision"] == "evidence-drift-detected", drift_payload assert drift_checks["overview-adoption-summary"]["status"] == "fail", drift_checks["overview-adoption-summary"] assert drift_checks["interpretation-adoption-summary"]["status"] == "pass", drift_checks["interpretation-adoption-summary"] + + claim_guard_drift_root = TMP / "claim-guard-drift-skill" + copy_reports(claim_guard_drift_root) + claim_guard_path = claim_guard_drift_root / "reports" / "world_class_claim_guard.json" + claim_guard = json.loads(claim_guard_path.read_text(encoding="utf-8")) + claim_guard["summary"]["package_claim_surface_count"] = 0 + claim_guard["scanned_surfaces"] = [ + item for item in claim_guard["scanned_surfaces"] if item.get("path") != "dist/manifest.json" + ] + claim_guard_path.write_text(json.dumps(claim_guard, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + claim_guard_drift_proc = run( + [ + sys.executable, + str(SCRIPT), + str(claim_guard_drift_root), + "--output-json", + str(TMP / "claim_guard_drift.json"), + "--output-md", + str(TMP / "claim_guard_drift.md"), + "--generated-at", + "2026-06-15", + ] + ) + assert claim_guard_drift_proc.returncode == 2, claim_guard_drift_proc.stdout + claim_guard_drift_payload = json.loads(claim_guard_drift_proc.stdout) + claim_guard_drift_checks = {item["key"]: item for item in claim_guard_drift_payload["checks"]} + assert claim_guard_drift_checks["claim-guard-package-runtime-surface"]["status"] == "fail", ( + claim_guard_drift_checks["claim-guard-package-runtime-surface"] + ) print(json.dumps({"ok": True}, ensure_ascii=False, indent=2))