diff --git a/reports/adaptation_proposals.json b/reports/adaptation_proposals.json index 594c30d..fe972de 100644 --- a/reports/adaptation_proposals.json +++ b/reports/adaptation_proposals.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-17", + "generated_at": "2026-06-16T16:14:10Z", "skill_dir": ".", "source_patterns": "reports/user_patterns.json", "pattern_count": 5, diff --git a/reports/adaptation_proposals.md b/reports/adaptation_proposals.md index 974c14d..5137665 100644 --- a/reports/adaptation_proposals.md +++ b/reports/adaptation_proposals.md @@ -1,6 +1,6 @@ # Adaptation Proposals -- Generated at: `2026-06-17` +- Generated at: `2026-06-16T16:14:10Z` - Pattern report: `reports/user_patterns.json` - Proposal only: `true` - Writes repository files: `false` diff --git a/reports/adoption_drift_report.json b/reports/adoption_drift_report.json index 3cc870e..ffa10b9 100644 --- a/reports/adoption_drift_report.json +++ b/reports/adoption_drift_report.json @@ -1,7 +1,7 @@ { "ok": true, "schema_version": "2.0", - "generated_at": "2026-06-16T16:07:15Z", + "generated_at": "2026-06-16T16:14:10Z", "skill_dir": ".", "privacy_contract": { "storage": "local-first", diff --git a/reports/architecture_maintainability.json b/reports/architecture_maintainability.json index 7927097..2fdf79f 100644 --- a/reports/architecture_maintainability.json +++ b/reports/architecture_maintainability.json @@ -87,7 +87,7 @@ }, { "path": "tests/verify_world_class_evidence_intake.py", - "lines": 610, + "lines": 628, "kind": "test", "severity": "pass", "recommendation": "Break broad integration assertions into focused verifier helpers when the next behavior change lands." diff --git a/reports/architecture_maintainability.md b/reports/architecture_maintainability.md index 7d12ee0..acade6a 100644 --- a/reports/architecture_maintainability.md +++ b/reports/architecture_maintainability.md @@ -42,7 +42,7 @@ No near-threshold files found. | `scripts/review_studio_gates.py` | `643` | `internal-module` | `pass` | | `scripts/cross_packager.py` | `638` | `cli-script` | `pass` | | `scripts/build_skill_atlas.py` | `637` | `cli-script` | `pass` | -| `tests/verify_world_class_evidence_intake.py` | `610` | `test` | `pass` | +| `tests/verify_world_class_evidence_intake.py` | `628` | `test` | `pass` | | `scripts/render_benchmark_reproducibility.py` | `595` | `cli-script` | `pass` | | `scripts/optimize_description.py` | `585` | `cli-script` | `pass` | diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json index 5a865c3..74e93d9 100644 --- a/reports/benchmark_reproducibility.json +++ b/reports/benchmark_reproducibility.json @@ -3,22 +3,35 @@ "ok": true, "generated_at": "2026-06-17", "skill_dir": ".", - "commit": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "commit": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "git_status": { "available": true, - "dirty": false, - "changed_file_count": 0, - "sample": [], + "dirty": true, + "changed_file_count": 30, + "sample": [ + " M reports/adaptation_proposals.json", + " M reports/adaptation_proposals.md", + " M reports/adoption_drift_report.json", + " M reports/architecture_maintainability.json", + " M reports/architecture_maintainability.md", + " M reports/benchmark_reproducibility.json", + " M reports/benchmark_reproducibility.md", + " M reports/context_budget.json", + " M reports/context_budget.md", + " M reports/context_budget_summary.json", + " M reports/output_execution_runs.json", + " M reports/output_execution_runs.md" + ], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91", - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "evidence_bundle_sha256": "c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -37,14 +50,15 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 30 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ + "release lock is not clean or commit is unavailable", "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", @@ -53,10 +67,10 @@ "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { - "ready": true, - "commit": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "ready": false, + "commit": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "status_scope": "generation-time status before this report is written", - "reason": "clean generation-time HEAD" + "reason": "working tree was dirty at generation time" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", @@ -64,7 +78,7 @@ "existing_count": 25, "missing_count": 0, "missing_paths": [], - "sha256": "62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91" + "sha256": "c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -138,7 +152,7 @@ "path": "reports/output_execution_runs.json", "exists": true, "bytes": 7966, - "sha256": "2ce010cd2fc2062d9a503e14394fae334b340d8c64192cf5536d4d579fd7b32d" + "sha256": "2c9409158a128e6d2ad4c96c499dd412591c7b287d907e8d88987132a0592633" }, { "label": "blind_review", @@ -173,7 +187,7 @@ "path": "reports/security_trust_report.json", "exists": true, "bytes": 129520, - "sha256": "893952155dde2bb2bc1b0d87fb1bf2c43ff4ed995e39c60bc3681f7f03b2a02d" + "sha256": "0d60cae961055a68ac84d83cc23d6d8456cc4b42b1a103dd963f3343a20987ff" }, { "label": "python_compatibility", @@ -270,8 +284,8 @@ "label": "world_class_claim_guard", "path": "reports/world_class_claim_guard.json", "exists": true, - "bytes": 18406, - "sha256": "e331d52d41166a07f068c44bf4c50c5dcdd9f1bce984e4303c832fe5f73d9d66" + "bytes": 18596, + "sha256": "abe7f7d60c0025e140373fadaefbed4063f285c140c74b9c7cfb464e354bb526" } ], "missing_artifacts": [], diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md index d923d30..c9a0240 100644 --- a/reports/benchmark_reproducibility.md +++ b/reports/benchmark_reproducibility.md @@ -1,18 +1,18 @@ # Benchmark Reproducibility Generated at: `2026-06-17` -Commit: `4a5880bea1a07966e0d914c453d22cf6132c5781` -Working tree dirty at generation: `false` -Evidence bundle SHA256: `62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91` +Commit: `500f8cc34ef5ff9a3c8125a72509faabbe95ac9a` +Working tree dirty at generation: `true` +Evidence bundle SHA256: `c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c` ## Summary - reproducibility ready: `true` -- release lock ready: `true` +- release lock ready: `false` - methodology complete: `true` - required artifacts: `25` - missing artifacts: `0` -- source contract sha256: `fcfe5f3c7222` +- source contract sha256: `d50f6ac9714b` - archive sha256: `5802e5f52255` - output cases: `5` - disclosed failure cases: `3` @@ -22,8 +22,8 @@ Evidence bundle SHA256: `62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcf - world-class ready: `false` - world-class source checks: `6` pass / `13` total; `7` blocked - public claim ready: `false` -- public claim blockers: `4` -- changed files at generation: `0` +- public claim blockers: `5` +- changed files at generation: `30` This report proves local benchmark reproducibility only. It keeps external provider and human-review gaps visible instead of counting them as complete. The git commit is generation-time context; the evidence bundle SHA is the durable anchor for the artifacts listed below. @@ -35,6 +35,7 @@ This report proves local benchmark reproducibility only. It keeps external provi | Blocker | | --- | +| release lock is not clean or commit is unavailable | | provider-backed model holdout evidence is incomplete | | human blind-review adjudication is incomplete | | world-class evidence is not accepted yet (4 open gaps, 4 ledger pending) | @@ -42,15 +43,15 @@ This report proves local benchmark reproducibility only. It keeps external provi ## Release Lock -- ready: `true` -- reason: clean generation-time HEAD +- ready: `false` +- reason: working tree was dirty at generation time - status scope: generation-time status before this report is written ## Evidence Bundle - algorithm: `sha256(path,label,exists,artifact_sha256)` - artifacts: `25` / `25` -- sha256: `62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91` +- sha256: `c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c` ## Methodology Sections @@ -72,12 +73,12 @@ This report proves local benchmark reproducibility only. It keeps external provi | output_cases | `evals/output/cases.jsonl` | present | `a6ae96857116` | | output_schema | `evals/output/schema.json` | present | `8ee340c95064` | | output_scorecard | `reports/output_quality_scorecard.json` | present | `0806258a8e08` | -| output_execution | `reports/output_execution_runs.json` | present | `2ce010cd2fc2` | +| output_execution | `reports/output_execution_runs.json` | present | `2c9409158a12` | | blind_review | `reports/output_blind_review_pack.json` | present | `bbe2db8ec277` | | review_adjudication | `reports/output_review_adjudication.json` | present | `bb8c72a9291e` | | trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` | | runtime_conformance | `reports/conformance_matrix.json` | present | `97f9ba949c23` | -| trust_report | `reports/security_trust_report.json` | present | `893952155dde` | +| trust_report | `reports/security_trust_report.json` | present | `0d60cae96105` | | python_compatibility | `reports/python_compatibility.json` | present | `8b48340618cc` | | registry_audit | `reports/registry_audit.json` | present | `e75a341d15e4` | | package_verification | `reports/package_verification.json` | present | `a27941fdb865` | @@ -91,7 +92,7 @@ This report proves local benchmark reproducibility only. It keeps external provi | world_class_operator_runbook | `reports/world_class_operator_runbook.json` | present | `f90d59c3e586` | | world_class_operator_runbook_markdown | `reports/world_class_operator_runbook.md` | present | `be79ee0f70a3` | | world_class_operator_runbook_html | `reports/world_class_operator_runbook.html` | present | `9a7be02a6990` | -| world_class_claim_guard | `reports/world_class_claim_guard.json` | present | `e331d52d4116` | +| world_class_claim_guard | `reports/world_class_claim_guard.json` | present | `abe7f7d60c00` | ## Reproduction Commands diff --git a/reports/context_budget.json b/reports/context_budget.json index 414d2f4..951489c 100644 --- a/reports/context_budget.json +++ b/reports/context_budget.json @@ -6,15 +6,15 @@ "context_budget_tier": "production", "context_budget_limit": 1000, "skill_body_tokens": 797, - "other_text_tokens": 1076491, + "other_text_tokens": 1076748, "estimated_initial_load_tokens": 990, - "estimated_total_text_tokens": 1077288, - "deferred_resource_tokens": 495322, + "estimated_total_text_tokens": 1077545, + "deferred_resource_tokens": 495216, "deferred_resource_warn_threshold": 120000, "deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 435224, + "estimated_tokens": 435118, "file_count": 140 }, { @@ -36,7 +36,7 @@ "large_deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 435224, + "estimated_tokens": 435118, "file_count": 140 } ], @@ -59,7 +59,7 @@ ], "missing": [], "path": "scripts", - "estimated_tokens": 435224, + "estimated_tokens": 435118, "file_count": 140, "rationale": "Script resources are deterministic deferred tools, not initial-load prompt context." } diff --git a/reports/context_budget.md b/reports/context_budget.md index b48f7da..e15688a 100644 --- a/reports/context_budget.md +++ b/reports/context_budget.md @@ -2,7 +2,7 @@ | Target | Path | Tier | Limit | Initial | SKILL | Deferred | Resource Governance | Large Deferred Dirs | Quality Density | Unused Dirs | Status | | --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- | ---: | --- | --- | -| root | `.` | `production` | 1000 | 990 | 797 | 495322 | `governed` | scripts:435224 | 131.3 | - | ok | +| root | `.` | `production` | 1000 | 990 | 797 | 495216 | `governed` | scripts:435118 | 131.3 | - | ok | | complex-release-orchestrator | `examples/complex-release-orchestrator/generated-skill` | `production` | 1000 | 790 | 718 | 1657 | `not-required` | - | 164.6 | - | ok | | governed-incident-command | `examples/governed-incident-command/generated-skill` | `production` | 1000 | 760 | 658 | 1030 | `not-required` | - | 171.1 | - | ok | diff --git a/reports/context_budget_summary.json b/reports/context_budget_summary.json index d251a08..03a62aa 100644 --- a/reports/context_budget_summary.json +++ b/reports/context_budget_summary.json @@ -8,11 +8,11 @@ "budget_limit": 1000, "initial_tokens": 990, "skill_body_tokens": 797, - "deferred_resource_tokens": 495322, + "deferred_resource_tokens": 495216, "large_deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 435224, + "estimated_tokens": 435118, "file_count": 140 } ], @@ -35,7 +35,7 @@ ], "missing": [], "path": "scripts", - "estimated_tokens": 435224, + "estimated_tokens": 435118, "file_count": 140, "rationale": "Script resources are deterministic deferred tools, not initial-load prompt context." } diff --git a/reports/evidence_consistency.json b/reports/evidence_consistency.json index f11155a..2dccc8b 100644 --- a/reports/evidence_consistency.json +++ b/reports/evidence_consistency.json @@ -189,12 +189,12 @@ "status": "pass", "expected": { "status": "pass", - "detail": "initial load 990/1000; deferred 495322/120000; top deferred scripts 435224; resource governance governed; quality density 131.3", + "detail": "initial load 990/1000; deferred 495216/120000; top deferred scripts 435118; resource governance governed; quality density 131.3", "evidence": "reports/context_budget.json" }, "actual": { "status": "pass", - "detail": "initial load 990/1000; deferred 495322/120000; top deferred scripts 435224; resource governance governed; quality density 131.3", + "detail": "initial load 990/1000; deferred 495216/120000; top deferred scripts 435118; resource governance governed; quality density 131.3", "evidence": "reports/context_budget.json" }, "paths": [ @@ -207,8 +207,8 @@ "key": "benchmark-release-lock-self-consistency", "label": "Benchmark release lock matches git dirty state", "status": "pass", - "expected": true, - "actual": true, + "expected": false, + "actual": false, "paths": [ "reports/benchmark_reproducibility.json" ], @@ -248,8 +248,8 @@ "key": "overview-benchmark-commit", "label": "overview embeds the benchmark commit", "status": "pass", - "expected": "4a5880bea1a07966e0d914c453d22cf6132c5781", - "actual": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "expected": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", + "actual": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-overview.json" @@ -261,30 +261,30 @@ "label": "overview embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 25, "missing_artifact_count": 0, - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "actual": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 25, "missing_artifact_count": 0, - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "paths": [ "reports/benchmark_reproducibility.json", @@ -394,8 +394,8 @@ "key": "interpretation-benchmark-commit", "label": "interpretation embeds the benchmark commit", "status": "pass", - "expected": "4a5880bea1a07966e0d914c453d22cf6132c5781", - "actual": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "expected": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", + "actual": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-interpretation.json" @@ -407,30 +407,30 @@ "label": "interpretation embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 25, "missing_artifact_count": 0, - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "actual": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 25, "missing_artifact_count": 0, - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "paths": [ "reports/benchmark_reproducibility.json", diff --git a/reports/output_execution_runs.json b/reports/output_execution_runs.json index db37f3a..154db78 100644 --- a/reports/output_execution_runs.json +++ b/reports/output_execution_runs.json @@ -34,7 +34,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.21, + "duration_ms": 25.96, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -62,7 +62,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.58, + "duration_ms": 26.2, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -85,7 +85,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.38, + "duration_ms": 26.36, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -113,7 +113,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.17, + "duration_ms": 26.09, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -136,7 +136,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.22, + "duration_ms": 26.43, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -164,7 +164,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.2, + "duration_ms": 26.16, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -187,7 +187,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.05, + "duration_ms": 27.55, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -214,7 +214,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 25.83, + "duration_ms": 28.86, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -237,7 +237,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.02, + "duration_ms": 28.87, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -266,7 +266,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.32, + "duration_ms": 30.17, "provider": "local-output-eval-runner", "model": "", "usage": { diff --git a/reports/output_execution_runs.md b/reports/output_execution_runs.md index dda40e9..c27422a 100644 --- a/reports/output_execution_runs.md +++ b/reports/output_execution_runs.md @@ -23,16 +23,16 @@ Command runner evidence is present. This proves the eval harness executed an ext | Case | Variant | Mode | Model | Duration ms | Tokens | Score | Status | | --- | --- | --- | --- | ---: | ---: | ---: | --- | -| skill-package-contract | baseline | command | local-output-eval-runner | 27.21 | 33 | 0.0 | pass | -| skill-package-contract | with_skill | command | local-output-eval-runner | 26.58 | 73 | 100.0 | pass | -| output-eval-expectation | baseline | command | local-output-eval-runner | 26.38 | 36 | 0.0 | pass | -| output-eval-expectation | with_skill | command | local-output-eval-runner | 26.17 | 80 | 100.0 | pass | -| ir-before-packaging | baseline | command | local-output-eval-runner | 26.22 | 33 | 0.0 | pass | -| ir-before-packaging | with_skill | command | local-output-eval-runner | 26.2 | 80 | 100.0 | pass | -| near-neighbor-boundary | baseline | command | local-output-eval-runner | 26.05 | 36 | 0.0 | pass | -| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 25.83 | 65 | 100.0 | pass | -| file-backed-governed-package | baseline | command | local-output-eval-runner | 26.02 | 37 | 0.0 | pass | -| file-backed-governed-package | with_skill | command | local-output-eval-runner | 26.32 | 98 | 100.0 | pass | +| skill-package-contract | baseline | command | local-output-eval-runner | 25.96 | 33 | 0.0 | pass | +| skill-package-contract | with_skill | command | local-output-eval-runner | 26.2 | 73 | 100.0 | pass | +| output-eval-expectation | baseline | command | local-output-eval-runner | 26.36 | 36 | 0.0 | pass | +| output-eval-expectation | with_skill | command | local-output-eval-runner | 26.09 | 80 | 100.0 | pass | +| ir-before-packaging | baseline | command | local-output-eval-runner | 26.43 | 33 | 0.0 | pass | +| ir-before-packaging | with_skill | command | local-output-eval-runner | 26.16 | 80 | 100.0 | pass | +| near-neighbor-boundary | baseline | command | local-output-eval-runner | 27.55 | 36 | 0.0 | pass | +| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 28.86 | 65 | 100.0 | pass | +| file-backed-governed-package | baseline | command | local-output-eval-runner | 28.87 | 37 | 0.0 | pass | +| file-backed-governed-package | with_skill | command | local-output-eval-runner | 30.17 | 98 | 100.0 | pass | ## Next Fixes diff --git a/reports/review-studio.html b/reports/review-studio.html index 062d60b..172d9f4 100644 --- a/reports/review-studio.html +++ b/reports/review-studio.html @@ -740,12 +740,12 @@

核心指标

-
Skill IR2.0.0

5 targets in platform-neutral contract

Compiler5/5

target contracts compiled from Skill IR

Output Delta100.0

5 cases; 1 file-backed

Exec Runs10

command 10; model 0; recorded 0

Blind A/B5

review pairs hide baseline vs with-skill labels

Review Kit0/5

pending 5; answer key hidden

Review A/B0/5

adjudication decisions; pending 5

Public Claimblocked

4 blockers; local reproducible true

Blueprint21/21

2.0 coverage; extensions partial 0, planned 0; evidence pending 4

Runtime5/5

target conformance pass rate

Perm Probe4/4

0 native; 4 installer-enforced

Trust0

140 scripts scanned; secrets found

Py Compat0

218 files scanned for Python 3.11

Arch Debt0

696 largest lines; 0 watchlist; 68 CLI handlers; 18 in entrypoint

Atlas5

12 scanned skills; route collisions

Driftlow

1 metadata events; 0 missed triggers

Daily Ops5

proposal-review; approval 0; release lock false

Weekly Queue5

curator-review; ready 1; top score 88

Waivers0

0 gates covered; human risk decisions

Intake4/4

0 valid submissions; 0 invalid

Claim Guard0

180 public surfaces scanned

Notes0/0

0 open blocker annotations

Registry1.1.0

5 targets; MIT license

Archivepass

658 zip entries; package verification

Installpass

4 adapters; 12 permissions enforced; 0 permission failures

Upgrademinor

declared minor; 0 breaking changes

+
Skill IR2.0.0

5 targets in platform-neutral contract

Compiler5/5

target contracts compiled from Skill IR

Output Delta100.0

5 cases; 1 file-backed

Exec Runs10

command 10; model 0; recorded 0

Blind A/B5

review pairs hide baseline vs with-skill labels

Review Kit0/5

pending 5; answer key hidden

Review A/B0/5

adjudication decisions; pending 5

Public Claimblocked

5 blockers; local reproducible true

Blueprint21/21

2.0 coverage; extensions partial 0, planned 0; evidence pending 4

Runtime5/5

target conformance pass rate

Perm Probe4/4

0 native; 4 installer-enforced

Trust0

140 scripts scanned; secrets found

Py Compat0

218 files scanned for Python 3.11

Arch Debt0

696 largest lines; 0 watchlist; 68 CLI handlers; 18 in entrypoint

Atlas5

12 scanned skills; route collisions

Driftlow

1 metadata events; 0 missed triggers

Daily Ops5

proposal-review; approval 0; release lock false

Weekly Queue5

curator-review; ready 1; top score 88

Waivers0

0 gates covered; human risk decisions

Intake4/4

0 valid submissions; 0 invalid

Claim Guard0

182 public surfaces scanned

Notes0/0

0 open blocker annotations

Registry1.1.0

5 targets; MIT license

Archivepass

658 zip entries; package verification

Installpass

4 adapters; 12 permissions enforced; 0 permission failures

Upgrademinor

declared minor; 0 breaking changes

审查闸门

-
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 990/1000; deferred 495322/120000; top deferred scripts 435224; resource governance governed; quality density 131.3

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 140 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

Python 兼容

Python 3.11; 218 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards

reports/python_compatibility.json 证据
通过

架构维护

215 Python files; 0 hotspots; 0 watchlist files; 0 blockers; largest 696 lines; 68 CLI handlers; 18 in entrypoint

reports/architecture_maintainability.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low; daily proposals 5; daily decision proposal-review; daily release lock false; weekly queue 5 unique; weekly ready 1; weekly top 88; weekly release lock false

reports/adoption_drift_report.json + reports/skillops/daily + reports/skillops/weekly 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

reports/promotion_decisions.json + reports/upgrade_check.json + docs/migration-v2.md 证据
+
通过

意图画布

intent confidence 100/100; Intent is clear enough to package the first routeable version.

reports/intent-confidence.json 证据
通过

触发实验

13 trigger cases; 0 misroutes; 0 ambiguous

reports/route_scorecard.json 证据
关注

输出实验

5/5 cases; with-skill 100.0; baseline 0.0; file-backed 1; near-neighbor 1; blind A/B 5; exec 10; command 10; model 0; recorded 0; reviewed 0/5; review pending 5

reports/output_quality_scorecard.json 证据
通过

上下文

initial load 990/1000; deferred 495216/120000; top deferred scripts 435118; resource governance governed; quality density 131.3

reports/context_budget.json 证据
通过

运行矩阵

5 / 5 targets pass

reports/conformance_matrix.json 证据
通过

信任报告

0 secrets; 140 scripts; 3 network-capable scripts; 0 help smoke failures

reports/security_trust_report.json 证据
通过

Python 兼容

Python 3.11; 218 files; 0 compatibility issues; 0 syntax; 0 f-string 3.11 hazards

reports/python_compatibility.json 证据
通过

架构维护

215 Python files; 0 hotspots; 0 watchlist files; 0 blockers; largest 696 lines; 68 CLI handlers; 18 in entrypoint

reports/architecture_maintainability.json 证据
通过

权限批准

3/3 permissions approved; gaps 0; required file_write, network, subprocess

reports/security_trust_report.json + security/permission_policy.json 证据
通过

权限探针

4/4 targets probed; native 0; metadata fallback 4; installer 4; residual risks 4

reports/runtime_permission_probes.json 证据
通过

组合治理

12 skills, 1 actionable; 0 actionable route collisions; 0 actionable owner gaps; 0 actionable stale; 0 actionable drift; 24 scoped non-actionable issues

reports/skill_atlas.json 证据
通过

运营回路

1 metadata events; adoption 0; missed 0; bad-output 0; risk low; daily proposals 5; daily decision proposal-review; daily release lock false; weekly queue 5 unique; weekly ready 1; weekly top 88; weekly release lock false

reports/adoption_drift_report.json + reports/skillops/daily + reports/skillops/weekly 证据
关注

人工批准

0 active waivers; 1 warning gates still need reviewer decision

reports/review_waivers.json 证据
关注

世界证据

4 pending world-class evidence entries; 1 human pending; 3 external pending; source checks 6/13 pass; 7 blocked; overclaim guard true

reports/world_class_evidence_ledger.json 证据
通过

注册审计

yao-meta-skill 1.1.0; 6/6 compatibility entries pass; install pass with 4 adapters; installer permissions 12 enforced / 0 failures

reports/registry_audit.json + reports/install_simulation.json 证据
通过

发布路线

0 promote; 3 keep current; 0 blocked; upgrade minor declared / minor recommended

@@ -812,12 +812,12 @@
-

上下文

initial load 990/1000; deferred 495322/120000; top deferred scripts 435224; resource governance governed; quality density 131.3

+

上下文

initial load 990/1000; deferred 495216/120000; top deferred scripts 435118; resource governance governed; quality density 131.3

编译证据

Review reports/compiled_targets.md before packaging to inspect target adapter modes, generated files, preserved semantics, warnings, and unsupported features.

-

信任报告

Secret
0
脚本数
140
网络脚本
3
Help 失败
0
包体哈希
fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627
+

信任报告

Secret
0
脚本数
140
网络脚本
3
Help 失败
0
包体哈希
d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69

安全边界

高风险 secret、远程 inline execution、缺失依赖策略或无法解释的脚本接口应阻断 governed release。

@@ -882,8 +882,8 @@
-

公开声明

本地复现
发布锁
可公开声明
声明阻断
4
Provider 证据
人审完成
世界级就绪
-

声明阻断

+

公开声明

本地复现
发布锁
可公开声明
声明阻断
5
Provider 证据
人审完成
世界级就绪
+

声明阻断

@@ -897,7 +897,7 @@
-

声明守卫

台账可声明
台账待补
4
声明面
180
违规数
0
Overclaim Guard Active
+

声明守卫

台账可声明
台账待补
4
声明面
182
违规数
0
Overclaim Guard Active

声明边界

claim guard 扫描 README、docs 和 reports 中的完成态表述;ledger 未 ready 时,任何英文完成断言、true 状态声明或中文完成态都会阻断发布审查。

diff --git a/reports/review-studio.json b/reports/review-studio.json index 7aed7ff..48d6391 100644 --- a/reports/review-studio.json +++ b/reports/review-studio.json @@ -44,7 +44,7 @@ "key": "context-budget", "label": "上下文", "status": "pass", - "detail": "initial load 990/1000; deferred 495322/120000; top deferred scripts 435224; resource governance governed; quality density 131.3", + "detail": "initial load 990/1000; deferred 495216/120000; top deferred scripts 435118; resource governance governed; quality density 131.3", "evidence": "reports/context_budget.json", "link": "context_budget.md" }, @@ -1968,12 +1968,12 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91", - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "evidence_bundle_sha256": "c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1992,11 +1992,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 30 }, - "commit": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "commit": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -2101,7 +2101,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 231, - "package_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627" + "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" }, "skill_atlas": { "skill_count": 12, @@ -4701,7 +4701,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 27.21, + "duration_ms": 25.96, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4729,7 +4729,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.58, + "duration_ms": 26.2, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4752,7 +4752,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.38, + "duration_ms": 26.36, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4780,7 +4780,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.17, + "duration_ms": 26.09, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4803,7 +4803,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.22, + "duration_ms": 26.43, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4831,7 +4831,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.2, + "duration_ms": 26.16, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4854,7 +4854,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.05, + "duration_ms": 27.55, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4881,7 +4881,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 25.83, + "duration_ms": 28.86, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4904,7 +4904,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.02, + "duration_ms": 28.87, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -4933,7 +4933,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 26.32, + "duration_ms": 30.17, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -5659,22 +5659,35 @@ "ok": true, "generated_at": "2026-06-17", "skill_dir": ".", - "commit": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "commit": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "git_status": { "available": true, - "dirty": false, - "changed_file_count": 0, - "sample": [], + "dirty": true, + "changed_file_count": 30, + "sample": [ + " M reports/adaptation_proposals.json", + " M reports/adaptation_proposals.md", + " M reports/adoption_drift_report.json", + " M reports/architecture_maintainability.json", + " M reports/architecture_maintainability.md", + " M reports/benchmark_reproducibility.json", + " M reports/benchmark_reproducibility.md", + " M reports/context_budget.json", + " M reports/context_budget.md", + " M reports/context_budget_summary.json", + " M reports/output_execution_runs.json", + " M reports/output_execution_runs.md" + ], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91", - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "evidence_bundle_sha256": "c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -5693,14 +5706,15 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 30 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ + "release lock is not clean or commit is unavailable", "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", @@ -5709,10 +5723,10 @@ "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { - "ready": true, - "commit": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "ready": false, + "commit": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "status_scope": "generation-time status before this report is written", - "reason": "clean generation-time HEAD" + "reason": "working tree was dirty at generation time" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", @@ -5720,7 +5734,7 @@ "existing_count": 25, "missing_count": 0, "missing_paths": [], - "sha256": "62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91" + "sha256": "c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -5794,7 +5808,7 @@ "path": "reports/output_execution_runs.json", "exists": true, "bytes": 7966, - "sha256": "2ce010cd2fc2062d9a503e14394fae334b340d8c64192cf5536d4d579fd7b32d" + "sha256": "2c9409158a128e6d2ad4c96c499dd412591c7b287d907e8d88987132a0592633" }, { "label": "blind_review", @@ -5829,7 +5843,7 @@ "path": "reports/security_trust_report.json", "exists": true, "bytes": 129520, - "sha256": "893952155dde2bb2bc1b0d87fb1bf2c43ff4ed995e39c60bc3681f7f03b2a02d" + "sha256": "0d60cae961055a68ac84d83cc23d6d8456cc4b42b1a103dd963f3343a20987ff" }, { "label": "python_compatibility", @@ -5926,8 +5940,8 @@ "label": "world_class_claim_guard", "path": "reports/world_class_claim_guard.json", "exists": true, - "bytes": 18406, - "sha256": "e331d52d41166a07f068c44bf4c50c5dcdd9f1bce984e4303c832fe5f73d9d66" + "bytes": 18596, + "sha256": "abe7f7d60c0025e140373fadaefbed4063f285c140c74b9c7cfb464e354bb526" } ], "missing_artifacts": [], @@ -6815,11 +6829,11 @@ "exists": true }, { - "path": "reports/skillops/daily/2026-06-16.json", + "path": "reports/skillops/daily/2026-06-17.json", "exists": true }, { - "path": "reports/skillops/daily/2026-06-16.md", + "path": "reports/skillops/daily/2026-06-17.md", "exists": true } ], @@ -12104,7 +12118,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 231, - "package_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627" + "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" }, "failures": [], "warnings": [], @@ -16819,7 +16833,7 @@ }, { "path": "tests/verify_world_class_evidence_intake.py", - "lines": 610, + "lines": 628, "kind": "test", "severity": "pass", "recommendation": "Break broad integration assertions into focused verifier helpers when the next behavior change lands." @@ -16855,15 +16869,15 @@ "context_budget_tier": "production", "context_budget_limit": 1000, "skill_body_tokens": 797, - "other_text_tokens": 1076491, + "other_text_tokens": 1076748, "estimated_initial_load_tokens": 990, - "estimated_total_text_tokens": 1077288, - "deferred_resource_tokens": 495322, + "estimated_total_text_tokens": 1077545, + "deferred_resource_tokens": 495216, "deferred_resource_warn_threshold": 120000, "deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 435224, + "estimated_tokens": 435118, "file_count": 140 }, { @@ -16885,7 +16899,7 @@ "large_deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 435224, + "estimated_tokens": 435118, "file_count": 140 } ], @@ -16908,7 +16922,7 @@ ], "missing": [], "path": "scripts", - "estimated_tokens": 435224, + "estimated_tokens": 435118, "file_count": 140, "rationale": "Script resources are deterministic deferred tools, not initial-load prompt context." } @@ -18811,7 +18825,7 @@ "adoption_drift": { "ok": true, "schema_version": "2.0", - "generated_at": "2026-06-16T16:07:15Z", + "generated_at": "2026-06-16T16:14:10Z", "skill_dir": ".", "privacy_contract": { "storage": "local-first", @@ -19409,7 +19423,7 @@ "weekly_curator": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-17", + "generated_at": "2026-06-16T16:14:10Z", "skill_dir": ".", "decision": "curator-review", "week_id": "2026-W25", @@ -19874,7 +19888,7 @@ "adaptation_proposals": { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-17", + "generated_at": "2026-06-16T16:14:10Z", "skill_dir": ".", "source_patterns": "reports/user_patterns.json", "pattern_count": 5, @@ -23043,9 +23057,9 @@ "summary": { "ledger_ready_to_claim_world_class": false, "ledger_pending_count": 4, - "claim_surface_count": 180, - "json_claim_surface_count": 88, - "metadata_claim_surface_count": 89, + "claim_surface_count": 182, + "json_claim_surface_count": 89, + "metadata_claim_surface_count": 90, "package_claim_surface_count": 17, "violation_count": 0, "overclaim_guard_active": true, @@ -23655,6 +23669,14 @@ "path": "reports/skillops/daily/2026-06-16.md", "violation_count": 0 }, + { + "path": "reports/skillops/daily/2026-06-17.json", + "violation_count": 0 + }, + { + "path": "reports/skillops/daily/2026-06-17.md", + "violation_count": 0 + }, { "path": "reports/skillops/weekly/2026-W25.json", "violation_count": 0 diff --git a/reports/review-viewer.json b/reports/review-viewer.json index 81dd1bc..2e026fc 100644 --- a/reports/review-viewer.json +++ b/reports/review-viewer.json @@ -997,12 +997,12 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91", - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "evidence_bundle_sha256": "c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1021,11 +1021,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 30 }, - "commit": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "commit": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1130,7 +1130,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 231, - "package_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627" + "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" }, "skill_atlas": { "skill_count": 12, diff --git a/reports/security_trust_report.json b/reports/security_trust_report.json index 3284e10..8e08b53 100644 --- a/reports/security_trust_report.json +++ b/reports/security_trust_report.json @@ -23,7 +23,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 231, - "package_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627" + "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" }, "failures": [], "warnings": [], diff --git a/reports/security_trust_report.md b/reports/security_trust_report.md index ea64f0a..2d01bed 100644 --- a/reports/security_trust_report.md +++ b/reports/security_trust_report.md @@ -16,7 +16,7 @@ - Interactive scripts: `0` - Package hash scope: `source-contract-without-generated-reports` - Package hash files: `231` -- Package SHA256: `fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627` +- Package SHA256: `d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69` ## Failures diff --git a/reports/skill-interpretation.json b/reports/skill-interpretation.json index f9a4c0d..bf842b9 100644 --- a/reports/skill-interpretation.json +++ b/reports/skill-interpretation.json @@ -1001,12 +1001,12 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91", - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "evidence_bundle_sha256": "c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1025,11 +1025,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 30 }, - "commit": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "commit": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1134,7 +1134,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 231, - "package_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627" + "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" }, "skill_atlas": { "skill_count": 12, diff --git a/reports/skill-os-2-review.md b/reports/skill-os-2-review.md index 5e26c77..a0122cd 100644 --- a/reports/skill-os-2-review.md +++ b/reports/skill-os-2-review.md @@ -23,6 +23,7 @@ Yao Meta Skill is no longer only a Meta Skill factory. The current working tree - World-Class Intake Contract Hardening v0 so real evidence submissions must use the ledger's canonical `.json` filename and are recursively rejected when they include raw prompt, output, transcript, message, credential, secret, token, or API-key fields. - World-Class Human Evidence Guard v0 so human-adjudication decision artifacts recursively reject raw content, credential, secret, token, and answer-key fields before blind A/B review evidence can be accepted. - Output Review Privacy Guard v0 so the blind-review importer and world-class human evidence validator share one recursive blocked-field contract for raw content, credential, secret, token, and answer-key leakage. +- World-Class Submission Privacy Guard v0 so real evidence submission packets reuse the same blocked-field contract and reject nested answer-key leakage before ledger review. - World-Class Provider Evidence Guard v0 so provider holdout submissions must reconcile `summary` counts with `runs` rows and include at least one passing model run whose provider, model, timing, non-estimated usage, and output hash match the submitted provenance. - World-Class Native Permission Evidence Guard v0 so native-permission submissions must reconcile runtime probe summary counts with target rows, require at least one native-enforced target row, and keep installer permission checks failure-free. - World-Class Native Telemetry Evidence Guard v0 so native-client-telemetry submissions must reconcile adoption summary counts with external metadata event rows and keep hook recipes metadata-only without claiming native auto-capture. @@ -71,7 +72,7 @@ This is still not the final world-class state. Target-native behavior contracts | Skill OS 2.0 Audit | `scripts/render_skill_os2_audit.py`, `reports/skill_os2_audit.md`, `tests/verify_skill_os2_audit.py` | v0 landed | | World-Class Evidence Plan | `scripts/render_world_class_evidence_plan.py`, `reports/world_class_evidence_plan.md`, `tests/verify_world_class_evidence_plan.py` | v0 landed | | World-Class Evidence Ledger | `scripts/render_world_class_evidence_ledger.py`, `reports/world_class_evidence_ledger.md`, `tests/verify_world_class_evidence_ledger.py` | v0 landed | -| World-Class Evidence Intake | `scripts/world_class_evidence_contract.py`, `scripts/world_class_human_evidence.py`, `scripts/output_review_privacy.py`, `scripts/world_class_provider_evidence.py`, `scripts/world_class_native_permission_evidence.py`, `scripts/world_class_native_telemetry_evidence.py`, `scripts/render_world_class_evidence_intake.py`, `evidence/world_class/intake.schema.json`, `tests/verify_world_class_evidence_intake.py` with canonical filename, source-artifact validation, recursive human decision privacy and answer-key validation, provider run-row validation, native permission target-row validation, native telemetry event-row validation, and nested raw-field rejection | v0 landed | +| World-Class Evidence Intake | `scripts/world_class_evidence_contract.py`, `scripts/world_class_human_evidence.py`, `scripts/output_review_privacy.py`, `scripts/world_class_provider_evidence.py`, `scripts/world_class_native_permission_evidence.py`, `scripts/world_class_native_telemetry_evidence.py`, `scripts/render_world_class_evidence_intake.py`, `evidence/world_class/intake.schema.json`, `tests/verify_world_class_evidence_intake.py` with canonical filename, source-artifact validation, recursive human decision privacy and answer-key validation, provider run-row validation, native permission target-row validation, native telemetry event-row validation, nested raw-field rejection, and real-submission answer-key leakage rejection | v0 landed | | World-Class Submission Kit | `scripts/prepare_world_class_submission_kit.py`, `scripts/world_class_submission_matrix.py`, `scripts/world_class_submission_kit_rendering.py`, `tests/verify_world_class_submission_kit.py` with draft, artifact, source-check, next-action matrix evidence, and separated Markdown/HTML rendering | v0 landed | | Runtime Conformance | `scripts/run_conformance_suite.py`, `reports/conformance_matrix.md` | v0 landed | | Trust & Security | `scripts/trust_check.py`, `reports/security_trust_report.md`, `security/*.md` | v0 landed | diff --git a/reports/skill-overview.json b/reports/skill-overview.json index 67c25ef..929a4e4 100644 --- a/reports/skill-overview.json +++ b/reports/skill-overview.json @@ -996,12 +996,12 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 25, "missing_artifact_count": 0, - "evidence_bundle_sha256": "62e7b774ed1b2bd66e28986ad09ffecb9da8ae4d9008641cfe63dcfe354f0c91", - "source_contract_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627", + "evidence_bundle_sha256": "c76666b64b01fbc6f421863b68fd621585935574fe0ce3959a472c26604e1c2c", + "source_contract_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69", "archive_sha256": "5802e5f522558cf77666ae59390a28845cf3affd22fa1267b6629891ddeba76b", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1020,11 +1020,11 @@ "world_class_source_pass_count": 6, "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 30 }, - "commit": "4a5880bea1a07966e0d914c453d22cf6132c5781", + "commit": "500f8cc34ef5ff9a3c8125a72509faabbe95ac9a", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1129,7 +1129,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 231, - "package_sha256": "fcfe5f3c722285daf14b6f83118ad71bdca21694d86d870d234d9f5df2f4e627" + "package_sha256": "d50f6ac9714b62b985f9f9edb181f87bb57246be1e4249cb92e7e4a87ae5aa69" }, "skill_atlas": { "skill_count": 12, diff --git a/reports/skill_os2_coverage.json b/reports/skill_os2_coverage.json index 344278b..24bc4ce 100644 --- a/reports/skill_os2_coverage.json +++ b/reports/skill_os2_coverage.json @@ -749,11 +749,11 @@ "exists": true }, { - "path": "reports/skillops/daily/2026-06-16.json", + "path": "reports/skillops/daily/2026-06-17.json", "exists": true }, { - "path": "reports/skillops/daily/2026-06-16.md", + "path": "reports/skillops/daily/2026-06-17.md", "exists": true } ], diff --git a/reports/skill_os2_coverage.md b/reports/skill_os2_coverage.md index 169ee9d..fd915be 100644 --- a/reports/skill_os2_coverage.md +++ b/reports/skill_os2_coverage.md @@ -235,7 +235,7 @@ These extension tracks come from the user-supplied 2.0 reference plan. They are - objective: Daily operations layer summarizes explicit-source conversation patterns, proposal-only adaptation work, approval state, release locks, and world-class evidence gaps. - status: `covered` -- existing evidence: `scripts/render_daily_skillops_report.py`, `tests/verify_daily_skillops.py`, `reports/skillops/daily/2026-06-16.json`, `reports/skillops/daily/2026-06-16.md` +- existing evidence: `scripts/render_daily_skillops_report.py`, `tests/verify_daily_skillops.py`, `reports/skillops/daily/2026-06-17.json`, `reports/skillops/daily/2026-06-17.md` - next action: Keep Daily SkillOps report aligned with proposal, approval, coverage, and world-class ledger contracts as the operations layer evolves. ### Weekly Curator Report diff --git a/reports/skillops/daily/2026-06-16.json b/reports/skillops/daily/2026-06-16.json index 52faf36..a21b7f6 100644 --- a/reports/skillops/daily/2026-06-16.json +++ b/reports/skillops/daily/2026-06-16.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-16T16:07:16Z", + "generated_at": "2026-06-16T16:14:10Z", "skill_dir": ".", "decision": "proposal-review", "source_supplied": false, diff --git a/reports/skillops/daily/2026-06-16.md b/reports/skillops/daily/2026-06-16.md index f54b449..de03fe2 100644 --- a/reports/skillops/daily/2026-06-16.md +++ b/reports/skillops/daily/2026-06-16.md @@ -1,6 +1,6 @@ # Daily SkillOps Report -Generated at: `2026-06-16T16:07:16Z` +Generated at: `2026-06-16T16:14:10Z` ## Summary diff --git a/reports/skillops/weekly/2026-W25.json b/reports/skillops/weekly/2026-W25.json index 8c1ed2f..ef43813 100644 --- a/reports/skillops/weekly/2026-W25.json +++ b/reports/skillops/weekly/2026-W25.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-17", + "generated_at": "2026-06-16T16:14:10Z", "skill_dir": ".", "decision": "curator-review", "week_id": "2026-W25", diff --git a/reports/skillops/weekly/2026-W25.md b/reports/skillops/weekly/2026-W25.md index e8b03e0..6bcefe1 100644 --- a/reports/skillops/weekly/2026-W25.md +++ b/reports/skillops/weekly/2026-W25.md @@ -1,6 +1,6 @@ # Weekly SkillOps Curator Report -Generated at: `2026-06-17` +Generated at: `2026-06-16T16:14:10Z` Week: `2026-W25` ## Summary diff --git a/reports/world_class_claim_guard.json b/reports/world_class_claim_guard.json index 192bb9f..22da949 100644 --- a/reports/world_class_claim_guard.json +++ b/reports/world_class_claim_guard.json @@ -6,9 +6,9 @@ "summary": { "ledger_ready_to_claim_world_class": false, "ledger_pending_count": 4, - "claim_surface_count": 180, - "json_claim_surface_count": 88, - "metadata_claim_surface_count": 89, + "claim_surface_count": 182, + "json_claim_surface_count": 89, + "metadata_claim_surface_count": 90, "package_claim_surface_count": 17, "violation_count": 0, "overclaim_guard_active": true, @@ -618,6 +618,14 @@ "path": "reports/skillops/daily/2026-06-16.md", "violation_count": 0 }, + { + "path": "reports/skillops/daily/2026-06-17.json", + "violation_count": 0 + }, + { + "path": "reports/skillops/daily/2026-06-17.md", + "violation_count": 0 + }, { "path": "reports/skillops/weekly/2026-W25.json", "violation_count": 0 diff --git a/reports/world_class_claim_guard.md b/reports/world_class_claim_guard.md index a41564b..d2d4d66 100644 --- a/reports/world_class_claim_guard.md +++ b/reports/world_class_claim_guard.md @@ -7,9 +7,9 @@ Generated at: `2026-06-17` - decision: `claim-guard-pass-evidence-pending` - ledger ready to claim world-class: `false` - ledger pending evidence: `4` -- claim surfaces scanned: `180` -- JSON claim surfaces scanned: `88` -- metadata claim surfaces scanned: `89` +- claim surfaces scanned: `182` +- JSON claim surfaces scanned: `89` +- metadata claim surfaces scanned: `90` - package/runtime claim surfaces scanned: `17` - violations: `0` - overclaim guard active: `true` diff --git a/scripts/world_class_evidence_contract.py b/scripts/world_class_evidence_contract.py index bfb3bba..1010e5b 100644 --- a/scripts/world_class_evidence_contract.py +++ b/scripts/world_class_evidence_contract.py @@ -6,6 +6,7 @@ from datetime import datetime, timezone from pathlib import Path from typing import Any +from output_review_privacy import BLOCKED_DECISION_FIELDS from world_class_human_evidence import validate_human_adjudication_report from world_class_native_permission_evidence import validate_native_permission_report from world_class_native_telemetry_evidence import validate_native_telemetry_report @@ -102,36 +103,7 @@ PLACEHOLDER_FRAGMENTS = ( "client or installer component", "/local/path/not/committed", ) -FORBIDDEN_REAL_SUBMISSION_FIELDS = { - "api_key", - "assistant_message", - "assistant_messages", - "baseline_output", - "credential", - "credentials", - "input", - "inputs", - "message", - "messages", - "model_output", - "output", - "outputs", - "prompt", - "prompts", - "raw_content", - "raw_output", - "raw_prompt", - "raw_provider_prompt", - "raw_user_content", - "secret", - "secrets", - "token", - "transcript", - "transcripts", - "user_message", - "user_messages", - "with_skill_output", -} +FORBIDDEN_REAL_SUBMISSION_FIELDS = BLOCKED_DECISION_FIELDS def load_json(path: Path) -> dict[str, Any]: @@ -236,7 +208,7 @@ def validate_real_submission_privacy_fields( add_error( errors, not blocked_paths, - "real submission must not include raw content, credential, secret, token, prompt, output, transcript, or message fields: " + "real submission must not include raw content, credential, secret, token, prompt, output, transcript, message, or answer-key fields: " + ", ".join(blocked_paths[:8]), ) diff --git a/tests/verify_world_class_evidence_intake.py b/tests/verify_world_class_evidence_intake.py index 929879f..c1d0f75 100644 --- a/tests/verify_world_class_evidence_intake.py +++ b/tests/verify_world_class_evidence_intake.py @@ -381,6 +381,24 @@ def assert_external_contract_artifact_validation() -> None: assert provider_leak_result["status"] == "fail", provider_leak_result assert any("raw content, credential, secret" in error for error in provider_leak_result["errors"]), provider_leak_result["errors"] assert any("$.raw_prompt" in error and "$.provenance.messages" in error for error in provider_leak_result["errors"]), provider_leak_result["errors"] + provider_answer_key_leak = provider_artifact_submission(skill_root) + provider_answer_key_leak["provenance"]["Expected_Winner_Variant"] = "A" + provider_answer_key_leak["review_notes"] = [{"answer_key": "blind answer key must not be embedded"}] + provider_answer_key_leak_result = validate_payload( + provider_answer_key_leak, + provider_entry, + path=skill_root / "evidence" / "world_class" / "submissions" / "provider-holdout.json", + root=skill_root, + template_expected=False, + ) + assert provider_answer_key_leak_result["status"] == "fail", provider_answer_key_leak_result + assert any("answer-key fields" in error for error in provider_answer_key_leak_result["errors"]), ( + provider_answer_key_leak_result["errors"] + ) + assert any( + "$.provenance.Expected_Winner_Variant" in error and "$.review_notes[0].answer_key" in error + for error in provider_answer_key_leak_result["errors"] + ), provider_answer_key_leak_result["errors"] write_provider_artifact(skill_root, complete=False) provider_invalid = validate_payload( provider_artifact_submission(skill_root),