diff --git a/reports/adoption_drift_report.json b/reports/adoption_drift_report.json index 0cc85b7..6c2e3e4 100644 --- a/reports/adoption_drift_report.json +++ b/reports/adoption_drift_report.json @@ -1,7 +1,7 @@ { "ok": true, "schema_version": "2.0", - "generated_at": "2026-06-15T14:07:59Z", + "generated_at": "2026-06-15T14:15:32Z", "skill_dir": ".", "privacy_contract": { "storage": "local-first", @@ -25,14 +25,14 @@ }, "summary": { "event_count": 1, - "adoption_sample_count": 1, - "activation_count": 1, - "accepted_count": 1, + "adoption_sample_count": 0, + "activation_count": 0, + "accepted_count": 0, "edited_count": 0, "rejected_count": 0, "missed_count": 0, "failed_count": 0, - "adoption_rate": 100.0, + "adoption_rate": 0, "missed_trigger_count": 0, "wrong_trigger_count": 0, "bad_output_count": 0, @@ -41,7 +41,7 @@ "review_overdue_count": 0, "risk_band": "low", "event_types": { - "skill_activation": 1 + "review_event": 1 }, "failure_types": {}, "source_types": { @@ -53,31 +53,31 @@ { "skill": "yao-meta-skill", "events": 1, - "adoption_events": 1, - "accepted": 1, + "adoption_events": 0, + "accepted": 0, "edited": 0, "rejected": 0, "missed": 0, - "adoption_rate": 100.0 + "adoption_rate": 0 } ], "next_iteration_candidates": [], "recent_events": [ { "command": "unknown", - "event": "skill_activation", + "event": "review_event", "skill": "yao-meta-skill", "source": "manual", "version": "1.1.0", - "activation_type": "explicit", - "outcome": "accepted", + "activation_type": "manual", + "outcome": "reviewed", "failure_type": "none", - "timestamp": "2026-06-13T10:00:00Z" + "timestamp": "2026-06-13T12:00:00Z" } ], "failures": [], "artifacts": { - "events_jsonl": "tests/tmp_review_studio/telemetry_events.jsonl", + "events_jsonl": "reports/telemetry_events.jsonl", "json": "reports/adoption_drift_report.json", "markdown": "reports/adoption_drift_report.md" } diff --git a/reports/adoption_drift_report.md b/reports/adoption_drift_report.md index 206372c..91b3198 100644 --- a/reports/adoption_drift_report.md +++ b/reports/adoption_drift_report.md @@ -5,9 +5,9 @@ Local-first, metadata-only telemetry for skill operations. Raw prompts, outputs, ## Summary - Events: `1` -- Adoption samples: `1` -- Activation events: `1` -- Adoption rate: `100.0` +- Adoption samples: `0` +- Activation events: `0` +- Adoption rate: `0` - Missed trigger signals: `0` - Bad output signals: `0` - Script error signals: `0` @@ -25,7 +25,7 @@ Local-first, metadata-only telemetry for skill operations. Raw prompts, outputs, | Skill | Events | Adoption Samples | Accepted | Edited | Rejected | Missed | Adoption Rate | | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | -| `yao-meta-skill` | 1 | 1 | 1 | 0 | 0 | 0 | 100.0 | +| `yao-meta-skill` | 1 | 0 | 0 | 0 | 0 | 0 | 0 | ## Next Iteration Candidates @@ -33,4 +33,4 @@ Local-first, metadata-only telemetry for skill operations. Raw prompts, outputs, ## Recent Metadata Events -- `2026-06-13T10:00:00Z` `yao-meta-skill` event=`skill_activation` source=`manual` command=`unknown` activation=`explicit` outcome=`accepted` failure=`none` +- `2026-06-13T12:00:00Z` `yao-meta-skill` event=`review_event` source=`manual` command=`unknown` activation=`manual` outcome=`reviewed` failure=`none` diff --git a/reports/architecture_maintainability.json b/reports/architecture_maintainability.json index 9f8f1a1..12fd0d1 100644 --- a/reports/architecture_maintainability.json +++ b/reports/architecture_maintainability.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "python_file_count": 170, diff --git a/reports/architecture_maintainability.md b/reports/architecture_maintainability.md index 8c7686f..86e9fe4 100644 --- a/reports/architecture_maintainability.md +++ b/reports/architecture_maintainability.md @@ -1,6 +1,6 @@ # Architecture Maintainability -Generated at: `2026-06-13` +Generated at: `2026-06-15` ## Summary diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json index 0bb6ccf..df528c4 100644 --- a/reports/benchmark_reproducibility.json +++ b/reports/benchmark_reproducibility.json @@ -3,21 +3,34 @@ "ok": true, "generated_at": "2026-06-15", "skill_dir": ".", - "commit": "c169e61dd493bc11f6b920323e25a31be888a67a", + "commit": "db1b24cbda5eb00f71b35846dff0f178625aa31d", "git_status": { "available": true, - "dirty": false, - "changed_file_count": 0, - "sample": [], + "dirty": true, + "changed_file_count": 38, + "sample": [ + " M reports/adoption_drift_report.json", + " M reports/adoption_drift_report.md", + " M reports/architecture_maintainability.json", + " M reports/architecture_maintainability.md", + " M reports/benchmark_reproducibility.json", + " M reports/benchmark_reproducibility.md", + " M reports/compiled_targets.json", + " M reports/context_budget.json", + " M reports/context_budget.md", + " M reports/context_budget_summary.json", + " M reports/output_execution_runs.json", + " M reports/output_execution_runs.md" + ], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "27da0809c59bcea3757a4f60fd43b515a8f51b11ab9ef805ba423b13a22f19e4", + "evidence_bundle_sha256": "0d1c762a722d1bbc83339a0365efcfcdf027849fc83c7079da92a524b905f5ab", "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "output_case_count": 5, @@ -34,29 +47,30 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 7, - "world_class_source_blocked_count": 6, + "world_class_source_pass_count": 6, + "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 38 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ + "release lock is not clean or commit is unavailable", "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", - "world-class source checks are not all accepted (7/13 pass, 6 blocked)" + "world-class source checks are not all accepted (6/13 pass, 7 blocked)" ], "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { - "ready": true, - "commit": "c169e61dd493bc11f6b920323e25a31be888a67a", + "ready": false, + "commit": "db1b24cbda5eb00f71b35846dff0f178625aa31d", "status_scope": "generation-time status before this report is written", - "reason": "clean generation-time HEAD" + "reason": "working tree was dirty at generation time" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", @@ -64,7 +78,7 @@ "existing_count": 24, "missing_count": 0, "missing_paths": [], - "sha256": "27da0809c59bcea3757a4f60fd43b515a8f51b11ab9ef805ba423b13a22f19e4" + "sha256": "0d1c762a722d1bbc83339a0365efcfcdf027849fc83c7079da92a524b905f5ab" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -137,8 +151,8 @@ "label": "output_execution", "path": "reports/output_execution_runs.json", "exists": true, - "bytes": 7966, - "sha256": "0fdf9c4498785556f8bc64f6418aba96d58fda70c40a6d8677cb343988897c63" + "bytes": 7965, + "sha256": "3f13fd99af6576302e9fffc8ef78161b2f9a7ef95bf94a22d361a3cb7bda2843" }, { "label": "blind_review", @@ -180,7 +194,7 @@ "path": "reports/python_compatibility.json", "exists": true, "bytes": 22510, - "sha256": "d9a52fb5a8d7fbc8f3e6550d26f44a73f0ffafbc2adc682330b550b0f9fe27f3" + "sha256": "73c6c2a81af9980cc1d8e2d1bb5dfb30bd8daccac48125fe71ec359dbde824dd" }, { "label": "registry_audit", @@ -215,56 +229,56 @@ "path": "reports/world_class_evidence_plan.json", "exists": true, "bytes": 19940, - "sha256": "b408af112c784fdc3e96ea9ecf374990b9f58a0ebfa4df18394ccd9e93dc5545" + "sha256": "933cdb0021818c0a8c5fc199af7dfa7879777ebf7959ed3a36bba52a0aed9c6c" }, { "label": "world_class_evidence_ledger", "path": "reports/world_class_evidence_ledger.json", "exists": true, - "bytes": 20003, - "sha256": "cc886713ba5b99bdb2f740217f0979a30312aef7539d2a22af0e96ec5c3e90bf" + "bytes": 20006, + "sha256": "5407409841eb8e1174a07ea0af17ddf709155f0c548c897b82c2f75b869b333a" }, { "label": "world_class_evidence_intake", "path": "reports/world_class_evidence_intake.json", "exists": true, "bytes": 18865, - "sha256": "2fbecc603d8dc6382fde6c61869aaf4b56fc9d1be5afcd07b345c419dce6d55a" + "sha256": "b10e1ce0a5a17df72462ce8f0c468356f58ff88bd7ceab349c82a1fb3b6697ef" }, { "label": "world_class_submission_review", "path": "reports/world_class_submission_review.json", "exists": true, - "bytes": 12410, - "sha256": "bc690472684dcdb8c77803b730857f98e6b6eb11f1e741d5fcc1aa9303b77e47" + "bytes": 12413, + "sha256": "3bce5f072d037a6c383cfd597f8530c2bc87ba31781fa90c843a9186dd2f8fea" }, { "label": "world_class_operator_runbook", "path": "reports/world_class_operator_runbook.json", "exists": true, - "bytes": 23957, - "sha256": "b4640688026ef9199e96f80bd2d9971aaca3ec6d092d8ecd052a5debc0aa94c7" + "bytes": 24021, + "sha256": "d377b8d99831ce297011e4e270fc7287a53727500f5d503dd3dc4b984d26f0ad" }, { "label": "world_class_operator_runbook_markdown", "path": "reports/world_class_operator_runbook.md", "exists": true, - "bytes": 15025, - "sha256": "3a916ac568377d8df7d612426f7358283eddfa2ab5fedb7ed12db0097a8b3c48" + "bytes": 15080, + "sha256": "9f141f09bf485a4299b37eb8c21ebab9f212ea130a0daea8b0d7d5bd7ab020aa" }, { "label": "world_class_operator_runbook_html", "path": "reports/world_class_operator_runbook.html", "exists": true, - "bytes": 20969, - "sha256": "886b915da0f27fb46f4d8cebb6211ab5066a9fd44e1ec2e3206487d9a8abfff2" + "bytes": 21030, + "sha256": "04cc091b113f018072d07a1d71ad142773c415a49e633493689903e7b399b5e6" }, { "label": "world_class_claim_guard", "path": "reports/world_class_claim_guard.json", "exists": true, "bytes": 8814, - "sha256": "9420a85386ea9044b679d8a5b8771a34efef19d7c58c84ce10653410e4f13e4b" + "sha256": "7e5a2eac1020f4f58688b52d3c59b5df1c74d7bc04f4e599b48e5be4cc23d786" } ], "missing_artifacts": [], diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md index 58b00f7..f405685 100644 --- a/reports/benchmark_reproducibility.md +++ b/reports/benchmark_reproducibility.md @@ -1,14 +1,14 @@ # Benchmark Reproducibility Generated at: `2026-06-15` -Commit: `c169e61dd493bc11f6b920323e25a31be888a67a` -Working tree dirty at generation: `false` -Evidence bundle SHA256: `27da0809c59bcea3757a4f60fd43b515a8f51b11ab9ef805ba423b13a22f19e4` +Commit: `db1b24cbda5eb00f71b35846dff0f178625aa31d` +Working tree dirty at generation: `true` +Evidence bundle SHA256: `0d1c762a722d1bbc83339a0365efcfcdf027849fc83c7079da92a524b905f5ab` ## Summary - reproducibility ready: `true` -- release lock ready: `true` +- release lock ready: `false` - methodology complete: `true` - required artifacts: `24` - missing artifacts: `0` @@ -20,10 +20,10 @@ Evidence bundle SHA256: `27da0809c59bcea3757a4f60fd43b515a8f51b11ab9ef805ba423b1 - provider evidence complete: `false` - human review complete: `false` - world-class ready: `false` -- world-class source checks: `7` pass / `13` total; `6` blocked +- world-class source checks: `6` pass / `13` total; `7` blocked - public claim ready: `false` -- public claim blockers: `4` -- changed files at generation: `0` +- public claim blockers: `5` +- changed files at generation: `38` This report proves local benchmark reproducibility only. It keeps external provider and human-review gaps visible instead of counting them as complete. The git commit is generation-time context; the evidence bundle SHA is the durable anchor for the artifacts listed below. @@ -35,22 +35,23 @@ This report proves local benchmark reproducibility only. It keeps external provi | Blocker | | --- | +| release lock is not clean or commit is unavailable | | provider-backed model holdout evidence is incomplete | | human blind-review adjudication is incomplete | | world-class evidence is not accepted yet (4 open gaps, 4 ledger pending) | -| world-class source checks are not all accepted (7/13 pass, 6 blocked) | +| world-class source checks are not all accepted (6/13 pass, 7 blocked) | ## Release Lock -- ready: `true` -- reason: clean generation-time HEAD +- ready: `false` +- reason: working tree was dirty at generation time - status scope: generation-time status before this report is written ## Evidence Bundle - algorithm: `sha256(path,label,exists,artifact_sha256)` - artifacts: `24` / `24` -- sha256: `27da0809c59bcea3757a4f60fd43b515a8f51b11ab9ef805ba423b13a22f19e4` +- sha256: `0d1c762a722d1bbc83339a0365efcfcdf027849fc83c7079da92a524b905f5ab` ## Methodology Sections @@ -72,25 +73,25 @@ This report proves local benchmark reproducibility only. It keeps external provi | output_cases | `evals/output/cases.jsonl` | present | `a6ae96857116` | | output_schema | `evals/output/schema.json` | present | `8ee340c95064` | | output_scorecard | `reports/output_quality_scorecard.json` | present | `0806258a8e08` | -| output_execution | `reports/output_execution_runs.json` | present | `0fdf9c449878` | +| output_execution | `reports/output_execution_runs.json` | present | `3f13fd99af65` | | blind_review | `reports/output_blind_review_pack.json` | present | `bbe2db8ec277` | | review_adjudication | `reports/output_review_adjudication.json` | present | `240485a721af` | | trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` | | runtime_conformance | `reports/conformance_matrix.json` | present | `8251329e663d` | | trust_report | `reports/security_trust_report.json` | present | `6409321f1c0d` | -| python_compatibility | `reports/python_compatibility.json` | present | `d9a52fb5a8d7` | +| python_compatibility | `reports/python_compatibility.json` | present | `73c6c2a81af9` | | registry_audit | `reports/registry_audit.json` | present | `76a55d6dfc15` | | package_verification | `reports/package_verification.json` | present | `2476ae8ec9c4` | | install_simulation | `reports/install_simulation.json` | present | `490e1f665580` | | skill_os2_audit | `reports/skill_os2_audit.json` | present | `a4cf40478f3a` | -| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `b408af112c78` | -| world_class_evidence_ledger | `reports/world_class_evidence_ledger.json` | present | `cc886713ba5b` | -| world_class_evidence_intake | `reports/world_class_evidence_intake.json` | present | `2fbecc603d8d` | -| world_class_submission_review | `reports/world_class_submission_review.json` | present | `bc690472684d` | -| world_class_operator_runbook | `reports/world_class_operator_runbook.json` | present | `b4640688026e` | -| world_class_operator_runbook_markdown | `reports/world_class_operator_runbook.md` | present | `3a916ac56837` | -| world_class_operator_runbook_html | `reports/world_class_operator_runbook.html` | present | `886b915da0f2` | -| world_class_claim_guard | `reports/world_class_claim_guard.json` | present | `9420a85386ea` | +| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `933cdb002181` | +| world_class_evidence_ledger | `reports/world_class_evidence_ledger.json` | present | `5407409841eb` | +| world_class_evidence_intake | `reports/world_class_evidence_intake.json` | present | `b10e1ce0a5a1` | +| world_class_submission_review | `reports/world_class_submission_review.json` | present | `3bce5f072d03` | +| world_class_operator_runbook | `reports/world_class_operator_runbook.json` | present | `d377b8d99831` | +| world_class_operator_runbook_markdown | `reports/world_class_operator_runbook.md` | present | `9f141f09bf48` | +| world_class_operator_runbook_html | `reports/world_class_operator_runbook.html` | present | `04cc091b113f` | +| world_class_claim_guard | `reports/world_class_claim_guard.json` | present | `7e5a2eac1020` | ## Reproduction Commands diff --git a/reports/compiled_targets.json b/reports/compiled_targets.json index 1897309..8f1ef92 100644 --- a/reports/compiled_targets.json +++ b/reports/compiled_targets.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "target_count": 5, diff --git a/reports/context_budget.json b/reports/context_budget.json index e786825..a8bcb18 100644 --- a/reports/context_budget.json +++ b/reports/context_budget.json @@ -6,15 +6,15 @@ "context_budget_tier": "production", "context_budget_limit": 1000, "skill_body_tokens": 767, - "other_text_tokens": 1332957, + "other_text_tokens": 1333919, "estimated_initial_load_tokens": 960, - "estimated_total_text_tokens": 1333724, - "deferred_resource_tokens": 421528, + "estimated_total_text_tokens": 1334686, + "deferred_resource_tokens": 421781, "deferred_resource_warn_threshold": 120000, "deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 372938, + "estimated_tokens": 373191, "file_count": 107 }, { @@ -31,7 +31,7 @@ "large_deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 372938, + "estimated_tokens": 373191, "file_count": 107 } ], @@ -54,14 +54,14 @@ ], "missing": [], "path": "scripts", - "estimated_tokens": 372938, + "estimated_tokens": 373191, "file_count": 107, "rationale": "Script resources are deterministic deferred tools, not initial-load prompt context." } ], "summary": "Large deferred resources are indexed and backed by evidence." }, - "relevant_file_count": 548, + "relevant_file_count": 549, "unused_resource_dirs": [], "quality_signal_points": 130, "quality_density": 135.4 diff --git a/reports/context_budget.md b/reports/context_budget.md index f447ebd..ac4e011 100644 --- a/reports/context_budget.md +++ b/reports/context_budget.md @@ -2,7 +2,7 @@ | Target | Path | Tier | Limit | Initial | SKILL | Deferred | Resource Governance | Large Deferred Dirs | Quality Density | Unused Dirs | Status | | --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- | ---: | --- | --- | -| root | `.` | `production` | 1000 | 960 | 767 | 421528 | `governed` | scripts:372938 | 135.4 | - | ok | +| root | `.` | `production` | 1000 | 960 | 767 | 421781 | `governed` | scripts:373191 | 135.4 | - | ok | | complex-release-orchestrator | `examples/complex-release-orchestrator/generated-skill` | `production` | 1000 | 790 | 718 | 1657 | `not-required` | - | 164.6 | - | ok | | governed-incident-command | `examples/governed-incident-command/generated-skill` | `production` | 1000 | 760 | 658 | 1030 | `not-required` | - | 171.1 | - | ok | diff --git a/reports/context_budget_summary.json b/reports/context_budget_summary.json index fe581f8..19092b3 100644 --- a/reports/context_budget_summary.json +++ b/reports/context_budget_summary.json @@ -8,11 +8,11 @@ "budget_limit": 1000, "initial_tokens": 960, "skill_body_tokens": 767, - "deferred_resource_tokens": 421528, + "deferred_resource_tokens": 421781, "large_deferred_resource_dirs": [ { "path": "scripts", - "estimated_tokens": 372938, + "estimated_tokens": 373191, "file_count": 107 } ], @@ -35,7 +35,7 @@ ], "missing": [], "path": "scripts", - "estimated_tokens": 372938, + "estimated_tokens": 373191, "file_count": 107, "rationale": "Script resources are deterministic deferred tools, not initial-load prompt context." } diff --git a/reports/evidence_consistency.json b/reports/evidence_consistency.json index 5ecdf67..9ae9ec5 100644 --- a/reports/evidence_consistency.json +++ b/reports/evidence_consistency.json @@ -37,8 +37,8 @@ "key": "benchmark-release-lock-self-consistency", "label": "Benchmark release lock matches git dirty state", "status": "pass", - "expected": true, - "actual": true, + "expected": false, + "actual": false, "paths": [ "reports/benchmark_reproducibility.json" ], @@ -48,8 +48,8 @@ "key": "overview-benchmark-commit", "label": "overview embeds the benchmark commit", "status": "pass", - "expected": "c169e61dd493bc11f6b920323e25a31be888a67a", - "actual": "c169e61dd493bc11f6b920323e25a31be888a67a", + "expected": "db1b24cbda5eb00f71b35846dff0f178625aa31d", + "actual": "db1b24cbda5eb00f71b35846dff0f178625aa31d", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-overview.json" @@ -61,30 +61,30 @@ "label": "overview embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 7, - "world_class_source_blocked_count": 6, + "world_class_source_pass_count": 6, + "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "actual": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 7, - "world_class_source_blocked_count": 6, + "world_class_source_pass_count": 6, + "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "paths": [ "reports/benchmark_reproducibility.json", @@ -98,13 +98,13 @@ "status": "pass", "expected": { "event_count": 1, - "adoption_sample_count": 1, - "activation_count": 1, - "accepted_count": 1, - "adoption_rate": 100.0, + "adoption_sample_count": 0, + "activation_count": 0, + "accepted_count": 0, + "adoption_rate": 0, "risk_band": "low", "event_types": { - "skill_activation": 1 + "review_event": 1 }, "source_types": { "manual": 1 @@ -112,13 +112,13 @@ }, "actual": { "event_count": 1, - "adoption_sample_count": 1, - "activation_count": 1, - "accepted_count": 1, - "adoption_rate": 100.0, + "adoption_sample_count": 0, + "activation_count": 0, + "accepted_count": 0, + "adoption_rate": 0, "risk_band": "low", "event_types": { - "skill_activation": 1 + "review_event": 1 }, "source_types": { "manual": 1 @@ -141,8 +141,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "ready_to_claim_world_class": false, "decision": "evidence-pending" }, @@ -153,8 +153,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "ready_to_claim_world_class": false, "decision": "evidence-pending" }, @@ -174,7 +174,7 @@ "pending_count": 4, "accepted_count": 0, "source_check_count": 13, - "source_pass_count": 7 + "source_pass_count": 6 }, "actual": { "ready": false, @@ -182,7 +182,7 @@ "pending_count": 4, "accepted_count": 0, "source_check_count": 13, - "source_pass_count": 7 + "source_pass_count": 6 }, "paths": [ "reports/world_class_evidence_ledger.json", @@ -194,8 +194,8 @@ "key": "interpretation-benchmark-commit", "label": "interpretation embeds the benchmark commit", "status": "pass", - "expected": "c169e61dd493bc11f6b920323e25a31be888a67a", - "actual": "c169e61dd493bc11f6b920323e25a31be888a67a", + "expected": "db1b24cbda5eb00f71b35846dff0f178625aa31d", + "actual": "db1b24cbda5eb00f71b35846dff0f178625aa31d", "paths": [ "reports/benchmark_reproducibility.json", "reports/skill-interpretation.json" @@ -207,30 +207,30 @@ "label": "interpretation embeds benchmark summary fields", "status": "pass", "expected": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 7, - "world_class_source_blocked_count": 6, + "world_class_source_pass_count": 6, + "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "actual": { - "release_lock_ready": true, + "release_lock_ready": false, "required_artifact_count": 24, "missing_artifact_count": 0, "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 7, - "world_class_source_blocked_count": 6, + "world_class_source_pass_count": 6, + "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4 + "public_claim_blocker_count": 5 }, "paths": [ "reports/benchmark_reproducibility.json", @@ -244,13 +244,13 @@ "status": "pass", "expected": { "event_count": 1, - "adoption_sample_count": 1, - "activation_count": 1, - "accepted_count": 1, - "adoption_rate": 100.0, + "adoption_sample_count": 0, + "activation_count": 0, + "accepted_count": 0, + "adoption_rate": 0, "risk_band": "low", "event_types": { - "skill_activation": 1 + "review_event": 1 }, "source_types": { "manual": 1 @@ -258,13 +258,13 @@ }, "actual": { "event_count": 1, - "adoption_sample_count": 1, - "activation_count": 1, - "accepted_count": 1, - "adoption_rate": 100.0, + "adoption_sample_count": 0, + "activation_count": 0, + "accepted_count": 0, + "adoption_rate": 0, "risk_band": "low", "event_types": { - "skill_activation": 1 + "review_event": 1 }, "source_types": { "manual": 1 @@ -287,8 +287,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "ready_to_claim_world_class": false, "decision": "evidence-pending" }, @@ -299,8 +299,8 @@ "human_pending_count": 1, "external_pending_count": 3, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "ready_to_claim_world_class": false, "decision": "evidence-pending" }, @@ -320,7 +320,7 @@ "pending_count": 4, "accepted_count": 0, "source_check_count": 13, - "source_pass_count": 7 + "source_pass_count": 6 }, "actual": { "ready": false, @@ -328,7 +328,7 @@ "pending_count": 4, "accepted_count": 0, "source_check_count": 13, - "source_pass_count": 7 + "source_pass_count": 6 }, "paths": [ "reports/world_class_evidence_ledger.json", @@ -1127,7 +1127,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 7, + "source_pass_count": 6, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -1186,7 +1186,8 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events" + "External events", + "Adoption sample" ] } ] @@ -1200,7 +1201,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 7, + "source_pass_count": 6, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -1259,7 +1260,8 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events" + "External events", + "Adoption sample" ] } ] @@ -1609,15 +1611,15 @@ "expected": { "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 7, - "world_class_source_blocked_count": 6, + "world_class_source_pass_count": 6, + "world_class_source_blocked_count": 7, "public_claim_ready": false }, "actual": { "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 7, - "world_class_source_blocked_count": 6, + "world_class_source_pass_count": 6, + "world_class_source_blocked_count": 7, "public_claim_ready": false }, "paths": [ diff --git a/reports/output_execution_runs.json b/reports/output_execution_runs.json index be75241..3079576 100644 --- a/reports/output_execution_runs.json +++ b/reports/output_execution_runs.json @@ -34,7 +34,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 36.06, + "duration_ms": 28.76, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -62,7 +62,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 33.24, + "duration_ms": 27.9, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -85,7 +85,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 34.12, + "duration_ms": 27.6, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -113,7 +113,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 35.2, + "duration_ms": 28.21, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -136,7 +136,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 36.31, + "duration_ms": 28.73, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -164,7 +164,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 40.88, + "duration_ms": 28.14, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -187,7 +187,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 64.44, + "duration_ms": 28.71, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -214,7 +214,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 66.38, + "duration_ms": 28.58, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -237,7 +237,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 44.12, + "duration_ms": 28.49, "provider": "local-output-eval-runner", "model": "", "usage": { @@ -266,7 +266,7 @@ "execution_mode": "command", "model_executed": false, "command_executed": true, - "duration_ms": 45.77, + "duration_ms": 27.46, "provider": "local-output-eval-runner", "model": "", "usage": { diff --git a/reports/output_execution_runs.md b/reports/output_execution_runs.md index db86bf1..7b6460c 100644 --- a/reports/output_execution_runs.md +++ b/reports/output_execution_runs.md @@ -23,16 +23,16 @@ Command runner evidence is present. This proves the eval harness executed an ext | Case | Variant | Mode | Model | Duration ms | Tokens | Score | Status | | --- | --- | --- | --- | ---: | ---: | ---: | --- | -| skill-package-contract | baseline | command | local-output-eval-runner | 36.06 | 33 | 0.0 | pass | -| skill-package-contract | with_skill | command | local-output-eval-runner | 33.24 | 73 | 100.0 | pass | -| output-eval-expectation | baseline | command | local-output-eval-runner | 34.12 | 36 | 0.0 | pass | -| output-eval-expectation | with_skill | command | local-output-eval-runner | 35.2 | 80 | 100.0 | pass | -| ir-before-packaging | baseline | command | local-output-eval-runner | 36.31 | 33 | 0.0 | pass | -| ir-before-packaging | with_skill | command | local-output-eval-runner | 40.88 | 80 | 100.0 | pass | -| near-neighbor-boundary | baseline | command | local-output-eval-runner | 64.44 | 36 | 0.0 | pass | -| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 66.38 | 65 | 100.0 | pass | -| file-backed-governed-package | baseline | command | local-output-eval-runner | 44.12 | 37 | 0.0 | pass | -| file-backed-governed-package | with_skill | command | local-output-eval-runner | 45.77 | 98 | 100.0 | pass | +| skill-package-contract | baseline | command | local-output-eval-runner | 28.76 | 33 | 0.0 | pass | +| skill-package-contract | with_skill | command | local-output-eval-runner | 27.9 | 73 | 100.0 | pass | +| output-eval-expectation | baseline | command | local-output-eval-runner | 27.6 | 36 | 0.0 | pass | +| output-eval-expectation | with_skill | command | local-output-eval-runner | 28.21 | 80 | 100.0 | pass | +| ir-before-packaging | baseline | command | local-output-eval-runner | 28.73 | 33 | 0.0 | pass | +| ir-before-packaging | with_skill | command | local-output-eval-runner | 28.14 | 80 | 100.0 | pass | +| near-neighbor-boundary | baseline | command | local-output-eval-runner | 28.71 | 36 | 0.0 | pass | +| near-neighbor-boundary | with_skill | command | local-output-eval-runner | 28.58 | 65 | 100.0 | pass | +| file-backed-governed-package | baseline | command | local-output-eval-runner | 28.49 | 37 | 0.0 | pass | +| file-backed-governed-package | with_skill | command | local-output-eval-runner | 27.46 | 98 | 100.0 | pass | ## Next Fixes diff --git a/reports/python_compatibility.json b/reports/python_compatibility.json index b3b97e6..3a19430 100644 --- a/reports/python_compatibility.json +++ b/reports/python_compatibility.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "root": ".", "summary": { "target_python": "3.11", diff --git a/reports/python_compatibility.md b/reports/python_compatibility.md index f1252bb..8fcded0 100644 --- a/reports/python_compatibility.md +++ b/reports/python_compatibility.md @@ -1,6 +1,6 @@ # Python Compatibility -Generated at: `2026-06-13` +Generated at: `2026-06-15` ## Summary diff --git a/reports/review-viewer.json b/reports/review-viewer.json index 8b51190..e5b450e 100644 --- a/reports/review-viewer.json +++ b/reports/review-viewer.json @@ -1000,8 +1000,8 @@ "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "1a11230c3274b89fc877d2e243970b3c15508fa913d7efc8f26d53949395c4a2", - "source_contract_sha256": "278b4ff6c0c7e8406dfa62e5813013532af58e35d12b22f8bef25d59ac544713", + "evidence_bundle_sha256": "0d1c762a722d1bbc83339a0365efcfcdf027849fc83c7079da92a524b905f5ab", + "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1022,9 +1022,9 @@ "public_claim_ready": false, "public_claim_blocker_count": 5, "working_tree_dirty": true, - "changed_file_count": 16 + "changed_file_count": 38 }, - "commit": "84b761abce9f0a361ddea7d446d0454666de0885", + "commit": "db1b24cbda5eb00f71b35846dff0f178625aa31d", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1129,7 +1129,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 194, - "package_sha256": "278b4ff6c0c7e8406dfa62e5813013532af58e35d12b22f8bef25d59ac544713" + "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306" }, "skill_atlas": { "skill_count": 12, @@ -1167,7 +1167,7 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "278b4ff6c0c7e8406dfa62e5813013532af58e35d12b22f8bef25d59ac544713", + "package_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc" }, "compatibility": { @@ -1299,7 +1299,7 @@ { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "278b4ff6c0c7e8406dfa62e5813013532af58e35d12b22f8bef25d59ac544713" + "to": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306" } ] }, diff --git a/reports/review_annotations.json b/reports/review_annotations.json index 66f2b1d..76fadc1 100644 --- a/reports/review_annotations.json +++ b/reports/review_annotations.json @@ -2,7 +2,7 @@ "schema_version": "1.0", "ok": true, "skill_dir": ".", - "source": "tests/tmp_review_studio/empty_review_annotations_input.json", + "source": "reports/review_annotations_input.json", "summary": { "annotation_count": 0, "open_count": 0, diff --git a/reports/review_waivers.json b/reports/review_waivers.json index 07eeccb..3acbd17 100644 --- a/reports/review_waivers.json +++ b/reports/review_waivers.json @@ -2,7 +2,7 @@ "schema_version": "1.0", "ok": true, "skill_dir": ".", - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "summary": { "waiver_count": 0, "active_count": 0, @@ -52,7 +52,7 @@ "Reviewer links output_review_adjudication or output_execution evidence." ], "suggested_evidence": "reports/output_review_adjudication.md", - "suggested_command": "python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer \"\" --reason \"Output Lab has pending human/provider evidence; accepted only for this bounded review scope.\" --expires-at 2027-06-13 --evidence reports/output_review_adjudication.md", + "suggested_command": "python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer \"\" --reason \"Output Lab has pending human/provider evidence; accepted only for this bounded review scope.\" --expires-at 2027-06-15 --evidence reports/output_review_adjudication.md", "world_class_boundary": "Does not count as provider, human, or public world-class completion evidence." }, { diff --git a/reports/review_waivers.md b/reports/review_waivers.md index dbc942d..2fcb054 100644 --- a/reports/review_waivers.md +++ b/reports/review_waivers.md @@ -35,7 +35,7 @@ - waiver allowed: `true` - risk: review pending 5; model-executed 0; output failures 0 - evidence: `reports/output_review_adjudication.md` -- verification: `python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer "" --reason "Output Lab has pending human/provider evidence; accepted only for this bounded review scope." --expires-at 2027-06-13 --evidence reports/output_review_adjudication.md` +- verification: `python3 scripts/yao.py review-waivers . --add-waiver --gate-key output-lab --reviewer "" --reason "Output Lab has pending human/provider evidence; accepted only for this bounded review scope." --expires-at 2027-06-15 --evidence reports/output_review_adjudication.md` - world-class boundary: Does not count as provider, human, or public world-class completion evidence. #### Required Review diff --git a/reports/skill-interpretation.html b/reports/skill-interpretation.html index d215364..6e8bccf 100644 --- a/reports/skill-interpretation.html +++ b/reports/skill-interpretation.html @@ -919,7 +919,7 @@ -

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks7 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
+

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks6 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
  • 采用样本Adoption sample
diff --git a/reports/skill-interpretation.json b/reports/skill-interpretation.json index 9451eff..22fa66f 100644 --- a/reports/skill-interpretation.json +++ b/reports/skill-interpretation.json @@ -412,7 +412,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 7, + "source_pass_count": 6, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -471,7 +471,8 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events" + "External events", + "Adoption sample" ] } ] @@ -999,11 +1000,11 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "27da0809c59bcea3757a4f60fd43b515a8f51b11ab9ef805ba423b13a22f19e4", + "evidence_bundle_sha256": "0d1c762a722d1bbc83339a0365efcfcdf027849fc83c7079da92a524b905f5ab", "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "output_case_count": 5, @@ -1020,14 +1021,14 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 7, - "world_class_source_blocked_count": 6, + "world_class_source_pass_count": 6, + "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 38 }, - "commit": "c169e61dd493bc11f6b920323e25a31be888a67a", + "commit": "db1b24cbda5eb00f71b35846dff0f178625aa31d", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1319,14 +1320,14 @@ "ok": true, "summary": { "event_count": 1, - "adoption_sample_count": 1, - "activation_count": 1, - "accepted_count": 1, + "adoption_sample_count": 0, + "activation_count": 0, + "accepted_count": 0, "edited_count": 0, "rejected_count": 0, "missed_count": 0, "failed_count": 0, - "adoption_rate": 100.0, + "adoption_rate": 0, "missed_trigger_count": 0, "wrong_trigger_count": 0, "bad_output_count": 0, @@ -1335,7 +1336,7 @@ "review_overdue_count": 0, "risk_band": "low", "event_types": { - "skill_activation": 1 + "review_event": 1 }, "failure_types": {}, "source_types": { @@ -1557,7 +1558,7 @@ "status": "external_required", "category": "external", "owner": "Browser/Chrome/IDE/provider client integrator", - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1609,8 +1610,8 @@ "missing_submission_count": 4, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "submitted_but_pending_count": 0, "source_accepted_without_valid_submission_count": 0, "overclaim_guard_active": true, @@ -1946,7 +1947,7 @@ "status": "pending", "source_status": "external_required", "source_accepted": false, - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1983,7 +1984,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -2004,8 +2005,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 1, - "status": "pass", + "actual": 0, + "status": "blocked", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -2021,8 +2022,8 @@ } ], "source_check_count": 3, - "source_pass_count": 2, - "source_blocked_count": 1, + "source_pass_count": 1, + "source_blocked_count": 2, "submission_state": { "status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json", diff --git a/reports/skill-overview.html b/reports/skill-overview.html index 9615b69..4936f8c 100644 --- a/reports/skill-overview.html +++ b/reports/skill-overview.html @@ -919,7 +919,7 @@ -

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks7 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
+

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks6 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
  • 采用样本Adoption sample
diff --git a/reports/skill-overview.json b/reports/skill-overview.json index 3835d4f..e3164eb 100644 --- a/reports/skill-overview.json +++ b/reports/skill-overview.json @@ -411,7 +411,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 7, + "source_pass_count": 6, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -470,7 +470,8 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events" + "External events", + "Adoption sample" ] } ] @@ -994,11 +995,11 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": true, + "release_lock_ready": false, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "27da0809c59bcea3757a4f60fd43b515a8f51b11ab9ef805ba423b13a22f19e4", + "evidence_bundle_sha256": "0d1c762a722d1bbc83339a0365efcfcdf027849fc83c7079da92a524b905f5ab", "source_contract_sha256": "4660a11db94947ab603dca42eddc447698785d08f0df2972bf2ca43454683306", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "output_case_count": 5, @@ -1015,14 +1016,14 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 7, - "world_class_source_blocked_count": 6, + "world_class_source_pass_count": 6, + "world_class_source_blocked_count": 7, "public_claim_ready": false, - "public_claim_blocker_count": 4, - "working_tree_dirty": false, - "changed_file_count": 0 + "public_claim_blocker_count": 5, + "working_tree_dirty": true, + "changed_file_count": 38 }, - "commit": "c169e61dd493bc11f6b920323e25a31be888a67a", + "commit": "db1b24cbda5eb00f71b35846dff0f178625aa31d", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1314,14 +1315,14 @@ "ok": true, "summary": { "event_count": 1, - "adoption_sample_count": 1, - "activation_count": 1, - "accepted_count": 1, + "adoption_sample_count": 0, + "activation_count": 0, + "accepted_count": 0, "edited_count": 0, "rejected_count": 0, "missed_count": 0, "failed_count": 0, - "adoption_rate": 100.0, + "adoption_rate": 0, "missed_trigger_count": 0, "wrong_trigger_count": 0, "bad_output_count": 0, @@ -1330,7 +1331,7 @@ "review_overdue_count": 0, "risk_band": "low", "event_types": { - "skill_activation": 1 + "review_event": 1 }, "failure_types": {}, "source_types": { @@ -1552,7 +1553,7 @@ "status": "external_required", "category": "external", "owner": "Browser/Chrome/IDE/provider client integrator", - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1604,8 +1605,8 @@ "missing_submission_count": 4, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "submitted_but_pending_count": 0, "source_accepted_without_valid_submission_count": 0, "overclaim_guard_active": true, @@ -1941,7 +1942,7 @@ "status": "pending", "source_status": "external_required", "source_accepted": false, - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1978,7 +1979,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -1999,8 +2000,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 1, - "status": "pass", + "actual": 0, + "status": "blocked", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -2016,8 +2017,8 @@ } ], "source_check_count": 3, - "source_pass_count": 2, - "source_blocked_count": 1, + "source_pass_count": 1, + "source_blocked_count": 2, "submission_state": { "status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json", diff --git a/reports/world_class_claim_guard.json b/reports/world_class_claim_guard.json index 2774d79..ad0e3b9 100644 --- a/reports/world_class_claim_guard.json +++ b/reports/world_class_claim_guard.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "ledger_ready_to_claim_world_class": false, diff --git a/reports/world_class_claim_guard.md b/reports/world_class_claim_guard.md index 56c163e..f5142e2 100644 --- a/reports/world_class_claim_guard.md +++ b/reports/world_class_claim_guard.md @@ -1,6 +1,6 @@ # World-Class Claim Guard -Generated at: `2026-06-13` +Generated at: `2026-06-15` ## Summary diff --git a/reports/world_class_evidence_intake.json b/reports/world_class_evidence_intake.json index ed002e8..c773fa5 100644 --- a/reports/world_class_evidence_intake.json +++ b/reports/world_class_evidence_intake.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "schema_present": true, @@ -307,7 +307,7 @@ "source_accepted": false, "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false diff --git a/reports/world_class_evidence_intake.md b/reports/world_class_evidence_intake.md index 62c2d7e..bc5fd17 100644 --- a/reports/world_class_evidence_intake.md +++ b/reports/world_class_evidence_intake.md @@ -1,6 +1,6 @@ # World-Class Evidence Intake -Generated at: `2026-06-13` +Generated at: `2026-06-15` ## Summary diff --git a/reports/world_class_evidence_ledger.json b/reports/world_class_evidence_ledger.json index 6637eb6..e245338 100644 --- a/reports/world_class_evidence_ledger.json +++ b/reports/world_class_evidence_ledger.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "ledger_entry_count": 4, @@ -14,8 +14,8 @@ "missing_submission_count": 4, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "submitted_but_pending_count": 0, "source_accepted_without_valid_submission_count": 0, "overclaim_guard_active": true, @@ -351,7 +351,7 @@ "status": "pending", "source_status": "external_required", "source_accepted": false, - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -388,7 +388,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -409,8 +409,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 1, - "status": "pass", + "actual": 0, + "status": "blocked", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -426,8 +426,8 @@ } ], "source_check_count": 3, - "source_pass_count": 2, - "source_blocked_count": 1, + "source_pass_count": 1, + "source_blocked_count": 2, "submission_state": { "status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json", diff --git a/reports/world_class_evidence_ledger.md b/reports/world_class_evidence_ledger.md index 8ec440f..1bc98d0 100644 --- a/reports/world_class_evidence_ledger.md +++ b/reports/world_class_evidence_ledger.md @@ -1,6 +1,6 @@ # World-Class Evidence Ledger -Generated at: `2026-06-13` +Generated at: `2026-06-15` ## Summary @@ -8,8 +8,8 @@ Generated at: `2026-06-13` - ready to claim world-class: `false` - entries: `4` - source accepted: `0` -- source checks: `7` pass / `13` total -- source blocked: `6` +- source checks: `6` pass / `13` total +- source blocked: `7` - accepted: `0` - pending: `4` - human pending: `1` @@ -29,7 +29,7 @@ This ledger records the current evidence state. It requires both passing source | `provider-holdout` | `pending` | `missing` | `external` | model-executed 0; token-observed 0 | Run provider-backed holdout cases with real credentials and commit only aggregate evidence. | | `human-adjudication` | `pending` | `missing` | `human` | 0/5 decisions; pending 5 | Record real A/B choices in the decision template, then regenerate adjudication. | | `native-permission-enforcement` | `pending` | `missing` | `external` | native-enforced targets 0; installer-enforced targets 4 | Integrate a real target-client or external installer runtime guard before claiming native permission enforcement. | -| `native-client-telemetry` | `pending` | `missing` | `external` | external source events 0; adoption samples 1 | Install a real client against the native host and import production metadata-only events. | +| `native-client-telemetry` | `pending` | `missing` | `external` | external source events 0; adoption samples 0 | Install a real client against the native host and import production metadata-only events. | ## Provider Holdout @@ -167,8 +167,8 @@ This ledger records the current evidence state. It requires both passing source - objective: Import production metadata-only events from a real external client into the local drift loop. - source status: `external_required` -- observed state: `{"external_source_events": 0, "adoption_sample_count": 1, "raw_content_allowed": false, "risk_band": "low", "accepted": false}` -- source checks: `2` pass / `3` total +- observed state: `{"external_source_events": 0, "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false}` +- source checks: `1` pass / `3` total - submission state: `{"status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json", "artifact_ref_count": 0, "attested_real_evidence": false, "privacy_contract_satisfied": false, "ledger_counts_as_completion": false}` ### Provenance Requirements @@ -192,7 +192,7 @@ This ledger records the current evidence state. It requires both passing source | Check | Current | Expected | Status | | --- | --- | --- | --- | | External events | `0` | `>0` | `blocked` | -| Adoption sample | `1` | `>0` | `pass` | +| Adoption sample | `0` | `>0` | `blocked` | | Raw content blocked | `False` | `false` | `pass` | ### Completion Assertions diff --git a/reports/world_class_evidence_plan.json b/reports/world_class_evidence_plan.json index 236c40f..20b0d4e 100644 --- a/reports/world_class_evidence_plan.json +++ b/reports/world_class_evidence_plan.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "audit_decision": "continue-iteration", @@ -141,7 +141,7 @@ "status": "external_required", "category": "external", "owner": "Browser/Chrome/IDE/provider client integrator", - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -299,7 +299,7 @@ "status": "external_required", "category": "external", "owner": "Browser/Chrome/IDE/provider client integrator", - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", diff --git a/reports/world_class_evidence_plan.md b/reports/world_class_evidence_plan.md index c0db846..6af4cf3 100644 --- a/reports/world_class_evidence_plan.md +++ b/reports/world_class_evidence_plan.md @@ -1,6 +1,6 @@ # World-Class Evidence Plan -Generated at: `2026-06-13` +Generated at: `2026-06-15` ## Summary @@ -22,7 +22,7 @@ This report is an execution plan for the remaining world-class evidence gaps. It | `provider-holdout` | `external_required` | `external` | operator with provider credentials | model-executed 0; token-observed 0 | | `human-adjudication` | `human_required` | `human` | human reviewer | 0/5 decisions; pending 5 | | `native-permission-enforcement` | `external_required` | `external` | target client or installer integrator | native-enforced targets 0; installer-enforced targets 4 | -| `native-client-telemetry` | `external_required` | `external` | Browser/Chrome/IDE/provider client integrator | external source events 0; adoption samples 1 | +| `native-client-telemetry` | `external_required` | `external` | Browser/Chrome/IDE/provider client integrator | external source events 0; adoption samples 0 | ## Provider Holdout diff --git a/reports/world_class_operator_runbook.html b/reports/world_class_operator_runbook.html index e239453..be9f267 100644 --- a/reports/world_class_operator_runbook.html +++ b/reports/world_class_operator_runbook.html @@ -57,7 +57,7 @@ Evidence Operations

World-Class Operator Runbook

A single operating page for collecting the remaining human and external evidence. It coordinates action, but does not accept evidence or change the ledger.

-
Pending4
Awaiting4
Ready0
Source7/13
Blocked6
Invalid0
+
Pending4
Awaiting4
Ready0
Source6/13
Blocked7
Invalid0

Fast Path

  1. Run the real external or human work for one evidence item.
  2. Generate and fill the matching submission draft.
  3. Validate intake and inspect the submission review queue.
  4. Refresh the ledger and run the claim guard before making any completion claim.

Evidence Items

@@ -120,7 +120,7 @@
Owner
Browser/Chrome/IDE/provider client integrator
Ledger
pending
-
Blocked
1
+
Blocked
2
Submission
evidence/world_class/submissions/native-client-telemetry.json

Source Runbook

  • python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension://<extension-id>/
  • Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event.
  • python3 scripts/yao.py telemetry-import . --input-jsonl .yao/telemetry_spool/external_events.jsonl
  • python3 scripts/yao.py skill-atlas --workspace-root .
  • python3 scripts/yao.py skill-os2-audit . --generated-at <YYYY-MM-DD>
  • Copy evidence/world_class/templates/native-client-telemetry.intake.json to evidence/world_class/submissions/native-client-telemetry.json and fill only real evidence fields.
  • python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions
@@ -130,8 +130,8 @@

Success Checks

  • reports/adoption_drift_report.json summary.source_types.external > 0
  • reports/adoption_drift_report.json summary.adoption_sample_count > 0
  • reports/skill_os2_audit.json item native-client-telemetry status becomes pass

Privacy

  • Telemetry must remain metadata-only and local-first.
  • Do not package reports/telemetry_events.jsonl or any raw prompt, output, transcript, note, or message field.
-

Next Source Actions

  • Import at least one metadata-only event from a real client.
-

Source Evidence Snapshot

  • External eventsexternal_source_events: 0 / >0Import at least one metadata-only event from a real client.
  • Adoption sampleadoption_sample_count: 1 / >0Telemetry must include adoption outcome evidence.
  • Raw content blockedraw_content_allowed: False / falseTelemetry must stay metadata-only.
+

Next Source Actions

  • Import at least one metadata-only event from a real client.
  • Telemetry must include adoption outcome evidence.
+

Source Evidence Snapshot

  • External eventsexternal_source_events: 0 / >0Import at least one metadata-only event from a real client.
  • Adoption sampleadoption_sample_count: 0 / >0Telemetry must include adoption outcome evidence.
  • Raw content blockedraw_content_allowed: False / falseTelemetry must stay metadata-only.

Boundary

  • Planned work, draft packets, metadata fallback, pending human decisions, and local command runners do not count as completion.
  • Valid intake means ready for submission review; ledger review still requires passing source evidence.
  • The world-class ledger and claim guard remain the source of truth.
diff --git a/reports/world_class_operator_runbook.json b/reports/world_class_operator_runbook.json index eb0e185..dfb9049 100644 --- a/reports/world_class_operator_runbook.json +++ b/reports/world_class_operator_runbook.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "evidence_item_count": 4, @@ -12,8 +12,8 @@ "valid_packet_source_incomplete_count": 0, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "ready_to_claim_world_class": false, "runbook_counts_as_completion": false, "decision": "collect-evidence" @@ -394,7 +394,7 @@ "review_state": "awaiting-submission", "source_accepted": false, "objective": "Import production metadata-only events from a real external client into the local drift loop.", - "current": "external source events 0; adoption samples 1", + "current": "external source events 0; adoption samples 0", "execution_runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", "Install the generated native messaging manifest for the real client and send at least one accepted skill_activation or skill_output event.", @@ -442,7 +442,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -463,8 +463,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 1, - "status": "pass", + "actual": 0, + "status": "blocked", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -479,9 +479,10 @@ "next_action": "Telemetry must stay metadata-only." } ], - "blocked_source_check_count": 1, + "blocked_source_check_count": 2, "next_source_actions": [ - "Import at least one metadata-only event from a real client." + "Import at least one metadata-only event from a real client.", + "Telemetry must include adoption outcome evidence." ], "submission_state": { "status": "missing", diff --git a/reports/world_class_operator_runbook.md b/reports/world_class_operator_runbook.md index bef6278..3abb08c 100644 --- a/reports/world_class_operator_runbook.md +++ b/reports/world_class_operator_runbook.md @@ -1,6 +1,6 @@ # World-Class Operator Runbook -Generated at: `2026-06-13` +Generated at: `2026-06-15` ## Summary @@ -28,7 +28,7 @@ This runbook coordinates evidence collection only. It does not accept submission | `provider-holdout` | `pending` | `awaiting-submission` | `awaiting-submission` | `2` | Run provider-backed output-exec with real credentials. | operator with provider credentials | | `human-adjudication` | `pending` | `awaiting-submission` | `awaiting-submission` | `2` | Record a reviewer choice for every pair. | human reviewer | | `native-permission-enforcement` | `pending` | `awaiting-submission` | `awaiting-submission` | `1` | Collect real target-client or external runtime guard proof. | target client or installer integrator | -| `native-client-telemetry` | `pending` | `awaiting-submission` | `awaiting-submission` | `1` | Import at least one metadata-only event from a real client. | Browser/Chrome/IDE/provider client integrator | +| `native-client-telemetry` | `pending` | `awaiting-submission` | `awaiting-submission` | `2` | Import at least one metadata-only event from a real client. | Browser/Chrome/IDE/provider client integrator | ## Provider Holdout @@ -239,7 +239,7 @@ This runbook coordinates evidence collection only. It does not accept submission - objective: Import production metadata-only events from a real external client into the local drift loop. - blocking reason: No evidence packet has been submitted for review. -- blocked source checks: `1` +- blocked source checks: `2` - submission: `evidence/world_class/submissions/native-client-telemetry.json` - template: `evidence/world_class/templates/native-client-telemetry.intake.json` @@ -292,13 +292,14 @@ This runbook coordinates evidence collection only. It does not accept submission ### Next Source Actions - Import at least one metadata-only event from a real client. +- Telemetry must include adoption outcome evidence. ### Source Evidence Snapshot | Check | Current | Expected | Status | Next action | | --- | --- | --- | --- | --- | | External events | `0` | `>0` | `blocked` | Import at least one metadata-only event from a real client. | -| Adoption sample | `1` | `>0` | `pass` | Telemetry must include adoption outcome evidence. | +| Adoption sample | `0` | `>0` | `blocked` | Telemetry must include adoption outcome evidence. | | Raw content blocked | `False` | `false` | `pass` | Telemetry must stay metadata-only. | ## Boundary diff --git a/reports/world_class_submission_review.json b/reports/world_class_submission_review.json index cb10f86..5b79d65 100644 --- a/reports/world_class_submission_review.json +++ b/reports/world_class_submission_review.json @@ -1,7 +1,7 @@ { "schema_version": "1.0", "ok": true, - "generated_at": "2026-06-13", + "generated_at": "2026-06-15", "skill_dir": ".", "summary": { "review_item_count": 4, @@ -13,8 +13,8 @@ "unmatched_submission_count": 0, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 7, - "source_blocked_count": 6, + "source_pass_count": 6, + "source_blocked_count": 7, "ready_to_claim_world_class": false, "review_counts_submission_as_completion": false, "decision": "awaiting-submissions" @@ -265,7 +265,7 @@ "intake_errors": [], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 1, + "adoption_sample_count": 0, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -286,8 +286,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 1, - "status": "pass", + "actual": 0, + "status": "blocked", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -303,8 +303,8 @@ } ], "source_check_count": 3, - "source_pass_count": 2, - "source_blocked_count": 1, + "source_pass_count": 1, + "source_blocked_count": 2, "success_checks": [ "reports/adoption_drift_report.json summary.source_types.external > 0", "reports/adoption_drift_report.json summary.adoption_sample_count > 0", diff --git a/reports/world_class_submission_review.md b/reports/world_class_submission_review.md index 7fe1e47..a896c21 100644 --- a/reports/world_class_submission_review.md +++ b/reports/world_class_submission_review.md @@ -1,6 +1,6 @@ # World-Class Submission Review -Generated at: `2026-06-13` +Generated at: `2026-06-15` ## Summary @@ -138,7 +138,7 @@ This report is a read-only reviewer queue. It does not accept evidence or make w #### Source Checks - External events: 0 / >0 => blocked -- Adoption sample: 1 / >0 => pass +- Adoption sample: 0 / >0 => blocked - Raw content blocked: False / false => pass #### Completion Assertions diff --git a/scripts/cross_packager.py b/scripts/cross_packager.py index 74b016f..d89aa7f 100644 --- a/scripts/cross_packager.py +++ b/scripts/cross_packager.py @@ -3,7 +3,7 @@ import argparse import json import shutil import zipfile -from pathlib import Path +from pathlib import Path, PurePosixPath import yaml from compile_skill import compile_target_contract @@ -64,6 +64,14 @@ def find_skill_ir(skill_dir: Path, name: str) -> tuple[dict, str]: return {}, "frontmatter-fallback" +def package_name_from_manifest(manifest: dict, skill_dir: Path) -> str: + name = str(manifest.get("name") or "").strip() + if name: + return name + frontmatter = read_frontmatter(skill_dir / "SKILL.md") + return str(frontmatter.get("name") or skill_dir.name) + + def require_fields(payload: dict, fields: list[str], label: str) -> None: missing = [field for field in fields if not payload.get(field)] if missing: @@ -213,7 +221,7 @@ def build_manifest(skill_dir: Path, platform: str) -> dict: "description": semantic["description"], "version": manifest.get("version") or frontmatter.get("version", "1.0.0"), "platform": platform, - "skill_root": skill_dir.name, + "skill_root": semantic["name"], "job_to_be_done": semantic["job_to_be_done"], "ir_source": semantic["ir_source"], "ir_schema_version": semantic["ir_schema_version"], @@ -496,7 +504,7 @@ def write_adapter(skill_dir: Path, out_dir: Path, platform: str) -> Path: notes = target_dir / "README.md" native = payload["target_native_contract"] notes.write_text( - f"# Claude-Compatible Package\n\nUse `{skill_dir.name}` with its neutral source files. This target does not require vendor metadata by default.\n\n" + f"# Claude-Compatible Package\n\nUse `{payload['name']}` with its neutral source files. This target does not require vendor metadata by default.\n\n" f"Native surface: {native['native_surface']}.\n\n" f"Activation: {native['activation']['policy']}\n\n" f"Resources: {native['resources']['strategy']}\n\n" @@ -510,7 +518,7 @@ def write_adapter(skill_dir: Path, out_dir: Path, platform: str) -> Path: native = payload["target_native_contract"] notes.write_text( f"# VS Code / Copilot Agent Skills Package\n\n" - f"Install `{skill_dir.name}` as a VS Code user or project scoped Agent Skill. Keep the folder name aligned with `SKILL.md` frontmatter name.\n\n" + f"Install `{payload['name']}` as a VS Code user or project scoped Agent Skill. Keep the folder name aligned with `SKILL.md` frontmatter name.\n\n" f"Native surface: {native['native_surface']}.\n\n" f"Activation: {native['activation']['policy']}\n\n" f"Resources: {native['resources']['strategy']}\n\n" @@ -524,15 +532,15 @@ def write_adapter(skill_dir: Path, out_dir: Path, platform: str) -> Path: "Install the package as a VS Code user or project scoped Agent Skill; use targets/vscode/README.md for scope and trust notes." ) else: - payload["install_hint"] = f"Use {skill_dir.name} as an Agent Skills compatible package." + payload["install_hint"] = f"Use {payload['name']} as an Agent Skills compatible package." path = target_dir / "adapter.json" payload["contract"] = PLATFORM_CONTRACTS.get(platform, PLATFORM_CONTRACTS["generic"]) path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") return path -def make_zip(skill_dir: Path, out_dir: Path) -> Path: - zip_path = out_dir / f"{skill_dir.name}.zip" +def make_zip(skill_dir: Path, out_dir: Path, package_name: str) -> Path: + zip_path = out_dir / f"{package_name}.zip" skill_root = skill_dir.resolve() out_root = out_dir.resolve() with zipfile.ZipFile(zip_path, "w", compression=zipfile.ZIP_DEFLATED) as zf: @@ -547,17 +555,18 @@ def make_zip(skill_dir: Path, out_dir: Path) -> Path: rel_path = path.relative_to(skill_dir) if should_skip_archive_path(rel_path): continue - zf.write(path, arcname=str(path.relative_to(skill_dir.parent))) + zf.write(path, arcname=str(PurePosixPath(package_name, *rel_path.parts))) return zip_path -def copy_manifest(skill_dir: Path, out_dir: Path) -> Path: +def copy_manifest(skill_dir: Path, out_dir: Path) -> tuple[Path, str]: manifest_path = out_dir / "manifest.json" + manifest = build_manifest(skill_dir, "generic") manifest_path.write_text( - json.dumps(build_manifest(skill_dir, "generic"), ensure_ascii=False, indent=2), + json.dumps(manifest, ensure_ascii=False, indent=2), encoding="utf-8", ) - return manifest_path + return manifest_path, package_name_from_manifest(manifest, skill_dir) def load_expectations(path: Path | None) -> dict: @@ -608,12 +617,12 @@ def main() -> None: if out_dir.exists(): shutil.rmtree(out_dir) out_dir.mkdir(parents=True) - manifest = copy_manifest(skill_dir, out_dir) + manifest, package_name = copy_manifest(skill_dir, out_dir) generated.append(str(manifest)) for platform in (args.platform or ["generic"]): generated.append(str(write_adapter(skill_dir, out_dir, platform))) if args.zip: - generated.append(str(make_zip(skill_dir, out_dir))) + generated.append(str(make_zip(skill_dir, out_dir, package_name))) except (FileNotFoundError, ValueError, yaml.YAMLError) as exc: failures.append(str(exc)) diff --git a/scripts/simulate_install.py b/scripts/simulate_install.py index 69a585c..e6f6c10 100644 --- a/scripts/simulate_install.py +++ b/scripts/simulate_install.py @@ -91,6 +91,10 @@ def add_check(checks: list[dict[str, str]], failures: list[str], check_id: str, failures.append(detail) +def package_name(package_manifest: dict[str, Any], skill_dir: Path) -> str: + return str(package_manifest.get("name") or skill_dir.name) + + def adapter_targets(adapter_root: Path) -> list[str]: targets_dir = adapter_root / "targets" if not targets_dir.exists(): @@ -188,8 +192,9 @@ def simulate_install(skill_dir: Path, package_dir: Path, install_root: Path | No checks: list[dict[str, str]] = [] failures: list[str] = [] warnings: list[str] = [] - archive_path = package_dir / f"{skill_dir.name}.zip" package_manifest = load_json(package_dir / "manifest.json") + package_root = package_name(package_manifest, skill_dir) + archive_path = package_dir / f"{package_root}.zip" installed_dir: Path | None = None archive_entries: list[str] = [] @@ -201,7 +206,7 @@ def simulate_install(skill_dir: Path, package_dir: Path, install_root: Path | No else: requested_root = install_root.resolve() requested_root.mkdir(parents=True, exist_ok=True) - install_base = requested_root / f"simulate-{skill_dir.name}" + install_base = requested_root / f"simulate-{package_root}" if install_base.exists(): shutil.rmtree(install_base) install_base.mkdir(parents=True, exist_ok=True) @@ -218,7 +223,7 @@ def simulate_install(skill_dir: Path, package_dir: Path, install_root: Path | No unsafe_entries = unsafe_zip_entries(archive_entries) add_check(checks, failures, "archive-safe-paths", not unsafe_entries, "Archive has no absolute or parent-traversal entries") roots = top_level_dirs(archive_entries) - add_check(checks, failures, "single-top-level", roots == [skill_dir.name], f"Archive top-level directory is {skill_dir.name}") + add_check(checks, failures, "single-top-level", roots == [package_root], f"Archive top-level directory is {package_root}") if not unsafe_entries and roots: with zipfile.ZipFile(archive_path) as archive: archive.extractall(install_base) @@ -228,7 +233,7 @@ def simulate_install(skill_dir: Path, package_dir: Path, install_root: Path | No source_manifest = load_json(installed_dir / "manifest.json") if installed_dir else {} interface_doc = load_yaml(installed_dir / "agents" / "interface.yaml") if installed_dir else {} add_check(checks, failures, "entrypoint-load", bool(frontmatter), "Installed SKILL.md frontmatter is readable") - add_check(checks, failures, "entrypoint-name", frontmatter.get("name") == skill_dir.name, "Installed SKILL.md name matches package directory") + add_check(checks, failures, "entrypoint-name", frontmatter.get("name") == package_root, "Installed SKILL.md name matches package directory") add_check(checks, failures, "entrypoint-description", bool(frontmatter.get("description")), "Installed SKILL.md description is present") add_check(checks, failures, "manifest-load", bool(source_manifest), "Installed manifest.json is readable") add_check(checks, failures, "manifest-name", source_manifest.get("name") == package_manifest.get("name"), "Installed manifest name matches package manifest") diff --git a/scripts/verify_package.py b/scripts/verify_package.py index cf14d7a..c84f078 100644 --- a/scripts/verify_package.py +++ b/scripts/verify_package.py @@ -75,6 +75,10 @@ def add_check(checks: list[dict[str, str]], failures: list[str], check_id: str, failures.append(detail) +def package_name(manifest: dict[str, Any], skill_dir: Path) -> str: + return str(manifest.get("name") or skill_dir.name) + + def verify_package( skill_dir: Path, package_dir: Path, @@ -91,6 +95,7 @@ def verify_package( manifest_path = package_dir / "manifest.json" manifest = load_json(manifest_path) + package_root = package_name(manifest, skill_dir) add_check(checks, failures, "package-manifest", bool(manifest), f"Package manifest exists: {display_path(manifest_path)}") targets = required_targets(expectations, package_dir) @@ -117,7 +122,7 @@ def verify_package( for rel in required_files: add_check(checks, failures, f"{target}-file-{rel}", (package_dir / rel).exists(), f"Package contains {rel}") - archive_path = package_dir / f"{skill_dir.name}.zip" + archive_path = package_dir / f"{package_root}.zip" archive_sha = "" archive_entries: list[str] = [] if archive_path.exists(): @@ -129,9 +134,9 @@ def verify_package( else: unsafe_entries = unsafe_zip_entries(archive_entries) required_entries = [ - f"{skill_dir.name}/SKILL.md", - f"{skill_dir.name}/manifest.json", - f"{skill_dir.name}/agents/interface.yaml", + f"{package_root}/SKILL.md", + f"{package_root}/manifest.json", + f"{package_root}/agents/interface.yaml", ] add_check(checks, failures, "archive-safe-paths", not unsafe_entries, "Archive has no absolute or parent-traversal entries") for entry in required_entries: diff --git a/tests/verify_install_simulation.py b/tests/verify_install_simulation.py index 282186b..f141c99 100644 --- a/tests/verify_install_simulation.py +++ b/tests/verify_install_simulation.py @@ -3,6 +3,7 @@ import json import shutil import subprocess import sys +import tempfile import zipfile from pathlib import Path @@ -28,12 +29,12 @@ def run(cmd: list[str]) -> dict: } -def build_package(out_dir: Path) -> dict: +def build_package(out_dir: Path, skill_root: Path = ROOT) -> dict: return run( [ sys.executable, str(PACKAGER), - str(ROOT), + str(skill_root), "--platform", "openai", "--platform", @@ -51,12 +52,12 @@ def build_package(out_dir: Path) -> dict: ) -def simulate(package_dir: Path, output_json: Path, output_md: Path) -> dict: +def simulate(package_dir: Path, output_json: Path, output_md: Path, skill_root: Path = ROOT) -> dict: return run( [ sys.executable, str(SIMULATOR), - str(ROOT), + str(skill_root), "--package-dir", str(package_dir), "--install-root", @@ -114,6 +115,22 @@ def main() -> None: assert "Install Simulation" in valid_markdown assert "Installer permissions enforced" in valid_markdown + with tempfile.TemporaryDirectory(prefix="renamed-install-root-") as temp_root: + renamed_root = Path(temp_root) / "checkout-alias" + shutil.copytree( + ROOT, + renamed_root, + ignore=shutil.ignore_patterns(".git", ".previews", "dist", "__pycache__", ".pytest_cache", "tmp*"), + ) + renamed_dir = TMP / "renamed-dist" + renamed_build = build_package(renamed_dir, renamed_root) + assert renamed_build["ok"], renamed_build + assert (renamed_dir / "yao-meta-skill.zip").exists(), renamed_build + renamed_valid = simulate(renamed_dir, TMP / "renamed_install_simulation.json", TMP / "renamed_install_simulation.md", renamed_root) + assert renamed_valid["ok"], renamed_valid + assert renamed_valid["payload"]["summary"]["archive_extracted"], renamed_valid + assert renamed_valid["payload"]["installed_skill_dir"].endswith("simulate-yao-meta-skill/yao-meta-skill"), renamed_valid + policy_gap_dir = TMP / "policy-gap-dist" shutil.copytree(valid_dir, policy_gap_dir) diff --git a/tests/verify_package_verification.py b/tests/verify_package_verification.py index 687b668..b0f8138 100644 --- a/tests/verify_package_verification.py +++ b/tests/verify_package_verification.py @@ -3,6 +3,7 @@ import json import shutil import subprocess import sys +import tempfile import zipfile from pathlib import Path @@ -28,12 +29,12 @@ def run(cmd: list[str]) -> dict: } -def build_package(out_dir: Path) -> dict: +def build_package(out_dir: Path, skill_root: Path = ROOT) -> dict: return run( [ sys.executable, str(PACKAGER), - str(ROOT), + str(skill_root), "--platform", "openai", "--platform", @@ -51,12 +52,12 @@ def build_package(out_dir: Path) -> dict: ) -def verify_package(out_dir: Path, output_json: Path, output_md: Path) -> dict: +def verify_package(out_dir: Path, output_json: Path, output_md: Path, skill_root: Path = ROOT) -> dict: return run( [ sys.executable, str(VERIFIER), - str(ROOT), + str(skill_root), "--package-dir", str(out_dir), "--expectations", @@ -93,6 +94,23 @@ def main() -> None: assert not payload["failures"], payload assert (TMP / "package_verification.md").exists(), TMP + with tempfile.TemporaryDirectory(prefix="renamed-package-root-") as temp_root: + renamed_root = Path(temp_root) / "checkout-alias" + shutil.copytree( + ROOT, + renamed_root, + ignore=shutil.ignore_patterns(".git", ".previews", "dist", "__pycache__", ".pytest_cache", "tmp*"), + ) + renamed_dir = TMP / "renamed-dist" + renamed_build = build_package(renamed_dir, renamed_root) + assert renamed_build["ok"], renamed_build + assert (renamed_dir / "yao-meta-skill.zip").exists(), renamed_build + with zipfile.ZipFile(renamed_dir / "yao-meta-skill.zip") as archive: + names = set(archive.namelist()) + assert "yao-meta-skill/SKILL.md" in names, sorted(list(names))[:10] + renamed_valid = verify_package(renamed_dir, TMP / "renamed_package_verification.json", TMP / "renamed_package_verification.md", renamed_root) + assert renamed_valid["ok"], renamed_valid + unsafe_dir = TMP / "unsafe-dist" shutil.copytree(valid_dir, unsafe_dir) with zipfile.ZipFile(unsafe_dir / "yao-meta-skill.zip", "a", compression=zipfile.ZIP_DEFLATED) as archive: