diff --git a/reports/benchmark_reproducibility.json b/reports/benchmark_reproducibility.json index a224fb3..84bd8ea 100644 --- a/reports/benchmark_reproducibility.json +++ b/reports/benchmark_reproducibility.json @@ -3,35 +3,22 @@ "ok": true, "generated_at": "2026-06-15", "skill_dir": ".", - "commit": "5495734e7e33b19b84a932626112d783a3edf271", + "commit": "038891715d7843daf7960c2ec61bf7d9ea91c9e4", "git_status": { "available": true, - "dirty": true, - "changed_file_count": 73, - "sample": [ - " M Makefile", - " M README.md", - " M registry/index.json", - " M registry/packages/yao-meta-skill.json", - " M reports/adoption_drift_report.json", - " M reports/adoption_drift_report.md", - " M reports/architecture_maintainability.json", - " M reports/architecture_maintainability.md", - " M reports/benchmark_reproducibility.json", - " M reports/benchmark_reproducibility.md", - " M reports/compiled_targets.json", - " M reports/context_budget.json" - ], + "dirty": false, + "changed_file_count": 0, + "sample": [], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "60df1cebf40b1627f6d313c233f377d0d36228a9525b70099623563c7efb7804", - "source_contract_sha256": "c0fe8976bca1ceac0b325c882003335c66487b8f152868af471f3bdd79b9f0af", + "evidence_bundle_sha256": "b8858dc09a54ae5d02549e2a4ad6fb9f6c0237ebb6be7a54c61b147aaff7f592", + "source_contract_sha256": "652c3a4a1f091d875bfacdd44da9a91bc02fec90af3fe76cf3a71a6dc80aba58", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "output_case_count": 5, "failure_disclosure_count": 3, @@ -47,30 +34,29 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 73 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ - "release lock is not clean or commit is unavailable", "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)", - "world-class source checks are not all accepted (6/13 pass, 7 blocked)" + "world-class source checks are not all accepted (7/13 pass, 6 blocked)" ], "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks." }, "release_lock": { - "ready": false, - "commit": "5495734e7e33b19b84a932626112d783a3edf271", + "ready": true, + "commit": "038891715d7843daf7960c2ec61bf7d9ea91c9e4", "status_scope": "generation-time status before this report is written", - "reason": "working tree was dirty at generation time" + "reason": "clean generation-time HEAD" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", @@ -78,7 +64,7 @@ "existing_count": 24, "missing_count": 0, "missing_paths": [], - "sha256": "60df1cebf40b1627f6d313c233f377d0d36228a9525b70099623563c7efb7804" + "sha256": "b8858dc09a54ae5d02549e2a4ad6fb9f6c0237ebb6be7a54c61b147aaff7f592" }, "methodology": { "path": "reports/benchmark_methodology.md", @@ -151,8 +137,8 @@ "label": "output_execution", "path": "reports/output_execution_runs.json", "exists": true, - "bytes": 7966, - "sha256": "406fa9dd65ac5a870ff4678e05b9e145fff9eb8a37bdc5f7a2e556e1dcc0c369" + "bytes": 7967, + "sha256": "60a9a34e5e2ea2c239edb5037ce126a5e1b0fbfc3ae6fbb7d686fa411a225068" }, { "label": "blind_review", @@ -187,21 +173,21 @@ "path": "reports/security_trust_report.json", "exists": true, "bytes": 105694, - "sha256": "cfabb1008f1ce3cd638c1f3251e860557102cd01963e58aa8cce2c4c92dfee33" + "sha256": "05b721899b8196240ba3dbdb442654d03cadecce6b162dac315d7fb238783c24" }, { "label": "python_compatibility", "path": "reports/python_compatibility.json", "exists": true, "bytes": 22252, - "sha256": "fcf288346cb5159e070b2df75a8e371cca2501945c971aead521e7ac3557a4bc" + "sha256": "2b8c4b338b5c70e1643ab20f8ef42329d69dd9876f964fdaaa454f90520adddc" }, { "label": "registry_audit", "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, - "sha256": "5765778984e60fd98c2459e5402bc7780391753c7353253701b279dd2b310eef" + "sha256": "efc0d48985c1fbb06205a32b3ef40cb5e69795b39c1385953bdc16f0a58f726b" }, { "label": "package_verification", @@ -229,56 +215,56 @@ "path": "reports/world_class_evidence_plan.json", "exists": true, "bytes": 19940, - "sha256": "933cdb0021818c0a8c5fc199af7dfa7879777ebf7959ed3a36bba52a0aed9c6c" + "sha256": "b408af112c784fdc3e96ea9ecf374990b9f58a0ebfa4df18394ccd9e93dc5545" }, { "label": "world_class_evidence_ledger", "path": "reports/world_class_evidence_ledger.json", "exists": true, - "bytes": 20006, - "sha256": "5407409841eb8e1174a07ea0af17ddf709155f0c548c897b82c2f75b869b333a" + "bytes": 20003, + "sha256": "cc886713ba5b99bdb2f740217f0979a30312aef7539d2a22af0e96ec5c3e90bf" }, { "label": "world_class_evidence_intake", "path": "reports/world_class_evidence_intake.json", "exists": true, "bytes": 18865, - "sha256": "b10e1ce0a5a17df72462ce8f0c468356f58ff88bd7ceab349c82a1fb3b6697ef" + "sha256": "2fbecc603d8dc6382fde6c61869aaf4b56fc9d1be5afcd07b345c419dce6d55a" }, { "label": "world_class_submission_review", "path": "reports/world_class_submission_review.json", "exists": true, - "bytes": 12413, - "sha256": "3bce5f072d037a6c383cfd597f8530c2bc87ba31781fa90c843a9186dd2f8fea" + "bytes": 12410, + "sha256": "bc690472684dcdb8c77803b730857f98e6b6eb11f1e741d5fcc1aa9303b77e47" }, { "label": "world_class_operator_runbook", "path": "reports/world_class_operator_runbook.json", "exists": true, - "bytes": 24021, - "sha256": "d377b8d99831ce297011e4e270fc7287a53727500f5d503dd3dc4b984d26f0ad" + "bytes": 23957, + "sha256": "b4640688026ef9199e96f80bd2d9971aaca3ec6d092d8ecd052a5debc0aa94c7" }, { "label": "world_class_operator_runbook_markdown", "path": "reports/world_class_operator_runbook.md", "exists": true, - "bytes": 15080, - "sha256": "9f141f09bf485a4299b37eb8c21ebab9f212ea130a0daea8b0d7d5bd7ab020aa" + "bytes": 15025, + "sha256": "3a916ac568377d8df7d612426f7358283eddfa2ab5fedb7ed12db0097a8b3c48" }, { "label": "world_class_operator_runbook_html", "path": "reports/world_class_operator_runbook.html", "exists": true, - "bytes": 21030, - "sha256": "04cc091b113f018072d07a1d71ad142773c415a49e633493689903e7b399b5e6" + "bytes": 20969, + "sha256": "886b915da0f27fb46f4d8cebb6211ab5066a9fd44e1ec2e3206487d9a8abfff2" }, { "label": "world_class_claim_guard", "path": "reports/world_class_claim_guard.json", "exists": true, "bytes": 8725, - "sha256": "7257fedb0a0407332265ef9ae7b1a6d789f4d2a075c6b62fc1a6f679b3a5a42d" + "sha256": "ba4a3f36c2a9468bbb286b2c69910ffbba1e9a3f622ddf13e39fcb29116952f9" } ], "missing_artifacts": [], diff --git a/reports/benchmark_reproducibility.md b/reports/benchmark_reproducibility.md index d18c4fb..5d6484a 100644 --- a/reports/benchmark_reproducibility.md +++ b/reports/benchmark_reproducibility.md @@ -1,18 +1,18 @@ # Benchmark Reproducibility Generated at: `2026-06-15` -Commit: `5495734e7e33b19b84a932626112d783a3edf271` -Working tree dirty at generation: `true` -Evidence bundle SHA256: `60df1cebf40b1627f6d313c233f377d0d36228a9525b70099623563c7efb7804` +Commit: `038891715d7843daf7960c2ec61bf7d9ea91c9e4` +Working tree dirty at generation: `false` +Evidence bundle SHA256: `b8858dc09a54ae5d02549e2a4ad6fb9f6c0237ebb6be7a54c61b147aaff7f592` ## Summary - reproducibility ready: `true` -- release lock ready: `false` +- release lock ready: `true` - methodology complete: `true` - required artifacts: `24` - missing artifacts: `0` -- source contract sha256: `c0fe8976bca1` +- source contract sha256: `652c3a4a1f09` - archive sha256: `6852cf91a74d` - output cases: `5` - disclosed failure cases: `3` @@ -20,10 +20,10 @@ Evidence bundle SHA256: `60df1cebf40b1627f6d313c233f377d0d36228a9525b70099623563 - provider evidence complete: `false` - human review complete: `false` - world-class ready: `false` -- world-class source checks: `6` pass / `13` total; `7` blocked +- world-class source checks: `7` pass / `13` total; `6` blocked - public claim ready: `false` -- public claim blockers: `5` -- changed files at generation: `73` +- public claim blockers: `4` +- changed files at generation: `0` This report proves local benchmark reproducibility only. It keeps external provider and human-review gaps visible instead of counting them as complete. The git commit is generation-time context; the evidence bundle SHA is the durable anchor for the artifacts listed below. @@ -35,23 +35,22 @@ This report proves local benchmark reproducibility only. It keeps external provi | Blocker | | --- | -| release lock is not clean or commit is unavailable | | provider-backed model holdout evidence is incomplete | | human blind-review adjudication is incomplete | | world-class evidence is not accepted yet (4 open gaps, 4 ledger pending) | -| world-class source checks are not all accepted (6/13 pass, 7 blocked) | +| world-class source checks are not all accepted (7/13 pass, 6 blocked) | ## Release Lock -- ready: `false` -- reason: working tree was dirty at generation time +- ready: `true` +- reason: clean generation-time HEAD - status scope: generation-time status before this report is written ## Evidence Bundle - algorithm: `sha256(path,label,exists,artifact_sha256)` - artifacts: `24` / `24` -- sha256: `60df1cebf40b1627f6d313c233f377d0d36228a9525b70099623563c7efb7804` +- sha256: `b8858dc09a54ae5d02549e2a4ad6fb9f6c0237ebb6be7a54c61b147aaff7f592` ## Methodology Sections @@ -73,25 +72,25 @@ This report proves local benchmark reproducibility only. It keeps external provi | output_cases | `evals/output/cases.jsonl` | present | `a6ae96857116` | | output_schema | `evals/output/schema.json` | present | `8ee340c95064` | | output_scorecard | `reports/output_quality_scorecard.json` | present | `0806258a8e08` | -| output_execution | `reports/output_execution_runs.json` | present | `406fa9dd65ac` | +| output_execution | `reports/output_execution_runs.json` | present | `60a9a34e5e2e` | | blind_review | `reports/output_blind_review_pack.json` | present | `bbe2db8ec277` | | review_adjudication | `reports/output_review_adjudication.json` | present | `240485a721af` | | trigger_scorecard | `reports/route_scorecard.json` | present | `c164e83e36d0` | | runtime_conformance | `reports/conformance_matrix.json` | present | `8251329e663d` | -| trust_report | `reports/security_trust_report.json` | present | `cfabb1008f1c` | -| python_compatibility | `reports/python_compatibility.json` | present | `fcf288346cb5` | -| registry_audit | `reports/registry_audit.json` | present | `5765778984e6` | +| trust_report | `reports/security_trust_report.json` | present | `05b721899b81` | +| python_compatibility | `reports/python_compatibility.json` | present | `2b8c4b338b5c` | +| registry_audit | `reports/registry_audit.json` | present | `efc0d48985c1` | | package_verification | `reports/package_verification.json` | present | `2476ae8ec9c4` | | install_simulation | `reports/install_simulation.json` | present | `46d0524a6968` | | skill_os2_audit | `reports/skill_os2_audit.json` | present | `8481f4d78f2a` | -| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `933cdb002181` | -| world_class_evidence_ledger | `reports/world_class_evidence_ledger.json` | present | `5407409841eb` | -| world_class_evidence_intake | `reports/world_class_evidence_intake.json` | present | `b10e1ce0a5a1` | -| world_class_submission_review | `reports/world_class_submission_review.json` | present | `3bce5f072d03` | -| world_class_operator_runbook | `reports/world_class_operator_runbook.json` | present | `d377b8d99831` | -| world_class_operator_runbook_markdown | `reports/world_class_operator_runbook.md` | present | `9f141f09bf48` | -| world_class_operator_runbook_html | `reports/world_class_operator_runbook.html` | present | `04cc091b113f` | -| world_class_claim_guard | `reports/world_class_claim_guard.json` | present | `7257fedb0a04` | +| world_class_evidence_plan | `reports/world_class_evidence_plan.json` | present | `b408af112c78` | +| world_class_evidence_ledger | `reports/world_class_evidence_ledger.json` | present | `cc886713ba5b` | +| world_class_evidence_intake | `reports/world_class_evidence_intake.json` | present | `2fbecc603d8d` | +| world_class_submission_review | `reports/world_class_submission_review.json` | present | `bc690472684d` | +| world_class_operator_runbook | `reports/world_class_operator_runbook.json` | present | `b4640688026e` | +| world_class_operator_runbook_markdown | `reports/world_class_operator_runbook.md` | present | `3a916ac56837` | +| world_class_operator_runbook_html | `reports/world_class_operator_runbook.html` | present | `886b915da0f2` | +| world_class_claim_guard | `reports/world_class_claim_guard.json` | present | `ba4a3f36c2a9` | ## Reproduction Commands diff --git a/reports/skill-interpretation.html b/reports/skill-interpretation.html index 3e58445..8585c71 100644 --- a/reports/skill-interpretation.html +++ b/reports/skill-interpretation.html @@ -919,7 +919,7 @@ -

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks6 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
  • 采用样本Adoption sample
+

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks7 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
@@ -930,7 +930,7 @@

让 reviewer 快速确认关键文件、目录和资产分布。Lets reviewers confirm key files, directories, and asset distribution quickly.

-
资产分布Asset Distribution386项386 itemsSKILL.mdSKILL.mdREADME.mdREADME.mdagents/interface.yamlagents/interface.yamlmanifest.jsonmanifest.jsonreferencesreferencesscriptsscripts
资产分布图展示当前包体的文件和目录重心。The asset distribution chart shows where files and directories are concentrated.
+
资产分布Asset Distribution303项303 itemsSKILL.mdSKILL.mdREADME.mdREADME.mdagents/interface.yamlagents/interface.yamlmanifest.jsonmanifest.jsonreferencesreferencesscriptsscripts
资产分布图展示当前包体的文件和目录重心。The asset distribution chart shows where files and directories are concentrated.
diff --git a/reports/skill-interpretation.json b/reports/skill-interpretation.json index 78ecc88..47cfabf 100644 --- a/reports/skill-interpretation.json +++ b/reports/skill-interpretation.json @@ -412,7 +412,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 6, + "source_pass_count": 7, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -471,8 +471,7 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events", - "Adoption sample" + "External events" ] } ] @@ -525,10 +524,10 @@ "path": "reports", "label": "Generated evidence and overview artifacts", "kind": "folder", - "file_count": 213 + "file_count": 130 } ], - "file_count": 386, + "file_count": 303, "folder_count": 4, "distribution": [ { @@ -561,7 +560,7 @@ }, { "label": "reports", - "value": 213 + "value": 130 } ] }, @@ -699,7 +698,7 @@ "path": "reports", "label": "Generated evidence and overview artifacts", "kind": "folder", - "file_count": 213 + "file_count": 130 } ], "strengths": [ @@ -1000,12 +999,12 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "60df1cebf40b1627f6d313c233f377d0d36228a9525b70099623563c7efb7804", - "source_contract_sha256": "c0fe8976bca1ceac0b325c882003335c66487b8f152868af471f3bdd79b9f0af", + "evidence_bundle_sha256": "b8858dc09a54ae5d02549e2a4ad6fb9f6c0237ebb6be7a54c61b147aaff7f592", + "source_contract_sha256": "652c3a4a1f091d875bfacdd44da9a91bc02fec90af3fe76cf3a71a6dc80aba58", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1021,14 +1020,14 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 73 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, - "commit": "5495734e7e33b19b84a932626112d783a3edf271", + "commit": "038891715d7843daf7960c2ec61bf7d9ea91c9e4", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1133,7 +1132,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 193, - "package_sha256": "c0fe8976bca1ceac0b325c882003335c66487b8f152868af471f3bdd79b9f0af" + "package_sha256": "652c3a4a1f091d875bfacdd44da9a91bc02fec90af3fe76cf3a71a6dc80aba58" }, "skill_atlas": { "skill_count": 12, @@ -1171,7 +1170,7 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "c0fe8976bca1ceac0b325c882003335c66487b8f152868af471f3bdd79b9f0af", + "package_sha256": "652c3a4a1f091d875bfacdd44da9a91bc02fec90af3fe76cf3a71a6dc80aba58", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc" }, "compatibility": { @@ -1303,7 +1302,7 @@ { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "c0fe8976bca1ceac0b325c882003335c66487b8f152868af471f3bdd79b9f0af" + "to": "652c3a4a1f091d875bfacdd44da9a91bc02fec90af3fe76cf3a71a6dc80aba58" } ] }, @@ -1320,14 +1319,14 @@ "ok": true, "summary": { "event_count": 1, - "adoption_sample_count": 0, - "activation_count": 0, - "accepted_count": 0, + "adoption_sample_count": 1, + "activation_count": 1, + "accepted_count": 1, "edited_count": 0, "rejected_count": 0, "missed_count": 0, "failed_count": 0, - "adoption_rate": 0, + "adoption_rate": 100.0, "missed_trigger_count": 0, "wrong_trigger_count": 0, "bad_output_count": 0, @@ -1336,7 +1335,7 @@ "review_overdue_count": 0, "risk_band": "low", "event_types": { - "review_event": 1 + "skill_activation": 1 }, "failure_types": {}, "source_types": { @@ -1558,7 +1557,7 @@ "status": "external_required", "category": "external", "owner": "Browser/Chrome/IDE/provider client integrator", - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1610,8 +1609,8 @@ "missing_submission_count": 4, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "submitted_but_pending_count": 0, "source_accepted_without_valid_submission_count": 0, "overclaim_guard_active": true, @@ -1947,7 +1946,7 @@ "status": "pending", "source_status": "external_required", "source_accepted": false, - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1984,7 +1983,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 0, + "adoption_sample_count": 1, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -2005,8 +2004,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 0, - "status": "blocked", + "actual": 1, + "status": "pass", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -2022,8 +2021,8 @@ } ], "source_check_count": 3, - "source_pass_count": 1, - "source_blocked_count": 2, + "source_pass_count": 2, + "source_blocked_count": 1, "submission_state": { "status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json", diff --git a/reports/skill-overview.html b/reports/skill-overview.html index c3b2dfe..e443492 100644 --- a/reports/skill-overview.html +++ b/reports/skill-overview.html @@ -919,7 +919,7 @@
路径Path作用Role类型Type
SKILL.mdSkill 入口文件Skill entrypoint文件file
README.md人类可读使用说明Human-readable usage guide文件file
agents/interface.yaml跨平台接口元数据Neutral interface metadata文件file
manifest.json生命周期与打包元数据Lifecycle and portability metadata文件file
references扩展指导与复用资料Extended guidance and reusable notes目录folder
scripts确定性脚本或本地工具Deterministic helpers or local tooling目录folder
evals触发与质量检查Trigger and quality checks目录folder
reports生成的证据与总结报告Generated evidence and overview artifacts目录folder
-

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks6 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
  • 采用样本Adoption sample
+

世界证据World Evidence

世界级证据尚未完成:4 项待补,0 项已接受。World-class evidence is not complete: 4 pending, 0 accepted.

证据待补Evidence pending
待补证据Pending4仍需外部或人工证据接受。External or human evidence still needs acceptance.
已接受Accepted0已通过 source check 与提交契约。Passed source checks and submission contract.
源检查Source Checks7 / 13通过数 / 总检查数。Passed checks / total checks.
外部证据External evidence

提供商留出Provider Holdout

缺少真实 provider 模型运行和 token metadata。Missing a real provider model run and token metadata.

阻塞检查Blocked Checks
  • 提供商实跑Provider model run
  • Token 用量Token usage observed
人工证据Human evidence

人工盲评Human Adjudication

盲评 pair 仍待真实 reviewer 决策。Blind-review pairs still need real reviewer decisions.

阻塞检查Blocked Checks
  • 无待判定No pending decisions
  • 盲评完成Judgments complete
外部证据External evidence

原生权限Native Permission

原生 runtime enforcement 仍待目标客户端或外部安装器证明。Native runtime enforcement still needs target-client or external-installer proof.

阻塞检查Blocked Checks
  • 原生执行Native enforcement
外部证据External evidence

原生遥测Native Telemetry

真实外部客户端 metadata-only 事件仍未导入。Real external-client metadata-only events have not been imported yet.

阻塞检查Blocked Checks
  • 外部事件External events
@@ -930,7 +930,7 @@

让 reviewer 快速确认关键文件、目录和资产分布。Lets reviewers confirm key files, directories, and asset distribution quickly.

-
资产分布Asset Distribution386项386 itemsSKILL.mdSKILL.mdREADME.mdREADME.mdagents/interface.yamlagents/interface.yamlmanifest.jsonmanifest.jsonreferencesreferencesscriptsscripts
资产分布图展示当前包体的文件和目录重心。The asset distribution chart shows where files and directories are concentrated.
+
资产分布Asset Distribution303项303 itemsSKILL.mdSKILL.mdREADME.mdREADME.mdagents/interface.yamlagents/interface.yamlmanifest.jsonmanifest.jsonreferencesreferencesscriptsscripts
资产分布图展示当前包体的文件和目录重心。The asset distribution chart shows where files and directories are concentrated.
diff --git a/reports/skill-overview.json b/reports/skill-overview.json index 67c20fe..2faf297 100644 --- a/reports/skill-overview.json +++ b/reports/skill-overview.json @@ -411,7 +411,7 @@ "external_pending_count": 3, "human_pending_count": 1, "source_check_count": 13, - "source_pass_count": 6, + "source_pass_count": 7, "conclusion_zh": "世界级证据尚未完成:4 项待补,0 项已接受。", "conclusion_en": "World-class evidence is not complete: 4 pending, 0 accepted.", "entries": [ @@ -470,8 +470,7 @@ "summary_zh": "真实外部客户端 metadata-only 事件仍未导入。", "summary_en": "Real external-client metadata-only events have not been imported yet.", "blocked_checks": [ - "External events", - "Adoption sample" + "External events" ] } ] @@ -524,10 +523,10 @@ "path": "reports", "label": "Generated evidence and overview artifacts", "kind": "folder", - "file_count": 213 + "file_count": 130 } ], - "file_count": 386, + "file_count": 303, "folder_count": 4, "distribution": [ { @@ -560,7 +559,7 @@ }, { "label": "reports", - "value": 213 + "value": 130 } ] }, @@ -694,7 +693,7 @@ "path": "reports", "label": "Generated evidence and overview artifacts", "kind": "folder", - "file_count": 213 + "file_count": 130 } ], "strengths": [ @@ -995,12 +994,12 @@ "ok": true, "summary": { "reproducibility_ready": true, - "release_lock_ready": false, + "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, - "evidence_bundle_sha256": "60df1cebf40b1627f6d313c233f377d0d36228a9525b70099623563c7efb7804", - "source_contract_sha256": "c0fe8976bca1ceac0b325c882003335c66487b8f152868af471f3bdd79b9f0af", + "evidence_bundle_sha256": "b8858dc09a54ae5d02549e2a4ad6fb9f6c0237ebb6be7a54c61b147aaff7f592", + "source_contract_sha256": "652c3a4a1f091d875bfacdd44da9a91bc02fec90af3fe76cf3a71a6dc80aba58", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc", "output_case_count": 5, "failure_disclosure_count": 3, @@ -1016,14 +1015,14 @@ "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "world_class_source_check_count": 13, - "world_class_source_pass_count": 6, - "world_class_source_blocked_count": 7, + "world_class_source_pass_count": 7, + "world_class_source_blocked_count": 6, "public_claim_ready": false, - "public_claim_blocker_count": 5, - "working_tree_dirty": true, - "changed_file_count": 73 + "public_claim_blocker_count": 4, + "working_tree_dirty": false, + "changed_file_count": 0 }, - "commit": "5495734e7e33b19b84a932626112d783a3edf271", + "commit": "038891715d7843daf7960c2ec61bf7d9ea91c9e4", "missing_artifacts": [], "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", @@ -1128,7 +1127,7 @@ "interactive_script_count": 0, "package_hash_scope": "source-contract-without-generated-reports", "package_hash_file_count": 193, - "package_sha256": "c0fe8976bca1ceac0b325c882003335c66487b8f152868af471f3bdd79b9f0af" + "package_sha256": "652c3a4a1f091d875bfacdd44da9a91bc02fec90af3fe76cf3a71a6dc80aba58" }, "skill_atlas": { "skill_count": 12, @@ -1166,7 +1165,7 @@ "trust_level": "local", "license": "MIT", "checksums": { - "package_sha256": "c0fe8976bca1ceac0b325c882003335c66487b8f152868af471f3bdd79b9f0af", + "package_sha256": "652c3a4a1f091d875bfacdd44da9a91bc02fec90af3fe76cf3a71a6dc80aba58", "archive_sha256": "6852cf91a74d232c32d732b7c159c971827abf23af50153987193b084ad3b5cc" }, "compatibility": { @@ -1298,7 +1297,7 @@ { "field": "package_sha256", "from": "0000000000000000000000000000000000000000000000000000000000000000", - "to": "c0fe8976bca1ceac0b325c882003335c66487b8f152868af471f3bdd79b9f0af" + "to": "652c3a4a1f091d875bfacdd44da9a91bc02fec90af3fe76cf3a71a6dc80aba58" } ] }, @@ -1315,14 +1314,14 @@ "ok": true, "summary": { "event_count": 1, - "adoption_sample_count": 0, - "activation_count": 0, - "accepted_count": 0, + "adoption_sample_count": 1, + "activation_count": 1, + "accepted_count": 1, "edited_count": 0, "rejected_count": 0, "missed_count": 0, "failed_count": 0, - "adoption_rate": 0, + "adoption_rate": 100.0, "missed_trigger_count": 0, "wrong_trigger_count": 0, "bad_output_count": 0, @@ -1331,7 +1330,7 @@ "review_overdue_count": 0, "risk_band": "low", "event_types": { - "review_event": 1 + "skill_activation": 1 }, "failure_types": {}, "source_types": { @@ -1553,7 +1552,7 @@ "status": "external_required", "category": "external", "owner": "Browser/Chrome/IDE/provider client integrator", - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1605,8 +1604,8 @@ "missing_submission_count": 4, "invalid_submission_count": 0, "source_check_count": 13, - "source_pass_count": 6, - "source_blocked_count": 7, + "source_pass_count": 7, + "source_blocked_count": 6, "submitted_but_pending_count": 0, "source_accepted_without_valid_submission_count": 0, "overclaim_guard_active": true, @@ -1942,7 +1941,7 @@ "status": "pending", "source_status": "external_required", "source_accepted": false, - "current": "external source events 0; adoption samples 0", + "current": "external source events 0; adoption samples 1", "objective": "Import production metadata-only events from a real external client into the local drift loop.", "runbook": [ "python3 scripts/telemetry_native_host.py . --write-launcher /tmp/yao-telemetry-host.sh --write-manifest /tmp/yao-telemetry-host.json --allowed-origin chrome-extension:///", @@ -1979,7 +1978,7 @@ ], "observed_state": { "external_source_events": 0, - "adoption_sample_count": 0, + "adoption_sample_count": 1, "raw_content_allowed": false, "risk_band": "low", "accepted": false @@ -2000,8 +1999,8 @@ "label": "Adoption sample", "field": "adoption_sample_count", "expected": ">0", - "actual": 0, - "status": "blocked", + "actual": 1, + "status": "pass", "source_accepted": false, "next_action": "Telemetry must include adoption outcome evidence." }, @@ -2017,8 +2016,8 @@ } ], "source_check_count": 3, - "source_pass_count": 1, - "source_blocked_count": 2, + "source_pass_count": 2, + "source_blocked_count": 1, "submission_state": { "status": "missing", "path": "evidence/world_class/submissions/native-client-telemetry.json",
路径Path作用Role类型Type
SKILL.mdSkill 入口文件Skill entrypoint文件file
README.md人类可读使用说明Human-readable usage guide文件file
agents/interface.yaml跨平台接口元数据Neutral interface metadata文件file
manifest.json生命周期与打包元数据Lifecycle and portability metadata文件file
references扩展指导与复用资料Extended guidance and reusable notes目录folder
scripts确定性脚本或本地工具Deterministic helpers or local tooling目录folder
evals触发与质量检查Trigger and quality checks目录folder
reports生成的证据与总结报告Generated evidence and overview artifacts目录folder