Files
yao-meta-skill/reports/benchmark_reproducibility.json
T
2026-06-17 04:24:04 +08:00

457 lines
16 KiB
JSON

{
"schema_version": "1.0",
"ok": true,
"generated_at": "2026-06-17",
"skill_dir": ".",
"commit": "ed6960f65dbba0a36c335835f2ac6e8133435ea7",
"git_status": {
"available": true,
"dirty": true,
"changed_file_count": 19,
"generated_dirty": true,
"generated_changed_file_count": 19,
"source_dirty": false,
"source_changed_file_count": 0,
"sample": [
" M reports/adoption_drift_report.json",
" M reports/benchmark_reproducibility.json",
" M reports/benchmark_reproducibility.md",
" M reports/context_budget.json",
" M reports/context_budget_summary.json",
" M reports/evidence_consistency.json",
" M reports/output_execution_runs.json",
" M reports/output_execution_runs.md",
" M reports/review-studio.html",
" M reports/review-studio.json",
" M reports/review-viewer.json",
" M reports/skill-interpretation.html"
],
"source_sample": [],
"generated_sample": [
" M reports/adoption_drift_report.json",
" M reports/benchmark_reproducibility.json",
" M reports/benchmark_reproducibility.md",
" M reports/context_budget.json",
" M reports/context_budget_summary.json",
" M reports/evidence_consistency.json",
" M reports/output_execution_runs.json",
" M reports/output_execution_runs.md",
" M reports/review-studio.html",
" M reports/review-studio.json",
" M reports/review-viewer.json",
" M reports/skill-interpretation.html"
],
"generated_dirty_prefixes": [
"dist/",
"registry/index.json",
"registry/packages/",
"reports/",
"skill_atlas/",
"skill-ir/examples/"
],
"scope": "generation-time status before this report is written"
},
"summary": {
"reproducibility_ready": true,
"release_lock_ready": true,
"methodology_complete": true,
"required_artifact_count": 25,
"missing_artifact_count": 0,
"evidence_bundle_sha256": "38a010bc2913be5c7aadd809095c436abf8e566c8c53b8b4a63a821fea6acc98",
"source_contract_sha256": "6fbdbed9dfdc8d272caa0a6596312fa524796132dfb3c9e9e1a9cbc801729f59",
"archive_sha256": "0db54f7880460ee162ea2779554315c225ad912737660379c0c503f068c636e8",
"output_case_count": 5,
"failure_disclosure_count": 3,
"command_count": 23,
"command_executed_count": 10,
"timing_observed_count": 10,
"model_executed_count": 0,
"token_observed_count": 0,
"human_review_complete": false,
"provider_evidence_complete": false,
"world_class_ready": false,
"world_class_open_gap_count": 4,
"world_class_task_count": 4,
"world_class_ledger_pending_count": 4,
"world_class_source_check_count": 19,
"world_class_source_pass_count": 10,
"world_class_source_blocked_count": 9,
"public_claim_ready": false,
"public_claim_blocker_count": 4,
"working_tree_dirty": true,
"changed_file_count": 19,
"source_tree_dirty": false,
"source_changed_file_count": 0,
"generated_tree_dirty": true,
"generated_changed_file_count": 19
},
"public_claim": {
"ready": false,
"scope": "public benchmark or world-class readiness claim",
"blockers": [
"provider-backed model holdout evidence is incomplete",
"human blind-review adjudication is incomplete",
"world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)",
"world-class source checks are not all accepted (10/19 pass, 9 blocked)"
],
"policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, accepted world-class evidence, and complete source checks."
},
"release_lock": {
"ready": true,
"commit": "ed6960f65dbba0a36c335835f2ac6e8133435ea7",
"status_scope": "generation-time status before this report is written",
"source_changed_file_count": 0,
"generated_changed_file_count": 19,
"reason": "only generated evidence artifacts were dirty at generation time"
},
"evidence_bundle": {
"algorithm": "sha256(path,label,exists,artifact_sha256)",
"artifact_count": 25,
"existing_count": 25,
"missing_count": 0,
"missing_paths": [],
"sha256": "38a010bc2913be5c7aadd809095c436abf8e566c8c53b8b4a63a821fea6acc98"
},
"methodology": {
"path": "reports/benchmark_methodology.md",
"exists": true,
"sections": [
{
"heading": "## Benchmark Types",
"exists": true
},
{
"heading": "## Sample Sources",
"exists": true
},
{
"heading": "## Evaluation Dimensions",
"exists": true
},
{
"heading": "## Weighting Rule",
"exists": true
},
{
"heading": "## Failure Disclosure",
"exists": true
},
{
"heading": "## Reproduction",
"exists": true
}
],
"missing_sections": []
},
"artifacts_checked": [
{
"label": "methodology",
"path": "reports/benchmark_methodology.md",
"exists": true,
"bytes": 2715,
"sha256": "57025e0123ce5d10401c5bff376d2eeeac7943c83897ae1ad3fb22cadf790f92"
},
{
"label": "failure_disclosure",
"path": "evals/failure-cases.md",
"exists": true,
"bytes": 889,
"sha256": "28833c0d4a217d612879d193fb5de199880dd5b3093ab5757e4315600fa4fb08"
},
{
"label": "output_cases",
"path": "evals/output/cases.jsonl",
"exists": true,
"bytes": 6555,
"sha256": "a6ae9685711620d7203b73ace4412194ae287689f945b5139ec8ad75b9eefe04"
},
{
"label": "output_schema",
"path": "evals/output/schema.json",
"exists": true,
"bytes": 2193,
"sha256": "8ee340c95064260c5e952be614e19841ac676162c6bf01d21b107e38cb04e0b9"
},
{
"label": "output_scorecard",
"path": "reports/output_quality_scorecard.json",
"exists": true,
"bytes": 25530,
"sha256": "0806258a8e084b27e112537faff0de64a8519ca90cfdc78b57c0e4c08a514cca"
},
{
"label": "output_execution",
"path": "reports/output_execution_runs.json",
"exists": true,
"bytes": 7967,
"sha256": "3b9c37bdf02259b0e87d9f9fc5980512a2e95d9bcca32fd109b7a622d453b652"
},
{
"label": "blind_review",
"path": "reports/output_blind_review_pack.json",
"exists": true,
"bytes": 7804,
"sha256": "bbe2db8ec2776fe289cd7d6bb78d48c2b8baa106ad37f79c15e43669b10c9390"
},
{
"label": "review_adjudication",
"path": "reports/output_review_adjudication.json",
"exists": true,
"bytes": 14084,
"sha256": "91fd88dd9b0f8876f68027a87893d1852dac59772613345bbbfabba658c9227e"
},
{
"label": "trigger_scorecard",
"path": "reports/route_scorecard.json",
"exists": true,
"bytes": 16961,
"sha256": "c164e83e36d0af276b6af2de2a5e026d7f0711b83eef2b5dcd0e760bd8bb28fc"
},
{
"label": "runtime_conformance",
"path": "reports/conformance_matrix.json",
"exists": true,
"bytes": 10342,
"sha256": "97f9ba949c23a60b00e9ba2ff279ca03ba517845cc7c62aa8a42645d58006c7e"
},
{
"label": "trust_report",
"path": "reports/security_trust_report.json",
"exists": true,
"bytes": 135139,
"sha256": "a18583795da7c3ffc0d355bc5465382b50f1a737a33e62b1e4b182496c9e4c5e"
},
{
"label": "python_compatibility",
"path": "reports/python_compatibility.json",
"exists": true,
"bytes": 29589,
"sha256": "364a150344f461934d36ca59c0f5f6e1ed487fbfa3b30cfdff548ded277dd5c5"
},
{
"label": "registry_audit",
"path": "reports/registry_audit.json",
"exists": true,
"bytes": 3183,
"sha256": "d0157e4c0242955a3870f11eb00649f77e14b75f393874ea110b8f8e6eca89fb"
},
{
"label": "package_verification",
"path": "reports/package_verification.json",
"exists": true,
"bytes": 19338,
"sha256": "755a9c6788ccc0624682e523fe0154ac8b471bc4e41ed132f9866143762f398e"
},
{
"label": "install_simulation",
"path": "reports/install_simulation.json",
"exists": true,
"bytes": 8758,
"sha256": "15b77ec2b4738df891457d1d65f914751f30d6cd816a377e5c50c5a0b6d35b78"
},
{
"label": "skill_os2_audit",
"path": "reports/skill_os2_audit.json",
"exists": true,
"bytes": 14466,
"sha256": "16cbe378cc1a5ef847022c274af42ec8b6f7c346c8a06f59e6f6e3060d666177"
},
{
"label": "world_class_evidence_plan",
"path": "reports/world_class_evidence_plan.json",
"exists": true,
"bytes": 22784,
"sha256": "4193115d881d4f68e61d2729633f88dd91d55d982af3db70691aafeba852f1ba"
},
{
"label": "world_class_evidence_ledger",
"path": "reports/world_class_evidence_ledger.json",
"exists": true,
"bytes": 26016,
"sha256": "cb66d3284045b5993d25447f011cc70737dc760765ea3a3b7ae653194f0c4ff4"
},
{
"label": "world_class_evidence_intake",
"path": "reports/world_class_evidence_intake.json",
"exists": true,
"bytes": 20646,
"sha256": "002fbf07989a804595fdcb51b6d0c8177dc035a1dc8d084b8294a274d21fbc1b"
},
{
"label": "world_class_evidence_preflight",
"path": "reports/world_class_evidence_preflight.json",
"exists": true,
"bytes": 66981,
"sha256": "758807fa5749e0dd46f46d311abfad8af7dd70663e8876569b1a4b8414b42ace"
},
{
"label": "world_class_submission_review",
"path": "reports/world_class_submission_review.json",
"exists": true,
"bytes": 17299,
"sha256": "f5b1e0e44aeb6e1999182d40fca421ab0a66b91154efc7b518e9ed6f6ac20dd2"
},
{
"label": "world_class_operator_runbook",
"path": "reports/world_class_operator_runbook.json",
"exists": true,
"bytes": 28871,
"sha256": "8da2e2cde3adecec955f7745a4b761edc5af871d1720409cbd4f3b4b4482510e"
},
{
"label": "world_class_operator_runbook_markdown",
"path": "reports/world_class_operator_runbook.md",
"exists": true,
"bytes": 17521,
"sha256": "bfac92bc14f3080b9d723f32a4848aa3cfc10d4c3ee2e3844b65153f34c87c0a"
},
{
"label": "world_class_operator_runbook_html",
"path": "reports/world_class_operator_runbook.html",
"exists": true,
"bytes": 23958,
"sha256": "46b435f98d8a91501580fa1426a8ab0d0f57b3bc92ecd17f42332e655fd2bd85"
},
{
"label": "world_class_claim_guard",
"path": "reports/world_class_claim_guard.json",
"exists": true,
"bytes": 18596,
"sha256": "c846ffc8565a40c270e711a58e968b005afd75be6df1a442b020deebe850386f"
}
],
"missing_artifacts": [],
"reproduction_commands": [
{
"label": "source commit",
"command": "git rev-parse HEAD",
"evidence": "git commit hash"
},
{
"label": "trigger eval",
"command": "make eval-suite",
"evidence": "reports/eval_suite.json"
},
{
"label": "output eval",
"command": "python3 scripts/yao.py output-eval",
"evidence": "reports/output_quality_scorecard.json"
},
{
"label": "output execution",
"command": "python3 scripts/yao.py output-exec --runner-command '[\"python3\",\"scripts/local_output_eval_runner.py\"]'",
"evidence": "reports/output_execution_runs.json"
},
{
"label": "blind review adjudication",
"command": "python3 scripts/yao.py output-review",
"evidence": "reports/output_review_adjudication.json"
},
{
"label": "skill ir",
"command": "python3 scripts/yao.py skill-ir . --output-json skill-ir/examples/yao-meta-skill.json",
"evidence": "skill-ir/examples/yao-meta-skill.json"
},
{
"label": "runtime conformance",
"command": "python3 scripts/yao.py conformance .",
"evidence": "reports/conformance_matrix.json"
},
{
"label": "trust report",
"command": "python3 scripts/yao.py trust .",
"evidence": "reports/security_trust_report.json"
},
{
"label": "python compatibility",
"command": "python3 scripts/yao.py python-compat .",
"evidence": "reports/python_compatibility.json"
},
{
"label": "package",
"command": "python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --expectations evals/packaging_expectations.json --output-dir dist --zip",
"evidence": "dist/yao-meta-skill.zip"
},
{
"label": "package verify",
"command": "python3 scripts/yao.py package-verify . --package-dir dist --require-zip",
"evidence": "reports/package_verification.json"
},
{
"label": "install simulate",
"command": "python3 scripts/yao.py install-simulate . --package-dir dist",
"evidence": "reports/install_simulation.json"
},
{
"label": "registry audit",
"command": "python3 scripts/yao.py registry-audit .",
"evidence": "reports/registry_audit.json"
},
{
"label": "skill os audit",
"command": "python3 scripts/yao.py skill-os2-audit .",
"evidence": "reports/skill_os2_audit.json"
},
{
"label": "world-class evidence plan",
"command": "python3 scripts/yao.py world-class-evidence .",
"evidence": "reports/world_class_evidence_plan.json"
},
{
"label": "world-class evidence ledger",
"command": "python3 scripts/yao.py world-class-ledger . --submissions-dir evidence/world_class/submissions",
"evidence": "reports/world_class_evidence_ledger.json"
},
{
"label": "world-class evidence intake",
"command": "python3 scripts/yao.py world-class-intake . --submissions-dir evidence/world_class/submissions",
"evidence": "reports/world_class_evidence_intake.json"
},
{
"label": "world-class evidence preflight",
"command": "python3 scripts/yao.py world-class-preflight . --submissions-dir evidence/world_class/submissions",
"evidence": "reports/world_class_evidence_preflight.json"
},
{
"label": "world-class submission review",
"command": "python3 scripts/yao.py world-class-submission-review . --submissions-dir evidence/world_class/submissions",
"evidence": "reports/world_class_submission_review.json"
},
{
"label": "world-class operator runbook",
"command": "python3 scripts/yao.py world-class-runbook . --submissions-dir evidence/world_class/submissions",
"evidence": "reports/world_class_operator_runbook.json"
},
{
"label": "world-class claim guard",
"command": "python3 scripts/yao.py world-class-claim-guard .",
"evidence": "reports/world_class_claim_guard.json"
},
{
"label": "evidence consistency",
"command": "python3 scripts/yao.py evidence-consistency .",
"evidence": "reports/evidence_consistency.json"
},
{
"label": "full ci",
"command": "make ci-test",
"evidence": "CI target output"
}
],
"failure_disclosure": {
"path": "evals/failure-cases.md",
"case_count": 3,
"policy": "Keep representative failures visible and tied to regression checks."
},
"limitations": [
"The git commit and dirty flags are generation-time context; release lock is blocked by source changes, while generated evidence artifacts are tracked separately.",
"Local command-runner evidence is reproducible but does not replace provider-backed model holdout evidence.",
"Pending blind-review decisions are visible but do not count as human adjudication.",
"World-class readiness remains false until external and human evidence gaps close."
],
"artifacts": {
"json": "reports/benchmark_reproducibility.json",
"markdown": "reports/benchmark_reproducibility.md"
}
}