{ "schema_version": "1.0", "ok": true, "generated_at": "2026-06-14", "skill_dir": ".", "commit": "ab58c97bce07ef00acefd499a6d4499041ebfa91", "git_status": { "available": true, "dirty": false, "changed_file_count": 0, "sample": [], "scope": "generation-time status before this report is written" }, "summary": { "reproducibility_ready": true, "release_lock_ready": true, "methodology_complete": true, "required_artifact_count": 24, "missing_artifact_count": 0, "evidence_bundle_sha256": "a358b81f0d5418371d422da439958fa541a864e928a0be915be95c74bc55b48f", "source_contract_sha256": "0de594f90374f72387eea382b32ad0dd220d0f1d6f809b8fdd39859a1f35d5b7", "archive_sha256": "9761608424bbab003651eb643921a13f86b51220de705a3255d7c201e267dc68", "output_case_count": 5, "failure_disclosure_count": 3, "command_count": 21, "command_executed_count": 10, "timing_observed_count": 10, "model_executed_count": 0, "token_observed_count": 0, "human_review_complete": false, "provider_evidence_complete": false, "world_class_ready": false, "world_class_open_gap_count": 4, "world_class_task_count": 4, "world_class_ledger_pending_count": 4, "public_claim_ready": false, "public_claim_blocker_count": 3, "working_tree_dirty": false, "changed_file_count": 0 }, "public_claim": { "ready": false, "scope": "public benchmark or world-class readiness claim", "blockers": [ "provider-backed model holdout evidence is incomplete", "human blind-review adjudication is incomplete", "world-class evidence is not accepted yet (4 open gaps, 4 ledger pending)" ], "policy": "Local reproducibility can pass before public claims; public claims require provider evidence, human adjudication, clean release lock, and accepted world-class evidence." }, "release_lock": { "ready": true, "commit": "ab58c97bce07ef00acefd499a6d4499041ebfa91", "status_scope": "generation-time status before this report is written", "reason": "clean generation-time HEAD" }, "evidence_bundle": { "algorithm": "sha256(path,label,exists,artifact_sha256)", "artifact_count": 24, "existing_count": 24, "missing_count": 0, "missing_paths": [], "sha256": "a358b81f0d5418371d422da439958fa541a864e928a0be915be95c74bc55b48f" }, "methodology": { "path": "reports/benchmark_methodology.md", "exists": true, "sections": [ { "heading": "## Benchmark Types", "exists": true }, { "heading": "## Sample Sources", "exists": true }, { "heading": "## Evaluation Dimensions", "exists": true }, { "heading": "## Weighting Rule", "exists": true }, { "heading": "## Failure Disclosure", "exists": true }, { "heading": "## Reproduction", "exists": true } ], "missing_sections": [] }, "artifacts_checked": [ { "label": "methodology", "path": "reports/benchmark_methodology.md", "exists": true, "bytes": 2715, "sha256": "57025e0123ce5d10401c5bff376d2eeeac7943c83897ae1ad3fb22cadf790f92" }, { "label": "failure_disclosure", "path": "evals/failure-cases.md", "exists": true, "bytes": 889, "sha256": "28833c0d4a217d612879d193fb5de199880dd5b3093ab5757e4315600fa4fb08" }, { "label": "output_cases", "path": "evals/output/cases.jsonl", "exists": true, "bytes": 6555, "sha256": "a6ae9685711620d7203b73ace4412194ae287689f945b5139ec8ad75b9eefe04" }, { "label": "output_schema", "path": "evals/output/schema.json", "exists": true, "bytes": 2193, "sha256": "8ee340c95064260c5e952be614e19841ac676162c6bf01d21b107e38cb04e0b9" }, { "label": "output_scorecard", "path": "reports/output_quality_scorecard.json", "exists": true, "bytes": 25530, "sha256": "0806258a8e084b27e112537faff0de64a8519ca90cfdc78b57c0e4c08a514cca" }, { "label": "output_execution", "path": "reports/output_execution_runs.json", "exists": true, "bytes": 7966, "sha256": "815dad0e8e7817fdb0092375f14b0e5f96a0645c52d762434e433e52123920b6" }, { "label": "blind_review", "path": "reports/output_blind_review_pack.json", "exists": true, "bytes": 7804, "sha256": "bbe2db8ec2776fe289cd7d6bb78d48c2b8baa106ad37f79c15e43669b10c9390" }, { "label": "review_adjudication", "path": "reports/output_review_adjudication.json", "exists": true, "bytes": 9495, "sha256": "240485a721af49d5c75fe8049e9ef74b50546f685aa6346ab9e20713fb1105f4" }, { "label": "trigger_scorecard", "path": "reports/route_scorecard.json", "exists": true, "bytes": 16961, "sha256": "c164e83e36d0af276b6af2de2a5e026d7f0711b83eef2b5dcd0e760bd8bb28fc" }, { "label": "runtime_conformance", "path": "reports/conformance_matrix.json", "exists": true, "bytes": 10313, "sha256": "8251329e663dda51472f29b7721e73d72ccbec9760d96fda022f6218a5a6e347" }, { "label": "trust_report", "path": "reports/security_trust_report.json", "exists": true, "bytes": 98830, "sha256": "dc06cc6e7c80726ba1158c214c660ef5e07d933d9783124ee114bfd3e2aaab49" }, { "label": "python_compatibility", "path": "reports/python_compatibility.json", "exists": true, "bytes": 20747, "sha256": "ae16e17266e4e7fed08f4423f8abca7902b0d1cc0839b5692925c62463b72e58" }, { "label": "registry_audit", "path": "reports/registry_audit.json", "exists": true, "bytes": 3183, "sha256": "9a288c41f8af0f5490c26662513695e8d8752600cbf2e7262471b8eec9932f92" }, { "label": "package_verification", "path": "reports/package_verification.json", "exists": true, "bytes": 19325, "sha256": "7ad16e1c57b7cb54832143d8d7f469c3c53771f053b27adbeaef9013e687c21a" }, { "label": "install_simulation", "path": "reports/install_simulation.json", "exists": true, "bytes": 8604, "sha256": "8f987e805c92bf4178e3f246356ab268e59f62e9b6948be09f5c53054e7780ec" }, { "label": "skill_os2_audit", "path": "reports/skill_os2_audit.json", "exists": true, "bytes": 14309, "sha256": "6bb2dcb0e1e590deacd04c28a86bc74769c82810ecca6a5f660d6615e8db7c18" }, { "label": "world_class_evidence_plan", "path": "reports/world_class_evidence_plan.json", "exists": true, "bytes": 19532, "sha256": "a9bd392cd45f3accabe9b5befb9090456f383b1edbc323cab7d20930179dc229" }, { "label": "world_class_evidence_ledger", "path": "reports/world_class_evidence_ledger.json", "exists": true, "bytes": 11613, "sha256": "0bff14542475657a1261c1afcd87ffabc5d5c5e75bcd20afcfeacc44b16b7c14" }, { "label": "world_class_evidence_intake", "path": "reports/world_class_evidence_intake.json", "exists": true, "bytes": 13743, "sha256": "9f382cb3717160cff3eccf63312e20fc670183139e2ce86ce5096a59cc1b9110" }, { "label": "world_class_submission_review", "path": "reports/world_class_submission_review.json", "exists": true, "bytes": 7263, "sha256": "4f03edb2ef28d9dfd27a393f8058d6e27d2b10fa8cbfe35f473e003556b9ffe0" }, { "label": "world_class_operator_runbook", "path": "reports/world_class_operator_runbook.json", "exists": true, "bytes": 15002, "sha256": "614742ea666c64cdb68e10fc77022c31e68abe82ed88325aea21acb2f277063e" }, { "label": "world_class_operator_runbook_markdown", "path": "reports/world_class_operator_runbook.md", "exists": true, "bytes": 9241, "sha256": "302cfaa160ab7b50afda58a9e330e1219f47786b39b1e14d81ccc09c0ba10319" }, { "label": "world_class_operator_runbook_html", "path": "reports/world_class_operator_runbook.html", "exists": true, "bytes": 13226, "sha256": "699da59fb4c5646a253e9688a35f24c692c9875083128220305d1d93a317f208" }, { "label": "world_class_claim_guard", "path": "reports/world_class_claim_guard.json", "exists": true, "bytes": 8463, "sha256": "250d616b028cf046e2033f9e2c5648c8d95c110d0d8990b35e4d481e14cfc558" } ], "missing_artifacts": [], "reproduction_commands": [ { "label": "source commit", "command": "git rev-parse HEAD", "evidence": "git commit hash" }, { "label": "trigger eval", "command": "make eval-suite", "evidence": "reports/eval_suite.json" }, { "label": "output eval", "command": "python3 scripts/yao.py output-eval", "evidence": "reports/output_quality_scorecard.json" }, { "label": "output execution", "command": "python3 scripts/yao.py output-exec --runner-command '[\"python3\",\"scripts/local_output_eval_runner.py\"]'", "evidence": "reports/output_execution_runs.json" }, { "label": "blind review adjudication", "command": "python3 scripts/yao.py output-review", "evidence": "reports/output_review_adjudication.json" }, { "label": "skill ir", "command": "python3 scripts/yao.py skill-ir . --output-json skill-ir/examples/yao-meta-skill.json", "evidence": "skill-ir/examples/yao-meta-skill.json" }, { "label": "runtime conformance", "command": "python3 scripts/yao.py conformance .", "evidence": "reports/conformance_matrix.json" }, { "label": "trust report", "command": "python3 scripts/yao.py trust .", "evidence": "reports/security_trust_report.json" }, { "label": "python compatibility", "command": "python3 scripts/yao.py python-compat .", "evidence": "reports/python_compatibility.json" }, { "label": "package", "command": "python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --expectations evals/packaging_expectations.json --output-dir dist --zip", "evidence": "dist/yao-meta-skill.zip" }, { "label": "package verify", "command": "python3 scripts/yao.py package-verify . --package-dir dist --require-zip", "evidence": "reports/package_verification.json" }, { "label": "install simulate", "command": "python3 scripts/yao.py install-simulate . --package-dir dist", "evidence": "reports/install_simulation.json" }, { "label": "registry audit", "command": "python3 scripts/yao.py registry-audit .", "evidence": "reports/registry_audit.json" }, { "label": "skill os audit", "command": "python3 scripts/yao.py skill-os2-audit .", "evidence": "reports/skill_os2_audit.json" }, { "label": "world-class evidence plan", "command": "python3 scripts/yao.py world-class-evidence .", "evidence": "reports/world_class_evidence_plan.json" }, { "label": "world-class evidence ledger", "command": "python3 scripts/yao.py world-class-ledger .", "evidence": "reports/world_class_evidence_ledger.json" }, { "label": "world-class evidence intake", "command": "python3 scripts/yao.py world-class-intake .", "evidence": "reports/world_class_evidence_intake.json" }, { "label": "world-class submission review", "command": "python3 scripts/yao.py world-class-submission-review .", "evidence": "reports/world_class_submission_review.json" }, { "label": "world-class operator runbook", "command": "python3 scripts/yao.py world-class-runbook .", "evidence": "reports/world_class_operator_runbook.json" }, { "label": "world-class claim guard", "command": "python3 scripts/yao.py world-class-claim-guard .", "evidence": "reports/world_class_claim_guard.json" }, { "label": "full ci", "command": "make ci-test", "evidence": "CI target output" } ], "failure_disclosure": { "path": "evals/failure-cases.md", "case_count": 3, "policy": "Keep representative failures visible and tied to regression checks." }, "limitations": [ "The git commit and dirty flag are generation-time context; the evidence bundle hash is the durable artifact anchor inside a committed report.", "Local command-runner evidence is reproducible but does not replace provider-backed model holdout evidence.", "Pending blind-review decisions are visible but do not count as human adjudication.", "World-class readiness remains false until external and human evidence gaps close." ], "artifacts": { "json": "reports/benchmark_reproducibility.json", "markdown": "reports/benchmark_reproducibility.md" } }