Files
yao-meta-skill/reports/benchmark_reproducibility.json
T
2026-06-14 08:06:56 +08:00

333 lines
10 KiB
JSON

{
"schema_version": "1.0",
"ok": true,
"generated_at": "2026-06-14",
"skill_dir": ".",
"commit": "08347d5c2cae9528ae1fe3c50ad39bb3c773a6e1",
"git_status": {
"available": true,
"dirty": true,
"changed_file_count": 61,
"sample": [
"M AGENTS.md",
"M Makefile",
"M registry/index.json",
"M registry/packages/yao-meta-skill.json",
"MM reports/adoption_drift_report.json",
" M reports/adoption_drift_report.md",
"MM reports/architecture_maintainability.json",
"MM reports/architecture_maintainability.md",
"MM reports/benchmark_reproducibility.json",
"MM reports/benchmark_reproducibility.md",
"MM reports/compiled_targets.json",
"MM reports/context_budget.json"
]
},
"summary": {
"reproducibility_ready": true,
"methodology_complete": true,
"required_artifact_count": 20,
"missing_artifact_count": 0,
"output_case_count": 5,
"failure_disclosure_count": 3,
"command_count": 19,
"command_executed_count": 10,
"timing_observed_count": 10,
"model_executed_count": 0,
"token_observed_count": 0,
"human_review_complete": false,
"provider_evidence_complete": false,
"world_class_ready": false,
"world_class_open_gap_count": 4,
"world_class_task_count": 4,
"world_class_ledger_pending_count": 4,
"working_tree_dirty": true,
"changed_file_count": 61
},
"methodology": {
"path": "reports/benchmark_methodology.md",
"exists": true,
"sections": [
{
"heading": "## Benchmark Types",
"exists": true
},
{
"heading": "## Sample Sources",
"exists": true
},
{
"heading": "## Evaluation Dimensions",
"exists": true
},
{
"heading": "## Weighting Rule",
"exists": true
},
{
"heading": "## Failure Disclosure",
"exists": true
},
{
"heading": "## Reproduction",
"exists": true
}
],
"missing_sections": []
},
"artifacts_checked": [
{
"label": "methodology",
"path": "reports/benchmark_methodology.md",
"exists": true,
"bytes": 2715,
"sha256": "57025e0123ce5d10401c5bff376d2eeeac7943c83897ae1ad3fb22cadf790f92"
},
{
"label": "failure_disclosure",
"path": "evals/failure-cases.md",
"exists": true,
"bytes": 889,
"sha256": "28833c0d4a217d612879d193fb5de199880dd5b3093ab5757e4315600fa4fb08"
},
{
"label": "output_cases",
"path": "evals/output/cases.jsonl",
"exists": true,
"bytes": 6555,
"sha256": "a6ae9685711620d7203b73ace4412194ae287689f945b5139ec8ad75b9eefe04"
},
{
"label": "output_schema",
"path": "evals/output/schema.json",
"exists": true,
"bytes": 2193,
"sha256": "8ee340c95064260c5e952be614e19841ac676162c6bf01d21b107e38cb04e0b9"
},
{
"label": "output_scorecard",
"path": "reports/output_quality_scorecard.json",
"exists": true,
"bytes": 25530,
"sha256": "0806258a8e084b27e112537faff0de64a8519ca90cfdc78b57c0e4c08a514cca"
},
{
"label": "output_execution",
"path": "reports/output_execution_runs.json",
"exists": true,
"bytes": 7965,
"sha256": "99b2e3f03710fdcee8fe66a476bc589f4e22a379bfcc1cd5561de49c184286e8"
},
{
"label": "blind_review",
"path": "reports/output_blind_review_pack.json",
"exists": true,
"bytes": 7804,
"sha256": "bbe2db8ec2776fe289cd7d6bb78d48c2b8baa106ad37f79c15e43669b10c9390"
},
{
"label": "review_adjudication",
"path": "reports/output_review_adjudication.json",
"exists": true,
"bytes": 9495,
"sha256": "240485a721af49d5c75fe8049e9ef74b50546f685aa6346ab9e20713fb1105f4"
},
{
"label": "trigger_scorecard",
"path": "reports/route_scorecard.json",
"exists": true,
"bytes": 16961,
"sha256": "c164e83e36d0af276b6af2de2a5e026d7f0711b83eef2b5dcd0e760bd8bb28fc"
},
{
"label": "runtime_conformance",
"path": "reports/conformance_matrix.json",
"exists": true,
"bytes": 10313,
"sha256": "8251329e663dda51472f29b7721e73d72ccbec9760d96fda022f6218a5a6e347"
},
{
"label": "trust_report",
"path": "reports/security_trust_report.json",
"exists": true,
"bytes": 97005,
"sha256": "252c580028df6ac2404b53cc10bb607fa82b98ef17965cb7354c4731f6c6b08e"
},
{
"label": "python_compatibility",
"path": "reports/python_compatibility.json",
"exists": true,
"bytes": 20367,
"sha256": "a8d14087863459534fadac760a64c213c9474413e92046a9ac2e772d4174d347"
},
{
"label": "registry_audit",
"path": "reports/registry_audit.json",
"exists": true,
"bytes": 3183,
"sha256": "6527b723a4bc367477e1090afbf7f1653f20d33fed0e0691725b7ee30896848b"
},
{
"label": "package_verification",
"path": "reports/package_verification.json",
"exists": true,
"bytes": 19325,
"sha256": "6bf19c9e29803ad8668914489d22b9e306ad0ad092c91c179c3aaffe81eed2c6"
},
{
"label": "install_simulation",
"path": "reports/install_simulation.json",
"exists": true,
"bytes": 8758,
"sha256": "270a28b82dc49d80ba9c5a82f912cc960e50a5ac42f022c6a2c345e4e84fbb36"
},
{
"label": "skill_os2_audit",
"path": "reports/skill_os2_audit.json",
"exists": true,
"bytes": 14309,
"sha256": "cad5296cd500a40cb0d83897d327b3542f31c1ce6bfe7eaa387ea2e88d229aa1"
},
{
"label": "world_class_evidence_plan",
"path": "reports/world_class_evidence_plan.json",
"exists": true,
"bytes": 10083,
"sha256": "02e49b66159d239309bd31be33f2119db23604b5506c4445ba6c6c84b6cdd1aa"
},
{
"label": "world_class_evidence_ledger",
"path": "reports/world_class_evidence_ledger.json",
"exists": true,
"bytes": 11308,
"sha256": "0b52d0d1528ba95a0fa7bead97cb8224b5b189d35497a05d90956fc90512b651"
},
{
"label": "world_class_evidence_intake",
"path": "reports/world_class_evidence_intake.json",
"exists": true,
"bytes": 12740,
"sha256": "5fbfcd35ac6afd494c64845a7130520fd1cffbdb86e4efed2eaf2fa3d381f8b3"
},
{
"label": "world_class_claim_guard",
"path": "reports/world_class_claim_guard.json",
"exists": true,
"bytes": 8463,
"sha256": "250d616b028cf046e2033f9e2c5648c8d95c110d0d8990b35e4d481e14cfc558"
}
],
"missing_artifacts": [],
"reproduction_commands": [
{
"label": "source commit",
"command": "git rev-parse HEAD",
"evidence": "git commit hash"
},
{
"label": "trigger eval",
"command": "make eval-suite",
"evidence": "reports/eval_suite.json"
},
{
"label": "output eval",
"command": "python3 scripts/yao.py output-eval",
"evidence": "reports/output_quality_scorecard.json"
},
{
"label": "output execution",
"command": "python3 scripts/yao.py output-exec --runner-command '[\"python3\",\"scripts/local_output_eval_runner.py\"]'",
"evidence": "reports/output_execution_runs.json"
},
{
"label": "blind review adjudication",
"command": "python3 scripts/yao.py output-review",
"evidence": "reports/output_review_adjudication.json"
},
{
"label": "skill ir",
"command": "python3 scripts/yao.py skill-ir . --output-json skill-ir/examples/yao-meta-skill.json",
"evidence": "skill-ir/examples/yao-meta-skill.json"
},
{
"label": "runtime conformance",
"command": "python3 scripts/yao.py conformance .",
"evidence": "reports/conformance_matrix.json"
},
{
"label": "trust report",
"command": "python3 scripts/yao.py trust .",
"evidence": "reports/security_trust_report.json"
},
{
"label": "python compatibility",
"command": "python3 scripts/yao.py python-compat .",
"evidence": "reports/python_compatibility.json"
},
{
"label": "package",
"command": "python3 scripts/yao.py package . --platform openai --platform claude --platform generic --platform vscode --expectations evals/packaging_expectations.json --output-dir dist --zip",
"evidence": "dist/yao-meta-skill.zip"
},
{
"label": "package verify",
"command": "python3 scripts/yao.py package-verify . --package-dir dist --require-zip",
"evidence": "reports/package_verification.json"
},
{
"label": "install simulate",
"command": "python3 scripts/yao.py install-simulate . --package-dir dist",
"evidence": "reports/install_simulation.json"
},
{
"label": "registry audit",
"command": "python3 scripts/yao.py registry-audit .",
"evidence": "reports/registry_audit.json"
},
{
"label": "skill os audit",
"command": "python3 scripts/yao.py skill-os2-audit .",
"evidence": "reports/skill_os2_audit.json"
},
{
"label": "world-class evidence plan",
"command": "python3 scripts/yao.py world-class-evidence .",
"evidence": "reports/world_class_evidence_plan.json"
},
{
"label": "world-class evidence ledger",
"command": "python3 scripts/yao.py world-class-ledger .",
"evidence": "reports/world_class_evidence_ledger.json"
},
{
"label": "world-class evidence intake",
"command": "python3 scripts/yao.py world-class-intake .",
"evidence": "reports/world_class_evidence_intake.json"
},
{
"label": "world-class claim guard",
"command": "python3 scripts/yao.py world-class-claim-guard .",
"evidence": "reports/world_class_claim_guard.json"
},
{
"label": "full ci",
"command": "make ci-test",
"evidence": "CI target output"
}
],
"failure_disclosure": {
"path": "evals/failure-cases.md",
"case_count": 3,
"policy": "Keep representative failures visible and tied to regression checks."
},
"limitations": [
"Local command-runner evidence is reproducible but does not replace provider-backed model holdout evidence.",
"Pending blind-review decisions are visible but do not count as human adjudication.",
"World-class readiness remains false until external and human evidence gaps close."
],
"artifacts": {
"json": "reports/benchmark_reproducibility.json",
"markdown": "reports/benchmark_reproducibility.md"
}
}