23f7624596
ADR-166 MCP Bridge Security Lock / Static-source security lock (push) Failing after 0s
ADR-166 MCP Bridge Security Lock / Compose default binds loopback + Mongo has auth (push) Failing after 2s
CodeQL Advanced / Analyze (rust) (push) Failing after 0s
ADR-166 MCP Bridge Security Lock / plugin-agent-federation bindHost default (push) Failing after 1s
ADR-166 MCP Bridge Security Lock / Runtime behavior — 401 + terminal gate + fail-closed (push) Failing after 4s
business-pods-smoke / smoke (push) Failing after 1s
all-plugins-smoke / smoke-all (push) Failing after 2s
CI/CD Pipeline / Security & Code Quality (push) Failing after 1s
CI/CD Pipeline / Test Suite (ubuntu-latest) (push) Failing after 1s
CI/CD Pipeline / Build & Package (macos-latest) (push) Has been skipped
CI/CD Pipeline / Build & Package (ubuntu-latest) (push) Has been skipped
CI/CD Pipeline / Build & Package (windows-latest) (push) Has been skipped
CI/CD Pipeline / Documentation & Examples (push) Failing after 1s
Clone Tracker (14-day rolling) / Snapshot clones for ruflo ecosystem (push) Failing after 1s
CodeQL Advanced / Analyze (actions) (push) Failing after 1s
CodeQL Advanced / Analyze (javascript-typescript) (push) Failing after 1s
federation-peer-rust / stable-noop (push) Failing after 1s
metaharness-ci / score (push) Failing after 1s
metaharness-ci / router-compat (push) Failing after 0s
metaharness-ci / similarity-tests (push) Failing after 0s
no-agentbbs-smoke / smoke-without-agentbbs (push) Failing after 1s
V3 CI/CD Pipeline / Build V3 (windows-latest) (push) Has been skipped
codex-integration-audit / Codex integration audit (push) Failing after 1s
helpers-manifest-guard / guard (push) Failing after 1s
🔗 Cross-Agent Integration Tests / 🤝 Agent Coordination Tests (push) Has been skipped
🔗 Cross-Agent Integration Tests / 🧠 Memory Sharing Integration (push) Has been skipped
🔗 Cross-Agent Integration Tests / 🛡️ Fault Tolerance Tests (push) Has been skipped
🔗 Cross-Agent Integration Tests / ⚡ Performance Integration Tests (push) Has been skipped
metaharness-ci / mcp-scan (push) Failing after 1s
metaharness-ci / eject-dryrun (push) Failing after 1s
metaharness-ci / metaharness-real-data (push) Failing after 0s
no-cli-optdep-bloat-2561 / guard (push) Failing after 1s
no-metaharness-smoke / smoke-without-metaharness (push) Failing after 1s
no-phantom-agentic-flow-subpath / guard (push) Failing after 1s
🔄 Automated Rollback Manager / 🚨 Failure Detection (push) Failing after 1s
V3 CI/CD Pipeline / Plugin hooks smoke / ubuntu-latest / Node 22 (push) Failing after 1s
V3 CI/CD Pipeline / ruflo-graph-intelligence build + test smoke (#2044, ADR-123) (push) Failing after 1s
CVE Audit Gate / Audit root (critical-blocking) (push) Failing after 2s
cost-tracker-smoke / smoke (push) Failing after 3s
oia-audit-weekly / audit (push) Failing after 2s
ruflo-agent-smoke / ruflo-agent structural smoke (push) Failing after 1s
📊 Status Badges Update / 📊 Update Status Badges (push) Failing after 1s
V3 CI/CD Pipeline / Static regression guards (#2267 YAML + (push) Failing after 1s
V3 CI/CD Pipeline / Test V3 Packages (push) Failing after 0s
V3 CI/CD Pipeline / agent_execute provider routing smoke (#2042) (push) Failing after 0s
CVE Audit Gate / Audit v3 (critical-blocking) (push) Failing after 1s
federation-peer-rust / stable-native (push) Failing after 2s
🔗 Cross-Agent Integration Tests / 🚀 Integration Test Setup (push) Failing after 2s
neural-trader-smoke / runtime-smoke (push) Failing after 1s
V3 CI/CD Pipeline / Build V3 (macos-latest) (push) Has been skipped
V3 CI/CD Pipeline / Build V3 (ubuntu-latest) (push) Has been skipped
V3 CI/CD Pipeline / Type Check V3 (push) Failing after 1s
V3 CI/CD Pipeline / Smoke (no better-sqlite3) / ubuntu-latest / Node 24 (push) Failing after 1s
V3 CI/CD Pipeline / Smoke (no better-sqlite3) / ubuntu-latest / Node 22 (push) Failing after 2s
V3 CI/CD Pipeline / browser rvf create flag smoke (#2015) (push) Failing after 0s
V3 CI/CD Pipeline / Dependency review (#2046) (push) Has been skipped
V3 CI/CD Pipeline / Supply-chain audit (#2046) (push) Failing after 0s
V3 CI/CD Pipeline / witness marker drift smoke (#2021) (push) Failing after 1s
V3 CI/CD Pipeline / neural-trader portfolio CG smoke (#2068, ADR-126 Phase 3) (push) Failing after 1s
V3 CI/CD Pipeline / neural-trader backtest signing smoke (#2068, ADR-126 Phase 4) (push) Failing after 1s
V3 CI/CD Pipeline / kg-extract type-import classification smoke (#2049) (push) Failing after 0s
V3 CI/CD Pipeline / witness verify precondition smoke (#1880) (push) Failing after 2s
V3 CI/CD Pipeline / neural-trader pipeline risk-gate smoke (#2068, ADR-126 Phase 5) (push) Failing after 0s
V3 CI/CD Pipeline / neural-trader feature attribution smoke (#2068, ADR-126 Phase 6) (push) Failing after 0s
V3 CI/CD Pipeline / plugin-registry signature verification smoke (#1922, CWE-347) (push) Failing after 4s
V3 CI/CD Pipeline / memory stats legacy-DB smoke (#2120) (push) Failing after 4s
V3 CI/CD Pipeline / github deprecated actions smoke (#2089, ADR-127 Phase 3) (push) Failing after 1s
V3 CI/CD Pipeline / graph query + pathfinder smoke (ADR-130 P2+P5) (push) Has been skipped
V3 CI/CD Pipeline / graph trajectory hooks smoke (ADR-130 P3) (push) Has been skipped
V3 CI/CD Pipeline / graph plugin adapter smoke (ADR-130 P4) (push) Has been skipped
V3 CI/CD Pipeline / graph benchmark (ADR-130 P6) (push) Has been skipped
V3 CI/CD Pipeline / statusline generator delegation smoke (#2195) (push) Failing after 1s
V3 CI/CD Pipeline / wizard init regression guard (#2206 (push) Failing after 1s
V3 CI/CD Pipeline / memory no-stray-db smoke (ADR-125 P7) (push) Failing after 1s
V3 CI/CD Pipeline / github-safe injection smoke (#2089, ADR-127 Phase 1) (push) Failing after 1s
V3 CI/CD Pipeline / github actions pin smoke (#2089, ADR-127 Phase 1) (push) Failing after 1s
V3 CI/CD Pipeline / github attribution opt-in smoke (#2089, ADR-127 Phase 4) (push) Failing after 1s
V3 CI/CD Pipeline / pre-bash hook safety smoke (#2017) (push) Failing after 1s
V3 CI/CD Pipeline / Memory import smoke / ubuntu-latest (push) Failing after 0s
V3 CI/CD Pipeline / MCP protocol smoke / ubuntu-latest (push) Failing after 2s
V3 CI/CD Pipeline / ruvllm WASM auto-init smoke (#2086) (push) Failing after 4s
V3 CI/CD Pipeline / MCP paired-tool round-trip smoke (#1889) (push) Failing after 1s
V3 CI/CD Pipeline / Plugin package install-safety (#1902/#1903/#1904) (push) Failing after 1s
V3 CI/CD Pipeline / Tool description discoverability (ADR-112) (push) Failing after 3s
V3 CI/CD Pipeline / CLI npx-install smoke (#1147 / (22) (push) Failing after 1s
V3 CI/CD Pipeline / CLI npx-install smoke (#1147 / (24) (push) Failing after 1s
V3 CI/CD Pipeline / Windows hook shim smoke (#2132) / ubuntu-latest (push) Failing after 2s
V3 CI/CD Pipeline / Windows hook execution smoke (#2132) / ubuntu-latest (push) Failing after 1s
V3 CI/CD Pipeline / Windows init hooks smoke (#2132) / ubuntu-latest (push) Failing after 1s
V3 CI/CD Pipeline / Vector-index dimension audit (#1947) (push) Failing after 0s
V3 CI/CD Pipeline / Hook-command install safety (#1921) (push) Failing after 1s
V3 CI/CD Pipeline / ToolOutputGuardrail smoke (ADR-131, (push) Failing after 1s
V3 CI/CD Pipeline / init-bundle invariants smoke (#2095, ADR-128 Phase 5) (push) Failing after 1s
V3 CI/CD Pipeline / wasm provider bridge smoke (ADR-129 P1) (push) Failing after 2s
V3 CI/CD Pipeline / wasm gallery CRUD smoke (ADR-129 P3) (push) Failing after 1s
V3 CI/CD Pipeline / wasm plugin bridge smoke (ADR-129 P4) (push) Failing after 0s
V3 CI/CD Pipeline / wasm compose smoke (ADR-129 P2) (push) Failing after 4s
V3 CI/CD Pipeline / graph schema smoke (ADR-130 P1) (push) Failing after 0s
Validate Marketplace / validate (push) Failing after 1s
🔍 Verification Pipeline / 🚀 Setup Verification (push) Failing after 1s
🔍 Verification Pipeline / 🛡️ Security Verification (push) Has been skipped
🔍 Verification Pipeline / 📝 Code Quality (push) Has been skipped
🔍 Verification Pipeline / 🧪 Test Verification (${{ matrix.os }}, Node ${{ matrix.node }}) (push) Has been skipped
🔍 Verification Pipeline / 🏗️ Build Verification (push) Has been skipped
🔍 Verification Pipeline / 📚 Documentation Verification (push) Has been skipped
CVE Audit Gate / High-severity report (warn only) (push) Has been cancelled
🔄 Automated Rollback Manager / 🔄 Execute Rollback (push) Has been cancelled
🔄 Automated Rollback Manager / ✅ Post-Rollback Verification (push) Has been cancelled
🔄 Automated Rollback Manager / 📊 Rollback Monitoring (push) Has been cancelled
V3 CI/CD Pipeline / Windows init hooks smoke (#2132) / windows-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows hook execution smoke (#2132) / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows hook execution smoke (#2132) / windows-latest (push) Has been cancelled
🔄 Automated Rollback Manager / ⏳ Manual Rollback Approval (push) Has been cancelled
V3 CI/CD Pipeline / MCP protocol smoke / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Memory import smoke / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows hook shim smoke (#2132) / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows hook shim smoke (#2132) / windows-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows init hooks smoke (#2132) / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Witness verify (signed manifest) / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Witness verify (signed manifest) / ubuntu-latest (push) Has been cancelled
V3 CI/CD Pipeline / Witness verify (signed manifest) / windows-latest (push) Has been cancelled
V3 CI/CD Pipeline / Publish to npm (alpha) (push) Has been cancelled
V3 CI/CD Pipeline / Smoke (no better-sqlite3) / macos-latest / Node 22 (push) Has been cancelled
V3 CI/CD Pipeline / Plugin hooks smoke / macos-latest / Node 22 (push) Has been cancelled
CI/CD Pipeline / Deploy & Release (push) Has been cancelled
CI/CD Pipeline / CI Status (push) Has been cancelled
🔗 Cross-Agent Integration Tests / 📊 Integration Test Report (push) Has been cancelled
🔄 Automated Rollback Manager / 🔍 Pre-Rollback Validation (push) Has been cancelled
🔍 Verification Pipeline / ⚡ Performance Verification (push) Has been cancelled
🔍 Verification Pipeline / 📊 Verification Report (push) Has been cancelled
74 lines
7.6 KiB
JSON
74 lines
7.6 KiB
JSON
{
|
||
"_meta": {
|
||
"purpose": "ADR-148 — per-Claude-tier OpenRouter alternate models. When `CLAUDE_FLOW_ROUTER_PROVIDER=openrouter` (or `OPENROUTER_API_KEY` is set and `CLAUDE_FLOW_ROUTER_PROVIDER` is not `anthropic`), the router suggests the OpenRouter slug below for the picked tier. Downstream `agent-execute-core` uses it to override the default MODEL_MAP entry.",
|
||
"generated": "2026-06-15",
|
||
"schema_version": 1,
|
||
"caveat": "These costs and choices are sensible starters, not measured. Override per-installation via $CLAUDE_FLOW_ROUTER_OPENROUTER_ALTS pointing at a custom JSON, or per-call via $OPENROUTER_DEFAULT_MODEL. Re-train an artifact with measured DRACO rows for your traffic to do better.",
|
||
"judge_bias_check_2026_06_15": {
|
||
"summary": "Cross-graded a 12-row cheap-tier slice with openai/gpt-4.1 as the judge instead of the default anthropic/claude-sonnet-4-6. gpt-4.1 marks the same answers 7-11 pp lower across all models (consistently harsher), but the RELATIVE ranking is preserved (Ling > gpt-4.1 > Opus > Sonnet > Haiku ≈ Llama). The Sonnet judge does NOT show systematic Anthropic-family favoritism: it rated gpt-4.1 at 89.6% (above Sonnet's own 88.5%), and gpt-4.1 (judging itself) rated itself 82.3% (below Ling's 87.5%). Net: single-judge measurements in this repo are mildly inflated by ~8-10 pp but ordinally honest. For absolute-quality claims, halve the absolute numbers and trust the ranking.",
|
||
"judge_diff_pp_by_model": {
|
||
"inclusionai/ling-2.6-flash": -6.3,
|
||
"openai/gpt-4.1": -7.3,
|
||
"google/gemini-2.5-flash-lite": -7.3,
|
||
"anthropic/claude-opus-4": -8.4,
|
||
"anthropic/claude-sonnet-4-6": -10.4,
|
||
"meta-llama/llama-3.3-70b-instruct": -10.4,
|
||
"anthropic/claude-haiku-4.5": -11.4
|
||
},
|
||
"bench": "docs/benchmarks/runs/seed-corpus-2026-06-15-23-0*.json (--judge openai/gpt-4.1 --max-rows 12)"
|
||
}
|
||
},
|
||
"tiers": {
|
||
"haiku": {
|
||
"anthropic_default": "anthropic/claude-haiku-4-5-20251001",
|
||
"openrouter_alt": "inclusionai/ling-2.6-flash",
|
||
"cost_per_m_tok_in": 0.01,
|
||
"cost_per_m_tok_out": 0.03,
|
||
"rationale": "Cheap-tier alt: Inclusion AI Ling 2.6 Flash. Variance-measured 100% pass rate over 45 runs (15 queries × 3 repeats), 684 ± 104 ms mean latency (lowest std-dev of all measured models), $0.001/1k passes — 151× cheaper than Anthropic Haiku 4.5. Nemotron-3 Super 120B free tier is faster (350 ms) but has free-tier rate limits (8/45 calls hit 429) and 97.8% pass rate; safe as a $0 fallback chain but not as the default. See docs/benchmarks/runs/cheap-models-2026-06-15-20-3*Z.json for the 4-model variance run.",
|
||
"alternates_ranked_by_dollar_per_1k_passes_measured_2026_06_15_repeat_3": [
|
||
{ "id": "nvidia/nemotron-3-super-120b-a12b:free", "pass_rate": 0.978, "latency_mean_ms": 350, "latency_stdev_ms": 171, "usd_per_1k_passes": 0.0000, "note": "FREE; 8/45 rate-limited in parallel run, 97.8% solo" },
|
||
{ "id": "inclusionai/ling-2.6-flash", "pass_rate": 1.000, "latency_mean_ms": 684, "latency_stdev_ms": 104, "usd_per_1k_passes": 0.0010, "note": "stable Pareto-best paid" },
|
||
{ "id": "google/gemini-2.5-flash-lite", "pass_rate": 1.000, "latency_mean_ms": 525, "latency_stdev_ms": 248, "usd_per_1k_passes": 0.0100, "note": "fastest paid 100%-pass, higher variance" },
|
||
{ "id": "meta-llama/llama-3.3-70b-instruct", "pass_rate": 1.000, "latency_mean_ms": 688, "latency_stdev_ms": null, "usd_per_1k_passes": 0.0121 },
|
||
{ "id": "openai/gpt-4o-mini", "pass_rate": 1.000, "latency_mean_ms": 1093, "latency_stdev_ms": null, "usd_per_1k_passes": 0.0150 },
|
||
{ "id": "anthropic/claude-haiku-4.5", "pass_rate": 1.000, "latency_mean_ms": 1022, "latency_stdev_ms": 226, "usd_per_1k_passes": 0.1511, "note": "control: most expensive and 1.5× slower than Ling" }
|
||
]
|
||
},
|
||
"sonnet": {
|
||
"anthropic_default": "anthropic/claude-sonnet-4-6",
|
||
"openrouter_alt": "openai/gpt-4.1",
|
||
"cost_per_m_tok_in": 2.00,
|
||
"cost_per_m_tok_out": 8.00,
|
||
"rationale": "Mid-tier alt: OpenAI GPT-4.1 — measured 81.0% quality (LLM-judged 5-criterion rubric) vs Sonnet 4.6's 76.7%, at 4× lower cost ($0.030 vs $0.112 per call) and 2.7× faster (582 ms vs 1593 ms). For maximum cost reduction at the price of ~10% quality, prefer meta-llama/llama-3.3-70b-instruct (69.6% quality at 70× cheaper $/quality). See docs/benchmarks/runs/midtier-models-2026-06-15-*.json.",
|
||
"alternates_ranked_by_quality_measured_2026_06_15": [
|
||
{ "id": "openai/gpt-4.1", "avg_score": 0.810, "structural_pass": 0.92, "latency_mean_ms": 582, "usd_per_run": 0.02975, "usd_per_quality": 0.0367 },
|
||
{ "id": "google/gemini-2.5-flash", "avg_score": 0.767, "structural_pass": 1.00, "latency_mean_ms": 997, "usd_per_run": 0.01377, "usd_per_quality": 0.0180 },
|
||
{ "id": "anthropic/claude-sonnet-4-6", "avg_score": 0.767, "structural_pass": 0.83, "latency_mean_ms": 1593, "usd_per_run": 0.11152, "usd_per_quality": 0.1455, "note": "control: 4-8× more expensive than measured alts" },
|
||
{ "id": "meta-llama/llama-3.3-70b-instruct", "avg_score": 0.696, "structural_pass": 0.92, "latency_mean_ms": 613, "usd_per_run": 0.00137, "usd_per_quality": 0.0020, "note": "Pareto $/quality leader — 70× cheaper than Sonnet for 91% of its quality" },
|
||
{ "id": "qwen/qwen3-32b", "avg_score": 0.367, "structural_pass": 0.50, "latency_mean_ms": 705, "usd_per_run": 0.00853, "usd_per_quality": 0.0233 },
|
||
{ "id": "openai/gpt-5-mini", "avg_score": 0.083, "structural_pass": 0.08, "latency_mean_ms": 586, "usd_per_run": 0.01873, "usd_per_quality": 0.2247, "note": "fails at max_tokens=768 — reasoning model consumes the budget before visible output. Re-bench with --max-tokens 4096 to evaluate fairly." },
|
||
{ "id": "google/gemini-2.5-pro", "avg_score": 0.023, "structural_pass": 0.17, "latency_mean_ms": 1996, "usd_per_run": 0.09316, "usd_per_quality": 4.0651, "note": "same caveat as gpt-5-mini — reasoning model + 768 cap" }
|
||
],
|
||
"alternates_ranked_at_max_tokens_4096_measured_2026_06_15": [
|
||
{ "id": "openai/gpt-4.1", "avg_score": 0.742, "structural_pass": 0.92, "latency_mean_ms": 506, "usd_per_run": 0.03136, "usd_per_quality": 0.0423, "note": "Pareto leader even at 4096 tokens" },
|
||
{ "id": "openai/gpt-5-mini", "avg_score": 0.721, "structural_pass": 0.75, "latency_mean_ms": 698, "usd_per_run": 0.04041, "usd_per_quality": 0.0561, "note": "competitive with gpt-4.1 at 4096 but pricier per quality — reasoning premium doesn't pay off here" },
|
||
{ "id": "google/gemini-2.5-pro", "avg_score": 0.683, "structural_pass": 0.75, "latency_mean_ms": 2161, "usd_per_run": 0.23994, "usd_per_quality": 0.3511, "note": "strictly Pareto-dominated even at 4096 tokens — slowest, lowest quality, 8× pricier per quality than gpt-4.1" }
|
||
]
|
||
},
|
||
"opus": {
|
||
"anthropic_default": "anthropic/claude-opus-4-8",
|
||
"openrouter_alt": "anthropic/claude-opus-4",
|
||
"cost_per_m_tok_in": 15.00,
|
||
"cost_per_m_tok_out": 75.00,
|
||
"rationale": "Strong-tier alt: same model family via OpenRouter (lets users with OR-only access still reach Opus). Real diversity here requires Phase B."
|
||
},
|
||
"inherit": {
|
||
"anthropic_default": "anthropic/claude-sonnet-4-6",
|
||
"openrouter_alt": "anthropic/claude-sonnet-4-6",
|
||
"cost_per_m_tok_in": 3.00,
|
||
"cost_per_m_tok_out": 15.00,
|
||
"rationale": "`inherit` is the caller-defined default; we don't override it. Both paths map to Sonnet 4.6."
|
||
}
|
||
}
|
||
}
|