Files
wehub-resource-sync 23f7624596
ADR-166 MCP Bridge Security Lock / Static-source security lock (push) Failing after 0s
ADR-166 MCP Bridge Security Lock / Compose default binds loopback + Mongo has auth (push) Failing after 2s
CodeQL Advanced / Analyze (rust) (push) Failing after 0s
ADR-166 MCP Bridge Security Lock / plugin-agent-federation bindHost default (push) Failing after 1s
ADR-166 MCP Bridge Security Lock / Runtime behavior — 401 + terminal gate + fail-closed (push) Failing after 4s
business-pods-smoke / smoke (push) Failing after 1s
all-plugins-smoke / smoke-all (push) Failing after 2s
CI/CD Pipeline / Security & Code Quality (push) Failing after 1s
CI/CD Pipeline / Test Suite (ubuntu-latest) (push) Failing after 1s
CI/CD Pipeline / Build & Package (macos-latest) (push) Has been skipped
CI/CD Pipeline / Build & Package (ubuntu-latest) (push) Has been skipped
CI/CD Pipeline / Build & Package (windows-latest) (push) Has been skipped
CI/CD Pipeline / Documentation & Examples (push) Failing after 1s
Clone Tracker (14-day rolling) / Snapshot clones for ruflo ecosystem (push) Failing after 1s
CodeQL Advanced / Analyze (actions) (push) Failing after 1s
CodeQL Advanced / Analyze (javascript-typescript) (push) Failing after 1s
federation-peer-rust / stable-noop (push) Failing after 1s
metaharness-ci / score (push) Failing after 1s
metaharness-ci / router-compat (push) Failing after 0s
metaharness-ci / similarity-tests (push) Failing after 0s
no-agentbbs-smoke / smoke-without-agentbbs (push) Failing after 1s
V3 CI/CD Pipeline / Build V3 (windows-latest) (push) Has been skipped
codex-integration-audit / Codex integration audit (push) Failing after 1s
helpers-manifest-guard / guard (push) Failing after 1s
🔗 Cross-Agent Integration Tests / 🤝 Agent Coordination Tests (push) Has been skipped
🔗 Cross-Agent Integration Tests / 🧠 Memory Sharing Integration (push) Has been skipped
🔗 Cross-Agent Integration Tests / 🛡️ Fault Tolerance Tests (push) Has been skipped
🔗 Cross-Agent Integration Tests / ⚡ Performance Integration Tests (push) Has been skipped
metaharness-ci / mcp-scan (push) Failing after 1s
metaharness-ci / eject-dryrun (push) Failing after 1s
metaharness-ci / metaharness-real-data (push) Failing after 0s
no-cli-optdep-bloat-2561 / guard (push) Failing after 1s
no-metaharness-smoke / smoke-without-metaharness (push) Failing after 1s
no-phantom-agentic-flow-subpath / guard (push) Failing after 1s
🔄 Automated Rollback Manager / 🚨 Failure Detection (push) Failing after 1s
V3 CI/CD Pipeline / Plugin hooks smoke / ubuntu-latest / Node 22 (push) Failing after 1s
V3 CI/CD Pipeline / ruflo-graph-intelligence build + test smoke (#2044, ADR-123) (push) Failing after 1s
CVE Audit Gate / Audit root (critical-blocking) (push) Failing after 2s
cost-tracker-smoke / smoke (push) Failing after 3s
oia-audit-weekly / audit (push) Failing after 2s
ruflo-agent-smoke / ruflo-agent structural smoke (push) Failing after 1s
📊 Status Badges Update / 📊 Update Status Badges (push) Failing after 1s
V3 CI/CD Pipeline / Static regression guards (#2267 YAML + (push) Failing after 1s
V3 CI/CD Pipeline / Test V3 Packages (push) Failing after 0s
V3 CI/CD Pipeline / agent_execute provider routing smoke (#2042) (push) Failing after 0s
CVE Audit Gate / Audit v3 (critical-blocking) (push) Failing after 1s
federation-peer-rust / stable-native (push) Failing after 2s
🔗 Cross-Agent Integration Tests / 🚀 Integration Test Setup (push) Failing after 2s
neural-trader-smoke / runtime-smoke (push) Failing after 1s
V3 CI/CD Pipeline / Build V3 (macos-latest) (push) Has been skipped
V3 CI/CD Pipeline / Build V3 (ubuntu-latest) (push) Has been skipped
V3 CI/CD Pipeline / Type Check V3 (push) Failing after 1s
V3 CI/CD Pipeline / Smoke (no better-sqlite3) / ubuntu-latest / Node 24 (push) Failing after 1s
V3 CI/CD Pipeline / Smoke (no better-sqlite3) / ubuntu-latest / Node 22 (push) Failing after 2s
V3 CI/CD Pipeline / browser rvf create flag smoke (#2015) (push) Failing after 0s
V3 CI/CD Pipeline / Dependency review (#2046) (push) Has been skipped
V3 CI/CD Pipeline / Supply-chain audit (#2046) (push) Failing after 0s
V3 CI/CD Pipeline / witness marker drift smoke (#2021) (push) Failing after 1s
V3 CI/CD Pipeline / neural-trader portfolio CG smoke (#2068, ADR-126 Phase 3) (push) Failing after 1s
V3 CI/CD Pipeline / neural-trader backtest signing smoke (#2068, ADR-126 Phase 4) (push) Failing after 1s
V3 CI/CD Pipeline / kg-extract type-import classification smoke (#2049) (push) Failing after 0s
V3 CI/CD Pipeline / witness verify precondition smoke (#1880) (push) Failing after 2s
V3 CI/CD Pipeline / neural-trader pipeline risk-gate smoke (#2068, ADR-126 Phase 5) (push) Failing after 0s
V3 CI/CD Pipeline / neural-trader feature attribution smoke (#2068, ADR-126 Phase 6) (push) Failing after 0s
V3 CI/CD Pipeline / plugin-registry signature verification smoke (#1922, CWE-347) (push) Failing after 4s
V3 CI/CD Pipeline / memory stats legacy-DB smoke (#2120) (push) Failing after 4s
V3 CI/CD Pipeline / github deprecated actions smoke (#2089, ADR-127 Phase 3) (push) Failing after 1s
V3 CI/CD Pipeline / graph query + pathfinder smoke (ADR-130 P2+P5) (push) Has been skipped
V3 CI/CD Pipeline / graph trajectory hooks smoke (ADR-130 P3) (push) Has been skipped
V3 CI/CD Pipeline / graph plugin adapter smoke (ADR-130 P4) (push) Has been skipped
V3 CI/CD Pipeline / graph benchmark (ADR-130 P6) (push) Has been skipped
V3 CI/CD Pipeline / statusline generator delegation smoke (#2195) (push) Failing after 1s
V3 CI/CD Pipeline / wizard init regression guard (#2206 (push) Failing after 1s
V3 CI/CD Pipeline / memory no-stray-db smoke (ADR-125 P7) (push) Failing after 1s
V3 CI/CD Pipeline / github-safe injection smoke (#2089, ADR-127 Phase 1) (push) Failing after 1s
V3 CI/CD Pipeline / github actions pin smoke (#2089, ADR-127 Phase 1) (push) Failing after 1s
V3 CI/CD Pipeline / github attribution opt-in smoke (#2089, ADR-127 Phase 4) (push) Failing after 1s
V3 CI/CD Pipeline / pre-bash hook safety smoke (#2017) (push) Failing after 1s
V3 CI/CD Pipeline / Memory import smoke / ubuntu-latest (push) Failing after 0s
V3 CI/CD Pipeline / MCP protocol smoke / ubuntu-latest (push) Failing after 2s
V3 CI/CD Pipeline / ruvllm WASM auto-init smoke (#2086) (push) Failing after 4s
V3 CI/CD Pipeline / MCP paired-tool round-trip smoke (#1889) (push) Failing after 1s
V3 CI/CD Pipeline / Plugin package install-safety (#1902/#1903/#1904) (push) Failing after 1s
V3 CI/CD Pipeline / Tool description discoverability (ADR-112) (push) Failing after 3s
V3 CI/CD Pipeline / CLI npx-install smoke (#1147 / (22) (push) Failing after 1s
V3 CI/CD Pipeline / CLI npx-install smoke (#1147 / (24) (push) Failing after 1s
V3 CI/CD Pipeline / Windows hook shim smoke (#2132) / ubuntu-latest (push) Failing after 2s
V3 CI/CD Pipeline / Windows hook execution smoke (#2132) / ubuntu-latest (push) Failing after 1s
V3 CI/CD Pipeline / Windows init hooks smoke (#2132) / ubuntu-latest (push) Failing after 1s
V3 CI/CD Pipeline / Vector-index dimension audit (#1947) (push) Failing after 0s
V3 CI/CD Pipeline / Hook-command install safety (#1921) (push) Failing after 1s
V3 CI/CD Pipeline / ToolOutputGuardrail smoke (ADR-131, (push) Failing after 1s
V3 CI/CD Pipeline / init-bundle invariants smoke (#2095, ADR-128 Phase 5) (push) Failing after 1s
V3 CI/CD Pipeline / wasm provider bridge smoke (ADR-129 P1) (push) Failing after 2s
V3 CI/CD Pipeline / wasm gallery CRUD smoke (ADR-129 P3) (push) Failing after 1s
V3 CI/CD Pipeline / wasm plugin bridge smoke (ADR-129 P4) (push) Failing after 0s
V3 CI/CD Pipeline / wasm compose smoke (ADR-129 P2) (push) Failing after 4s
V3 CI/CD Pipeline / graph schema smoke (ADR-130 P1) (push) Failing after 0s
Validate Marketplace / validate (push) Failing after 1s
🔍 Verification Pipeline / 🚀 Setup Verification (push) Failing after 1s
🔍 Verification Pipeline / 🛡️ Security Verification (push) Has been skipped
🔍 Verification Pipeline / 📝 Code Quality (push) Has been skipped
🔍 Verification Pipeline / 🧪 Test Verification (${{ matrix.os }}, Node ${{ matrix.node }}) (push) Has been skipped
🔍 Verification Pipeline / 🏗️ Build Verification (push) Has been skipped
🔍 Verification Pipeline / 📚 Documentation Verification (push) Has been skipped
CVE Audit Gate / High-severity report (warn only) (push) Has been cancelled
🔄 Automated Rollback Manager / 🔄 Execute Rollback (push) Has been cancelled
🔄 Automated Rollback Manager / ✅ Post-Rollback Verification (push) Has been cancelled
🔄 Automated Rollback Manager / 📊 Rollback Monitoring (push) Has been cancelled
V3 CI/CD Pipeline / Windows init hooks smoke (#2132) / windows-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows hook execution smoke (#2132) / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows hook execution smoke (#2132) / windows-latest (push) Has been cancelled
🔄 Automated Rollback Manager / ⏳ Manual Rollback Approval (push) Has been cancelled
V3 CI/CD Pipeline / MCP protocol smoke / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Memory import smoke / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows hook shim smoke (#2132) / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows hook shim smoke (#2132) / windows-latest (push) Has been cancelled
V3 CI/CD Pipeline / Windows init hooks smoke (#2132) / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Witness verify (signed manifest) / macos-latest (push) Has been cancelled
V3 CI/CD Pipeline / Witness verify (signed manifest) / ubuntu-latest (push) Has been cancelled
V3 CI/CD Pipeline / Witness verify (signed manifest) / windows-latest (push) Has been cancelled
V3 CI/CD Pipeline / Publish to npm (alpha) (push) Has been cancelled
V3 CI/CD Pipeline / Smoke (no better-sqlite3) / macos-latest / Node 22 (push) Has been cancelled
V3 CI/CD Pipeline / Plugin hooks smoke / macos-latest / Node 22 (push) Has been cancelled
CI/CD Pipeline / Deploy & Release (push) Has been cancelled
CI/CD Pipeline / CI Status (push) Has been cancelled
🔗 Cross-Agent Integration Tests / 📊 Integration Test Report (push) Has been cancelled
🔄 Automated Rollback Manager / 🔍 Pre-Rollback Validation (push) Has been cancelled
🔍 Verification Pipeline / ⚡ Performance Verification (push) Has been cancelled
🔍 Verification Pipeline / 📊 Verification Report (push) Has been cancelled
chore: import upstream snapshot with attribution
2026-07-13 12:02:19 +08:00

74 lines
7.6 KiB
JSON
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
{
"_meta": {
"purpose": "ADR-148 — per-Claude-tier OpenRouter alternate models. When `CLAUDE_FLOW_ROUTER_PROVIDER=openrouter` (or `OPENROUTER_API_KEY` is set and `CLAUDE_FLOW_ROUTER_PROVIDER` is not `anthropic`), the router suggests the OpenRouter slug below for the picked tier. Downstream `agent-execute-core` uses it to override the default MODEL_MAP entry.",
"generated": "2026-06-15",
"schema_version": 1,
"caveat": "These costs and choices are sensible starters, not measured. Override per-installation via $CLAUDE_FLOW_ROUTER_OPENROUTER_ALTS pointing at a custom JSON, or per-call via $OPENROUTER_DEFAULT_MODEL. Re-train an artifact with measured DRACO rows for your traffic to do better.",
"judge_bias_check_2026_06_15": {
"summary": "Cross-graded a 12-row cheap-tier slice with openai/gpt-4.1 as the judge instead of the default anthropic/claude-sonnet-4-6. gpt-4.1 marks the same answers 7-11 pp lower across all models (consistently harsher), but the RELATIVE ranking is preserved (Ling > gpt-4.1 > Opus > Sonnet > Haiku ≈ Llama). The Sonnet judge does NOT show systematic Anthropic-family favoritism: it rated gpt-4.1 at 89.6% (above Sonnet's own 88.5%), and gpt-4.1 (judging itself) rated itself 82.3% (below Ling's 87.5%). Net: single-judge measurements in this repo are mildly inflated by ~8-10 pp but ordinally honest. For absolute-quality claims, halve the absolute numbers and trust the ranking.",
"judge_diff_pp_by_model": {
"inclusionai/ling-2.6-flash": -6.3,
"openai/gpt-4.1": -7.3,
"google/gemini-2.5-flash-lite": -7.3,
"anthropic/claude-opus-4": -8.4,
"anthropic/claude-sonnet-4-6": -10.4,
"meta-llama/llama-3.3-70b-instruct": -10.4,
"anthropic/claude-haiku-4.5": -11.4
},
"bench": "docs/benchmarks/runs/seed-corpus-2026-06-15-23-0*.json (--judge openai/gpt-4.1 --max-rows 12)"
}
},
"tiers": {
"haiku": {
"anthropic_default": "anthropic/claude-haiku-4-5-20251001",
"openrouter_alt": "inclusionai/ling-2.6-flash",
"cost_per_m_tok_in": 0.01,
"cost_per_m_tok_out": 0.03,
"rationale": "Cheap-tier alt: Inclusion AI Ling 2.6 Flash. Variance-measured 100% pass rate over 45 runs (15 queries × 3 repeats), 684 ± 104 ms mean latency (lowest std-dev of all measured models), $0.001/1k passes — 151× cheaper than Anthropic Haiku 4.5. Nemotron-3 Super 120B free tier is faster (350 ms) but has free-tier rate limits (8/45 calls hit 429) and 97.8% pass rate; safe as a $0 fallback chain but not as the default. See docs/benchmarks/runs/cheap-models-2026-06-15-20-3*Z.json for the 4-model variance run.",
"alternates_ranked_by_dollar_per_1k_passes_measured_2026_06_15_repeat_3": [
{ "id": "nvidia/nemotron-3-super-120b-a12b:free", "pass_rate": 0.978, "latency_mean_ms": 350, "latency_stdev_ms": 171, "usd_per_1k_passes": 0.0000, "note": "FREE; 8/45 rate-limited in parallel run, 97.8% solo" },
{ "id": "inclusionai/ling-2.6-flash", "pass_rate": 1.000, "latency_mean_ms": 684, "latency_stdev_ms": 104, "usd_per_1k_passes": 0.0010, "note": "stable Pareto-best paid" },
{ "id": "google/gemini-2.5-flash-lite", "pass_rate": 1.000, "latency_mean_ms": 525, "latency_stdev_ms": 248, "usd_per_1k_passes": 0.0100, "note": "fastest paid 100%-pass, higher variance" },
{ "id": "meta-llama/llama-3.3-70b-instruct", "pass_rate": 1.000, "latency_mean_ms": 688, "latency_stdev_ms": null, "usd_per_1k_passes": 0.0121 },
{ "id": "openai/gpt-4o-mini", "pass_rate": 1.000, "latency_mean_ms": 1093, "latency_stdev_ms": null, "usd_per_1k_passes": 0.0150 },
{ "id": "anthropic/claude-haiku-4.5", "pass_rate": 1.000, "latency_mean_ms": 1022, "latency_stdev_ms": 226, "usd_per_1k_passes": 0.1511, "note": "control: most expensive and 1.5× slower than Ling" }
]
},
"sonnet": {
"anthropic_default": "anthropic/claude-sonnet-4-6",
"openrouter_alt": "openai/gpt-4.1",
"cost_per_m_tok_in": 2.00,
"cost_per_m_tok_out": 8.00,
"rationale": "Mid-tier alt: OpenAI GPT-4.1 — measured 81.0% quality (LLM-judged 5-criterion rubric) vs Sonnet 4.6's 76.7%, at 4× lower cost ($0.030 vs $0.112 per call) and 2.7× faster (582 ms vs 1593 ms). For maximum cost reduction at the price of ~10% quality, prefer meta-llama/llama-3.3-70b-instruct (69.6% quality at 70× cheaper $/quality). See docs/benchmarks/runs/midtier-models-2026-06-15-*.json.",
"alternates_ranked_by_quality_measured_2026_06_15": [
{ "id": "openai/gpt-4.1", "avg_score": 0.810, "structural_pass": 0.92, "latency_mean_ms": 582, "usd_per_run": 0.02975, "usd_per_quality": 0.0367 },
{ "id": "google/gemini-2.5-flash", "avg_score": 0.767, "structural_pass": 1.00, "latency_mean_ms": 997, "usd_per_run": 0.01377, "usd_per_quality": 0.0180 },
{ "id": "anthropic/claude-sonnet-4-6", "avg_score": 0.767, "structural_pass": 0.83, "latency_mean_ms": 1593, "usd_per_run": 0.11152, "usd_per_quality": 0.1455, "note": "control: 4-8× more expensive than measured alts" },
{ "id": "meta-llama/llama-3.3-70b-instruct", "avg_score": 0.696, "structural_pass": 0.92, "latency_mean_ms": 613, "usd_per_run": 0.00137, "usd_per_quality": 0.0020, "note": "Pareto $/quality leader — 70× cheaper than Sonnet for 91% of its quality" },
{ "id": "qwen/qwen3-32b", "avg_score": 0.367, "structural_pass": 0.50, "latency_mean_ms": 705, "usd_per_run": 0.00853, "usd_per_quality": 0.0233 },
{ "id": "openai/gpt-5-mini", "avg_score": 0.083, "structural_pass": 0.08, "latency_mean_ms": 586, "usd_per_run": 0.01873, "usd_per_quality": 0.2247, "note": "fails at max_tokens=768 — reasoning model consumes the budget before visible output. Re-bench with --max-tokens 4096 to evaluate fairly." },
{ "id": "google/gemini-2.5-pro", "avg_score": 0.023, "structural_pass": 0.17, "latency_mean_ms": 1996, "usd_per_run": 0.09316, "usd_per_quality": 4.0651, "note": "same caveat as gpt-5-mini — reasoning model + 768 cap" }
],
"alternates_ranked_at_max_tokens_4096_measured_2026_06_15": [
{ "id": "openai/gpt-4.1", "avg_score": 0.742, "structural_pass": 0.92, "latency_mean_ms": 506, "usd_per_run": 0.03136, "usd_per_quality": 0.0423, "note": "Pareto leader even at 4096 tokens" },
{ "id": "openai/gpt-5-mini", "avg_score": 0.721, "structural_pass": 0.75, "latency_mean_ms": 698, "usd_per_run": 0.04041, "usd_per_quality": 0.0561, "note": "competitive with gpt-4.1 at 4096 but pricier per quality — reasoning premium doesn't pay off here" },
{ "id": "google/gemini-2.5-pro", "avg_score": 0.683, "structural_pass": 0.75, "latency_mean_ms": 2161, "usd_per_run": 0.23994, "usd_per_quality": 0.3511, "note": "strictly Pareto-dominated even at 4096 tokens — slowest, lowest quality, 8× pricier per quality than gpt-4.1" }
]
},
"opus": {
"anthropic_default": "anthropic/claude-opus-4-8",
"openrouter_alt": "anthropic/claude-opus-4",
"cost_per_m_tok_in": 15.00,
"cost_per_m_tok_out": 75.00,
"rationale": "Strong-tier alt: same model family via OpenRouter (lets users with OR-only access still reach Opus). Real diversity here requires Phase B."
},
"inherit": {
"anthropic_default": "anthropic/claude-sonnet-4-6",
"openrouter_alt": "anthropic/claude-sonnet-4-6",
"cost_per_m_tok_in": 3.00,
"cost_per_m_tok_out": 15.00,
"rationale": "`inherit` is the caller-defined default; we don't override it. Both paths map to Sonnet 4.6."
}
}
}