chore: import upstream snapshot with attribution
OSV-Scanner (Scheduled) / scan-scheduled (push) Failing after 0s
Create Release / test-gate (push) Has been cancelled
Create Release / release-gate (push) Has been cancelled
Create Release / ci-gate (push) Has been cancelled
Create Release / version-check (push) Has been cancelled
Create Release / e2e-test-gate (push) Has been cancelled
Create Release / responsive-test-gate (push) Has been cancelled
Create Release / compat-test-gate (push) Has been cancelled
Create Release / compose-integration-gate (push) Has been cancelled
Create Release / vulture-gate (push) Has been cancelled
Create Release / build (push) Has been cancelled
Create Release / provenance (push) Has been cancelled
Create Release / prerelease-docker (push) Has been cancelled
Create Release / publish-docker (push) Has been cancelled
Create Release / create-release (push) Has been cancelled
Create Release / cleanup-changelog (push) Has been cancelled
Create Release / trigger-pypi (push) Has been cancelled
Create Release / monitor-pypi (push) Has been cancelled
Create Release / Clean up orphan prerelease tags and signatures (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [research-form] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [research-metrics] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [research-workflow] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [settings-core] (push) Has been cancelled
CodeQL Advanced / Analyze (javascript-typescript) (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [history-news] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [library] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [link-analytics] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [chat-core] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [chat-lifecycle] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [error-benchmark] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [settings-pages] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) (push) Has been cancelled
Docker Tests (Consolidated) / Accessibility Tests (push) Has been cancelled
Docker Tests (Consolidated) / LLM Unit Tests (push) Has been cancelled
Docker Tests (Consolidated) / LLM Example Tests (push) Has been cancelled
Docker Tests (Consolidated) / Production Image Smoke Test (push) Has been cancelled
Docker Tests (Consolidated) / Infrastructure Tests (push) Has been cancelled
OSSF Scorecard / OSSF Security Scorecard Analysis (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [mobile] (push) Has been cancelled
Backwards Compatibility / Verify Encryption Constants (push) Has been cancelled
Backwards Compatibility / PyPI Version Compatibility (push) Has been cancelled
Backwards Compatibility / Database Migration Tests (push) Has been cancelled
CodeQL Advanced / Analyze (python) (push) Has been cancelled
Docker Tests (Consolidated) / detect-changes (push) Has been cancelled
Docker Tests (Consolidated) / Build Test Image (push) Has been cancelled
Docker Tests (Consolidated) / All Pytest Tests + Coverage (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [accessibility] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [api-crud] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [auth-login] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [auth-pages] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [auth-register] (push) Has been cancelled
OSV-Scanner (Scheduled) / scan-scheduled (push) Failing after 0s
Create Release / test-gate (push) Has been cancelled
Create Release / release-gate (push) Has been cancelled
Create Release / ci-gate (push) Has been cancelled
Create Release / version-check (push) Has been cancelled
Create Release / e2e-test-gate (push) Has been cancelled
Create Release / responsive-test-gate (push) Has been cancelled
Create Release / compat-test-gate (push) Has been cancelled
Create Release / compose-integration-gate (push) Has been cancelled
Create Release / vulture-gate (push) Has been cancelled
Create Release / build (push) Has been cancelled
Create Release / provenance (push) Has been cancelled
Create Release / prerelease-docker (push) Has been cancelled
Create Release / publish-docker (push) Has been cancelled
Create Release / create-release (push) Has been cancelled
Create Release / cleanup-changelog (push) Has been cancelled
Create Release / trigger-pypi (push) Has been cancelled
Create Release / monitor-pypi (push) Has been cancelled
Create Release / Clean up orphan prerelease tags and signatures (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [research-form] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [research-metrics] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [research-workflow] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [settings-core] (push) Has been cancelled
CodeQL Advanced / Analyze (javascript-typescript) (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [history-news] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [library] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [link-analytics] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [chat-core] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [chat-lifecycle] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [error-benchmark] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [settings-pages] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) (push) Has been cancelled
Docker Tests (Consolidated) / Accessibility Tests (push) Has been cancelled
Docker Tests (Consolidated) / LLM Unit Tests (push) Has been cancelled
Docker Tests (Consolidated) / LLM Example Tests (push) Has been cancelled
Docker Tests (Consolidated) / Production Image Smoke Test (push) Has been cancelled
Docker Tests (Consolidated) / Infrastructure Tests (push) Has been cancelled
OSSF Scorecard / OSSF Security Scorecard Analysis (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [mobile] (push) Has been cancelled
Backwards Compatibility / Verify Encryption Constants (push) Has been cancelled
Backwards Compatibility / PyPI Version Compatibility (push) Has been cancelled
Backwards Compatibility / Database Migration Tests (push) Has been cancelled
CodeQL Advanced / Analyze (python) (push) Has been cancelled
Docker Tests (Consolidated) / detect-changes (push) Has been cancelled
Docker Tests (Consolidated) / Build Test Image (push) Has been cancelled
Docker Tests (Consolidated) / All Pytest Tests + Coverage (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [accessibility] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [api-crud] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [auth-login] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [auth-pages] (push) Has been cancelled
Docker Tests (Consolidated) / UI Tests (Puppeteer) [auth-register] (push) Has been cancelled
This commit is contained in:
@@ -0,0 +1,32 @@
|
||||
# Gemini Benchmark Examples
|
||||
|
||||
This directory contains example scripts for running benchmarks with Gemini models via OpenRouter.
|
||||
|
||||
## Scripts Included
|
||||
|
||||
### run_gemini_benchmark_fixed.py
|
||||
A comprehensive benchmark script that runs both SimpleQA and BrowseComp evaluations
|
||||
using Google's Gemini 2.0 Flash model via the OpenRouter API.
|
||||
|
||||
Key features:
|
||||
- Patches the LLM configuration to use Gemini for all evaluations
|
||||
- Supports both SimpleQA and BrowseComp benchmarks
|
||||
- Properly handles result collection and reporting
|
||||
|
||||
## Usage
|
||||
|
||||
To run the benchmark with Gemini:
|
||||
|
||||
```bash
|
||||
# Run with default settings (1 example)
|
||||
python run_gemini_benchmark_fixed.py
|
||||
|
||||
# Run with custom number of examples
|
||||
python run_gemini_benchmark_fixed.py --examples 5
|
||||
```
|
||||
|
||||
## Notes
|
||||
|
||||
These scripts assume you have:
|
||||
1. An OpenRouter API key configured in your LDR database
|
||||
2. The correct access permissions for the Gemini model
|
||||
+178
@@ -0,0 +1,178 @@
|
||||
#!/usr/bin/env python
|
||||
"""
|
||||
Fixed benchmark with Gemini 2.0 Flash via OpenRouter
|
||||
"""
|
||||
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime, UTC
|
||||
from pathlib import Path
|
||||
|
||||
# Import the benchmark functions
|
||||
from local_deep_research.benchmarks.benchmark_functions import (
|
||||
evaluate_browsecomp,
|
||||
evaluate_simpleqa,
|
||||
)
|
||||
|
||||
# Monkey patch the get_llm function to use Gemini
|
||||
from local_deep_research.config import llm_config
|
||||
|
||||
# Save original function
|
||||
original_get_llm = llm_config.get_llm
|
||||
|
||||
|
||||
def setup_gemini_config():
|
||||
"""
|
||||
Create a custom evaluation configuration using Gemini 2.0 Flash via OpenRouter
|
||||
"""
|
||||
# Configure to use Gemini 2.0 Flash via OpenRouter
|
||||
evaluation_config = {
|
||||
"model_name": "google/gemini-2.0-flash-001", # OpenRouter format for Gemini
|
||||
"provider": "openai_endpoint", # Use OpenRouter as endpoint
|
||||
"openai_endpoint_url": "https://openrouter.ai/api/v1",
|
||||
"temperature": 0, # Zero temp for consistent evaluation
|
||||
}
|
||||
|
||||
print(f"Using Gemini 2.0 Flash for evaluation: {evaluation_config}")
|
||||
return evaluation_config
|
||||
|
||||
|
||||
# Override get_llm to always use Gemini
|
||||
def patched_get_llm(
|
||||
model_name=None, temperature=None, provider=None, openai_endpoint_url=None
|
||||
):
|
||||
"""Patched version that always uses Gemini via OpenRouter"""
|
||||
if (
|
||||
model_name == "gemma3:12b"
|
||||
): # This is the default model that causes the error
|
||||
print("Overriding local model with Gemini 2.0 Flash")
|
||||
model_name = "google/gemini-2.0-flash-001"
|
||||
provider = "openai_endpoint"
|
||||
openai_endpoint_url = "https://openrouter.ai/api/v1"
|
||||
return original_get_llm(
|
||||
model_name, temperature, provider, openai_endpoint_url
|
||||
)
|
||||
|
||||
|
||||
# Apply the patch
|
||||
llm_config.get_llm = patched_get_llm
|
||||
|
||||
|
||||
def run_benchmark(examples=1):
|
||||
"""Run benchmarks with Gemini 2.0 Flash"""
|
||||
try:
|
||||
# Create timestamp for output
|
||||
timestamp = datetime.now(UTC).strftime("%Y%m%d_%H%M%S")
|
||||
output_dir = str(
|
||||
Path(__file__).parent.parent.parent
|
||||
/ "benchmark_results"
|
||||
/ f"gemini_eval_{timestamp}"
|
||||
)
|
||||
Path(output_dir).mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Setup the Gemini configuration
|
||||
gemini_config = setup_gemini_config()
|
||||
|
||||
# Run SimpleQA benchmark
|
||||
print(f"\n=== Running SimpleQA benchmark with {examples} examples ===")
|
||||
simpleqa_start = time.time()
|
||||
|
||||
simpleqa_results = evaluate_simpleqa(
|
||||
num_examples=examples,
|
||||
search_iterations=2,
|
||||
questions_per_iteration=3,
|
||||
search_tool="searxng",
|
||||
evaluation_model=gemini_config["model_name"],
|
||||
evaluation_provider=gemini_config["provider"],
|
||||
output_dir=str(Path(output_dir) / "simpleqa"),
|
||||
)
|
||||
|
||||
simpleqa_duration = time.time() - simpleqa_start
|
||||
print(
|
||||
f"SimpleQA evaluation complete in {simpleqa_duration:.1f} seconds"
|
||||
)
|
||||
if (
|
||||
isinstance(simpleqa_results, dict)
|
||||
and "accuracy" in simpleqa_results
|
||||
):
|
||||
print(f"SimpleQA accuracy: {simpleqa_results['accuracy']:.4f}")
|
||||
else:
|
||||
print("SimpleQA accuracy: N/A")
|
||||
|
||||
# Run BrowseComp benchmark
|
||||
print(
|
||||
f"\n=== Running BrowseComp benchmark with {examples} examples ==="
|
||||
)
|
||||
browsecomp_start = time.time()
|
||||
|
||||
browsecomp_results = evaluate_browsecomp(
|
||||
num_examples=examples,
|
||||
search_iterations=3,
|
||||
questions_per_iteration=3,
|
||||
search_tool="searxng",
|
||||
evaluation_model=gemini_config["model_name"],
|
||||
evaluation_provider=gemini_config["provider"],
|
||||
output_dir=str(Path(output_dir) / "browsecomp"),
|
||||
)
|
||||
|
||||
browsecomp_duration = time.time() - browsecomp_start
|
||||
print(
|
||||
f"BrowseComp evaluation complete in {browsecomp_duration:.1f} seconds"
|
||||
)
|
||||
if (
|
||||
isinstance(browsecomp_results, dict)
|
||||
and "accuracy" in browsecomp_results
|
||||
):
|
||||
print(f"BrowseComp accuracy: {browsecomp_results['accuracy']:.4f}")
|
||||
else:
|
||||
print("BrowseComp accuracy: N/A")
|
||||
|
||||
# Generate summary
|
||||
print("\n=== Evaluation Summary ===")
|
||||
print(f"Examples: {examples}")
|
||||
print(f"Model: {gemini_config.get('model_name', 'unknown')}")
|
||||
print(f"Provider: {gemini_config.get('provider', 'unknown')}")
|
||||
print(f"Results saved to: {output_dir}")
|
||||
|
||||
return {
|
||||
"simpleqa": simpleqa_results,
|
||||
"browsecomp": browsecomp_results,
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error running benchmark: {e}")
|
||||
import traceback
|
||||
|
||||
traceback.print_exc()
|
||||
return None
|
||||
|
||||
|
||||
def main():
|
||||
# Parse command line arguments
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Run benchmark with Gemini 2.0 Flash"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--examples",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Number of examples to evaluate (default: 1)",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
print(
|
||||
f"Starting benchmark with Gemini 2.0 Flash on {args.examples} examples"
|
||||
)
|
||||
|
||||
# Run the evaluation
|
||||
results = run_benchmark(examples=args.examples)
|
||||
|
||||
# Return success if benchmark completed
|
||||
return 0 if results else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user