Files
2026-07-13 12:58:18 +08:00

131 lines
3.4 KiB
Python

"""LangSmith evaluations for the Finance ERP agent.
Run with:
python -m evals.test_agent
Requires LANGCHAIN_API_KEY and LANGCHAIN_PROJECT environment variables.
"""
from __future__ import annotations
import os
from dotenv import load_dotenv
load_dotenv()
from langsmith import Client
from langsmith.evaluation import aevaluate
# Dataset of (input, expected_output) pairs for evaluation
EVAL_DATASET = [
{
"input": "How many overdue invoices do we have?",
"expected": "3 overdue invoice",
"tags": ["invoices", "query"],
},
{
"input": "What is our total revenue?",
"expected": "$2,847,350",
"tags": ["accounts", "query"],
},
{
"input": "Which inventory items are out of stock?",
"expected": "Cisco Catalyst 9300",
"tags": ["inventory", "query"],
},
{
"input": "Who is the CFO?",
"expected": "Sarah Chen",
"tags": ["hr", "query"],
},
{
"input": "Generate a balance sheet",
"expected": "BALANCE SHEET",
"tags": ["reports", "generation"],
},
{
"input": "What is our cash position?",
"expected": "$1,245,000",
"tags": ["accounts", "query"],
},
{
"input": "Forecast revenue for the next 4 quarters",
"expected": "Q2 2026",
"tags": ["analytics", "forecast"],
},
{
"input": "What are the low stock items?",
"expected": "MacBook Pro",
"tags": ["inventory", "alerts"],
},
]
DATASET_NAME = "finance-erp-agent-evals"
def create_or_update_dataset():
"""Push eval dataset to LangSmith."""
client = Client()
try:
dataset = client.read_dataset(dataset_name=DATASET_NAME)
except Exception:
dataset = client.create_dataset(
dataset_name=DATASET_NAME,
description="Evaluation dataset for the Finance ERP deep agent",
)
for example in EVAL_DATASET:
client.create_example(
inputs={"question": example["input"]},
outputs={"answer": example["expected"]},
dataset_id=dataset.id,
metadata={"tags": example["tags"]},
)
print(f"Dataset '{DATASET_NAME}' created with {len(EVAL_DATASET)} examples.")
return dataset
def contains_expected(run, example) -> dict:
"""Check if the agent output contains the expected substring."""
output = (run.outputs or {}).get("output", "")
expected = (example.outputs or {}).get("answer", "")
return {
"key": "contains_expected",
"score": 1.0 if expected.lower() in output.lower() else 0.0,
}
async def run_evaluation():
"""Run LangSmith evaluation against the agent."""
from agent import finance_erp_graph
async def predict(inputs: dict) -> dict:
result = await finance_erp_graph.ainvoke(
{"messages": [{"role": "user", "content": inputs["question"]}]}
)
last_message = result["messages"][-1]
return {"output": last_message.content}
results = await aevaluate(
predict,
data=DATASET_NAME,
evaluators=[contains_expected],
experiment_prefix="finance-erp-eval",
metadata={"version": "0.1.0"},
)
print(f"Evaluation complete. Results: {results}")
return results
if __name__ == "__main__":
import asyncio
import sys
if "--create-dataset" in sys.argv:
create_or_update_dataset()
else:
asyncio.run(run_evaluation())