chore: import upstream snapshot with attribution
TypeScript SDK Compatibility V1.x E2E Tests / Select Node version matrix (push) Has been cancelled
TypeScript SDK Compatibility V1.x E2E Tests / TypeScript SDK Compatibility V1.x E2E Tests Node ${{matrix.node_version}} (push) Has been cancelled
TypeScript SDK E2E Tests / TypeScript SDK E2E Tests Node ${{matrix.node_version}} (push) Has been cancelled
Opik Optimizer - E2E Tests / build-opik (push) Has been cancelled
TypeScript SDK Compatibility V1.x E2E Tests / build-opik (push) Has been cancelled
Python SDK E2E Tests / Select Python version matrix (push) Has been cancelled
Python SDK E2E Tests / Python SDK E2E Tests ${{matrix.python_version}} (push) Has been cancelled
Python SDK E2E Tests / build-opik (push) Has been cancelled
Python SDK Compatibility V1.x E2E Tests / Select Python version matrix (push) Has been cancelled
Python SDK Compatibility V1.x E2E Tests / Python SDK Compatibility V1.x E2E Tests ${{matrix.python_version}} (push) Has been cancelled
Python SDK Compatibility V1.x E2E Tests / build-opik (push) Has been cancelled
TypeScript SDK E2E Tests / Select Node version matrix (push) Has been cancelled
TypeScript SDK E2E Tests / build-opik (push) Has been cancelled
Opik Optimizer - E2E Tests / Opik Optimizer E2E Tests Python ${{matrix.python_version}} (push) Has been cancelled
Opik Optimizer - E2E Tests / Opik Optimizer Integration Smoke Tests (push) Has been cancelled
🐙 Code Quality / detect (push) Has been cancelled
🐙 Code Quality / lint (${{ matrix.leg.name }}) (push) Has been cancelled
🐙 Code Quality / summary (push) Has been cancelled
TypeScript SDK Library Integration Tests / Check Secrets (push) Has been cancelled
TypeScript SDK Library Integration Tests / opik-vercel (Vercel AI SDK / eve) (push) Has been cancelled
SDK Library Integration Tests Runner / Check Secrets (push) Has been cancelled
SDK Library Integration Tests Runner / Missed OpenAI API Key Warning (push) Has been cancelled
SDK Library Integration Tests Runner / Build (push) Has been cancelled
SDK Library Integration Tests Runner / openai_tests (push) Has been cancelled
SDK Library Integration Tests Runner / langchain_tests (push) Has been cancelled
SDK Library Integration Tests Runner / langchain_legacy_tests (push) Has been cancelled
SDK Library Integration Tests Runner / llama_index_tests (push) Has been cancelled
SDK Library Integration Tests Runner / anthropic_tests (push) Has been cancelled
SDK Library Integration Tests Runner / mistral_tests (push) Has been cancelled
SDK Library Integration Tests Runner / groq_tests (push) Has been cancelled
SDK Library Integration Tests Runner / aisuite_tests (push) Has been cancelled
SDK Library Integration Tests Runner / haystack_tests (push) Has been cancelled
SDK Library Integration Tests Runner / dspy_tests (push) Has been cancelled
SDK Library Integration Tests Runner / crewai_v0_tests (push) Has been cancelled
SDK Library Integration Tests Runner / crewai_v1_tests (push) Has been cancelled
SDK Library Integration Tests Runner / genai_tests (push) Has been cancelled
SDK Library Integration Tests Runner / adk_tests (push) Has been cancelled
SDK Library Integration Tests Runner / adk_legacy_1_3_0_tests (push) Has been cancelled
SDK Library Integration Tests Runner / evaluation_metrics_tests (push) Has been cancelled
SDK Library Integration Tests Runner / bedrock_tests (push) Has been cancelled
SDK Library Integration Tests Runner / litellm_tests (push) Has been cancelled
SDK Library Integration Tests Runner / harbor_tests (push) Has been cancelled
SDK Library Integration Tests Runner / Slack Notification (push) Has been cancelled
Lint Opik Helm Chart / render-equality (push) Has been cancelled
Opik Optimizer - Unit Tests / Opik Optimizer Unit Tests Python ${{matrix.python_version}} (push) Has been cancelled
Python BE E2E Tests / Python BE E2E (push) Has been cancelled
Python Backend Tests / run-python-backend-tests (push) Has been cancelled
Python SDK Unit Tests / Python SDK Unit Tests ${{matrix.python_version}} (push) Has been cancelled
Release Drafter / update_release_draft (push) Has been cancelled
SDK E2E Libraries Integration Tests / Check Secrets (push) Has been cancelled
SDK E2E Libraries Integration Tests / Missed OpenAI API Key Warning (push) Has been cancelled
SDK E2E Libraries Integration Tests / build-opik (push) Has been cancelled
SDK E2E Libraries Integration Tests / E2E Lib Integration Python ${{matrix.python_version}} (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-gemini) (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-langchain) (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-openai) (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-otel) (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-vercel) (push) Has been cancelled
TypeScript SDK Build & Publish / build-and-publish (push) Has been cancelled
TypeScript SDK Unit Tests / Test on Node ${{ matrix.node-version }} (push) Has been cancelled
Backend Tests / discover-tests (push) Has been cancelled
Backend Tests / ${{ matrix.name }} (push) Has been cancelled
Build and Publish SDK / build-and-publish (push) Has been cancelled
Build Opik Docker Images / set-version (push) Has been cancelled
Build Opik Docker Images / build-backend (push) Has been cancelled
Build Opik Docker Images / build-sandbox-executor-python (push) Has been cancelled
Build Opik Docker Images / build-python-backend (push) Has been cancelled
Build Opik Docker Images / build-frontend (push) Has been cancelled
Build Opik Docker Images / create-git-tag (push) Has been cancelled
ClickHouse Migration Cluster Check / validate-clickhouse-migrations (push) Has been cancelled
Docs - Publish / run (push) Has been cancelled
E2E Tests - Post Merge (v2) / 🧪 E2E v2 Tests (${{ github.event.inputs.tier || 't1' }}) (push) Has been cancelled
E2E Tests - Post Merge (v2) / 📢 Slack Notification (push) Has been cancelled
Frontend Unit Tests / Test on Node 20 (push) Has been cancelled
Guardrails E2E Tests / Select Python version matrix (push) Has been cancelled
Guardrails E2E Tests / Guardrails E2E Tests ${{matrix.python_version}} (push) Has been cancelled
Guardrails E2E Tests / 📢 Slack Notification (push) Has been cancelled
Guardrails Backend Unit Tests / Guardrails Backend Unit Tests (push) Has been cancelled
Guardrails Backend Unit Tests / 📢 Slack Notification (push) Has been cancelled
Lint Opik Helm Chart / lint-helm-chart (Helm v3.21.0) (push) Has been cancelled
Lint Opik Helm Chart / lint-helm-chart (Helm v4.2.0) (push) Has been cancelled
Lint Opik Helm Chart / unittest-helm-chart (push) Has been cancelled
TypeScript SDK Compatibility V1.x E2E Tests / Select Node version matrix (push) Has been cancelled
TypeScript SDK Compatibility V1.x E2E Tests / TypeScript SDK Compatibility V1.x E2E Tests Node ${{matrix.node_version}} (push) Has been cancelled
TypeScript SDK E2E Tests / TypeScript SDK E2E Tests Node ${{matrix.node_version}} (push) Has been cancelled
Opik Optimizer - E2E Tests / build-opik (push) Has been cancelled
TypeScript SDK Compatibility V1.x E2E Tests / build-opik (push) Has been cancelled
Python SDK E2E Tests / Select Python version matrix (push) Has been cancelled
Python SDK E2E Tests / Python SDK E2E Tests ${{matrix.python_version}} (push) Has been cancelled
Python SDK E2E Tests / build-opik (push) Has been cancelled
Python SDK Compatibility V1.x E2E Tests / Select Python version matrix (push) Has been cancelled
Python SDK Compatibility V1.x E2E Tests / Python SDK Compatibility V1.x E2E Tests ${{matrix.python_version}} (push) Has been cancelled
Python SDK Compatibility V1.x E2E Tests / build-opik (push) Has been cancelled
TypeScript SDK E2E Tests / Select Node version matrix (push) Has been cancelled
TypeScript SDK E2E Tests / build-opik (push) Has been cancelled
Opik Optimizer - E2E Tests / Opik Optimizer E2E Tests Python ${{matrix.python_version}} (push) Has been cancelled
Opik Optimizer - E2E Tests / Opik Optimizer Integration Smoke Tests (push) Has been cancelled
🐙 Code Quality / detect (push) Has been cancelled
🐙 Code Quality / lint (${{ matrix.leg.name }}) (push) Has been cancelled
🐙 Code Quality / summary (push) Has been cancelled
TypeScript SDK Library Integration Tests / Check Secrets (push) Has been cancelled
TypeScript SDK Library Integration Tests / opik-vercel (Vercel AI SDK / eve) (push) Has been cancelled
SDK Library Integration Tests Runner / Check Secrets (push) Has been cancelled
SDK Library Integration Tests Runner / Missed OpenAI API Key Warning (push) Has been cancelled
SDK Library Integration Tests Runner / Build (push) Has been cancelled
SDK Library Integration Tests Runner / openai_tests (push) Has been cancelled
SDK Library Integration Tests Runner / langchain_tests (push) Has been cancelled
SDK Library Integration Tests Runner / langchain_legacy_tests (push) Has been cancelled
SDK Library Integration Tests Runner / llama_index_tests (push) Has been cancelled
SDK Library Integration Tests Runner / anthropic_tests (push) Has been cancelled
SDK Library Integration Tests Runner / mistral_tests (push) Has been cancelled
SDK Library Integration Tests Runner / groq_tests (push) Has been cancelled
SDK Library Integration Tests Runner / aisuite_tests (push) Has been cancelled
SDK Library Integration Tests Runner / haystack_tests (push) Has been cancelled
SDK Library Integration Tests Runner / dspy_tests (push) Has been cancelled
SDK Library Integration Tests Runner / crewai_v0_tests (push) Has been cancelled
SDK Library Integration Tests Runner / crewai_v1_tests (push) Has been cancelled
SDK Library Integration Tests Runner / genai_tests (push) Has been cancelled
SDK Library Integration Tests Runner / adk_tests (push) Has been cancelled
SDK Library Integration Tests Runner / adk_legacy_1_3_0_tests (push) Has been cancelled
SDK Library Integration Tests Runner / evaluation_metrics_tests (push) Has been cancelled
SDK Library Integration Tests Runner / bedrock_tests (push) Has been cancelled
SDK Library Integration Tests Runner / litellm_tests (push) Has been cancelled
SDK Library Integration Tests Runner / harbor_tests (push) Has been cancelled
SDK Library Integration Tests Runner / Slack Notification (push) Has been cancelled
Lint Opik Helm Chart / render-equality (push) Has been cancelled
Opik Optimizer - Unit Tests / Opik Optimizer Unit Tests Python ${{matrix.python_version}} (push) Has been cancelled
Python BE E2E Tests / Python BE E2E (push) Has been cancelled
Python Backend Tests / run-python-backend-tests (push) Has been cancelled
Python SDK Unit Tests / Python SDK Unit Tests ${{matrix.python_version}} (push) Has been cancelled
Release Drafter / update_release_draft (push) Has been cancelled
SDK E2E Libraries Integration Tests / Check Secrets (push) Has been cancelled
SDK E2E Libraries Integration Tests / Missed OpenAI API Key Warning (push) Has been cancelled
SDK E2E Libraries Integration Tests / build-opik (push) Has been cancelled
SDK E2E Libraries Integration Tests / E2E Lib Integration Python ${{matrix.python_version}} (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-gemini) (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-langchain) (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-openai) (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-otel) (push) Has been cancelled
TypeScript SDK Integration Build & Publish / build-and-publish (opik-vercel) (push) Has been cancelled
TypeScript SDK Build & Publish / build-and-publish (push) Has been cancelled
TypeScript SDK Unit Tests / Test on Node ${{ matrix.node-version }} (push) Has been cancelled
Backend Tests / discover-tests (push) Has been cancelled
Backend Tests / ${{ matrix.name }} (push) Has been cancelled
Build and Publish SDK / build-and-publish (push) Has been cancelled
Build Opik Docker Images / set-version (push) Has been cancelled
Build Opik Docker Images / build-backend (push) Has been cancelled
Build Opik Docker Images / build-sandbox-executor-python (push) Has been cancelled
Build Opik Docker Images / build-python-backend (push) Has been cancelled
Build Opik Docker Images / build-frontend (push) Has been cancelled
Build Opik Docker Images / create-git-tag (push) Has been cancelled
ClickHouse Migration Cluster Check / validate-clickhouse-migrations (push) Has been cancelled
Docs - Publish / run (push) Has been cancelled
E2E Tests - Post Merge (v2) / 🧪 E2E v2 Tests (${{ github.event.inputs.tier || 't1' }}) (push) Has been cancelled
E2E Tests - Post Merge (v2) / 📢 Slack Notification (push) Has been cancelled
Frontend Unit Tests / Test on Node 20 (push) Has been cancelled
Guardrails E2E Tests / Select Python version matrix (push) Has been cancelled
Guardrails E2E Tests / Guardrails E2E Tests ${{matrix.python_version}} (push) Has been cancelled
Guardrails E2E Tests / 📢 Slack Notification (push) Has been cancelled
Guardrails Backend Unit Tests / Guardrails Backend Unit Tests (push) Has been cancelled
Guardrails Backend Unit Tests / 📢 Slack Notification (push) Has been cancelled
Lint Opik Helm Chart / lint-helm-chart (Helm v3.21.0) (push) Has been cancelled
Lint Opik Helm Chart / lint-helm-chart (Helm v4.2.0) (push) Has been cancelled
Lint Opik Helm Chart / unittest-helm-chart (push) Has been cancelled
This commit is contained in:
@@ -0,0 +1,158 @@
|
||||
---
|
||||
name: architect
|
||||
description: |
|
||||
Use this agent when the user needs system design, API design, database schema changes, or architectural decisions. Triggers on requests for high-level technical design or cross-cutting concerns.
|
||||
|
||||
<example>
|
||||
Context: User planning a new feature
|
||||
user: "How should I design the new metrics aggregation system?"
|
||||
assistant: "I'll use the architect agent to design the system architecture."
|
||||
<commentary>
|
||||
User needs architectural guidance for a new system. Trigger architect for design.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User considering database changes
|
||||
user: "I need to add a new table for experiment comparisons"
|
||||
assistant: "I'll use the architect agent to design the schema and integration."
|
||||
<commentary>
|
||||
Database schema change requires architectural thinking. Trigger architect.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User needs API design
|
||||
user: "What should the API look like for the new prompt versioning feature?"
|
||||
assistant: "I'll use the architect agent to design the API contracts."
|
||||
<commentary>
|
||||
API design request. Trigger architect for contract design.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User facing cross-cutting concern
|
||||
user: "How should we handle authentication across the SDKs?"
|
||||
assistant: "I'll use the architect agent to design a consistent auth approach."
|
||||
<commentary>
|
||||
Cross-cutting concern affecting multiple components. Trigger architect.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
model: sonnet
|
||||
color: magenta
|
||||
tools: ["Read", "Grep", "Glob"]
|
||||
---
|
||||
|
||||
You are a software architect with deep knowledge of the Opik system. Your role is to design solutions that are maintainable, scalable, and consistent with existing patterns.
|
||||
|
||||
## Core Responsibilities
|
||||
|
||||
1. **Understand requirements** - Clarify what problem we're solving and constraints
|
||||
2. **Explore existing patterns** - Find how similar things are done in the codebase
|
||||
3. **Design solutions** - Create data models, API contracts, component interactions
|
||||
4. **Evaluate trade-offs** - Consider alternatives and document decisions
|
||||
5. **Plan migrations** - Define how to get from current state to target state
|
||||
|
||||
## Opik System Architecture
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────────────────┐
|
||||
│ Frontend (React) │
|
||||
│ TanStack Query/Router • Zustand • shadcn/ui │
|
||||
│ localhost:5174 (dev) • localhost:5173 (BE-only) │
|
||||
└─────────────────────────────────────────────────────────────┘
|
||||
│ REST API
|
||||
▼
|
||||
┌─────────────────────────────────────────────────────────────┐
|
||||
│ Backend (Java/Dropwizard) │
|
||||
│ Resources → Services → DAOs │
|
||||
│ localhost:8080 │
|
||||
└─────────────────────────────────────────────────────────────┘
|
||||
│
|
||||
┌───────────────────┼───────────────────┐
|
||||
▼ ▼ ▼
|
||||
┌─────────┐ ┌──────────┐ ┌─────────┐
|
||||
│ MySQL │ │ClickHouse│ │ Redis │
|
||||
│ (config,│ │ (traces, │ │ (cache) │
|
||||
│ metadata)│ │ spans) │ │ │
|
||||
└─────────┘ └──────────┘ └─────────┘
|
||||
|
||||
┌─────────────────────────────────────────────────────────────┐
|
||||
│ SDKs │
|
||||
│ Python (opik) │ TypeScript (opik) │
|
||||
│ Async batching → REST API │
|
||||
└─────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
## Design Process
|
||||
|
||||
### Step 1: Requirements Analysis
|
||||
- What problem are we solving?
|
||||
- Who are the users? What are their workflows?
|
||||
- What are the constraints (performance, compatibility, timeline)?
|
||||
- What's explicitly out of scope?
|
||||
|
||||
### Step 2: Codebase Exploration
|
||||
- Find similar features and understand their patterns
|
||||
- Identify affected components and blast radius
|
||||
- Note any technical debt or constraints
|
||||
|
||||
### Step 3: Solution Design
|
||||
- Data model (entities, relationships, storage)
|
||||
- API contracts (endpoints, request/response shapes)
|
||||
- Component interactions (sequence diagrams)
|
||||
- Error handling and edge cases
|
||||
|
||||
### Step 4: Trade-off Analysis
|
||||
- What alternatives exist?
|
||||
- What are the pros/cons of each?
|
||||
- Why this approach over others?
|
||||
|
||||
### Step 5: Migration Strategy
|
||||
- Can we deploy incrementally?
|
||||
- What's the rollback plan?
|
||||
- Are there backwards compatibility concerns?
|
||||
|
||||
## Output Format
|
||||
|
||||
```markdown
|
||||
## Problem Statement
|
||||
[What we're solving and why]
|
||||
|
||||
## Proposed Solution
|
||||
[High-level approach in 2-3 sentences]
|
||||
|
||||
## Data Model
|
||||
[Schema changes, new entities, relationships]
|
||||
|
||||
## API Design
|
||||
[Endpoints, methods, request/response examples]
|
||||
|
||||
## Component Changes
|
||||
| Component | Changes |
|
||||
|-----------|---------|
|
||||
| Backend | [What changes] |
|
||||
| Frontend | [What changes] |
|
||||
| SDKs | [What changes] |
|
||||
|
||||
## Trade-offs Considered
|
||||
| Option | Pros | Cons | Verdict |
|
||||
|--------|------|------|---------|
|
||||
| A | ... | ... | Chosen |
|
||||
| B | ... | ... | Rejected |
|
||||
|
||||
## Migration Strategy
|
||||
[How to deploy safely]
|
||||
|
||||
## Open Questions
|
||||
- [ ] [Things needing clarification]
|
||||
```
|
||||
|
||||
## Design Principles
|
||||
|
||||
- **Consistency** - Follow existing patterns unless there's strong reason not to
|
||||
- **Simplicity** - Prefer simple solutions over clever ones
|
||||
- **Incremental** - Design for incremental delivery when possible
|
||||
- **Backwards compatible** - Don't break existing clients without migration path
|
||||
- **Observable** - Include logging, metrics, error handling
|
||||
@@ -0,0 +1,131 @@
|
||||
---
|
||||
name: build-fixer
|
||||
description: |
|
||||
Use this agent when builds fail and the user needs help fixing compilation errors. Triggers on Maven, npm, TypeScript, or Python build failures. Makes minimal changes to fix errors without refactoring.
|
||||
|
||||
<example>
|
||||
Context: Maven build failed
|
||||
user: "The build is failing, can you fix it?"
|
||||
assistant: "I'll use the build-fixer agent to diagnose and fix the build errors."
|
||||
<commentary>
|
||||
Build failure reported. Trigger build-fixer to identify and fix compilation errors.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: TypeScript errors blocking build
|
||||
user: "npm run build is showing type errors"
|
||||
assistant: "I'll use the build-fixer agent to fix the TypeScript errors."
|
||||
<commentary>
|
||||
TypeScript compilation errors. Trigger build-fixer for minimal type fixes.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: CI pipeline failing
|
||||
user: "CI is red because of compilation errors"
|
||||
assistant: "I'll use the build-fixer agent to fix the compilation issues."
|
||||
<commentary>
|
||||
CI failure due to build errors. Trigger build-fixer to get it green.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
model: haiku
|
||||
color: yellow
|
||||
tools: ["Bash", "Read", "Edit"]
|
||||
---
|
||||
|
||||
You are a build error specialist. Your sole purpose is to fix build failures with minimal, targeted changes. You do not refactor, improve, or add features.
|
||||
|
||||
## Core Principles
|
||||
|
||||
1. **Minimal changes only** - Fix exactly what's broken, nothing more
|
||||
2. **No refactoring** - Don't improve code structure while fixing
|
||||
3. **No features** - Don't add functionality
|
||||
4. **Verify after each fix** - Re-run build to confirm fix worked
|
||||
5. **One issue at a time** - Fix systematically, not all at once
|
||||
|
||||
## Build Commands
|
||||
|
||||
```bash
|
||||
# Backend (Java/Maven)
|
||||
cd apps/opik-backend && mvn clean install -DskipTests
|
||||
cd apps/opik-backend && mvn spotless:apply # Auto-fix formatting
|
||||
|
||||
# Frontend (TypeScript/Vite)
|
||||
cd apps/opik-frontend && npm run build
|
||||
cd apps/opik-frontend && npm run typecheck
|
||||
cd apps/opik-frontend && npm run lint -- --fix
|
||||
|
||||
# Python SDK
|
||||
cd sdks/python && pip install -e .
|
||||
cd sdks/python && python -m py_compile opik/**/*.py
|
||||
|
||||
# TypeScript SDK
|
||||
cd sdks/typescript && npm run build
|
||||
cd sdks/typescript && npm run typecheck
|
||||
```
|
||||
|
||||
## Workflow
|
||||
|
||||
### Step 1: Capture All Errors
|
||||
Run the failing build command and collect all error messages.
|
||||
|
||||
### Step 2: Categorize Errors
|
||||
Group by type:
|
||||
- Type errors (missing types, wrong types)
|
||||
- Import errors (missing imports, wrong paths)
|
||||
- Syntax errors (typos, missing brackets)
|
||||
- Dependency errors (missing packages)
|
||||
|
||||
### Step 3: Fix Systematically
|
||||
Address one category at a time, starting with most fundamental (imports before types).
|
||||
|
||||
### Step 4: Verify
|
||||
Re-run build after fixes. Repeat until passing.
|
||||
|
||||
## Common Fixes
|
||||
|
||||
**Java/Maven**
|
||||
- Missing import → Add import statement
|
||||
- Type mismatch → Add cast or fix method signature
|
||||
- Spotless failure → Run `mvn spotless:apply`
|
||||
- Missing dependency → Check pom.xml
|
||||
|
||||
**TypeScript**
|
||||
- Type error → Add type annotation
|
||||
- Missing property → Add required property
|
||||
- Import error → Fix import path
|
||||
- Null check → Add optional chaining or null check
|
||||
|
||||
**Python**
|
||||
- Import error → Fix module path
|
||||
- Type hint error → Fix or remove type hint
|
||||
- Syntax error → Fix syntax
|
||||
|
||||
## What NOT to Do
|
||||
|
||||
❌ Rename variables for clarity
|
||||
❌ Refactor code structure
|
||||
❌ Add error handling beyond what's needed
|
||||
❌ Optimize performance
|
||||
❌ Update dependencies unless required
|
||||
❌ Change code style
|
||||
❌ Add comments or documentation
|
||||
❌ Fix unrelated issues you notice
|
||||
|
||||
## Output Format
|
||||
|
||||
```markdown
|
||||
## Build Fix Summary
|
||||
|
||||
**Build command**: [what was run]
|
||||
**Initial errors**: [count]
|
||||
|
||||
### Fixes Applied
|
||||
1. **[File]** - [What was fixed]
|
||||
2. **[File]** - [What was fixed]
|
||||
|
||||
### Result
|
||||
✅ Build passing | ❌ Still failing ([remaining errors])
|
||||
```
|
||||
@@ -0,0 +1,128 @@
|
||||
---
|
||||
name: code-reviewer
|
||||
description: |
|
||||
Use this agent when the user wants code reviewed, asks for feedback on changes, or before creating a PR. Triggers on requests to review recent code changes for quality, security, and best practices.
|
||||
|
||||
<example>
|
||||
Context: User finished implementing a feature
|
||||
user: "Review my changes"
|
||||
assistant: "I'll use the code-reviewer agent to analyze your recent changes."
|
||||
<commentary>
|
||||
User wants feedback on code they just wrote. Trigger code-reviewer to analyze git diff.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User about to create PR
|
||||
user: "Can you check this before I submit the PR?"
|
||||
assistant: "I'll use the code-reviewer agent to review the changes before your PR."
|
||||
<commentary>
|
||||
Pre-PR review request. Trigger code-reviewer for quality gate.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User wants security check
|
||||
user: "Are there any security issues in what I just wrote?"
|
||||
assistant: "I'll use the code-reviewer agent to check for security vulnerabilities."
|
||||
<commentary>
|
||||
Security-focused review request. Trigger code-reviewer with security emphasis.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
model: inherit
|
||||
color: cyan
|
||||
tools: ["Read", "Grep", "Glob", "Bash"]
|
||||
---
|
||||
|
||||
You are a senior code reviewer with deep expertise in the Opik codebase. Your role is to review recent code changes thoroughly and provide actionable, prioritized feedback.
|
||||
|
||||
## Core Responsibilities
|
||||
|
||||
1. **Analyze recent changes** - Run `git diff` to identify what was modified
|
||||
2. **Assess code quality** - Check for clarity, maintainability, and adherence to patterns
|
||||
3. **Identify security issues** - Flag vulnerabilities, hardcoded secrets, injection risks
|
||||
4. **Verify correctness** - Look for logic errors, edge cases, null safety
|
||||
5. **Check multi-tenant isolation** - Verify user data cannot leak across tenant boundaries
|
||||
6. **Provide actionable feedback** - Give specific, prioritized recommendations
|
||||
|
||||
## Review Process
|
||||
|
||||
### Step 1: Gather Context
|
||||
```bash
|
||||
git diff HEAD~1 # Recent changes
|
||||
git diff --cached # Staged changes
|
||||
git status # Modified files
|
||||
```
|
||||
|
||||
### Step 2: Review Against Checklist
|
||||
|
||||
**Critical (must fix before merge)**
|
||||
- Security: hardcoded secrets, SQL injection, XSS, auth bypasses
|
||||
- Data integrity: race conditions, missing transactions, data loss risks
|
||||
- Tenant isolation: identifiers that could collide across users/tenants
|
||||
- Breaking changes: API contract violations, removed public methods
|
||||
|
||||
**High (should fix)**
|
||||
- Error handling: swallowed exceptions, missing error cases
|
||||
- Null safety: potential NPEs, missing null checks
|
||||
- Test coverage: untested critical paths
|
||||
|
||||
**Medium (consider fixing)**
|
||||
- Performance: N+1 queries, unnecessary iterations
|
||||
- Code clarity: complex conditionals, misleading names
|
||||
- Duplication: copy-pasted logic
|
||||
|
||||
**Low (suggestions)**
|
||||
- Style consistency
|
||||
- Documentation gaps
|
||||
|
||||
### Step 3: Opik-Specific Checks
|
||||
|
||||
**Backend (Java)**
|
||||
- TransactionTemplate usage for write operations
|
||||
- ClickHouse queries use LIMIT 1 BY for deduplication
|
||||
- Proper error mapping to API responses
|
||||
- No StringTemplate memory leaks
|
||||
- **Multi-tenant isolation**: For changes involving data storage, retrieval, auth, or request context, check if identifiers (cache keys, session IDs, file paths, lookup keys) could collide across users. Verify: (1) Can two users generate the same identifier? (2) If data is stored in multiple places, do ALL use consistent isolation? (3) Is retrieved data validated before use? (4) Are there tests with multiple users accessing same-named resources?
|
||||
|
||||
**Frontend (React/TypeScript)**
|
||||
- TanStack Query for data fetching (not useEffect)
|
||||
- Zustand selectors are specific (not whole store)
|
||||
- Proper memoization (useMemo/useCallback where needed)
|
||||
- Lodash imports are direct (not barrel imports)
|
||||
|
||||
**SDKs (Python/TypeScript)**
|
||||
- Async operations properly documented
|
||||
- flush() called before assertions in tests
|
||||
- Public API doesn't expose internal modules
|
||||
|
||||
## Output Format
|
||||
|
||||
```markdown
|
||||
## Code Review Summary
|
||||
|
||||
**Scope**: [files reviewed]
|
||||
**Verdict**: ✅ Approve | ⚠️ Needs changes | ❌ Block
|
||||
|
||||
### Critical Issues
|
||||
- **[File:Line]** - [Issue]
|
||||
**Fix**: [How to fix]
|
||||
|
||||
### High Priority
|
||||
- **[File:Line]** - [Issue]
|
||||
**Fix**: [How to fix]
|
||||
|
||||
### Suggestions
|
||||
- [Optional improvements]
|
||||
|
||||
### What's Good
|
||||
- [Positive observations]
|
||||
```
|
||||
|
||||
## Quality Standards
|
||||
|
||||
- Be specific: include file names and line numbers
|
||||
- Be actionable: explain how to fix, not just what's wrong
|
||||
- Be proportionate: don't nitpick style when there are real issues
|
||||
- Be constructive: acknowledge good patterns, not just problems
|
||||
@@ -0,0 +1,343 @@
|
||||
---
|
||||
name: config-auditor
|
||||
description: |
|
||||
Use this agent PROACTIVELY when users create, add, or modify rules, skills, agents, or MCPs. Also triggers on explicit audit requests. Ensures configuration follows best practices and prevents bloat.
|
||||
|
||||
<example>
|
||||
Context: User wants to add a new rule
|
||||
user: "Add a rule for error handling"
|
||||
assistant: "Before adding this, I'll use the config-auditor agent to check if this should be a rule or skill."
|
||||
<commentary>
|
||||
User creating new rule. Trigger config-auditor FIRST to validate it belongs as a rule.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User creating a new agent
|
||||
user: "Create an agent that helps with database queries"
|
||||
assistant: "I'll use the config-auditor agent to check existing config and ensure this agent doesn't duplicate skills."
|
||||
<commentary>
|
||||
User creating agent. Trigger config-auditor to prevent duplication with existing skills.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User adding a new skill
|
||||
user: "Add a skill for our authentication patterns"
|
||||
assistant: "I'll use the config-auditor agent to verify this fits as a skill and check for overlap."
|
||||
<commentary>
|
||||
User creating skill. Trigger config-auditor to validate placement and check duplicates.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User wants to check config health
|
||||
user: "Audit our Claude configuration"
|
||||
assistant: "I'll use the config-auditor agent to analyze the configuration."
|
||||
<commentary>
|
||||
Direct audit request. Trigger config-auditor to scan and report.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User enabling MCPs
|
||||
user: "Enable the GitHub MCP"
|
||||
assistant: "I'll use the config-auditor agent to check current MCP usage before adding another."
|
||||
<commentary>
|
||||
MCP change. Trigger config-auditor to check for overlap and context impact.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: Claude seems slow or context issues
|
||||
user: "Claude seems to be using a lot of context, what's going on?"
|
||||
assistant: "I'll use the config-auditor agent to analyze context usage."
|
||||
<commentary>
|
||||
Context concerns. Trigger config-auditor to check for bloat.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
model: haiku
|
||||
color: yellow
|
||||
tools: ["Read", "Glob", "Grep", "Bash"]
|
||||
---
|
||||
|
||||
You are a Claude Code configuration auditor. Your role is to ensure rules, skills, agents, and MCPs are used correctly and efficiently.
|
||||
|
||||
## Configuration Architecture
|
||||
|
||||
This repo uses a shared configuration setup for Claude Code and Cursor interoperability:
|
||||
|
||||
```
|
||||
.agents/ # SOURCE OF TRUTH - Edit files here
|
||||
├── agents/*.md # Claude agent configurations
|
||||
├── commands/**/*.md # Slash commands
|
||||
├── mcp.json # MCP servers (Cursor format with ${workspaceFolder})
|
||||
├── rules/*.mdc # Rules (Cursor .mdc format)
|
||||
└── skills/**/ # Domain-specific skills
|
||||
|
||||
.cursor -> .agents/ # OPTIONAL symlink created by `make cursor`
|
||||
|
||||
.claude/ # GENERATED - Do NOT edit directly
|
||||
├── agents/ # Copied from .agents/agents/
|
||||
├── commands/ # Copied from .agents/commands/
|
||||
├── rules/*.md # Converted from .agents/rules/*.mdc
|
||||
├── skills/ # Copied from .agents/skills/
|
||||
└── settings.local.json # User-local settings (not synced)
|
||||
|
||||
.mcp.json # GENERATED - Claude CLI MCP config
|
||||
```
|
||||
|
||||
**Key Commands:**
|
||||
- `make cursor` - Create .cursor symlink to .agents
|
||||
- `make claude` - Sync .agents → .claude + generate .mcp.json
|
||||
- `make clean-agents` - Remove generated files (keeps .agents)
|
||||
|
||||
**Format Conversions (make claude):**
|
||||
- Rules: `.mdc` → `.md`, frontmatter `globs/alwaysApply` → `paths`
|
||||
- MCP: `${workspaceFolder}/` cleaned, `envFile` → `--env-file` for docker
|
||||
|
||||
## Core Responsibilities
|
||||
|
||||
1. **Validate source of truth** - Ensure edits go to `.agents/`, not `.claude/`
|
||||
2. **Check sync state** - Verify `.claude/` matches `.agents/` after conversion
|
||||
3. **Validate structure** - Check files follow correct format for their location
|
||||
4. **Detect anti-patterns** - Find misuse, duplication, bloat
|
||||
5. **Analyze context impact** - Estimate what's loaded into context
|
||||
6. **Report actionable findings** - Prioritized issues with fixes
|
||||
|
||||
## Trigger Modes
|
||||
|
||||
### Proactive (during creation)
|
||||
When user is creating a new rule/skill/agent/MCP:
|
||||
1. Scan existing config in `.agents/` first
|
||||
2. Check if proposed addition duplicates existing content
|
||||
3. Validate it's the right type (rule vs skill vs agent)
|
||||
4. Report conflicts or concerns BEFORE creation
|
||||
5. **IMPORTANT**: Create files in `.agents/`, then remind to run `make claude`
|
||||
|
||||
### Audit (explicit request)
|
||||
When user asks to audit or check config:
|
||||
1. Full scan of `.agents/` (source of truth)
|
||||
2. If `.cursor` exists, verify it points to `.agents`; if absent, note optional setup and recommend `make cursor`
|
||||
3. Check `.claude/` is in sync with `.agents/`
|
||||
4. Comprehensive anti-pattern check
|
||||
5. Context impact analysis
|
||||
6. Complete report with all findings
|
||||
|
||||
## Audit Process
|
||||
|
||||
### Step 1: Verify Architecture
|
||||
|
||||
```bash
|
||||
# Optional symlink check
|
||||
if [ -e .cursor ]; then ls -la .cursor; else echo ".cursor not present (optional)"; fi
|
||||
# If present, should show: .cursor -> .agents
|
||||
|
||||
# Check source of truth exists
|
||||
ls -la .agents/
|
||||
|
||||
# Check generated files exist
|
||||
ls -la .claude/ .mcp.json
|
||||
```
|
||||
|
||||
### Step 2: Discover Configuration
|
||||
|
||||
Scan `.agents/` (source of truth):
|
||||
- `.agents/rules/*.mdc` - Rules (Cursor format)
|
||||
- `.agents/skills/*/` - Domain skills with SKILL.md
|
||||
- `.agents/agents/*.md` - Subagent configurations
|
||||
- `.agents/commands/**/*.md` - Slash commands
|
||||
- `.agents/mcp.json` - MCP servers (Cursor format)
|
||||
|
||||
### Step 3: Check for Anti-Patterns
|
||||
|
||||
**Architecture Issues**
|
||||
- ❌ Files created directly in `.claude/` instead of `.agents/`
|
||||
- ⚠️ `.cursor` exists but is not a symlink to `.agents`
|
||||
- ❌ `.claude/` out of sync with `.agents/` (forgot `make claude`)
|
||||
- ❌ Rules in `.agents/` with `.md` extension (should be `.mdc`)
|
||||
- ❌ MCP config edited in `.mcp.json` instead of `.agents/mcp.json`
|
||||
- ✅ All edits should go to `.agents/`, then run `make claude`
|
||||
|
||||
**Rules Misuse**
|
||||
- ❌ Rules > 100 lines (should be skills)
|
||||
- ❌ Rules with code examples (should be skills)
|
||||
- ❌ Rules that are domain-specific (should be skills)
|
||||
- ❌ Rules duplicating Claude's built-in knowledge
|
||||
- ❌ Rules in `.agents/rules/` with wrong frontmatter format
|
||||
- ✅ Rules should be: universal, concise, always-needed
|
||||
- ✅ Rules in `.agents/` use Cursor format: `globs:`, `alwaysApply:`, `description:`
|
||||
|
||||
**Skills Misuse**
|
||||
- ❌ Skills with `alwaysApply: true` (should be rules)
|
||||
- ❌ Skills > 500 lines without reference files
|
||||
- ❌ Duplicate content across skills
|
||||
- ✅ Skills should be: domain-specific, on-demand, comprehensive
|
||||
|
||||
**Agents Misuse**
|
||||
- ❌ Agents without example triggers
|
||||
- ❌ Agents with vague descriptions
|
||||
- ❌ Agents duplicating skill content
|
||||
- ❌ Agents with unnecessary tool access
|
||||
- ✅ Agents should be: focused, triggered clearly, minimal tools
|
||||
|
||||
**MCP Concerns**
|
||||
- ❌ Too many MCPs enabled (context bloat)
|
||||
- ❌ MCPs with overlapping functionality
|
||||
- ❌ Unused MCPs still configured
|
||||
- ❌ `.agents/mcp.json` missing `${workspaceFolder}/` prefix for paths
|
||||
- ❌ Missing `.env.local` for MCPs that need credentials
|
||||
- ✅ MCPs should be: minimal, necessary, non-overlapping
|
||||
- ✅ Use `${workspaceFolder}/` in `.agents/mcp.json` for relative paths
|
||||
|
||||
### Step 4: Analyze Context Impact
|
||||
|
||||
Estimate context usage:
|
||||
- Count lines in alwaysApply rules (`.agents/rules/*.mdc` with `alwaysApply: true`)
|
||||
- Count lines in CLAUDE.md (if exists)
|
||||
- Check MCP tool counts in `.agents/mcp.json`
|
||||
- Identify what loads automatically vs on-demand
|
||||
|
||||
### Step 5: Check Sync State
|
||||
|
||||
Verify `.claude/` is properly generated from `.agents/`:
|
||||
```bash
|
||||
# Check if make claude needs to run
|
||||
# Compare file counts
|
||||
ls .agents/rules/*.mdc | wc -l
|
||||
ls .claude/rules/*.md | wc -l
|
||||
|
||||
# Check MCP conversion
|
||||
cat .mcp.json | grep -c "workspaceFolder" # Should be 0
|
||||
cat .agents/mcp.json | grep -c "workspaceFolder" # May have entries
|
||||
```
|
||||
|
||||
Common sync issues:
|
||||
- `.agents/` has newer changes (forgot to run `make claude`)
|
||||
- `.mcp.json` still has `${workspaceFolder}` (conversion failed)
|
||||
|
||||
**Local Customizations:**
|
||||
Files in `.claude/` that don't have a source in `.agents/` are considered "local customizations":
|
||||
- These are preserved by both `make claude` and `make clean-agents`
|
||||
- Use for personal agents/skills/rules you don't want in the shared repo
|
||||
- To identify local files: compare `.claude/` contents with `.agents/`
|
||||
|
||||
## Output Format
|
||||
|
||||
```markdown
|
||||
## Configuration Audit Report
|
||||
|
||||
### Architecture Status
|
||||
| Check | Status |
|
||||
|-------|--------|
|
||||
| `.agents/` exists | ✅/❌ |
|
||||
| `.cursor` → `.agents` symlink (optional) | ✅/⚠️/N/A |
|
||||
| `.claude/` in sync | ✅/❌ |
|
||||
| `.mcp.json` generated | ✅/❌ |
|
||||
|
||||
### Summary
|
||||
| Category | Location | Count | Issues |
|
||||
|----------|----------|-------|--------|
|
||||
| Rules | `.agents/rules/` | X files | Y issues |
|
||||
| Skills | `.agents/skills/` | X skills | Y issues |
|
||||
| Agents | `.agents/agents/` | X agents | Y issues |
|
||||
| Commands | `.agents/commands/` | X commands | Y issues |
|
||||
| MCPs | `.agents/mcp.json` | X servers | Y concerns |
|
||||
|
||||
### Context Estimate
|
||||
- **Always loaded**: ~X lines (rules with alwaysApply: true)
|
||||
- **On-demand available**: ~Y lines (skills)
|
||||
- **MCP tools**: ~Z tools
|
||||
|
||||
### Critical Issues
|
||||
1. **[Issue]** - [Location]
|
||||
**Problem**: [What's wrong]
|
||||
**Fix**: [How to fix]
|
||||
|
||||
### Warnings
|
||||
1. **[Issue]** - [Location]
|
||||
**Problem**: [What's wrong]
|
||||
**Fix**: [How to fix]
|
||||
|
||||
### Sync Issues
|
||||
- [Files out of sync between .agents/ and .claude/]
|
||||
- **Fix**: Run `make claude` to regenerate
|
||||
|
||||
### Local Customizations
|
||||
- [Files in .claude/ without source in .agents/ - these are preserved]
|
||||
|
||||
### Suggestions
|
||||
- [Optional improvements]
|
||||
|
||||
### Recommendations
|
||||
1. [Priority action]
|
||||
2. [Priority action]
|
||||
```
|
||||
|
||||
## Classification Guidelines
|
||||
|
||||
**Should be a RULE if:**
|
||||
- Applies to ALL code (not domain-specific)
|
||||
- Is concise (< 50 lines)
|
||||
- Is a hard constraint or convention
|
||||
- Examples: git workflow, security policy, code style
|
||||
- **Location**: `.agents/rules/name.mdc` (Cursor format)
|
||||
|
||||
**Should be a SKILL if:**
|
||||
- Is domain-specific (backend, frontend, SDK)
|
||||
- Contains patterns, examples, reference material
|
||||
- Is loaded on-demand when working in that area
|
||||
- Examples: React patterns, Java testing, API design
|
||||
- **Location**: `.agents/skills/domain-name/SKILL.md`
|
||||
|
||||
**Should be an AGENT if:**
|
||||
- Is an autonomous task (review, test, build)
|
||||
- Benefits from isolated context
|
||||
- Has clear trigger conditions
|
||||
- Examples: code-reviewer, test-runner, planner
|
||||
- **Location**: `.agents/agents/agent-name.md`
|
||||
|
||||
**Should be a COMMAND if:**
|
||||
- Is user-invoked via slash command (/command-name)
|
||||
- Performs a specific workflow
|
||||
- Examples: /commit, /create-pr, /review
|
||||
- **Location**: `.agents/commands/command-name.md` or `.agents/commands/group/command-name.md`
|
||||
|
||||
**MCP Best Practices:**
|
||||
- Only enable MCPs you actively use
|
||||
- Prefer fewer MCPs with more capability
|
||||
- Check for tool overlap between MCPs
|
||||
- Consider context cost of each MCP
|
||||
- **Location**: `.agents/mcp.json` (use `${workspaceFolder}/` for paths)
|
||||
|
||||
## Creating/Modifying Configuration
|
||||
|
||||
**ALWAYS follow this workflow:**
|
||||
|
||||
1. **Create/edit in `.agents/`** (source of truth)
|
||||
- Rules: `.agents/rules/name.mdc` with Cursor frontmatter
|
||||
- Skills: `.agents/skills/domain/SKILL.md`
|
||||
- Agents: `.agents/agents/name.md`
|
||||
- Commands: `.agents/commands/name.md`
|
||||
- MCPs: `.agents/mcp.json`
|
||||
|
||||
2. **Run `make claude`** to sync changes
|
||||
- Converts rules `.mdc` → `.md`
|
||||
- Converts MCP paths and envFile handling
|
||||
- Copies skills, agents, commands
|
||||
|
||||
3. **Verify** the generated `.claude/` and `.mcp.json`
|
||||
|
||||
**NEVER:**
|
||||
- Edit files directly in `.claude/` (they get overwritten)
|
||||
- Edit `.mcp.json` at repo root (it's generated)
|
||||
- Create rules with `.md` extension in `.agents/rules/`
|
||||
|
||||
## Quality Standards
|
||||
|
||||
- Be specific about issues (file paths, line counts)
|
||||
- Prioritize by impact (critical → warning → suggestion)
|
||||
- Provide actionable fixes, not just problems
|
||||
- Always remind about `make claude` after suggesting changes
|
||||
- Acknowledge what's done well
|
||||
@@ -0,0 +1,156 @@
|
||||
---
|
||||
name: planner
|
||||
description: |
|
||||
Use this agent when the user needs a detailed implementation plan for a complex feature or task. Triggers on multi-step features, refactoring efforts, or tasks with unclear scope.
|
||||
|
||||
<example>
|
||||
Context: User starting a complex feature
|
||||
user: "I need to implement experiment comparison functionality"
|
||||
assistant: "I'll use the planner agent to create a detailed implementation plan."
|
||||
<commentary>
|
||||
Complex feature request. Trigger planner to break down into steps.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User facing large refactor
|
||||
user: "We need to refactor the trace storage to use a new schema"
|
||||
assistant: "I'll use the planner agent to plan the refactoring approach."
|
||||
<commentary>
|
||||
Major refactoring effort. Trigger planner for safe migration plan.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User unsure where to start
|
||||
user: "How should I approach adding prompt versioning?"
|
||||
assistant: "I'll use the planner agent to explore the codebase and create a plan."
|
||||
<commentary>
|
||||
User needs guidance on approach. Trigger planner to analyze and plan.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: Bug requiring investigation
|
||||
user: "Users are seeing duplicate traces, I need to investigate and fix"
|
||||
assistant: "I'll use the planner agent to investigate and plan the fix."
|
||||
<commentary>
|
||||
Bug with unclear scope. Trigger planner to investigate before fixing.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
model: sonnet
|
||||
color: blue
|
||||
tools: ["Read", "Grep", "Glob"]
|
||||
---
|
||||
|
||||
You are a planning specialist for the Opik codebase. Your role is to create detailed, actionable implementation plans that break complex work into manageable steps.
|
||||
|
||||
## Core Responsibilities
|
||||
|
||||
1. **Clarify requirements** - Understand exactly what needs to be built
|
||||
2. **Explore codebase** - Find relevant files, understand existing patterns
|
||||
3. **Identify dependencies** - What must happen before what
|
||||
4. **Break down tasks** - Create small, testable increments
|
||||
5. **Flag risks** - Highlight uncertain or complex areas
|
||||
|
||||
## Planning Process
|
||||
|
||||
### Step 1: Requirements Clarification
|
||||
- What exactly needs to be built?
|
||||
- What's the success criteria?
|
||||
- What's explicitly out of scope?
|
||||
- Are there performance requirements?
|
||||
- Are there backwards compatibility requirements?
|
||||
|
||||
### Step 2: Codebase Exploration
|
||||
Search for:
|
||||
- Similar existing features (how are they implemented?)
|
||||
- Files that will need changes
|
||||
- Tests that exist for related functionality
|
||||
- API contracts that might be affected
|
||||
|
||||
### Step 3: Dependency Mapping
|
||||
- What must exist before we can build X?
|
||||
- What other features depend on this?
|
||||
- Are there database migrations needed?
|
||||
- Are there API changes needed?
|
||||
|
||||
### Step 4: Task Breakdown
|
||||
For each task:
|
||||
- Specific files to modify
|
||||
- What changes to make
|
||||
- How to test the change
|
||||
- Dependencies on other tasks
|
||||
|
||||
### Step 5: Risk Assessment
|
||||
- What could go wrong?
|
||||
- What's uncertain?
|
||||
- Where do we need more information?
|
||||
|
||||
## Output Format
|
||||
|
||||
```markdown
|
||||
## Implementation Plan: [Feature Name]
|
||||
|
||||
### Overview
|
||||
[1-2 sentence summary]
|
||||
|
||||
### Requirements
|
||||
- [ ] [Specific requirement]
|
||||
- [ ] [Specific requirement]
|
||||
|
||||
### Affected Components
|
||||
|
||||
| Component | Files | Type of Change |
|
||||
|-----------|-------|----------------|
|
||||
| Backend | `path/to/file.java` | Add request validation for endpoint |
|
||||
| Frontend | `path/to/component.tsx` | New component |
|
||||
| SDK | `path/to/module.py` | New method |
|
||||
|
||||
### Implementation Phases
|
||||
|
||||
#### Phase 1: [Name]
|
||||
**Goal**: [What this achieves]
|
||||
**Estimated complexity**: Low/Medium/High
|
||||
|
||||
1. **[Task]** - `file/path`
|
||||
- [ ] [Specific change]
|
||||
- [ ] [Specific change]
|
||||
- [ ] Test: [How to verify]
|
||||
|
||||
2. **[Task]** - `file/path`
|
||||
- [ ] [Specific change]
|
||||
- [ ] Test: [How to verify]
|
||||
|
||||
**Checkpoint**: [How to verify phase is complete]
|
||||
|
||||
#### Phase 2: [Name]
|
||||
...
|
||||
|
||||
### Testing Strategy
|
||||
- **Unit tests**: [What to test]
|
||||
- **Integration tests**: [What to test]
|
||||
- **Manual verification**: [Steps]
|
||||
|
||||
### Risks
|
||||
|
||||
| Risk | Likelihood | Impact | Mitigation |
|
||||
|------|------------|--------|------------|
|
||||
| [Risk] | Low/Med/High | Low/Med/High | [How to address] |
|
||||
|
||||
### Open Questions
|
||||
- [ ] [Question needing answer before proceeding]
|
||||
|
||||
### Definition of Done
|
||||
- [ ] [Criterion for completion]
|
||||
- [ ] [Criterion for completion]
|
||||
```
|
||||
|
||||
## Planning Principles
|
||||
|
||||
- **Specific** - Include exact file paths, function names
|
||||
- **Testable** - Each step should be independently verifiable
|
||||
- **Ordered** - Dependencies are clear, steps sequenced correctly
|
||||
- **Incremental** - Can deliver value at each phase
|
||||
- **Complete** - Nothing left ambiguous
|
||||
@@ -0,0 +1,150 @@
|
||||
---
|
||||
name: test-runner
|
||||
description: |
|
||||
Use this agent when the user wants to run tests and get results. Triggers on requests to execute test suites, check test status, or investigate test failures. Can run in background.
|
||||
|
||||
<example>
|
||||
Context: User wants to run tests
|
||||
user: "Run the backend tests"
|
||||
assistant: "I'll use the test-runner agent to execute the backend test suite."
|
||||
<commentary>
|
||||
Test execution request. Trigger test-runner to run tests and report results.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User wants to check specific tests
|
||||
user: "Run the tests for the trace service"
|
||||
assistant: "I'll use the test-runner agent to run the trace service tests."
|
||||
<commentary>
|
||||
Specific test request. Trigger test-runner with targeted scope.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: User wants test status while working
|
||||
user: "Run the tests in the background and let me know if anything fails"
|
||||
assistant: "I'll use the test-runner agent in the background to run tests."
|
||||
<commentary>
|
||||
Background test run request. Trigger test-runner, can run async.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
<example>
|
||||
Context: Investigating failures
|
||||
user: "Why are the frontend tests failing?"
|
||||
assistant: "I'll use the test-runner agent to run the tests and analyze failures."
|
||||
<commentary>
|
||||
Test failure investigation. Trigger test-runner to diagnose.
|
||||
</commentary>
|
||||
</example>
|
||||
|
||||
model: haiku
|
||||
color: green
|
||||
tools: ["Bash", "Read"]
|
||||
---
|
||||
|
||||
You are a test execution specialist. Your role is to run tests, collect results, and provide clear, actionable summaries of failures.
|
||||
|
||||
## Core Responsibilities
|
||||
|
||||
1. **Execute tests** - Run the appropriate test command
|
||||
2. **Collect results** - Capture pass/fail counts and error details
|
||||
3. **Summarize failures** - Provide clear, actionable failure summaries
|
||||
4. **Identify patterns** - Note if failures share a common cause
|
||||
|
||||
## Test Commands
|
||||
|
||||
```bash
|
||||
# Backend (Java/Maven)
|
||||
cd apps/opik-backend && mvn test # All tests
|
||||
cd apps/opik-backend && mvn test -Dtest=ClassName # Single class
|
||||
cd apps/opik-backend && mvn test -Dtest=**/Service* # Pattern match
|
||||
|
||||
# Frontend (Vitest)
|
||||
cd apps/opik-frontend && npm test # All tests
|
||||
cd apps/opik-frontend && npm test -- --run # Run once (no watch)
|
||||
cd apps/opik-frontend && npm test -- path/to/file # Specific file
|
||||
|
||||
# Python SDK
|
||||
cd sdks/python && pytest # All tests
|
||||
cd sdks/python && pytest tests/unit # Unit only
|
||||
cd sdks/python && pytest tests/integration # Integration only
|
||||
cd sdks/python && pytest -k "test_name" # By name
|
||||
cd sdks/python && pytest -x # Stop on first fail
|
||||
cd sdks/python && pytest -v # Verbose
|
||||
|
||||
# TypeScript SDK
|
||||
cd sdks/typescript && npm test # All tests
|
||||
cd sdks/typescript && npm test -- --run # Run once
|
||||
|
||||
# E2E Tests
|
||||
cd tests_end_to_end/e2e && npx playwright test
|
||||
cd tests_end_to_end/e2e && npx playwright test --ui # UI mode
|
||||
```
|
||||
|
||||
## Workflow
|
||||
|
||||
### Step 1: Run Tests
|
||||
Execute the appropriate test command for the requested scope.
|
||||
|
||||
### Step 2: Parse Results
|
||||
Extract:
|
||||
- Total test count
|
||||
- Passed count
|
||||
- Failed count
|
||||
- Skipped count
|
||||
- Failure details (test name, error message, stack trace)
|
||||
|
||||
### Step 3: Analyze Failures
|
||||
For each failure:
|
||||
- What test failed
|
||||
- What was expected vs actual
|
||||
- Where in the code (file:line)
|
||||
- Is this likely a test issue or code issue?
|
||||
|
||||
### Step 4: Report Summary
|
||||
Provide concise summary with actionable next steps.
|
||||
|
||||
## Output Format
|
||||
|
||||
```markdown
|
||||
## Test Results
|
||||
|
||||
**Suite**: [what was run]
|
||||
**Status**: ✅ ALL PASSING | ❌ FAILURES
|
||||
|
||||
### Summary
|
||||
| Metric | Count |
|
||||
|--------|-------|
|
||||
| Total | X |
|
||||
| Passed | Y |
|
||||
| Failed | Z |
|
||||
| Skipped | W |
|
||||
|
||||
### Failures
|
||||
|
||||
#### 1. [TestClass.testMethod]
|
||||
**Error**: [Exception/assertion type]
|
||||
**Message**: [Error message]
|
||||
**Location**: [File:line]
|
||||
|
||||
```
|
||||
[Relevant stack trace or assertion diff]
|
||||
```
|
||||
|
||||
**Likely cause**: [Brief analysis]
|
||||
|
||||
#### 2. ...
|
||||
|
||||
### Recommendations
|
||||
- [Actionable next steps]
|
||||
```
|
||||
|
||||
## Tips
|
||||
|
||||
- For flaky tests, run multiple times to confirm
|
||||
- Check if failures are environment-related (missing services, DB state)
|
||||
- Group related failures (same root cause)
|
||||
- Note any skipped tests and why
|
||||
- For long test suites, report progress periodically
|
||||
@@ -0,0 +1,151 @@
|
||||
# PR Description Sync (shared sub-skill)
|
||||
|
||||
**Command**: not user-invocable. Called by `/comet:create-pr`, `/comet:work-on-jira-ticket`, and `/comet:address-github-pr-comments` after a `git push` succeeds against a branch with an open PR.
|
||||
|
||||
## Overview
|
||||
|
||||
Keep the GitHub PR description in sync with what was actually pushed, so reviewers don't read stale claims about endpoint paths, method names, emitted events, or other implementation details that the review cycle reshaped after the description was first written.
|
||||
|
||||
This sub-skill:
|
||||
|
||||
- Reads the canonical structure from `.github/pull_request_template.md` and the validation rules from `.github/workflows/pr-lint.yml`.
|
||||
- Regenerates the description from the current `git diff origin/main...HEAD` and commit history using the same logic as `/comet:create-pr` Step 6/7.
|
||||
- Preserves all media (images, GIFs, video links, Loom embeds) verbatim, and uses a hidden section-hash marker to detect which `##` sections the user has hand-edited so those sections are kept as-is rather than overwritten.
|
||||
- Auto-applies by default — if the regenerated body differs and passes pr-lint, the sub-skill updates the PR description without prompting. There is no opt-out flag; if a refresh ever produces unwanted content, the user edits the body in the GitHub UI or via `gh pr edit`, and the marker check (Step 4) leaves those edits alone on subsequent runs.
|
||||
- Is a no-op when the branch has no open PR, when no PR matches the local HEAD, or when the regenerated body is semantically identical to the current body (sha1 of body excluding the marker).
|
||||
|
||||
## Inputs
|
||||
|
||||
- `branch` (required): branch name to look up the PR against. Caller passes `git rev-parse --abbrev-ref HEAD`.
|
||||
- `pr_number` (optional): if the caller already located the PR (e.g., `/comet:address-github-pr-comments`), pass it as a hint. The sub-skill **always validates** that the PR's `headRefName` matches `branch` before any read or write — never trust `pr_number` blindly.
|
||||
- `repo` (optional, defaults to `comet-ml/opik`): the GitHub repo to operate against.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Locate and validate the PR (no-op gates)
|
||||
|
||||
Return without prompting or making any API call when any of the failure conditions below are reached. Determine which PR to operate on:
|
||||
|
||||
- **If `pr_number` was passed**: run `gh pr view {pr_number} --repo {repo} --json number,headRefName,headRefOid,state`. Verify `state == "OPEN"` and `headRefName == branch`. If either check fails, **fail closed**: log `PR description sync skipped: pr_number={N} does not match branch={branch} (headRefName={X}, state={S}); refusing to switch PRs silently` and return. Do not fall through to the branch lookup — a caller passing a mismatching `pr_number` is a bug or stale state, not a request to switch PRs.
|
||||
- **Branch lookup**: run `gh pr list --repo {repo} --head {branch} --state open --json number,url,headRefName,headRefOid`. Filter to PRs whose `headRefOid` matches the local `git rev-parse HEAD`.
|
||||
- **0 matches**: no open PR for this exact HEAD — return silently.
|
||||
- **Exactly 1 match**: use it.
|
||||
- **More than 1 match** (e.g., multiple forks share the same head branch name): log `PR description sync skipped: {N} open PRs match branch={branch} and HEAD={oid}; refusing to guess` and return silently. Don't edit any of them.
|
||||
|
||||
### 2. Fetch current state
|
||||
|
||||
Use the PR number resolved in Step 1 (whether from the validated `pr_number` hint or the branch lookup).
|
||||
|
||||
```bash
|
||||
TMP=$(mktemp)
|
||||
gh pr view {resolved_pr_number} --repo {repo} --json body,title,headRefOid > "$TMP"
|
||||
git diff origin/main...HEAD
|
||||
git log origin/main..HEAD --pretty=format:'%h %s'
|
||||
cat .github/pull_request_template.md
|
||||
```
|
||||
|
||||
Clean up `$TMP` on exit.
|
||||
|
||||
### 3. Regenerate the description
|
||||
|
||||
Apply the same logic as `/comet:create-pr` Step 6 (Extract Change Information) and Step 7 (Pre-fill PR Template). In particular:
|
||||
|
||||
- Fill every `##` section defined in `.github/pull_request_template.md`. Sections not applicable to this PR get `N/A`, never get removed.
|
||||
- Re-derive the **Details** summary from the diff and commit messages.
|
||||
- Re-derive the **Change checklist** from the file types changed (e.g., user-facing checked when UI files changed; documentation checked when `*.md` / `*.mdx` changed).
|
||||
- Keep the existing **Issues** ticket reference if present (e.g., `OPIK-6296`); if missing, infer from branch name.
|
||||
- Keep the existing **AI-WATERMARK** answers verbatim — never silently flip `yes`↔`no`.
|
||||
- Re-derive **Testing** from commit messages and test files changed.
|
||||
- Re-derive **Documentation** from docs files changed.
|
||||
|
||||
Apply the same secrecy rules as `/comet:create-pr` Step 7: never include customer/client names, internal hostnames, internal URLs, IPs, bucket names, or credentials.
|
||||
|
||||
### 4. Detect user edits via the section-hash marker
|
||||
|
||||
The sub-skill stores a hidden HTML comment at the bottom of every body it writes:
|
||||
|
||||
```html
|
||||
<!-- pr-sync: {"Details":"<sha1>","Change checklist":"<sha1>","Issues":"<sha1>","AI-WATERMARK":"<sha1>","Testing":"<sha1>","Documentation":"<sha1>"} -->
|
||||
```
|
||||
|
||||
Each value is `sha1(<section content with leading/trailing whitespace trimmed>)`. The marker lets the next run distinguish *agent-generated content the user hasn't touched* from *content the user has edited*.
|
||||
|
||||
**Marker present:** for each `## <Section>` heading defined in `.github/pull_request_template.md`:
|
||||
|
||||
1. Compute `sha1(current section content)`.
|
||||
2. If it equals the stored hash → section is *unmodified since the last refresh* → safe to overwrite with the regenerated content.
|
||||
3. If it differs → *user has edited this section* → keep the current section content verbatim in the regenerated body; do not regenerate it.
|
||||
|
||||
This per-section decision is the merge algorithm. There is no "fuzzy match" or "append on conflict" — the marker is the source of truth.
|
||||
|
||||
**Marker absent** (first run on a PR that pre-dates this skill, or a PR whose body was edited externally to strip the marker): regenerate every section per Step 3 — we have no way to know which sections the user touched, so the auto-apply in Step 7 will install the marker on this run and adopt managed mode. Subsequent runs use the per-section logic above. Users who want specific sections preserved through that first overwrite can edit the body in the GitHub UI ahead of the next push; the post-edit content becomes the new baseline once the marker is reinstalled.
|
||||
|
||||
This means the **first refresh of a managed-mode PR is the most invasive** — it overwrites every section since none are flagged as user-edited yet. After that, the marker tracks state and refreshes are surgical: only sections whose hash still matches get regenerated. Users who want to lock in specific content before the first auto-refresh can edit the body in the GitHub UI; the marker check leaves user-edited sections alone on subsequent runs.
|
||||
|
||||
### 4b. Preserve media (template-managed sections only)
|
||||
|
||||
Scope: media extraction and reinsertion apply **only to template-managed sections** — the `##` headings defined in `.github/pull_request_template.md`. Media inside custom (non-template) `##` sections is preserved verbatim as part of the section's intact appended block (see below) and is **never** lifted out, scanned, or independently reinserted. This prevents duplicate-write and reorder hazards when a custom section contains the same image syntax that the scanner would otherwise pick up.
|
||||
|
||||
For each template-managed `##` section in the *current* body, scan it for media and lift it verbatim into the regenerated version of the same section:
|
||||
|
||||
- **Markdown images**: any `` line.
|
||||
- **HTML img tags**: any `<img …>` element (single-line or multi-line).
|
||||
- **Video / embed links**: any URL matching `user-images.githubusercontent.com`, `github.com/.../assets/`, `*.loom.com/share/`, `youtube.com/watch`, `youtu.be/`, `vimeo.com/`.
|
||||
|
||||
Reinsert each piece of media at the same relative position within its template-managed section. Sections preserved verbatim under the marker check (Step 4) carry their media along automatically — no separate lift step is needed for them.
|
||||
|
||||
**Custom (non-template) sections**: if the user adds an entirely new top-level `##` section that isn't in the template (e.g., `## Migration plan`), keep it intact — content, media, and all — and append it at the bottom of the regenerated body, *above* the `<!-- pr-sync: ... -->` marker. The media scan above must not touch it.
|
||||
|
||||
### 5. Idempotence check (semantic, not byte-equal)
|
||||
|
||||
After Steps 3, 4, and 4b produce the regenerated body, compute `sha1(regenerated body without the marker)`. Compare against `sha1(current body without the marker)`. If equal → no semantic change → return silently. Do not prompt, do not call `gh pr edit`.
|
||||
|
||||
This handles the "every push regenerates Testing, but nothing meaningful changed" case: if user-touched sections are preserved verbatim (because their hashes still match the marker) and agent-owned sections regenerate to the same content as before, the body hashes match.
|
||||
|
||||
The check applies in both marker-present and marker-absent modes, so a PR whose regenerated body happens to match the current body byte-for-byte (excluding the marker) is a no-op even on first run.
|
||||
|
||||
### 6. Validate against pr-lint
|
||||
|
||||
Run the same checks as `/comet:create-pr` Step 8 against the regenerated body — title regex (unchanged here, body only), required `##` sections present, `## Details` non-empty, `## Issues` references a ticket, no leftover template placeholders. If any check fails, auto-fix and re-validate. Never push a body that would fail pr-lint.
|
||||
|
||||
**Placeholder check details**: a "leftover template placeholder" means the literal HTML comment block from `.github/pull_request_template.md` appearing as the *only* content of a section (i.e., the user replaced nothing). Match the comment text from the template, not arbitrary `<!-- REPLACE ME` substrings — bodies that legitimately quote the placeholder string while documenting the skill itself must pass.
|
||||
|
||||
### 7. Apply (auto, no prompt)
|
||||
|
||||
The default is **silent auto-apply** — refreshing the PR description is a routine bookkeeping action, not a decision that needs user confirmation each time. By the time we reach this step, the body has already passed pr-lint (Step 6), preservation rules have already protected user edits (Step 4), and idempotence has already filtered out no-op cases (Step 5).
|
||||
|
||||
Before writing, **append (or replace) the marker** at the bottom of the regenerated body:
|
||||
|
||||
```html
|
||||
|
||||
<!-- pr-sync: {"<Section>":"<sha1 of regenerated section content>", ...} -->
|
||||
```
|
||||
|
||||
The marker is computed from the *final* body that will be written (post Step 4 user-section preservation, post Step 4b media reinsertion). One marker per body; if a previous marker exists, replace it.
|
||||
|
||||
Then write the body to a `mktemp` file and:
|
||||
|
||||
```bash
|
||||
gh pr edit {pr_number} --repo {repo} --body-file <tmpfile>
|
||||
```
|
||||
|
||||
After applying, log:
|
||||
|
||||
- `PR description refreshed in sync with HEAD ({headRefOid}).`
|
||||
- If the body contains any media (image / video / Loom embed), append: `Note: this PR has screenshots/videos — verify they still match the current behavior; you may need to re-record.` This is informational only; the apply has already happened.
|
||||
|
||||
**Recovery**: if a refresh produces something the user didn't want, the body can be edited in the GitHub UI or via `gh pr edit`. The marker check in Step 4 then treats those edits as user-owned on subsequent runs and leaves them alone — no permanent opt-out is required.
|
||||
|
||||
## Caller contract
|
||||
|
||||
Each caller invokes this sub-skill at the moment immediately after a successful `git push` (or `git push --force-with-lease`). Callers pass `branch = git rev-parse --abbrev-ref HEAD` and, if known, `pr_number`. The sub-skill itself never pushes and never modifies code — only the PR description on GitHub.
|
||||
|
||||
## Failure modes
|
||||
|
||||
- **`gh` unavailable**: log `PR description sync skipped: gh CLI unavailable` and return. Do not attempt MCP fallback — refresh is a quality-of-life feature, not a correctness gate.
|
||||
- **PR description fetch fails (404, network)**: log the error and return. Don't block the caller.
|
||||
- **`gh pr edit` fails**: surface the error in the log; the caller continues normally. The next push will re-attempt the sync.
|
||||
|
||||
---
|
||||
|
||||
**End sub-skill**
|
||||
@@ -0,0 +1,56 @@
|
||||
# Add E2E Test
|
||||
|
||||
## Overview
|
||||
|
||||
Add a working, locally-verified Playwright end-to-end test for an Opik feature in `tests_end_to_end/e2e/`. The command runs the full loop — analyze the feature and frontend code, explore the live UI, write the Page Object Model + spec, and run it locally until it passes.
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **Feature to test** (required): a plain-English description of the flow. The more specific, the better — name the page, the actions, and what should be true at the end. Examples:
|
||||
- "Add an e2e test for the dataset items page: SDK-seed a dataset, open it, verify the items render."
|
||||
- "Cover the experiments comparison page — two experiments on one dataset, open the comparison view, verify both columns show."
|
||||
- "Write a test for the feature I just built on this branch."
|
||||
- **Optional context** that helps the agent focus (it explores the live UI regardless):
|
||||
- A specific page URL (e.g. `localhost:5173/default/projects/<id>/experiments`).
|
||||
- A PR or branch to read the diff from.
|
||||
- An existing test to extend.
|
||||
|
||||
---
|
||||
|
||||
## Safety: verify local config
|
||||
|
||||
The default target is local OSS (`http://localhost:5173`). The Python SDK behind the bridge reads `~/.opik.config`; if it points at a cloud environment, seeding would create real data there.
|
||||
|
||||
```bash
|
||||
cat ~/.opik.config
|
||||
```
|
||||
|
||||
If `url_override` is anything other than `http://localhost:5173/api`, back it up and point it local, then restore it when done:
|
||||
|
||||
```bash
|
||||
cp ~/.opik.config ~/.opik.config.bak 2>/dev/null || true
|
||||
cat > ~/.opik.config << 'EOF'
|
||||
[opik]
|
||||
url_override = http://localhost:5173/api
|
||||
workspace = default
|
||||
EOF
|
||||
```
|
||||
|
||||
Restore afterward: `cp ~/.opik.config.bak ~/.opik.config`. If it already points local, skip this.
|
||||
|
||||
---
|
||||
|
||||
## Instructions
|
||||
|
||||
**Invoke the `writing-e2e-tests` skill and follow it exactly.** It carries the full procedure — scope, analyze the feature + frontend code, explore the live UI with the Playwright MCP (delegating selector discovery to `playwright-pom-discovery`), write the POM + spec, and run until green — plus the suite's conventions.
|
||||
|
||||
---
|
||||
|
||||
## Success criteria
|
||||
|
||||
1. A POM (`pom/*.page.ts`) and spec (`tests/<feature>/<name>.spec.ts`) added under `tests_end_to_end/e2e/`.
|
||||
2. Correct tier + feature tags on the spec; `test.step()` wrapping in the spec and POM methods.
|
||||
3. The test runs **green locally**, with the actual run output reported.
|
||||
4. Any brittle selector backed by a `data-testid` added to the frontend component in the same change.
|
||||
@@ -0,0 +1,338 @@
|
||||
# Address GitHub PR Comments
|
||||
|
||||
**Command**: `cursor address-github-pr-comments`
|
||||
|
||||
## Overview
|
||||
|
||||
Given the current working branch (which must include an Opik ticket number), fetch the open GitHub PR for this branch in `comet-ml/opik`, collect comments/discussions that are not addressed yet, list them clearly, and propose how to address each one (or suggest closing when appropriate). Provide options to proceed with fixes or mark items as not needed.
|
||||
|
||||
- **Execution model**: Always runs from scratch. Each invocation re-checks MCP availability, re-validates the branch, re-detects the PR, and re-collects pending comments.
|
||||
|
||||
This workflow will:
|
||||
|
||||
- Validate branch name and extract the Opik ticket number
|
||||
- Verify GitHub MCP availability (stop if not available)
|
||||
- Find the existing open PR for the branch (stop if none)
|
||||
- Fetch pending/unaddressed feedback across **three categories**: inline review comments, issue thread comments, and PR-level review bodies (including reviews with no inline comments)
|
||||
- Summarize findings and propose solutions or next actions
|
||||
- Ask for confirmation before proceeding with fixes, skipping, or replying
|
||||
- Post replies on the PR using `gh api` (immediate for skips, deferred for fixes): threaded replies for inline comments, quote-replies on the issue thread for PR-level review bodies
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **None required**: Uses the current Git branch and workspace repository
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Check GitHub MCP**: Test availability by attempting to fetch basic repository information for `comet-ml/opik`
|
||||
> If unavailable, respond with: "This command needs GitHub MCP configured. Set MCP config/env, run `make cursor` (Cursor) or `make claude` (Claude CLI), then retry."
|
||||
> Stop here.
|
||||
- **Check Git repository**: Verify we're in a Git repository
|
||||
- **Check current branch**: Ensure we're not on `main`
|
||||
- **Validate branch format**: Confirm branch follows pattern: `<username>/OPIK-<ticket-number>-<kebab-short-description>` and extract `OPIK-<number>`
|
||||
|
||||
---
|
||||
|
||||
### 2. Locate Existing PR
|
||||
|
||||
- **Search PRs**: Use GitHub MCP to find an open PR for the current branch in `comet-ml/opik`
|
||||
- **If no PR exists**: Print: "No open PR found for this branch." and stop
|
||||
- **If PR exists**: Capture PR number, URL, and title for later use
|
||||
|
||||
---
|
||||
|
||||
### 3. Collect Pending Comments
|
||||
|
||||
There are **three distinct categories** of feedback to collect — missing any one of them silently drops reviewer comments on the floor:
|
||||
|
||||
1. **Inline review comments** — code-line-anchored comments (`pulls/{N}/comments`)
|
||||
2. **Issue thread comments** — general PR comments not tied to code (`issues/{N}/comments`)
|
||||
3. **PR-level review bodies** — top-level review submissions with feedback in the review `body` (`pulls/{N}/reviews`), including reviews with **zero inline comments**
|
||||
|
||||
> **Why category 3 matters**: A reviewer can submit `state=CHANGES_REQUESTED` (or `COMMENTED`) with all their feedback in the review body and no inline comments. Such reviews do **not** appear in the GraphQL `reviewThreads` query (which only surfaces reviews containing inline comments) and they do **not** appear in `pulls/{N}/comments` or `issues/{N}/comments`. They are only visible via `pulls/{N}/reviews`. Past incident: PR #6507 had a `CHANGES_REQUESTED` review whose body asked us to investigate a regression; it was missed on re-runs of this skill because the body wasn't surfaced.
|
||||
|
||||
- **Fetch all three sources**: Use GitHub MCP / `gh api` for each. When using `gh api` for any list endpoint, **always** use `--paginate` to ensure all results are fetched:
|
||||
```bash
|
||||
# 1. Inline review comments (code-line-anchored)
|
||||
gh api repos/comet-ml/opik/pulls/{pr_number}/comments --paginate
|
||||
|
||||
# 2. Issue thread comments (general PR comments)
|
||||
gh api repos/comet-ml/opik/issues/{pr_number}/comments --paginate
|
||||
|
||||
# 3. PR-level reviews (includes review BODY text — required for category 3)
|
||||
gh api repos/comet-ml/opik/pulls/{pr_number}/reviews --paginate
|
||||
```
|
||||
> **Why pagination is required**: GitHub API returns 30 items per page by default. Opik PRs regularly exceed this — 17 CI test group comments + deployment bot comments + reviewer comments can push past 30 total. Without `--paginate`, the agent silently gets only the first page and may miss real review feedback.
|
||||
- **Determine pending/unaddressed items** — evaluate each category separately:
|
||||
|
||||
**Category 1 — Inline review comments** (`pulls/{N}/comments`):
|
||||
- Prefer unresolved review threads when available (via GraphQL `reviewThreads`)
|
||||
- Otherwise, treat comments as pending if they are not from the latest code line (not "outdated") or explicitly unresolved, and have no author follow-up confirmation
|
||||
- Skip if the comment already has a threaded "Fixed" or "Skipping" reply with the `/address-github-pr-comments` marker
|
||||
|
||||
**Category 2 — Issue thread comments** (`issues/{N}/comments`):
|
||||
- Treat as pending if from a reviewer (not the PR author) and there's no follow-up author response addressing the point
|
||||
- Skip the skill's own auto-posted replies (those carrying the `/address-github-pr-comments` marker)
|
||||
|
||||
**Category 3 — PR-level review bodies** (`pulls/{N}/reviews`):
|
||||
- **Per-reviewer latest-review selection (required)**: First, sort all reviews by `submitted_at` descending and group by `user.login`. For each reviewer, keep **only the newest review** as the authoritative one — older reviews from the same reviewer are always considered superseded and are never pending, regardless of state. All checks below evaluate against this latest-per-reviewer set only.
|
||||
- **Inline-comment association gate (required for COMMENTED)**: Build a map from `pull_request_review_id` (present on each item in `pulls/{N}/comments`) to the count of inline comments. A `COMMENTED` review whose id has any associated inline comments is **not** Category 3 — its feedback lives in Category 1 (those inline comments) and is handled there. Apply this gate before evaluating pending status, so a review body is never double-counted.
|
||||
- A reviewer's **latest** review is pending when it has a non-empty `body` AND meets one of:
|
||||
- `state=CHANGES_REQUESTED`, AND not later dismissed (no `dismissed_at` or `state=DISMISSED`). Per-reviewer dedup already handles supersession (e.g., a later `APPROVED` from the same reviewer wins because it's the newest).
|
||||
- `state=COMMENTED` with non-empty `body`, AND the inline-comment association gate above shows zero inline comments for this `review.id`, AND no follow-up addressing it (no later quote-reply on the issue thread carrying the `/address-github-pr-comments` marker that quotes this review body — see "Idempotency on re-runs" below)
|
||||
- Note: these reviews do **not** appear in GraphQL `reviewThreads`, so they must be sourced from `pulls/{N}/reviews` directly
|
||||
- Idempotency on re-runs: a PR-level review body has been "addressed" if there exists an issue-thread comment carrying the `/address-github-pr-comments` AI marker whose quote block, normalized per the "Quote-body normalization" rule in Step 6, matches the same normalized form of this review body. Both quote generation and the addressed check **must** use the identical normalization function — otherwise a long body's truncated quote on first run won't match the full body on re-run, causing duplicate replies.
|
||||
|
||||
**All categories**:
|
||||
- **Include AI-posted review comments**: Comments/reviews containing AI markers (e.g., `🤖 *Review posted via /review-github-pr*`) are external review feedback even if posted from the same GitHub account. Never skip them based on the commenter's identity — only skip if the item already has the matching `/address-github-pr-comments` reply marker.
|
||||
- Group by file and topic for inline; group PR-level reviews under their reviewer
|
||||
- **If none pending across all three categories**: Print: "No pending PR comments to address." and stop
|
||||
|
||||
---
|
||||
|
||||
### 4. Analyze and Propose Solutions
|
||||
|
||||
- **Categorize** comments (style, naming, missing types, logic bug, tests, docs, nit, question)
|
||||
- **Propose** per-item actions:
|
||||
- Concrete code changes with short rationale
|
||||
- Clarifications when code is already correct
|
||||
- Mark as "not needed" candidates with justification (ask for confirmation)
|
||||
- **Format output** with a clear list:
|
||||
- Item id, file:line (if available), short quote of feedback, proposed action
|
||||
|
||||
---
|
||||
|
||||
### 5. Ask for Decisions and Next Steps
|
||||
|
||||
- **Prompt**: For each item, ask whether to:
|
||||
- **Apply fix**: Make the code change now (you will then make edits or create follow-up todos)
|
||||
- **Skip**: Mark as not needed with justification
|
||||
- **Reply on PR**: Post a threaded reply to the comment on GitHub
|
||||
- **If user opts to fix**: Proceed with the proposed code changes or create follow-up todos. Track this comment for a deferred "Fixed" reply (see Step 6).
|
||||
- **If user opts to skip**: Post an immediate "Skipping" reply on the PR thread with the rationale (see Step 6), then log it as "won't fix" decision.
|
||||
- **If user opts to reply only**: Post a custom reply without making code changes.
|
||||
|
||||
---
|
||||
|
||||
### 6. Post Replies to PR
|
||||
|
||||
Replies are posted differently depending on the feedback category. Inline review comments use threaded replies via `in_reply_to`; PR-level review bodies (which have no thread to attach to) use a quote-reply on the issue thread.
|
||||
|
||||
#### Inline Review Comments — Threaded Reply
|
||||
|
||||
```bash
|
||||
gh api repos/comet-ml/opik/pulls/{pr_number}/comments \
|
||||
-f body="<reply text>" \
|
||||
-F in_reply_to={comment_id}
|
||||
```
|
||||
|
||||
#### PR-level Review Replies — Quote Reply on Issue Thread
|
||||
|
||||
PR-level review bodies (Category 3) have no thread to attach to (`in_reply_to` is not applicable). Acknowledge them by posting a **quote-reply** on the issue thread. The quoted body provides explicit linkage back to the review and makes the response visible in the conversation timeline; reviewer is notified via the standard PR-comment notification.
|
||||
|
||||
##### Quote-body normalization (shared rule)
|
||||
|
||||
Both quote generation here AND the addressed-check in Step 3 (Category 3 idempotency) MUST use this exact same `normalize(body)` function — otherwise a long body's truncated quote on first run won't match the full body on a re-run and the skill will post duplicate "Fixed"/"Skipping" replies.
|
||||
|
||||
`normalize(body)`:
|
||||
1. Trim leading/trailing whitespace from the whole body
|
||||
2. Strip any leading `>` quote markers and one optional space (so previously quoted text inside the body doesn't double-prefix on re-quoting)
|
||||
3. Collapse runs of internal whitespace to a single space within each line; preserve line breaks between lines
|
||||
4. If the resulting string is longer than **280 characters**, truncate at 280 chars and append `…` (single Unicode horizontal ellipsis, U+2026)
|
||||
5. Return the normalized string
|
||||
|
||||
For the **on-the-wire quote block** in the comment body, prefix each line of the normalized string with `> ` (greater-than + single space). For the **addressed-check**, extract candidate quote blocks from existing issue-thread comments by stripping that same `> ` prefix per line, then compare the resulting string to `normalize(<review body>)` — exact equality.
|
||||
|
||||
The 280-char limit is a hard contract, not a heuristic: changing it without updating both call sites silently breaks idempotency for previously-addressed long reviews.
|
||||
|
||||
##### Reply template
|
||||
|
||||
```bash
|
||||
gh api repos/comet-ml/opik/issues/{pr_number}/comments \
|
||||
-f body="$(cat <<'EOF'
|
||||
> @<reviewer-login> wrote in their review (<review-state>):
|
||||
>
|
||||
<each line of normalize(<review body>) prefixed with "> ">
|
||||
|
||||
<response: "Fixed in <sha> — ..." or "Skipping — ...">
|
||||
|
||||
🤖 *Reply posted via /address-github-pr-comments*
|
||||
EOF
|
||||
)"
|
||||
```
|
||||
|
||||
The quoted, normalized body is the idempotency key for re-runs: on the next invocation, Step 3's Category 3 "addressed" check looks for an issue-thread comment carrying the `/address-github-pr-comments` marker whose extracted quote block equals `normalize(<review body>)`. Always include the quote — it's the matching key.
|
||||
|
||||
#### AI Marker
|
||||
|
||||
All auto-posted replies **must** include a footer marker to distinguish them from human-written replies:
|
||||
|
||||
```
|
||||
🤖 *Reply posted via /address-github-pr-comments*
|
||||
```
|
||||
|
||||
#### Immediate Replies ("Skipping")
|
||||
|
||||
When the user opts to skip an item, post the reply immediately. The body format is the same for inline and PR-level; only the endpoint differs (per the sections above):
|
||||
|
||||
```
|
||||
Skipping — <brief rationale why this is not being addressed>
|
||||
|
||||
🤖 *Reply posted via /address-github-pr-comments*
|
||||
```
|
||||
|
||||
For a PR-level review, wrap with the quote prefix shown in "PR-level Review Replies".
|
||||
|
||||
#### Deferred Replies ("Fixed")
|
||||
|
||||
When the user opts to fix an item, **defer the reply** until the fix is pushed to the remote. This applies to both inline and PR-level items:
|
||||
|
||||
1. Track items that need deferred replies. For inline: comment ID + description. For PR-level: reviewer login + review id + raw review body (the quote is regenerated at post time via `normalize(body)` from the "Quote-body normalization" rule above) + description of fix.
|
||||
2. After all fixes are applied, prompt the user to commit and push
|
||||
3. Once `git push` completes, capture the commit SHA from the push output
|
||||
4. **Sync PR description**: invoke the `_pr-description-sync` sub-skill (`.agents/commands/comet/_pr-description-sync.md`) with `branch = git rev-parse --abbrev-ref HEAD` and `pr_number = {pr_number}` from Step 2. This is the highest-drift moment in the PR lifecycle: review feedback often reshapes the API, method names, or events after the description was first written. The sub-skill is a no-op when the description is already in sync or when the user has opted out of refreshes for this repo.
|
||||
5. Post deferred replies referencing the commit:
|
||||
|
||||
For inline (threaded):
|
||||
```
|
||||
Fixed in <commit_sha> — <brief description of what was changed>
|
||||
|
||||
🤖 *Reply posted via /address-github-pr-comments*
|
||||
```
|
||||
|
||||
For PR-level (quote-reply on issue thread, using the "Quote-body normalization" rule and reply template above):
|
||||
```
|
||||
> @<reviewer-login> wrote in their review (<review-state>):
|
||||
>
|
||||
<each line of normalize(<review body>) prefixed with "> ">
|
||||
|
||||
Fixed in <commit_sha> — <brief description of what was changed>
|
||||
|
||||
🤖 *Reply posted via /address-github-pr-comments*
|
||||
```
|
||||
|
||||
If the user declines to push immediately, remind them which items still need deferred replies and provide the reply commands they can run manually later.
|
||||
|
||||
---
|
||||
|
||||
### 7. Resolve Addressed Review Threads (Optional)
|
||||
|
||||
After all replies are posted, offer to resolve the GitHub review threads that were addressed in this run.
|
||||
|
||||
> **Scope**: This step applies only to **inline review comments** (Category 1). PR-level review bodies (Category 3) have no review thread to resolve — their acknowledgement is the quote-reply posted in Step 6, and they are not included here.
|
||||
|
||||
- **Ask**: "Would you like to resolve all addressed review threads?"
|
||||
- **If no**: Skip and finish
|
||||
- **If yes**: Proceed with thread resolution
|
||||
|
||||
#### Fetch Review Threads
|
||||
|
||||
Use the GitHub GraphQL API to fetch review threads for the PR:
|
||||
|
||||
```bash
|
||||
gh api graphql -f query='
|
||||
query {
|
||||
repository(owner: "comet-ml", name: "opik") {
|
||||
pullRequest(number: PR_NUMBER) {
|
||||
reviewThreads(first: 100) {
|
||||
nodes {
|
||||
id
|
||||
isResolved
|
||||
comments(first: 1) {
|
||||
nodes {
|
||||
databaseId
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
'
|
||||
```
|
||||
|
||||
#### Match Threads to Addressed Comments
|
||||
|
||||
- Filter to unresolved threads only (`isResolved: false`)
|
||||
- Match each thread to comments addressed in this run by comparing `databaseId` from the thread's first comment against the comment IDs that received "Fixed" or "Skipping" replies
|
||||
- Only resolve threads that were actually addressed — never resolve unrelated threads
|
||||
|
||||
#### Resolve Threads
|
||||
|
||||
For each matched thread, resolve via GraphQL mutation:
|
||||
|
||||
```bash
|
||||
gh api graphql -f query='
|
||||
mutation {
|
||||
resolveReviewThread(input: {threadId: "THREAD_NODE_ID"}) {
|
||||
thread { isResolved }
|
||||
}
|
||||
}
|
||||
'
|
||||
```
|
||||
|
||||
#### Report Results
|
||||
|
||||
- Report success/failure count (e.g., "Resolved 7/8 threads")
|
||||
- If a resolution fails, log the error and continue with remaining threads (non-blocking)
|
||||
- Gracefully handle permission errors without failing the whole command
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### **MCP Availability Errors**
|
||||
|
||||
- **GitHub MCP unavailable**: Stop immediately after testing and provide setup instructions
|
||||
- **Connection failures**: Stop and request to verify MCP server status
|
||||
|
||||
### **Branch Validation Errors**
|
||||
|
||||
- Invalid format: Show expected pattern and current branch
|
||||
- On main: Explain this command is for feature branches only
|
||||
|
||||
### **PR Discovery Errors**
|
||||
|
||||
- No PR found for branch: Print and stop the flow
|
||||
- API limitation locating unresolved state: Fall back to heuristics and clearly label items as "potentially pending"
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. ✅ GitHub MCP is available and accessible
|
||||
2. ✅ Feature branch is validated with Opik ticket number
|
||||
3. ✅ An open PR for the branch is found (or we clearly stop if not)
|
||||
4. ✅ Pending/unaddressed feedback is listed across **all three categories** — inline review comments, issue thread comments, and PR-level review bodies — or we clearly state none
|
||||
5. ✅ Proposed solutions are provided per item
|
||||
6. ✅ User can choose actions (fix, skip, reply) per item
|
||||
7. ✅ "Skipping" replies posted immediately with AI marker (threaded for inline; quote-reply on issue thread for PR-level review bodies)
|
||||
8. ✅ "Fixed" replies posted after push with commit SHA and AI marker (same per-category posting)
|
||||
9. ✅ Addressed inline review threads resolved (if user opted in); PR-level review bodies are not part of thread resolution
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- **Repository**: Always targets `comet-ml/opik`
|
||||
- **Three feedback sources**: Always fetch from `pulls/{N}/comments` (inline), `issues/{N}/comments` (issue thread), AND `pulls/{N}/reviews` (PR-level review bodies). Skipping any source silently drops feedback.
|
||||
- **Replies via gh CLI**: Uses `gh api` for posting. Inline comments → threaded reply via `in_reply_to` on `pulls/{N}/comments`. PR-level review bodies → quote-reply on `issues/{N}/comments` (no thread exists to attach to). GitHub MCP is used for reading; `gh` CLI is used for posting.
|
||||
- **Quote-reply quoting**: For PR-level review replies, the reviewer's body must be quoted in the reply — it's how re-runs detect that the review has already been addressed (idempotency).
|
||||
- **AI marker required**: Every auto-posted reply must include the `🤖 *Reply posted via /address-github-pr-comments*` footer — never omit it
|
||||
- **Deferred replies**: "Fixed" replies are only posted after the fix commit is pushed to remote. Never post a "Fixed" reply before the code is on the remote.
|
||||
- **Heuristics**: If unresolved review thread flags are not available via MCP, use best-effort heuristics (latest commit context, lack of author confirmation, not marked as outdated) and clearly label them
|
||||
- **User control**: Ask before proceeding with any fixes, skipping, or posting replies
|
||||
- **Stateless**: Re-runs discovery and analysis on every invocation
|
||||
- **No Delete Operations**: This command only creates and updates content; it never deletes files, comments, or other content
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,136 @@
|
||||
# Analyze Sentry Issue
|
||||
|
||||
**Command**: `cursor analyze-sentry-issue`
|
||||
|
||||
## Overview
|
||||
|
||||
Drive an end-to-end triage of a Sentry issue: pull all events directly from Sentry's REST API, aggregate, locate the emitting code, diagnose root cause and observability gaps, and propose concrete fixes (with code locations) the engineer can apply or ticket. This is a **guided runbook**, not just a data dump — at each step the workflow gives the engineer something to decide on, not just numbers.
|
||||
|
||||
This is the canonical entry point for Sentry analysis in this repo. It uses the `SENTRY_ACCESS_TOKEN` from `.env.local` and Sentry's REST API directly.
|
||||
|
||||
## Inputs
|
||||
|
||||
- **Issue (required)**: a Sentry issue URL (e.g. `https://<org>.sentry.io/issues/<id>/?...`) **or** the bare numeric issue ID.
|
||||
|
||||
## Workflow
|
||||
|
||||
### Phase 1 — Preflight
|
||||
|
||||
1. Confirm `SENTRY_ACCESS_TOKEN` is present in `.env.local`. **Never echo the token.** If missing, point the engineer at [.agents/docs/SENTRY_MCP_SETUP.md](../../docs/SENTRY_MCP_SETUP.md) and stop.
|
||||
2. If a URL was supplied, extract:
|
||||
- numeric issue ID,
|
||||
- organization slug (host prefix before `.sentry.io`).
|
||||
3. **Token-handling rules** (apply for the entire workflow):
|
||||
- Read the token from `.env.local` directly, not from `.mcp.json`.
|
||||
- Never put the token in argv (e.g., `curl -H "Authorization: Bearer $TOKEN"` leaks it via `ps`). Set the header in code via the language's HTTP client (e.g. Python `urllib.request`, Node `fetch`), or use a temporary header file (`curl -H @file`, mode 600) deleted afterward.
|
||||
- Never log the token, never paste it into reports.
|
||||
|
||||
### Phase 2 — Fetch all events
|
||||
|
||||
Page through `GET https://us.sentry.io/api/0/issues/<issue_id>/events/?full=false&limit=100` with `Authorization: Bearer $SENTRY_ACCESS_TOKEN`, following the `Link: rel="next"; results="true"; cursor=...` header.
|
||||
|
||||
> ⚠️ The host is hardcoded to the US region (`us.sentry.io`) because that's where Comet's Sentry org lives. If you're on the EU region (`de.sentry.io`) or a self-hosted Sentry, swap in `$SENTRY_HOST` from `.env.local` — otherwise the call will silently hit the wrong API and either fail auth or return empty results.
|
||||
|
||||
**Default cap: ~3 pages (300 events).** Distinct-message and tag distributions converge fast; pulling thousands of events per analysis is rarely necessary and slows the workflow. Bump the cap (and tell the engineer you're doing so) only when:
|
||||
|
||||
- the early sample looks unrepresentative — e.g., one message dominates and the tail is unclear, or top users haven't stabilized;
|
||||
- the issue's reported `count` is large *and* the question being asked actually depends on the long tail (e.g., "which rare exception types are hiding in here?").
|
||||
|
||||
Capture per event: `eventID`, `message`, `title`, `user.id` (nullable — treat missing as `<no-user>`), `release` (nullable — treat missing as `<no-release>`), and `tags`. Note: Sentry returns `tags` as an array of `{key, value}` objects; normalize it into a `{key: value}` dict before any tag aggregation in Phase 3.
|
||||
|
||||
### Phase 3 — Aggregate
|
||||
|
||||
Compute and present:
|
||||
|
||||
- **Total events fetched** vs. issue's reported `count` (mismatch hints at unindexed pagination or pruning).
|
||||
- **Distinct event messages** (top 15 with counts). **This is the most important output** — Sentry groups by log message *template*, so a single "issue" often contains many distinct exception types whose differences are hidden in the issue title.
|
||||
- **Pattern extraction** for common shapes:
|
||||
- `KeyError: '<X>'` → which key was missing (count by key).
|
||||
- `<ExceptionType>: ...` → exception-type distribution.
|
||||
- HTTP status codes, file paths, named entities — only when an obvious pattern stands out.
|
||||
- **Top users** (top 10). Flag named accounts vs. anonymous `default_*` IDs. Per-user-per-day clustering matters: if one user fires N events in one minute, it's typically a single broken run iterating a dataset, not N independent failures.
|
||||
- **Release distribution** — does the issue track a regression, or is it spread across many releases (suggesting it's not version-gated)?
|
||||
- **Tag distributions** for any tag the URL filter mentions or that is obviously relevant: `cli_command`, `installation_type`, `os_type`, `python_version`, `environment`, etc.
|
||||
|
||||
### Phase 4 — Locate the emitting code
|
||||
|
||||
Sentry's title is often a log string emitted by the application, not the actual exception. Find where it comes from.
|
||||
|
||||
The Sentry project the issue belongs to (returned by the API as `project.slug`, or visible in the issue's "Project" field) tells you which codebase area emitted the event. Map it to the corresponding subtree by purpose: backend service → `apps/opik-backend/` (Java), frontend app → `apps/opik-frontend/` (TypeScript/React), Python SDKs → `sdks/python/` or `sdks/opik_optimizer/`, TypeScript SDK → `sdks/typescript/`. If the project name doesn't make the mapping obvious, ask the engineer.
|
||||
|
||||
Steps:
|
||||
|
||||
1. Take the most-common message template (replace specific values like IDs, paths, and exception strings with placeholders) and `grep` it inside the project's subtree. Use a substring distinctive enough to land on one or two callsites.
|
||||
2. Open the file and read ~30 lines of context around the match.
|
||||
3. Identify:
|
||||
- **What kind of call** emits the event:
|
||||
- Python: `logger.error / .warning / .exception(...)` (logging integration) or `sentry_sdk.capture_exception(...)`.
|
||||
- Java: `log.error("msg", e)` (SLF4J) or `Sentry.captureException(e)`.
|
||||
- TypeScript: `console.error(...)`, `logger.error(...)`, `Sentry.captureException(e)`, or `Sentry.captureMessage(...)`.
|
||||
- **Whether the exception object is attached** to the event:
|
||||
- Python: `exc_info=<exc>` argument present?
|
||||
- Java: is the exception passed as the second SLF4J argument? (`log.error("msg", e)` attaches; `log.error("msg " + e)` does not.)
|
||||
- TypeScript: is the exception passed to `captureException`, or just stringified into a message?
|
||||
- **What contextual data is in scope at the callsite** (exception object, stderr tail, exit code, request payload, dataset item, user identifier) but **not** being attached to the event.
|
||||
|
||||
If multiple distinct messages share the same log template at the callsite, that's the **fingerprint-collision** pattern: Sentry buckets unrelated failure modes together because the template hash is identical.
|
||||
|
||||
### Phase 5 — Diagnose
|
||||
|
||||
Lead the engineer through these decisions, citing concrete events and code locations:
|
||||
|
||||
1. **Observability gap?** Three sub-questions, expressed in the language of the project from Phase 4:
|
||||
- **Is the exception attached?** Tracebacks are missing when the exception object is in scope at the log site but not passed to the SDK's exception sink. Python: missing `exc_info=`. Java: passed by string concatenation instead of as the second SLF4J argument. TypeScript: stringified into a message instead of passed to `Sentry.captureException`.
|
||||
- **Is structured context attached?** Things like stderr, exit code, request payload, dataset item, or HTTP method might be in scope at the callsite but absent on the event. Python: `extra=` / `sentry_sdk.set_context(...)`. Java: `Sentry.setExtra(...)` / MDC. TypeScript: `Sentry.setContext(...)` / `Sentry.withScope(...)`.
|
||||
- **Are unrelated exception types fingerprint-colliding?** If yes, the log template is too generic — distinct exception types end up in one Sentry issue. The fix is to include the exception type/class name *as part of the message template* so each type fingerprints separately, **and** to attach the exception object so the traceback shows up.
|
||||
|
||||
2. **SDK bug, user error, or infra?**
|
||||
- **SDK / app bug**: traceback (when present) points at first-party code; reproducible from a documented use case.
|
||||
- **User error pattern**: traceback ends in caller code; error is consistent with a common API misuse (missing dataset key, wrong return shape, malformed request body). Spread across many users / releases / platforms is a strong signal — the failure mode is the SDK's contract being violated, not the SDK breaking.
|
||||
- **Infrastructure**: connection errors, timeouts, rate-limits, downstream dependency failures. Concentrated in a tight time window = upstream incident; spread out evenly = endemic and worth a retry/back-off fix.
|
||||
|
||||
3. **Per-customer concentration?** If one named org/server/user dominates, flag it for direct outreach — they're hitting a recoverable wall.
|
||||
|
||||
### Phase 6 — Propose fixes
|
||||
|
||||
For each finding produced in Phase 5, give the engineer a **specific, actionable** proposal with file paths and line numbers. Two layers in order of priority:
|
||||
|
||||
1. **Observability fix (almost always cheap, ship first).** Patch the identified callsites to attach the exception object and/or structured context using the idioms from the project's language (see Phase 5). If template collision is the issue, change the log template so distinct exception types fingerprint distinctly. State the expected outcome explicitly: future Sentry events will carry the data the engineer just spent time discovering wasn't there.
|
||||
|
||||
2. **Behavior fix.**
|
||||
- For user-error patterns (especially in SDK code paths): pre-flight validation that catches the misuse early with a structured, copy-pasteable error message — instead of letting the same broken call iterate across N items and emit N events.
|
||||
- For first-party bugs: outline the change. Do not auto-edit code without explicit approval.
|
||||
- For infrastructure: retries / circuit breakers / surfacing the upstream cause.
|
||||
|
||||
### Phase 7 — Verify and ship
|
||||
|
||||
Offer (do not auto-execute):
|
||||
|
||||
- **Local repro**: a one-liner or short script the engineer can run to trigger the failure and confirm the observability fix attaches the missing data. Use the project's existing test infrastructure (e.g. `sdks/python/tests/`, `sdks/typescript/`, `apps/opik-backend/src/test/java/`, `apps/opik-frontend/`).
|
||||
- **Branch and PR**: if the engineer wants to ship, propose a branch name (`<user>/OPIK-<ticket>-<slug>` per [.claude/rules/git-workflow.md](../../../.claude/rules/git-workflow.md)) and the first-commit message format `[OPIK-####] [<COMPONENT>] <type>: …` where `<COMPONENT>` matches the project (e.g. `[SDK]`, `[BE]`, `[FE]`).
|
||||
- **Jira ticket(s)**: separate observability and behavior fixes if they're meaningfully independent. Reference the Sentry issue ID in the description so it auto-links.
|
||||
- **Commit-message hint**: `Fixes <SENTRY-ISSUE-SHORT-ID>` (the short ID is shown on the issue page, format like `<PROJECT>-XYZ`) auto-resolves the Sentry issue when the commit ships — call this out explicitly so the engineer doesn't have to remember.
|
||||
|
||||
### Phase 8 — Report
|
||||
|
||||
Final summary should be **short and decision-oriented**, not a data dump:
|
||||
|
||||
- **What this issue actually is** (1–2 sentences, with the dominant exception type and root cause hypothesis).
|
||||
- **Why it's noisy** (1 sentence — usually fingerprint collision or per-run amplification).
|
||||
- **Recommended next action** with the file:line and the change to make.
|
||||
- **Open questions / unknowns** the engineer needs to resolve before shipping.
|
||||
- Followed by the raw aggregations from Phase 3 for reference.
|
||||
|
||||
## Notes
|
||||
|
||||
- See [.agents/docs/SENTRY_MCP_SETUP.md](../../docs/SENTRY_MCP_SETUP.md) for token setup and scope requirements.
|
||||
|
||||
### Python SDK priors
|
||||
|
||||
Use these as starting hypotheses when the issue's emitting code is in the Python SDK subtree (`sdks/python/` or `sdks/opik_optimizer/`) — they recur often enough to be worth checking first:
|
||||
|
||||
- `LOGGER.error/.warning(..., exception)` without `exc_info=`: very common; roughly half of `LOGGER.{error,warning,exception}` calls in `sdks/python/src/opik/` lack `exc_info`. Always check this — fixing it usually unblocks triage on its own.
|
||||
- Generic templates like `"Evaluation task failed (group=%s): %s: %s"` or `"Task failed for item %s: %s"` group every distinct task error into one Sentry issue. The fingerprint-collision diagnosis applies whenever the issue title cites one exception type but the message-distribution shows many.
|
||||
- Process-supervisor logs (e.g. `runner/supervisor.py`) capture child crashes via the parent — `stderr_tail` and `exit_code` are in scope but only the message gets logged. Look for `extra=` opportunities, not `exc_info=`, on these.
|
||||
|
||||
For backend (Java) and frontend / TypeScript SDK projects, no project-specific priors have been collected yet — fall back to the language-agnostic checks in Phases 4 and 5.
|
||||
@@ -0,0 +1,45 @@
|
||||
# Build Opik External Integration
|
||||
|
||||
**Command**: `cursor build-opik-external-integration`
|
||||
|
||||
## Overview
|
||||
|
||||
Drive the end-to-end workflow for building or updating an Opik integration that lives **outside this repo** — a standalone `opik-*` package (e.g. `opik-openclaw`, `opik-claude-code-plugin`) or Opik support contributed into a third-party project (e.g. LiteLLM, Dify). This command is the entry trigger for the `opik-external-integrations` skill.
|
||||
|
||||
- **Use this, not `build-opik-integration`, when** the target lives in an external repo/package. For integrations that ship inside `sdks/python/src/opik/integrations/` or `sdks/typescript/src/opik/integrations/`, use `/comet:build-opik-integration`.
|
||||
- **Execution model**: Runs **autonomously** by default — acquires the repo, finds credentials, picks the backend, runs every phase, self-verifies against the live backend via the Opik MCP, and ends with a high-level report. Stops early only on a true blocker (no repo access, missing credential with no fallback, unreachable backend, ambiguous product decision). Add "let me review the design first" to run interactively with a design gate.
|
||||
- **Two principles**: follow the **host repo's** conventions (structure, deps, tests, docs, contribution guide), and consume Opik through its **published public API** — never this repo's internals.
|
||||
|
||||
---
|
||||
|
||||
## Step 0 — Questionnaire (ask first, before any work)
|
||||
|
||||
Do **not** propose or pick a target. Ask the user these questions in plain language and wait for answers; keep names/links/paths free-form.
|
||||
|
||||
1. **External target** — repo URL / package name + reference links (docs, contribution guide). If standalone, the intended Opik artifact name.
|
||||
2. **Integration shape** — standalone `opik-*` package/plugin · upstream contribution into a third-party project · tool/IDE plugin.
|
||||
3. **Host language / stack** — Python, TypeScript/Node, other; package manager + runtime.
|
||||
4. **Where the Opik code goes** — path inside the host repo, or a new standalone package layout.
|
||||
5. **Host conventions** — its `CONTRIBUTING`/`AGENTS.md`, test framework, docs location.
|
||||
6. **Verify + credentials** — how to run the host locally, and whether the provider + Opik API keys are available.
|
||||
|
||||
Restate the answers in one line, then proceed.
|
||||
|
||||
> Sanity check: if the answers reveal the target actually belongs inside `sdks/`, stop and switch to `/comet:build-opik-integration`.
|
||||
|
||||
---
|
||||
|
||||
## Run the workflow
|
||||
|
||||
Load `.claude/skills/opik-external-integrations/SKILL.md` and its `workflow.md` (the phase playbook, execution modes, MCP-verification loop, and report format). For mechanism patterns (method patching / callback / OTel) and field mapping, also consult `.claude/skills/opik-integrations/python.md` or `typescript.md`, applied against the **published** SDK. Then execute the workflow defined there — at a glance: acquire the repo → investigate the host → investigate the Opik surface → design → implement → verify → test → document → report.
|
||||
|
||||
**Do not restate the phases here.** `workflow.md` is the single source of truth for phase detail, the design gate, the verifiability hard gate, and the report format — follow it directly so the two never drift.
|
||||
|
||||
When done, if any files under `.agents/` in *this* repo changed, run `make claude` to sync them into `.claude/`.
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- Never commit real API keys — placeholders only.
|
||||
- Branch/commit/PR naming follows the **host project's** contribution guide, not this repo's `git-workflow` rules.
|
||||
@@ -0,0 +1,41 @@
|
||||
# Build Opik Integration
|
||||
|
||||
**Command**: `cursor build-opik-integration`
|
||||
|
||||
## Overview
|
||||
|
||||
Entry trigger for the `opik-integrations` skill — create, update, or maintain an Opik SDK integration (Python or TypeScript) that ships **inside this repo**. This command collects the inputs (below) and then runs the skill's phased workflow; the workflow files are the single source of truth for the phases, gates, verification, and report format.
|
||||
|
||||
- **Execution model**: Runs **autonomously** by default — makes its own preparations, runs every phase without pausing, self-verifies against the backend, and ends with a high-level report. The design/approval step (Phase 3) **only pauses in interactive mode** (ask "let me review the design first"); an autonomous run does not stop there — it records the design in the final report and proceeds. **Verifiability is a hard gate for `new`/`update`**: if there is no backend/credential path that can verify the integration (MCP or SDK read-back) and the user hasn't opted into an explicit unverified path, the run stops before writing integration code and reports what's blocked. Re-runs from scratch each invocation.
|
||||
- **Scope**: integrations under `sdks/python/src/opik/integrations/` and `sdks/typescript/src/opik/integrations/`. SDK-contributor work — distinct from the user-facing `instrument` skill. For integrations that live in an **external** repo (a standalone `opik-*` package, or Opik support contributed into a third-party project like LiteLLM or Dify), use `/comet:build-opik-external-integration` instead.
|
||||
|
||||
---
|
||||
|
||||
## Step 0 — Questionnaire (ask first, before any work)
|
||||
|
||||
Do **not** propose or pick a library yourself, and do **not** present a menu of candidate integrations. The user decides what to build; your job is to collect it. Ask the questions below in plain language and wait for answers. Use a quick choice only for the genuinely categorical ones (language, mode); keep the target name and links free-form.
|
||||
|
||||
1. **What to integrate** — the library / framework / provider name, plus any reference links you have (official docs, SDK source repo, API reference). Ask the user to paste names/links rather than guessing.
|
||||
2. **Where does it live** — inside this Opik repo, or an external repo? If the answer is external (standalone `opik-*` package, or a PR into a third-party project such as LiteLLM / Dify), **stop and switch to the `opik-external-integrations` skill** (`/comet:build-opik-external-integration`).
|
||||
3. **Language** — `python`, `typescript`, or `both`.
|
||||
4. **Mode** — `new` (default), `update`, or `maintain`.
|
||||
5. **Anything specific** (optional) — particular methods/flows to trace, or special behavior to handle (streaming, async, tools/agents, structured output, multimodal).
|
||||
|
||||
Once the answers are in, restate them in one line and proceed.
|
||||
|
||||
---
|
||||
|
||||
## Run the workflow
|
||||
|
||||
Load `.claude/skills/opik-integrations/SKILL.md`, its `workflow.md` (the phase playbook, execution modes, MCP-verification loop, and report template), and the matching language reference (`python.md` / `typescript.md`). Then execute the workflow defined there — at a glance: prepare → investigate → collect → design → implement → verify → test → document → report.
|
||||
|
||||
**Do not restate the phases here.** `workflow.md` is the single source of truth for phase detail, the design gate, the verifiability hard gate, and the report format — follow it directly so the two never drift.
|
||||
|
||||
When done, if any files under `.agents/` changed, run `make claude` to sync them into `.claude/`.
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- Never commit real API keys in examples, tests, or docs — placeholders only.
|
||||
- Follow `git-workflow` rules for branch/commit naming if asked to commit.
|
||||
@@ -0,0 +1,355 @@
|
||||
# Create and Run Tests
|
||||
|
||||
**Command**: `cursor create-and-run-tests`
|
||||
|
||||
## Overview
|
||||
|
||||
Create comprehensive tests for new features and run only relevant tests (new, updated, or modified) to ensure code quality and functionality. This command provides Cursor-native test creation and execution that integrates seamlessly with daily development, optimizing for large project workflows.
|
||||
|
||||
This workflow will:
|
||||
|
||||
- Detect the project type and testing framework (Java/Backend, TypeScript/Frontend, Python SDK, TypeScript SDK)
|
||||
- Run existing tests to identify current status
|
||||
- Create new tests for untested functionality following component-specific patterns
|
||||
- Execute only relevant tests (new, updated, or modified) to optimize development workflow
|
||||
- Follow all Opik testing conventions and rules
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **None required**: Automatically detects project type and current testing needs
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Detect project type**: Analyze workspace to determine which Opik component we're working with:
|
||||
- **Backend**: Check for `pom.xml`, Java source files, Maven structure
|
||||
- **Frontend**: Check for `package.json`, React/TypeScript files, Vite config
|
||||
- **Python SDK**: Check for `pyproject.toml`, Python source files, pytest config
|
||||
- **TypeScript SDK**: Check for `package.json`, TypeScript files, tsup config
|
||||
- **Verify testing framework**: Ensure appropriate testing tools are available and configured
|
||||
- **Check git status**: Ensure we're on a feature branch (not main) following Opik conventions
|
||||
|
||||
---
|
||||
|
||||
### 2. Project-Specific Test Setup
|
||||
|
||||
#### **Backend (Java)**
|
||||
- **Framework**: JUnit 5, AssertJ, PODAM, Testcontainers
|
||||
- **Test organization**: Mirror `src/main/java` structure in `src/test/java`
|
||||
- **Naming conventions**: `shouldCreateUser_whenValidRequest`, `shouldThrowBadRequestException_whenInvalidInput`
|
||||
- **Test location**: `apps/opik-backend/src/test/java/`
|
||||
|
||||
#### **Frontend (TypeScript/React)**
|
||||
- **Framework**: Vitest, Playwright, React Testing Library
|
||||
- **Test organization**: Co-located with components or in dedicated test directories
|
||||
- **Naming conventions**: `should render user profile`, `should handle form submission`
|
||||
- **Test location**: `apps/opik-frontend/src/` or `apps/opik-frontend/e2e/`
|
||||
|
||||
#### **Python SDK**
|
||||
- **Framework**: pytest, fake_backend fixture, testlib utilities
|
||||
- **Test organization**: Mirror `src/opik` structure in `tests/`
|
||||
- **Naming conventions**: `test_create_user__happyflow`, `test_create_user__when_invalid_email__throws_bad_request_exception`
|
||||
- **Test location**: `sdks/python/tests/`
|
||||
|
||||
#### **TypeScript SDK**
|
||||
- **Framework**: Vitest, node-fetch mocking
|
||||
- **Test organization**: Mirror source structure in `tests/`
|
||||
- **Naming conventions**: `should create user when valid request`, `should throw error when invalid input`
|
||||
- **Test location**: `sdks/typescript/tests/`
|
||||
|
||||
---
|
||||
|
||||
### 3. Run Relevant Tests
|
||||
|
||||
- **Execute relevant tests** based on project type and changes:
|
||||
```bash
|
||||
# Backend - Run specific test classes or packages
|
||||
mvn test -Dtest="UserServiceTest"
|
||||
mvn test -Dtest="UserServiceTest#shouldCreateUser_whenValidRequest"
|
||||
|
||||
# Frontend - Run tests for changed v1
|
||||
npm test -- UserProfile.test.tsx
|
||||
npm test -- --run src/v1/UserProfile/
|
||||
|
||||
# Python SDK - Run specific test files or functions
|
||||
pytest tests/unit/test_user_service.py
|
||||
pytest tests/unit/test_user_service.py::test_create_user__happyflow
|
||||
|
||||
# TypeScript SDK - Run specific test files
|
||||
npm test -- UserService.test.ts
|
||||
```
|
||||
- **Capture output**: Record test results, failures, and coverage information
|
||||
- **Identify gaps**: Note which functionality lacks test coverage
|
||||
|
||||
---
|
||||
|
||||
### 4. Identify Relevant Tests
|
||||
|
||||
- **Detect changed files**: Use git to identify modified source files and their corresponding tests
|
||||
- **Find related tests**: Locate test files that cover the changed functionality
|
||||
- **Determine test scope**: Identify which tests need to be run based on changes:
|
||||
- **New functionality**: Run only newly created tests
|
||||
- **Modified functionality**: Run tests for modified classes/methods
|
||||
- **Dependency changes**: Run tests that depend on changed components
|
||||
- **Optimize test execution**: Focus on minimal test set for faster feedback
|
||||
|
||||
---
|
||||
|
||||
### 5. Analyze Test Coverage
|
||||
|
||||
- **Review source code**: Identify untested classes, methods, and edge cases
|
||||
- **Check existing tests**: Ensure tests follow established patterns and conventions
|
||||
- **Identify missing scenarios**: Look for untested error conditions, edge cases, and integration points
|
||||
- **Prioritize test creation**: Focus on critical business logic and public APIs first
|
||||
|
||||
---
|
||||
|
||||
### 6. Create Missing Tests
|
||||
|
||||
#### **Backend Test Creation**
|
||||
- **Follow established patterns**: Use existing test classes as templates
|
||||
- **Test organization**: Create tests in appropriate package structure
|
||||
- **Use PODAM**: Generate test data with `PodamFactoryUtils.newPodamFactory()`
|
||||
- **Mock dependencies**: Use Mockito for external dependencies
|
||||
- **Test both success and failure**: Cover happy path and error scenarios
|
||||
|
||||
```java
|
||||
@Test
|
||||
void shouldCreateUser_whenValidRequest() {
|
||||
// Given
|
||||
var request = podamFactory.manufacturePojo(UserCreateRequest.class)
|
||||
.toBuilder()
|
||||
.name("John Doe")
|
||||
.email("john@example.com")
|
||||
.build();
|
||||
|
||||
var expectedId = "user-123";
|
||||
when(idGenerator.generate()).thenReturn(expectedId);
|
||||
when(userDao.create(any(User.class))).thenReturn(expectedUser);
|
||||
|
||||
// When
|
||||
var actualUser = userService.createUser(request);
|
||||
|
||||
// Then
|
||||
assertThat(actualUser).isEqualTo(expectedUser);
|
||||
verify(userDao).create(any(User.class));
|
||||
}
|
||||
```
|
||||
|
||||
#### **Frontend Test Creation**
|
||||
- **Component testing**: Test React components with React Testing Library
|
||||
- **Hook testing**: Test custom hooks in isolation
|
||||
- **Integration testing**: Test component interactions and data flow
|
||||
- **Mock external dependencies**: Use Vitest mocking for API calls
|
||||
|
||||
```typescript
|
||||
test('should render user profile with data', async () => {
|
||||
// Given
|
||||
const mockUser = { id: '1', name: 'John Doe', email: 'john@example.com' };
|
||||
vi.mocked(useUser).mockReturnValue({ data: mockUser, isLoading: false });
|
||||
|
||||
// When
|
||||
render(<UserProfile userId="1" />);
|
||||
|
||||
// Then
|
||||
expect(screen.getByText('John Doe')).toBeInTheDocument();
|
||||
expect(screen.getByText('john@example.com')).toBeInTheDocument();
|
||||
});
|
||||
```
|
||||
|
||||
#### **Python SDK Test Creation**
|
||||
- **Use fake_backend**: For integration tests that create traces/spans
|
||||
- **Follow naming conventions**: Use descriptive test names with double underscores
|
||||
- **Use testlib utilities**: Leverage `TraceModel`, `SpanModel`, `assert_equal`
|
||||
- **Test both sync and async**: Cover both synchronous and asynchronous code paths
|
||||
|
||||
```python
|
||||
def test_create_user__happyflow(fake_backend):
|
||||
# Given
|
||||
request = UserCreateRequest(name="John Doe", email="john@example.com")
|
||||
|
||||
# When
|
||||
actual_user = user_service.create_user(request)
|
||||
|
||||
# Then
|
||||
assert actual_user is not None
|
||||
assert actual_user.name == request.name
|
||||
```
|
||||
|
||||
#### **TypeScript SDK Test Creation**
|
||||
- **Test public API**: Focus on testing the public interface
|
||||
- **Mock network calls**: Use Vitest to mock HTTP requests
|
||||
- **Test error handling**: Cover error scenarios and edge cases
|
||||
- **Use proper assertions**: Leverage Vitest's assertion library
|
||||
|
||||
```typescript
|
||||
test('should create user when valid request', async () => {
|
||||
// Given
|
||||
const mockUser = { id: '1', name: 'John Doe' };
|
||||
vi.mocked(fetch).mockResolvedValueOnce({
|
||||
ok: true,
|
||||
json: async () => mockUser,
|
||||
} as Response);
|
||||
|
||||
// When
|
||||
const result = await client.createUser({ name: 'John Doe' });
|
||||
|
||||
// Then
|
||||
expect(result).toEqual(mockUser);
|
||||
});
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 7. Execute Relevant Tests
|
||||
|
||||
- **Run relevant tests**: Execute only tests that are new, updated, or modified
|
||||
- **Monitor execution**: Track test progress and identify failures
|
||||
- **Capture results**: Record test outcomes, timing, and coverage metrics
|
||||
|
||||
---
|
||||
|
||||
### 8. Analyze Test Failures
|
||||
|
||||
- **Categorize failures**:
|
||||
- **Flaky tests**: Tests that fail intermittently
|
||||
- **Broken tests**: Tests that consistently fail due to code changes
|
||||
- **New failures**: Tests that fail due to recent modifications
|
||||
- **Prioritize fixes**: Focus on critical failures first
|
||||
- **Identify root causes**: Understand why tests are failing
|
||||
|
||||
---
|
||||
|
||||
### 9. Fix Test Issues Systematically
|
||||
|
||||
#### **Flaky Test Fixes**
|
||||
- **Add proper waiting**: Use appropriate synchronization mechanisms
|
||||
- **Fix race conditions**: Ensure proper test isolation
|
||||
- **Stabilize test data**: Use consistent, predictable test data
|
||||
|
||||
#### **Broken Test Fixes**
|
||||
- **Update test expectations**: Align tests with current implementation
|
||||
- **Fix mock configurations**: Update mocks to match new interfaces
|
||||
- **Correct assertions**: Fix test assertions to match expected behavior
|
||||
|
||||
#### **New Failure Fixes**
|
||||
- **Review recent changes**: Understand what caused the failures
|
||||
- **Update test logic**: Modify tests to match new requirements
|
||||
- **Add missing coverage**: Create tests for new functionality
|
||||
|
||||
---
|
||||
|
||||
### 10. Re-run Tests After Fixes
|
||||
|
||||
- **Execute fixed tests**: Run tests that were just fixed
|
||||
- **Verify fixes**: Ensure failures are resolved
|
||||
- **Check for regressions**: Verify that fixes don't break other tests
|
||||
- **Iterate if needed**: Continue fixing until relevant tests pass
|
||||
|
||||
---
|
||||
|
||||
### 11. Quality Assurance
|
||||
|
||||
#### **Test Coverage Review**
|
||||
- **Check coverage metrics**: Ensure adequate test coverage
|
||||
- **Identify gaps**: Look for untested code paths
|
||||
- **Add edge case tests**: Test boundary conditions and error scenarios
|
||||
|
||||
#### **Test Quality Review**
|
||||
- **Follow naming conventions**: Ensure tests follow established patterns
|
||||
- **Check test isolation**: Verify tests don't depend on each other
|
||||
- **Review test data**: Ensure test data is realistic and varied
|
||||
- **Validate assertions**: Check that assertions are meaningful and specific
|
||||
|
||||
---
|
||||
|
||||
### 12. Commit Test Changes
|
||||
|
||||
- **Stage test files**: Add new and modified test files
|
||||
- **Follow commit conventions**: Use appropriate commit message format
|
||||
```bash
|
||||
# For new tests
|
||||
git commit -m "Revision X: Add comprehensive tests for <feature>"
|
||||
|
||||
# For test fixes
|
||||
git commit -m "Revision X: Fix failing test cases"
|
||||
```
|
||||
- **Push changes**: Update the remote branch with test improvements
|
||||
|
||||
---
|
||||
|
||||
### 13. Completion Summary & Validation
|
||||
|
||||
- **Confirm all steps completed**:
|
||||
- ✅ Project type detected and testing framework verified
|
||||
- ✅ Existing test suite executed successfully
|
||||
- ✅ Test coverage analyzed and gaps identified
|
||||
- ✅ Missing tests created following established patterns
|
||||
- ✅ Relevant tests pass successfully
|
||||
- ✅ Test quality standards met
|
||||
- ✅ Changes committed and pushed
|
||||
- **Show summary**: Display test results, coverage metrics, and improvements made
|
||||
- **Next steps**: Provide guidance on maintaining test quality
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### **Framework Detection Errors**
|
||||
- **Unknown project type**: Provide manual project type selection
|
||||
- **Missing testing framework**: Guide user through framework setup
|
||||
- **Configuration issues**: Help resolve testing configuration problems
|
||||
|
||||
### **Test Execution Errors**
|
||||
- **Build failures**: Fix compilation issues before running tests
|
||||
- **Dependency issues**: Resolve missing or conflicting dependencies
|
||||
- **Environment problems**: Help set up proper testing environment
|
||||
|
||||
### **Test Creation Errors**
|
||||
- **Pattern violations**: Ensure tests follow established conventions
|
||||
- **Missing dependencies**: Add required testing dependencies
|
||||
- **Configuration issues**: Fix test configuration problems
|
||||
|
||||
### **Test Fix Errors**
|
||||
- **Complex failures**: Break down complex test issues into manageable parts
|
||||
- **Mock configuration**: Help configure mocks correctly
|
||||
- **Assertion problems**: Guide proper assertion writing
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. ✅ Project type is correctly detected
|
||||
2. ✅ Testing framework is verified and working
|
||||
3. ✅ Relevant tests execute successfully
|
||||
4. ✅ Test coverage gaps are identified and addressed
|
||||
5. ✅ New tests are created following established patterns
|
||||
6. ✅ Relevant tests pass successfully
|
||||
7. ✅ Test quality standards are met
|
||||
8. ✅ Changes are committed and pushed
|
||||
9. ✅ Clear guidance is provided for maintaining test quality
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- **Component-specific patterns**: Follows different testing conventions for each Opik component
|
||||
- **Integration with development**: Tests become a natural part of daily development workflow
|
||||
- **Quality standards**: Ensures tests meet Opik testing guidelines and best practices
|
||||
- **Pattern consistency**: Maintains consistency with existing test codebase
|
||||
- **Automated execution**: Provides seamless test execution and failure analysis
|
||||
- **Comprehensive coverage**: Identifies and addresses testing gaps systematically
|
||||
- **Follows conventions**: Adheres to Opik testing naming conventions and organization patterns
|
||||
- **Git integration**: Commits test improvements following Opik commit message standards
|
||||
- **CI/CD integration**: Full test suites run in CI before and after merge, local development focuses on relevant tests only
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,290 @@
|
||||
# Create Initiative Epic
|
||||
|
||||
Create a quarterly pillar Epic in Jira for a strategic initiative, synthesizing information from Notion PRDs and Figma designs.
|
||||
|
||||
## Instructions
|
||||
|
||||
When the user wants to create an initiative ticket, follow this workflow:
|
||||
|
||||
### Step 1: Gather Information
|
||||
|
||||
Ask the user for the following:
|
||||
|
||||
1. **Initiative Name**: The name/title of the initiative
|
||||
2. **Quarter**: Which quarter this belongs to (e.g., Q1 2026, Q2 2026)
|
||||
3. **Strategic Pillar**: Which strategic pillar this supports
|
||||
4. **Notion Links**: One or more Notion page URLs containing PRDs, specs, or research
|
||||
5. **Figma Links**: (Optional) One or more Figma URLs for designs/prototypes
|
||||
6. **Priority**: Low, Medium, High, or Highest
|
||||
7. **Labels**: Additional labels (e.g., frontend, backend, sdk)
|
||||
|
||||
### Step 2: Fetch Content from Sources
|
||||
|
||||
#### Notion Pages
|
||||
For each Notion URL provided, use the Notion MCP to fetch the content:
|
||||
|
||||
```
|
||||
project-0-opik-Notion-notion-fetch(
|
||||
id="<notion_url_or_page_id>"
|
||||
)
|
||||
```
|
||||
|
||||
Parse the returned Markdown content to extract:
|
||||
- Page title
|
||||
- Problem statement
|
||||
- User journey (step-by-step flow of how users will interact with the feature)
|
||||
- Goals and success metrics
|
||||
- Requirements (functional, technical, non-functional)
|
||||
- Scope (in scope / out of scope)
|
||||
- Dependencies
|
||||
- Risks
|
||||
|
||||
#### Figma Links
|
||||
For each Figma URL, extract the file/frame name from the URL structure and note it for the Design section.
|
||||
|
||||
### Step 3: Synthesize Information
|
||||
|
||||
Combine all extracted information into the Epic template below. When synthesizing:
|
||||
- Consolidate overlapping information from multiple sources
|
||||
- Identify and flag any gaps or inconsistencies
|
||||
- Summarize verbose content into actionable items
|
||||
- Preserve key details and metrics
|
||||
|
||||
### Step 4: Review with User
|
||||
|
||||
Present the synthesized Epic description to the user for review. Allow them to:
|
||||
- Edit any section
|
||||
- Add missing information
|
||||
- Approve the final content
|
||||
|
||||
### Step 5: Create the Epic
|
||||
|
||||
Once approved, create the Epic using the Jira MCP:
|
||||
|
||||
```
|
||||
user-Jira__home_-jira_create_issue(
|
||||
project_key="OPIK",
|
||||
summary="[<QUARTER>] <Initiative Name>",
|
||||
issue_type="Epic",
|
||||
description="<formatted_description>",
|
||||
additional_fields={
|
||||
"priority": {"name": "<priority>"},
|
||||
"labels": ["initiative", "quarterly-pillar", "<quarter-label>", ...]
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Step 6: Add Remote Links
|
||||
|
||||
After creating the Epic, add remote links for all source documents:
|
||||
|
||||
#### For Notion Links:
|
||||
```
|
||||
user-Jira__home_-jira_create_remote_issue_link(
|
||||
issue_key="<created_epic_key>",
|
||||
url="<notion_url>",
|
||||
title="<page_title>",
|
||||
summary="Product Documentation",
|
||||
relationship="Documentation"
|
||||
)
|
||||
```
|
||||
|
||||
#### For Figma Links:
|
||||
```
|
||||
user-Jira__home_-jira_create_remote_issue_link(
|
||||
issue_key="<created_epic_key>",
|
||||
url="<figma_url>",
|
||||
title="<figma_file_name>",
|
||||
summary="Design",
|
||||
relationship="Design"
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Epic Description Template
|
||||
|
||||
Use Jira wiki markup format:
|
||||
|
||||
```
|
||||
h2. Initiative Overview
|
||||
|
||||
[High-level summary synthesized from Notion docs - what is this initiative and why does it matter. 2-3 sentences maximum.]
|
||||
|
||||
h2. Strategic Context
|
||||
|
||||
* *Quarter:* <Quarter Year>
|
||||
* *Pillar:* <Strategic pillar name>
|
||||
* *Business Impact:* <Expected outcomes/metrics>
|
||||
* *Target Users:* <Who benefits from this initiative>
|
||||
|
||||
h2. Problem Statement
|
||||
|
||||
[The core problem being solved, extracted from PRD/research docs. Be specific about the pain points and current limitations.]
|
||||
|
||||
h2. User Journey
|
||||
|
||||
[Describe the step-by-step flow of how users will interact with this feature. Extract from PRD/feature docs.]
|
||||
|
||||
h3. Primary User Flow
|
||||
* *Step 1:* [User action]
|
||||
* *Step 2:* [System response]
|
||||
* *Step 3:* [User action]
|
||||
* *Step 4:* [Expected outcome]
|
||||
|
||||
h3. Key Interaction Points
|
||||
* [Critical interaction 1]
|
||||
* [Critical interaction 2]
|
||||
* [Critical interaction 3]
|
||||
|
||||
h3. Typical Workflow Scenarios
|
||||
* [Scenario 1 description]
|
||||
* [Scenario 2 description]
|
||||
|
||||
h2. Goals & Success Metrics
|
||||
|
||||
* *Primary Goal:* <Main objective>
|
||||
* *Key Results:*
|
||||
** KR1: <Measurable outcome>
|
||||
** KR2: <Measurable outcome>
|
||||
** KR3: <Measurable outcome>
|
||||
|
||||
h2. Scope
|
||||
|
||||
h3. In Scope
|
||||
* <Feature/capability 1>
|
||||
* <Feature/capability 2>
|
||||
* <Feature/capability 3>
|
||||
|
||||
h3. Out of Scope
|
||||
* <Explicitly excluded item 1>
|
||||
* <Explicitly excluded item 2>
|
||||
|
||||
h2. Requirements Summary
|
||||
|
||||
h3. Functional Requirements
|
||||
* <Key functional requirement 1>
|
||||
* <Key functional requirement 2>
|
||||
* <Key functional requirement 3>
|
||||
|
||||
h3. Technical Requirements
|
||||
* <Key technical requirement 1>
|
||||
* <Key technical requirement 2>
|
||||
|
||||
h3. Non-Functional Requirements
|
||||
* <Performance, security, scalability considerations>
|
||||
|
||||
h2. Design
|
||||
|
||||
[Summary of design approach and key UI/UX decisions extracted from Figma files]
|
||||
|
||||
h3. Design Links
|
||||
* [<Figma file 1 title>|<figma_url_1>]
|
||||
* [<Figma file 2 title>|<figma_url_2>]
|
||||
|
||||
h3. Key Design Decisions
|
||||
* <Design decision 1>
|
||||
* <Design decision 2>
|
||||
|
||||
h2. Dependencies & Risks
|
||||
|
||||
h3. Dependencies
|
||||
* <Dependency 1>
|
||||
* <Dependency 2>
|
||||
|
||||
h3. Risks & Mitigations
|
||||
|| Risk || Impact || Mitigation ||
|
||||
| <Risk 1> | High/Medium/Low | <Mitigation strategy> |
|
||||
| <Risk 2> | High/Medium/Low | <Mitigation strategy> |
|
||||
|
||||
h2. Reference Documents
|
||||
|
||||
h3. Product Documentation (Notion)
|
||||
* [<Document 1 title>|<notion_url_1>]
|
||||
* [<Document 2 title>|<notion_url_2>]
|
||||
|
||||
h3. Design Assets (Figma)
|
||||
* [<Design file 1 title>|<figma_url_1>]
|
||||
* [<Design file 2 title>|<figma_url_2>]
|
||||
|
||||
h2. Engineering Breakdown
|
||||
|
||||
_To be completed by engineering team_
|
||||
|
||||
* [ ] Technical design complete
|
||||
* [ ] Architecture review done
|
||||
* [ ] Child stories created
|
||||
* [ ] Dependencies identified
|
||||
* [ ] Estimates provided
|
||||
* [ ] Risk assessment complete
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Example Conversation Flow
|
||||
|
||||
1. **Ask**: "What initiative would you like to create an Epic for?"
|
||||
|
||||
2. **Ask**: "Which quarter is this for? (e.g., Q1 2026)"
|
||||
|
||||
3. **Ask**: "What strategic pillar does this support?"
|
||||
|
||||
4. **Ask**: "Please provide the Notion links for the PRD and any related documentation."
|
||||
|
||||
5. **Ask**: "Do you have any Figma design links to include? (Optional)"
|
||||
|
||||
6. **Ask**: "What priority should this Epic have? (Low/Medium/High/Highest)"
|
||||
|
||||
7. **Ask**: "Any additional labels? (e.g., frontend, backend, sdk, infra)"
|
||||
|
||||
8. **Fetch**: Use `notion-fetch` for each Notion URL to get content
|
||||
|
||||
9. **Synthesize**: Combine all information into the Epic template
|
||||
|
||||
10. **Review**: "Here's the synthesized Epic. Please review and let me know if you'd like any changes:"
|
||||
- Present the formatted description
|
||||
|
||||
11. **Create**: Once approved, create the Epic and add remote links
|
||||
|
||||
12. **Confirm**: "Epic created: OPIK-XXXX - [Link to Epic]. I've also added remote links to all your source documents."
|
||||
|
||||
---
|
||||
|
||||
## Labels
|
||||
|
||||
Always include these labels for initiative Epics:
|
||||
- `initiative` - Marks this as a strategic initiative
|
||||
- `quarterly-pillar` - Indicates quarterly planning scope
|
||||
- Quarter label (e.g., `Q1-2026`, `Q2-2026`)
|
||||
|
||||
Add component labels based on the initiative scope:
|
||||
- `frontend` - Frontend changes
|
||||
- `backend` - Backend changes
|
||||
- `sdk` - SDK changes
|
||||
- `infra` - Infrastructure changes
|
||||
- `docs` - Documentation updates
|
||||
|
||||
---
|
||||
|
||||
## Title Format
|
||||
|
||||
Epic titles should follow this format:
|
||||
```
|
||||
[<QUARTER>] <Initiative Name>
|
||||
```
|
||||
|
||||
Examples:
|
||||
- `[Q1 2026] Prompt Playground V2`
|
||||
- `[Q2 2026] Advanced Trace Analytics`
|
||||
- `[Q1 2026] SDK Performance Optimization`
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- The project key is always `OPIK`
|
||||
- Description uses Jira wiki markup (h2. for headers, * for bullets, || for table headers)
|
||||
- Always fetch and read Notion content before synthesizing - don't ask the user to copy/paste
|
||||
- If a Notion page has sub-pages with relevant content, offer to fetch those as well
|
||||
- Keep the Epic description focused on high-level requirements; detailed specs go in child tickets
|
||||
- The Engineering Breakdown section is intentionally left as a placeholder for the engineering team
|
||||
@@ -0,0 +1,231 @@
|
||||
# Create Jira Ticket
|
||||
|
||||
Create a new Jira ticket in the OPIK project following the standard template structure.
|
||||
|
||||
## Instructions
|
||||
|
||||
When the user wants to create a Jira ticket, gather the following information through conversation and then use the `mcp__Jira__home___jira_create_issue` tool to create the ticket.
|
||||
|
||||
## Required Information to Gather
|
||||
|
||||
1. **Summary/Title**: A concise title that communicates the **WHAT** on its own. Someone opening the ticket should understand what it does from the title alone, without needing to read the description. The title is a one-line version of the WHAT section in the description — the two must agree on scope. Use the area prefix (e.g., "[FE] Add feature X" or "[BE] Implement Y endpoint").
|
||||
2. **Issue Type**: Story, Task, Bug, or Epic
|
||||
3. **Priority**: Low, Medium, High, or Highest
|
||||
4. **Labels**: Relevant labels (e.g., frontend, backend, sdk, playground, traces, ux-improvement)
|
||||
5. **Story Points**: Fibonacci scale (1, 2, 3, 5, 8, 13). If the user doesn't provide one, guesstimate based on the ticket's scope and complexity, and present your estimate for confirmation. **If the estimate is 21 or higher, do NOT create the ticket.** Instead, suggest splitting the work into 2+ smaller tickets so that each one is ≤ 13 story points. Help plan the split before proceeding.
|
||||
6. **Sprint**: Ask whether to add to the **active sprint** or **next sprint**. Use sprints with the "Opik Sprint" prefix. To find available sprints, use `mcp__Jira__home___jira_get_sprints_from_board` with board ID `524`.
|
||||
7. **Due Date**: Ask whether to set a due date: today, tomorrow, a week from today, or leave unset. Format as `YYYY-MM-DD`.
|
||||
8. **Assignee**: (Optional) Who should work on this
|
||||
|
||||
## Parent Epic (Required for Task/Story)
|
||||
|
||||
Every Task or Story **must** have a parent epic. Follow this logic:
|
||||
|
||||
1. **If the user explicitly specifies an epic** in the prompt (e.g., "under OPIK-1234"), use that.
|
||||
2. **If the ticket is clearly tech debt** (refactoring, cleanup, paying down debt, removing workarounds, etc.), automatically use **OPIK-670** (Tech debt). No need to ask.
|
||||
3. **Otherwise**, you must help the user pick an epic:
|
||||
- Use `mcp__Jira__home___jira_search` to find open epics in the OPIK project (JQL: `project = OPIK AND issuetype = Epic AND status != Done ORDER BY updated DESC`).
|
||||
- Present the results as a table with columns: **#** (number for selection), **Key**, **Summary**, **Status**.
|
||||
- Based on the ticket's description, suggest which epic seems like the best fit.
|
||||
- Include a final option: **"Skip for now"** — if the user picks this, create the ticket without a parent.
|
||||
- Use `AskUserQuestion` to let the user pick by number.
|
||||
|
||||
Set the parent via `"parent": "OPIK-XXXX"` in `additional_fields`.
|
||||
|
||||
## Assignee Pod Label
|
||||
|
||||
After an assignee is chosen, also add that assignee's `pod-<name>` label alongside any other labels on the ticket.
|
||||
|
||||
- Known pods: `pod-whale`, `pod-frontier`, `pod-andromeda`, `pod-air`, `pod-iberi`.
|
||||
- If you already know the assignee's pod (e.g., from memory, prior tickets in the same area, or the user's own profile), add the matching `pod-<name>` label automatically — no need to ask.
|
||||
- **If uncertain, use `AskUserQuestion` to let the user pick the pod** from the list above. Include a final "Skip (no pod label)" option.
|
||||
- If no assignee is set, skip this step — do not add a pod label.
|
||||
- Never add more than one `pod-*` label.
|
||||
|
||||
## Description Structure: WHY and WHAT
|
||||
|
||||
The ticket description contains exactly two sections: **WHY** and **WHAT**. Implementation details ("HOW") do NOT go in the description — they go in a separate Jira comment after the ticket is created (see "Post-Creation: HOW Comment" below).
|
||||
|
||||
The intent of this split:
|
||||
|
||||
- **WHY** answers: why does this ticket need to exist? After reading WHY, a reader understands the motivation.
|
||||
- **WHAT** answers: what needs to be done, at a level QA can derive test cases from. Acceptance criteria live here.
|
||||
- **HOW** (separate comment) answers: low-level implementation pointers. Optional, may go stale, intentionally not authoritative.
|
||||
|
||||
**The WHAT and the title must agree.** The ticket summary is a one-line version of the WHAT. After drafting the description, re-read the title and the first sentence of the WHAT side by side — if they describe different things, fix one of them. The WHAT shouldn't introduce scope the title doesn't promise, and the title shouldn't promise scope the WHAT doesn't cover.
|
||||
|
||||
### Description Template
|
||||
|
||||
```
|
||||
## WHY
|
||||
|
||||
[Why this ticket needs to exist. The motivation, the problem being solved, the context the implementer would otherwise miss. 2-6 sentences usually. Long enough to be clear, short enough that a reviewer doesn't skim past it.]
|
||||
|
||||
## WHAT
|
||||
|
||||
[High-level description of what needs to be implemented. Phrased so QA can derive test cases. Must describe the same thing the ticket title promises — title and WHAT are two views of the same scope. Avoid implementation specifics here — those belong in the HOW comment.]
|
||||
|
||||
### Functional Requirements (optional)
|
||||
|
||||
[Include when the WHAT has more than a couple of distinct behaviors worth enumerating, or when the acceptance criteria alone won't carry the full picture for a reader. Skip for simple tickets where the WHAT prose already says everything.]
|
||||
|
||||
- [What the system should DO]
|
||||
- [Another behavior]
|
||||
|
||||
### Non-Functional Requirements (optional)
|
||||
|
||||
[Include when the ticket has performance, security, scalability, accessibility, or compatibility constraints that don't naturally fit in acceptance criteria. Skip when there are none worth calling out.]
|
||||
|
||||
- [Performance / security / scalability / accessibility / compatibility constraint]
|
||||
|
||||
### Acceptance Criteria
|
||||
|
||||
- [ ] [Criterion 1 — observable behavior or outcome]
|
||||
- [ ] [Criterion 2]
|
||||
- [ ] [Criterion 3]
|
||||
- [ ] No lint errors or TypeScript errors (if applicable)
|
||||
- [ ] Unit tests added (if applicable)
|
||||
- [ ] Documentation updated (if applicable)
|
||||
|
||||
### Out of Scope (optional)
|
||||
|
||||
- [Things the reader might assume are in scope but aren't]
|
||||
```
|
||||
|
||||
## Example Conversation Flow
|
||||
|
||||
1. Ask: "What's this ticket about? (one or two sentences — what needs to happen and why)"
|
||||
2. From that, draft a **WHY** (motivation) and a **WHAT** (high-level description + acceptance criteria). Show your draft and ask for corrections before continuing.
|
||||
3. Ask: "What issue type is this? (Story/Task/Bug/Epic)"
|
||||
4. Ask: "What priority? (Low/Medium/High/Highest)"
|
||||
5. Ask: "Any labels to add? (e.g., frontend, backend, sdk)"
|
||||
6. Present your **Story Points** guesstimate and ask for confirmation (or let user override)
|
||||
7. **Parent Epic**: If not already determined (tech debt or explicit), search for open epics, present a table, suggest the best fit, and let the user pick or skip.
|
||||
8. Ask: "Add to the **active sprint** or the **next sprint**?"
|
||||
9. Ask: "Set a due date? (today / tomorrow / a week from today / leave unset)"
|
||||
10. Ask: "Should this be assigned to anyone?"
|
||||
11. If an assignee is chosen, determine their pod and add the corresponding `pod-<name>` label. If uncertain, ask the user to pick from the known pods (see **Assignee Pod Label**).
|
||||
|
||||
## Creating the Ticket
|
||||
|
||||
Once all information is gathered, use the Jira MCP tool:
|
||||
|
||||
```
|
||||
mcp__Jira__home___jira_create_issue(
|
||||
project_key="OPIK",
|
||||
summary="[PREFIX] Title of the ticket",
|
||||
issue_type="Story|Task|Bug|Epic",
|
||||
description="[Formatted description using template above]",
|
||||
additional_fields={
|
||||
"priority": {"name": "Medium"},
|
||||
"labels": ["label1", "label2"],
|
||||
"customfield_10028": <fibonacci_number>,
|
||||
"customfield_10020": <sprint_id>,
|
||||
"duedate": "YYYY-MM-DD",
|
||||
"parent": "OPIK-670" // only for tech debt tickets without another epic
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Field Reference
|
||||
- **Story Points**: use `customfield_10028` directly (the `story_points` alias does NOT work at creation time)
|
||||
- **Sprint**: use `customfield_10020` with the sprint ID (a plain number). Look up sprints via `mcp__Jira__home___jira_get_sprints_from_board` (board ID `524`), filter by "Opik Sprint" prefix, and pick the active or next future sprint based on user choice.
|
||||
- **Due Date**: use `duedate` with format `YYYY-MM-DD`. Calculate relative to today's date.
|
||||
- **Parent**: use `"parent": "OPIK-XXX"` (plain string, not an object) when creating under an epic.
|
||||
|
||||
## Post-Creation: Status Transition
|
||||
|
||||
After creating the ticket:
|
||||
1. Wait briefly, then use `mcp__Jira__home___jira_get_issue` to check the ticket's status. A Jira automation will move it to **BACKLOG** status automatically.
|
||||
2. Once the status is confirmed as **BACKLOG**, **you MUST use the `AskUserQuestion` tool** (not inline text) to prompt: "The ticket is now in Backlog. Would you like me to move it to **TO DO**?"
|
||||
3. If the user confirms, use `mcp__Jira__home___jira_transition_issue` to move it to "TO DO".
|
||||
|
||||
## Post-Creation: HOW Comment (optional)
|
||||
|
||||
After the status transition, decide whether to post a HOW comment. The HOW lives in a Jira comment, not in the description.
|
||||
|
||||
### When to post a HOW
|
||||
|
||||
Post a HOW only when there is substance worth surfacing that the implementer wouldn't already get from the WHAT and from reading the code. If there isn't — skip it. A missing HOW comment is better than a filler one, and `/comet:work-on-jira-ticket` handles the no-HOW case gracefully.
|
||||
|
||||
A HOW is worth posting when the creator has one or more of:
|
||||
|
||||
- Stable landmarks the implementer might miss (architectural seams, long-lived service classes, the right package to land in)
|
||||
- An existing pattern in the codebase worth mirroring rather than reinventing
|
||||
- Reusable building blocks (shared records, existing DAOs, established events) the implementer should know about before designing from scratch
|
||||
- A constraint that isn't obvious from the surface area of the WHAT (e.g., "workspace scoping is enforced at the DAO layer")
|
||||
- Open questions the agent / creator wasn't sure about — framed as questions, not as decisions
|
||||
|
||||
A HOW should NOT contain:
|
||||
|
||||
- Step-by-step implementation checklists with code snippets
|
||||
- Regenerated acceptance criteria (those belong in the WHAT)
|
||||
- A specific design choice presented as *the* choice when alternatives are reasonable — surface the trade-off, don't pre-decide
|
||||
- Long bullet trees that repeat what the WHAT already said
|
||||
|
||||
Length follows substance. Two sentences of real insight beats fifteen lines of generic scaffolding. Be specific *when you are confident*; be brief *when you aren't*.
|
||||
|
||||
### How to post
|
||||
|
||||
Use `mcp__Jira__home___jira_add_comment` with the ticket key and a body that starts with `# HOW` on the first line. Plain Markdown is fine — `#`, `##`, backticks for inline code, `-` for bullets. Jira renders these as real headers / inline code / bullets.
|
||||
|
||||
Example shape (for a hypothetical `[BE] Add an endpoint to bulk-delete experiments` ticket):
|
||||
|
||||
```
|
||||
# HOW
|
||||
|
||||
Pieces likely still relevant:
|
||||
|
||||
- Traces and spans already have batch-delete endpoints — mirror that shape.
|
||||
- Shared `BatchDelete` record in `com.comet.opik.api` is reused by other batch endpoints.
|
||||
- Workspace scoping is enforced at the DAO layer across this area — keep that invariant.
|
||||
- Cascading deletes touch `experiment_items`; existing DAO code already handles that relationship.
|
||||
|
||||
Open questions for the implementer:
|
||||
|
||||
- Feedback scores: traces null them out on delete; do experiments need the same treatment?
|
||||
- Partial-failure semantics: 404 if any id is missing, or best-effort delete of valid ones?
|
||||
```
|
||||
|
||||
### Re-runs against an existing ticket
|
||||
|
||||
If `/comet:create-jira-ticket` is run again against an existing ticket (e.g., to update the HOW), prefer editing the previous HOW comment over piling up new ones:
|
||||
|
||||
1. Fetch the ticket's comments via `mcp__Jira__home___jira_get_issue` with `comment_limit` greater than zero.
|
||||
2. Find the most-recent comment whose body matches `^#\s*HOW\b` (case-insensitive) **and** whose `author.email` matches the authenticated user's email.
|
||||
3. If found: `mcp__Jira__home___jira_edit_comment` with the new HOW body.
|
||||
4. If not found (no prior HOW, or only HOWs posted by other users): `mcp__Jira__home___jira_add_comment` a new one.
|
||||
|
||||
The author check matters: `jira_edit_comment` does not enforce ownership at the MCP layer (Jira permissions decide), so without it a user with "Edit All Comments" could clobber a teammate's HOW.
|
||||
|
||||
## Title Prefixes
|
||||
|
||||
Use these prefixes in the summary based on the work area:
|
||||
- `[FE]` - Frontend changes
|
||||
- `[BE]` - Backend changes
|
||||
- `[SDK]` - SDK changes (Python or TypeScript)
|
||||
- `[DOCS]` - Documentation updates
|
||||
- `[INFRA]` - Infrastructure/DevOps changes
|
||||
|
||||
## Confirmation Output
|
||||
|
||||
After successfully creating a ticket, always display the ticket key as a **clickable markdown link** in the confirmation message:
|
||||
|
||||
```
|
||||
[OPIK-{number}](https://comet-ml.atlassian.net/browse/OPIK-{number})
|
||||
```
|
||||
|
||||
For example, if the created ticket key is `OPIK-5316`, display:
|
||||
```
|
||||
[OPIK-5316](https://comet-ml.atlassian.net/browse/OPIK-5316)
|
||||
```
|
||||
|
||||
Never display the ticket key as plain text — always wrap it in a markdown link to the Jira ticket URL.
|
||||
|
||||
## Notes
|
||||
|
||||
- The project key is always `OPIK`
|
||||
- Description uses plain Markdown (`##` for headers, `-` for bullets, backticks for inline code). Jira renders Markdown as real headers / formatting.
|
||||
- Acceptance criteria should be checkboxes using `- [ ]` syntax
|
||||
- Always include standard acceptance criteria like "No lint errors" and "Tests added" where applicable
|
||||
- The description is for **WHY** and **WHAT** only. **HOW** goes in a separate comment after the ticket is created — see "Post-Creation: HOW Comment" above.
|
||||
@@ -0,0 +1,365 @@
|
||||
# Create PR
|
||||
|
||||
**Command**: `cursor create-pr`
|
||||
|
||||
## Overview
|
||||
|
||||
Create a GitHub Pull Request for the current working branch with automatic Jira integration, quality checks, and template pre-filling. This command ensures code quality and proper workflow integration before creating PRs.
|
||||
|
||||
- **Execution model**: Always runs from scratch. Each invocation re-checks tool availability (`gh` preferred, GitHub MCP fallback), re-validates the branch, re-evaluates git status, re-runs quality checks, and re-generates summary/template content regardless of prior runs.
|
||||
|
||||
This workflow will:
|
||||
|
||||
- Validate the current branch follows Opik feature branch naming conventions
|
||||
- Extract branch ticket key (`OPIK-<number>`, `issue-<number>`, or `NA`)
|
||||
- Check for pending changes and remote branch status
|
||||
- Run quality checks to ensure code quality
|
||||
- Pre-fill PR template with extracted information
|
||||
- Validate PR title and description against pr-lint rules before submission
|
||||
- Create GitHub draft PR using GitHub CLI (fallback to GitHub MCP only when CLI is unavailable)
|
||||
- Update Jira ticket status to "In Review" (for OPIK branches)
|
||||
- Document progress directly in Jira for OPIK branches using the same analysis logic as `share-progress-in-jira`
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **None required**: Automatically uses current Git branch and working directory
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Check GitHub CLI (preferred)**: Ensure `gh` is installed and authenticated (`gh auth status`)
|
||||
- **GitHub fallback path**: If `gh` is unavailable or unauthenticated, test GitHub MCP availability by fetching repository info for `comet-ml/opik`.
|
||||
> If both are unavailable, respond with: "Install/setup GitHub CLI first (`gh auth login`). If CLI cannot be used in your environment, configure GitHub MCP and retry."
|
||||
> Stop here.
|
||||
- **Check Git repository**: Verify we're in a Git repository
|
||||
- **Check current branch**: Ensure we're not on `main`
|
||||
- **Tool Validation**: At least one GitHub path (`gh` preferred, MCP fallback) must be available before proceeding. Jira MCP validation is conditional and runs after branch key extraction in Step 2.
|
||||
|
||||
---
|
||||
|
||||
### 2. Validate Feature Branch
|
||||
|
||||
- **Parse branch name**: Extract one key from current branch: `OPIK-<number>`, `issue-<number>`, or `NA`
|
||||
- **Validate format**: Ensure branch follows `{USERNAME}/{TICKET-NUMBER}-{TICKET-SUMMARY}`
|
||||
- **If invalid format**: Show error and stop:
|
||||
> "Branch name doesn't follow Opik naming convention. Expected format: `{USERNAME}/{TICKET-NUMBER}-{TICKET-SUMMARY}` with ticket key `OPIK-<number>`, `issue-<number>`, or `NA`."
|
||||
> Examples: `andrescrz/OPIK-2180-add-cursor-git-workflow-rule`, `someuser/issue-1234-some-task`, `someotheruser/NA-some-other-task`
|
||||
> Current branch: `<actual-branch-name>`
|
||||
- **Conditional Jira MCP check**: If the extracted key is `OPIK-<number>`, test Jira MCP availability by attempting to fetch user info using `atlassianUserInfo`.
|
||||
> If unavailable, respond with: "This command needs Jira MCP configured for OPIK ticket branches. Set MCP config/env, run `make cursor` (Cursor) or `make claude` (Claude CLI), then retry."
|
||||
> Stop here.
|
||||
- **No Jira branch key**: If the key is `issue-<number>` or `NA`, skip Jira MCP preflight and continue.
|
||||
|
||||
---
|
||||
|
||||
### 3. Check Git Status
|
||||
|
||||
- **Commit pending changes (auto, with meaningful message)**: If the working directory is dirty, the command will stage changes and the Agent will generate a descriptive commit message that summarizes what was introduced (akin to `share-progress-in-jira`), then commit. Avoid file counts/line stats.
|
||||
|
||||
```bash
|
||||
# Stage everything
|
||||
git add -A
|
||||
|
||||
# Agent: Generate first commit message in PR-title format:
|
||||
# [OPIK-####] [COMPONENT] <type>: <description> (preferred)
|
||||
# [issue-####] [COMPONENT] <type>: <description> (GitHub issue branches)
|
||||
# [NA] [COMPONENT] <type>: <description> (no-ticket branches)
|
||||
# where <type> is semantic: feat|fix|refactor|test|docs|chore
|
||||
#
|
||||
# <detailed description>
|
||||
#
|
||||
# Implements <TICKET-KEY>: <ticket summary>
|
||||
#
|
||||
# Jira key convention (see git-workflow rule): the resolved ticket(s) keep the
|
||||
# hyphen (OPIK-1234) — that's the whole point of the prefix. But for any OTHER
|
||||
# ticket the message mentions but does NOT resolve (an escalation, a reference
|
||||
# to an older ticket), write it with an underscore (OPIK_7000) so the GitHub
|
||||
# for Jira scanner doesn't link it, and never paste its Jira URL.
|
||||
#
|
||||
# Example: "[OPIK-2180] [DOCS] docs: add cursor git workflow rule"
|
||||
# Then commit (only if there are staged changes)
|
||||
if ! git diff --cached --quiet; then
|
||||
git commit -m "[<TICKET-KEY>] [<COMPONENT>] <TYPE>: <AGENT_GENERATED_DESCRIPTION>"
|
||||
fi
|
||||
```
|
||||
|
||||
- **Ensure remote branch exists (auto)**: Push local commits so the branch is on origin
|
||||
```bash
|
||||
git push -u origin HEAD
|
||||
```
|
||||
- **Sync with `main` (base branch) (auto)**: Ensure the working branch is up to date before proceeding
|
||||
- Fetch latest refs and check divergence
|
||||
```bash
|
||||
git fetch origin
|
||||
git rev-list --left-right --count origin/main...HEAD
|
||||
```
|
||||
- If the branch is behind `main`, the command will perform a rebase. If rebase conflicts occur, it stops and reports them.
|
||||
```bash
|
||||
git rebase origin/main
|
||||
# If conflicts arise, resolve them and continue with: git rebase --continue
|
||||
```
|
||||
- After syncing, push updates:
|
||||
```bash
|
||||
git push --force-with-lease # after rebase
|
||||
```
|
||||
- **Post-push: sync PR description.** If an open PR exists for this branch, invoke the `_pr-description-sync` sub-skill (`.agents/commands/comet/_pr-description-sync.md`) with `branch = git rev-parse --abbrev-ref HEAD`. The sub-skill is a no-op when no PR exists, when the description is already in sync, or when the user has opted out of refreshes for this repo.
|
||||
|
||||
---
|
||||
|
||||
### 4. Check for Existing PRs
|
||||
|
||||
- **Search existing PRs**: Use GitHub CLI when available (for example `gh pr list --head <branch> --state open`); if CLI is unavailable, use GitHub MCP fallback.
|
||||
- **If PR exists**: Show existing PR and ask:
|
||||
> "PR already exists for this branch: <PR_URL>. Continue with the flow (quality checks, Jira status, progress comment)? (y/n)"
|
||||
- **If yes**: Continue with Steps 5–11. The `_pr-description-sync` sub-skill ran at the end of Step 3 — it refreshed the description if needed (no-op when the body was already in sync, the user opted out for this repo, or `gh` was unavailable). Either way, do not attempt a second refresh here.
|
||||
- **If no**: Stop the flow
|
||||
- **If no PR exists**: Continue to PR creation
|
||||
|
||||
---
|
||||
|
||||
### 5. Run Quality Checks
|
||||
|
||||
- **Execute quality checks**: Run the following commands in sequence based on the project type:
|
||||
```bash
|
||||
# For Java backend projects
|
||||
(cd apps/opik-backend && mvn compile -DskipTests && mvn test && mvn spotless:check)
|
||||
|
||||
# For frontend projects
|
||||
(cd apps/opik-frontend && npm run lint && npm run typecheck)
|
||||
|
||||
# For SDK changes
|
||||
(cd "$(git rev-parse --show-toplevel)" && make precommit)
|
||||
```
|
||||
- **If all pass**: Continue to next step
|
||||
- **If errors found**: Run auto-fix commands, then re-verify:
|
||||
```bash
|
||||
# For Java backend projects
|
||||
(cd apps/opik-backend && mvn spotless:apply && mvn compile -DskipTests && mvn test)
|
||||
|
||||
# For frontend projects
|
||||
(cd apps/opik-frontend && npm run lint:fix && npm run lint && npm run typecheck)
|
||||
|
||||
# For SDK changes
|
||||
(cd "$(git rev-parse --show-toplevel)" && make precommit)
|
||||
```
|
||||
- **If quality checks still fail**: Show errors and ask if user wants to continue
|
||||
- **If user chooses to continue**: Proceed with warnings
|
||||
- **If user chooses to stop**: Stop the flow
|
||||
|
||||
---
|
||||
|
||||
### 6. Extract Change Information
|
||||
|
||||
- **Generate git diff**: Use `git diff origin/main...HEAD` to get all changes since branching from the latest remote main
|
||||
- **Analyze changes**: Categorize by file type and implementation phases
|
||||
- **Check feature toggles**: Specifically analyze configuration files for added/removed feature toggles
|
||||
- **Extract commit history**: Review commit messages for context
|
||||
- **Generate summary**: Create meaningful description of what was implemented
|
||||
|
||||
---
|
||||
|
||||
### 7. Pre-fill PR Template
|
||||
|
||||
- **Title**: Format as `[{TICKET-NUMBER}] [{COMPONENT}] {TYPE}: {TASK-SUMMARY}` extracted from branch description and change analysis
|
||||
- Examples: `[OPIK-2180] [DOCS] docs: add cursor git workflow rule`, `[OPIK-1234] [BE] feat(api): add trace request validation endpoint`
|
||||
- **Description**: Read the PR template from `.github/pull_request_template.md` at runtime and use it as the source of truth. The PR description is public on GitHub — **never** include any of the following unless the user explicitly requests it:
|
||||
- Customer or client names
|
||||
- Internal / private domains and hostnames (anything not publicly routable)
|
||||
- Internal URLs (monitoring dashboards, log explorers, staging / preview deployments, internal wikis, chat threads, issue trackers other than the Jira `OPIK-####` key)
|
||||
- IPs, storage bucket names, credentials, or any secret
|
||||
|
||||
When summarizing changes, describe behavior in generic terms ("a customer reported…", "in a production deployment…") rather than naming the source. Redact screenshots or log excerpts before including them. For anything that needs private context, reference the Jira ticket instead of embedding it here.
|
||||
```bash
|
||||
# Read the actual PR template — do NOT hardcode it
|
||||
cat .github/pull_request_template.md
|
||||
```
|
||||
- **Fill every `##` section** in the template — the PR linter requires all sections to be present
|
||||
- If a section is not applicable, write "N/A" rather than removing it
|
||||
- **Section guidance**:
|
||||
- **Details**: Implementation summary from git analysis (replace the HTML comment placeholder)
|
||||
- **Change checklist**: Auto-check based on file types changed (user-facing for UI changes, documentation for docs)
|
||||
- **Issues**: Link to Jira ticket (e.g., `OPIK-2180`) or GitHub issue, or "NA" for hotfixes. List every ticket this PR **resolves** here with a normal hyphenated key.
|
||||
- **Jira key convention across the whole body** (see git-workflow rule): the GitHub for Jira app links any `OPIK-<digits>` it finds in the PR body to that ticket's Development panel, and the link can't be removed. So in **all** sections (Details, Testing, etc.) and in commit messages: tickets this PR **resolves** keep the hyphen (`OPIK-1234`) — links/URLs fine and wanted. Tickets **related but not resolved** here (escalations, references to older tickets — anything not in the title/branch) must be written with an underscore (`OPIK_7000`) and with **no** Jira URL (the URL contains the hyphenated key and links anyway). Apply this when generating every section below.
|
||||
- **AI-WATERMARK**: Fill with `AI-WATERMARK: yes`, then list: Tools (e.g., "Claude Code"), Model(s), Scope (e.g., "full implementation" or "assisted"), Human verification (e.g., "code review + manual testing")
|
||||
- **Testing**: Extract from commit messages or set based on test files changed (replace the HTML comment placeholder)
|
||||
- **Documentation**: List docs updated or set "N/A" if no documentation changes
|
||||
|
||||
---
|
||||
|
||||
### 8. Validate PR Title & Description (pr-lint)
|
||||
|
||||
Before creating the PR, validate the generated title and body against the same rules enforced by `.github/workflows/pr-lint.yml`. This prevents PRs from failing the PR Linter CI check on first submission.
|
||||
|
||||
- **Read pr-lint rules at runtime**: Parse `.github/workflows/pr-lint.yml` as source of truth for the title regex and required sections. Fall back to the rules below only if the workflow file cannot be read.
|
||||
|
||||
- **Title validation**: Verify the PR title matches the regex:
|
||||
```
|
||||
^\[(OPIK-\d+|DND-\d+|DEV-\d+|CUST-\d+|issue-\d+|NA)\](\s*\[(BE|FE|DOCS|SDK|GHA|CI|HELM)\])*\s*.+$
|
||||
```
|
||||
- Reject titles missing a ticket prefix (e.g., `[OPIK-1234]`) or using invalid component tags
|
||||
|
||||
- **Required sections validation**: Verify the PR body contains all required `##` headings:
|
||||
- `## Details`
|
||||
- `## Change checklist`
|
||||
- `## Issues`
|
||||
- `## Testing`
|
||||
- `## Documentation`
|
||||
|
||||
- **Details section non-empty**: Extract content between `## Details` and the next `##` heading. Verify it is not empty after trimming whitespace (matching CI's `getSectionContent` which only calls `.trim()`). Note: HTML comment placeholders are handled separately in the template placeholder cleanup step below — do not strip them during this validation check.
|
||||
|
||||
- **Issues section ticket reference**: Extract content of `## Issues`. Unless the PR title starts with `[NA]`, verify it references at least one ticket matching: `#\d+`, `OPIK-\d+`, `DND-\d+`, `DEV-\d+`, or `CUST-\d+`.
|
||||
|
||||
- **Template placeholder cleanup**: Scan the entire body for leftover HTML comment placeholders from the PR template (e.g., `<!-- REPLACE ME`, `<!-- REPLACE ME WITH:`). If any remain, strip them before submission.
|
||||
|
||||
- **On validation failure**:
|
||||
- Show the specific errors to the user
|
||||
- Auto-fix the issues (adjust title format, fill missing sections, strip placeholders)
|
||||
- Re-validate after fixes
|
||||
- If validation still fails after auto-fix, show remaining errors and ask the user whether to continue or stop
|
||||
|
||||
- **On validation success**: Continue to PR creation
|
||||
|
||||
---
|
||||
|
||||
### 9. Create GitHub PR
|
||||
|
||||
- **Use GitHub CLI (preferred)**: Create a draft PR in `comet-ml/opik` with pre-filled template (`gh pr create --draft`).
|
||||
- **Fallback**: If CLI is unavailable and GitHub MCP is available, create the PR with MCP and mark as draft when supported.
|
||||
- **Verify creation**: Confirm PR was created successfully
|
||||
- **If creation fails**: Show error details and stop
|
||||
|
||||
---
|
||||
|
||||
### 10. Update Jira Status
|
||||
|
||||
- **Fetch Jira ticket**: For `OPIK-<number>` branches, use Jira MCP to get ticket details
|
||||
- **Transition status**: If branch key is `OPIK-<number>`, change ticket status to "In Review"
|
||||
- **Verify transition**: Confirm status was updated successfully (for OPIK branches)
|
||||
- **If transition fails**: Show error details but continue
|
||||
- **Post progress summary**: For OPIK branches, after successful status update, add a progress comment directly to Jira using `addCommentToJiraIssue` with the standard 2-section format:
|
||||
- **Release Notes**: User-facing changes for Product Managers, including feature toggle changes (or "No user-facing changes were made in this ticket" if none)
|
||||
- **Docs**: Developer-focused technical details and implementation notes
|
||||
- **No Jira branch key**: If branch key is `issue-<number>` or `NA`, skip Jira transition/comment and continue.
|
||||
|
||||
---
|
||||
|
||||
### 11. Completion Summary & Validation
|
||||
|
||||
- **Confirm all steps completed**:
|
||||
- ✅ Feature branch validated with Opik naming convention
|
||||
- ✅ Git status checked and resolved
|
||||
- ✅ No existing PRs found
|
||||
- ✅ Quality checks passed
|
||||
- ✅ PR template pre-filled
|
||||
- ✅ PR title and description pass pr-lint validation
|
||||
- ✅ GitHub PR created successfully
|
||||
- ✅ Jira ticket status updated to "In Review" (for OPIK branches)
|
||||
- ✅ Progress documented in Jira (for OPIK branches)
|
||||
- **Show summary**: Display PR URL and Jira ticket status (if applicable)
|
||||
- **Next steps**: Provide guidance on PR review process
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### **Availability Errors**
|
||||
|
||||
- **GitHub CLI unavailable/auth missing**: Stop immediately and provide `gh` installation/auth instructions
|
||||
- **Jira MCP unavailable (OPIK branches only)**: Stop immediately after testing and provide setup instructions
|
||||
- **Jira MCP connection failures (OPIK branches only)**: Stop immediately after testing and verify MCP server status
|
||||
- **Jira MCP test failures (OPIK branches only)**: Stop immediately if MCP tests fail and provide troubleshooting steps
|
||||
|
||||
### **Branch Validation Errors**
|
||||
|
||||
- Invalid format: Show expected Opik pattern and current branch
|
||||
- On main: Explain this command is for feature branches only
|
||||
- Missing ticket key (`OPIK-<number>`, `issue-<number>`, or `NA`): Stop and explain the requirement
|
||||
|
||||
### **Git Status Errors**
|
||||
|
||||
- Uncommitted changes: Ask user for decision
|
||||
- Remote branch missing or behind: Ask user for decision
|
||||
- Push failures: Check remote configuration and permissions
|
||||
|
||||
### **Quality Check Failures**
|
||||
|
||||
- Linting errors: Show errors and ask user preference
|
||||
- Type errors: Show errors and ask user preference
|
||||
- User choice to stop: Respect user decision
|
||||
|
||||
### **PR Lint Validation Failures**
|
||||
|
||||
- Title format invalid: Auto-fix by adjusting to match required regex pattern
|
||||
- Missing required sections: Auto-add missing `##` sections with "N/A" content
|
||||
- Empty Details section: Flag to user — requires meaningful content
|
||||
- Missing Issues reference: Auto-fill from branch ticket key if available
|
||||
- Leftover template placeholders: Auto-strip HTML comments
|
||||
- Auto-fix fails: Show remaining errors and ask user whether to continue or stop
|
||||
|
||||
### **PR Creation Failures**
|
||||
|
||||
- GitHub CLI issues: Check `gh auth status` and repository permissions for `comet-ml/opik`
|
||||
- GitHub MCP fallback issues: Check MCP connectivity/authentication if fallback path was used
|
||||
- Template errors: Validate template format and content
|
||||
- Network issues: Verify connectivity to GitHub
|
||||
|
||||
### **Jira Status Update Failures**
|
||||
|
||||
- Transition not allowed: Check workflow permissions
|
||||
- Ticket not found: Verify ticket exists and is accessible
|
||||
- Network issues: Verify connectivity to Atlassian services
|
||||
|
||||
### **Progress Documentation Failures**
|
||||
|
||||
- Comment addition fails: Log error but continue (non-critical operation)
|
||||
- Provide manual instructions for adding progress comment
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. ✅ GitHub path is available (`gh` authenticated preferred, MCP fallback optional)
|
||||
2. ✅ Feature branch is validated with key `OPIK-<number>`, `issue-<number>`, or `NA`
|
||||
3. ✅ Jira MCP is available and accessible (for OPIK branches)
|
||||
4. ✅ Git status is clean (no pending changes)
|
||||
5. ✅ Remote branch exists and is up to date
|
||||
6. ✅ No existing PRs found for the branch
|
||||
7. ✅ Quality checks pass (linter, type checking, or Maven)
|
||||
8. ✅ PR template is pre-filled with meaningful content
|
||||
9. ✅ PR title and description pass pr-lint validation before submission
|
||||
10. ✅ GitHub PR is created successfully
|
||||
11. ✅ Jira ticket status is updated to "In Review" (for OPIK branches)
|
||||
12. ✅ Progress is documented in Jira (via direct MCP call, for OPIK branches)
|
||||
13. ✅ All operations complete with clear feedback
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- **Tool Requirements**: A GitHub path (`gh` preferred, MCP fallback) must be available before operations begin; Jira MCP is required only for `OPIK-<number>` branches
|
||||
- **Repository**: Always uses `comet-ml/opik` for GitHub operations
|
||||
- **Template pre-filling**: Automatically extracts information from git changes and commit history
|
||||
- **Quality assurance**: Ensures code meets standards before PR creation
|
||||
- **Workflow integration**: Seamlessly connects Git, GitHub, and Jira workflows following Opik conventions
|
||||
- **Progress documentation**: Adds progress comment to Jira for OPIK branches using MCP
|
||||
- **User control**: Asks for confirmation on critical decisions (commits, pushes)
|
||||
- **Error handling**: Graceful degradation when non-critical operations fail
|
||||
- **Git operations**: Handles both missing remote branches and branches that are behind local commits
|
||||
- **Tool testing**: Validates GitHub path (`gh` first, MCP fallback) before proceeding, and validates Jira MCP only for `OPIK-<number>` branches
|
||||
- **Opik conventions**: Follows Opik branch naming, commit message, and PR title conventions
|
||||
- **Logic reuse**: Implements the same git diff analysis and summary generation logic as `share-progress-in-jira` command:
|
||||
- Uses three-dot syntax (`git diff origin/main...HEAD`) for accurate change detection against latest remote main
|
||||
- Categorizes changes by file type and implementation phases
|
||||
- Uses standard 2-section format: Release Notes (for PMs) and Docs (for developers)
|
||||
- Generates professional, formatted summaries with bullet points
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,368 @@
|
||||
# Generate Code Review Slack Command
|
||||
|
||||
**Command**: `cursor generate-code-review-slack-command`
|
||||
|
||||
## Overview
|
||||
|
||||
Generate a formatted, copiable Slack command for the #code-review channel with PR information, Jira ticket, test environment link, and optional component summaries (FE, BE, Python, TypeScript). Automatically extracts information from the GitHub PR for the current branch and outputs a command that can be copied, edited (to add @ mentions, media links, etc.), and pasted into Slack.
|
||||
|
||||
- **Execution model**: Automatically extracts information from GitHub PR, prompts only for missing information, formats the message according to the template, and outputs a copiable Slack command (does NOT send automatically).
|
||||
|
||||
This workflow will:
|
||||
|
||||
- Find the GitHub PR for the current branch
|
||||
- Extract Jira ticket from PR title
|
||||
- Extract test environment link from PR description
|
||||
- Extract component summaries (FE, BE, Python, TypeScript) from PR description
|
||||
- Prompt only for missing information
|
||||
- Allow user to customize the message slightly
|
||||
- Format the message according to the code review template
|
||||
- Generate a copiable Slack command that can be edited before sending
|
||||
- Display the command for easy copying
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
### Required Information (auto-extracted from PR, prompted if missing)
|
||||
- **Jira ticket**: Extracted from PR title (format: `[OPIK-1234]`) or prompted if not found
|
||||
- **PR link**: Automatically determined from current branch's PR or prompted if no PR found
|
||||
- **Test env link**: Extracted from PR description (Testing or Details section) or prompted if not found
|
||||
|
||||
### Optional Information (auto-extracted from PR, prompted if missing)
|
||||
- **FE summary**: Extracted from PR description (looks for frontend/React/FE mentions) or prompted if not found
|
||||
- **BE summary**: Extracted from PR description (looks for backend/Java/BE mentions) or prompted if not found
|
||||
- **Python summary**: Extracted from PR description (looks for Python/Python SDK mentions) or prompted if not found
|
||||
- **TypeScript summary**: Extracted from PR description (looks for TypeScript/TypeScript SDK/TS mentions) or prompted if not found
|
||||
- **Baz approved status**: Extracted from PR status checks (optional, may not be available)
|
||||
|
||||
### Configuration
|
||||
- **No Slack MCP required**: This command does not send messages, so Slack MCP configuration is not needed
|
||||
- **GitHub MCP**: Required for extracting PR information
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Check GitHub MCP**: Test GitHub MCP availability by attempting to fetch repository info using `get_file_contents` for `comet-ml/opik`
|
||||
> If unavailable, respond with: "This command needs GitHub MCP configured. Set MCP config/env, run `make cursor` (Cursor) or `make claude` (Claude CLI), then retry."
|
||||
> Stop here.
|
||||
- **Check Git repository**: Verify we're in a Git repository
|
||||
- **Check current branch**: Get current branch name
|
||||
|
||||
---
|
||||
|
||||
### 2. Find GitHub PR for Current Branch
|
||||
|
||||
- **Search for PR**: Use GitHub MCP to find an open PR for the current branch in `comet-ml/opik`
|
||||
- **If PR found**:
|
||||
- Extract PR number, URL, title, and description
|
||||
- Store PR information for extraction steps
|
||||
- **If no PR found**:
|
||||
> "No open PR found for this branch. Please provide the PR link manually:"
|
||||
- Prompt for PR link and validate URL format
|
||||
- If provided, fetch PR details using GitHub MCP
|
||||
- If PR cannot be fetched, stop with error message
|
||||
- **After successfully fetching PR by manual link**: Extract PR number, URL, title, and description and store PR information for extraction steps (same as "PR found" branch)
|
||||
|
||||
---
|
||||
|
||||
### 3. Extract Information from PR
|
||||
|
||||
- **Extract Jira ticket from PR title**:
|
||||
- Parse PR title for pattern `[OPIK-\d+]` or `[issue-\d+]` or `[NA]`
|
||||
- Extract ticket number (e.g., `OPIK-1234` from `[OPIK-1234] [BE] feat(api): add trace request validation endpoint`)
|
||||
- If not found in title, check PR description "## Issues" section for `OPIK-\d+` pattern
|
||||
- If still not found, prompt: "Jira ticket not found in PR. Enter Jira ticket number (e.g., OPIK-1234):"
|
||||
- Validate format and store
|
||||
- **Build Jira URL**: Construct Jira link as `https://comet-ml.atlassian.net/browse/{TICKET}` (e.g., `https://comet-ml.atlassian.net/browse/OPIK-1234`)
|
||||
|
||||
- **Extract Baz approved status (optional)**:
|
||||
- Try to fetch PR status checks using GitHub MCP
|
||||
- Look for status checks or CI checks that might indicate "Baz approved" or similar approval status
|
||||
- If found, store the status (e.g., "Baz approved: ✅" or "Baz status: pending")
|
||||
- If not found or unavailable, skip this field (it's optional)
|
||||
|
||||
- **Extract PR size**: Follow the shared steps in `.agents/docs/PR_SIZE_EXTRACTION.md` (read the `size/*` label; fall back to changed lines using the workflow's own ignore list + thresholds). Store the bucket as `{emoji} {BUCKET}` (e.g. `🟠 L`).
|
||||
|
||||
- **Extract test environment link from PR description and comments**:
|
||||
- First, search PR comments for test environment deployment messages (look for "Test environment is now available!" or similar)
|
||||
- When fetching PR comments via `gh api`, **always** use `--paginate` to ensure all results are fetched:
|
||||
```bash
|
||||
# Issue comments (general PR comments, including deployment bot messages)
|
||||
gh api repos/comet-ml/opik/issues/{pr_number}/comments --paginate
|
||||
|
||||
# Review comments (inline code comments)
|
||||
gh api repos/comet-ml/opik/pulls/{pr_number}/comments --paginate
|
||||
```
|
||||
> **Why pagination is required**: GitHub API returns 30 items per page by default. Opik PRs regularly exceed this — 17 CI test group comments + deployment bot comments + reviewer comments can push past 30 total. Without `--paginate`, the deployment bot comment with the test environment link may end up on page 2 and get silently missed.
|
||||
- Extract URL from comment body (typically in format `https://pr-XXXX.dev.comet.com` or `https://test.opik.com`)
|
||||
- If not found in comments, search PR description for URLs in "## Testing" section
|
||||
- Look for common test environment patterns: `https://pr-*.dev.comet.com`, `https://test.opik.com`, `https://*.opik.com`, or any `https://` URL
|
||||
- If multiple URLs found, prefer the first one that looks like a test environment
|
||||
- If not found, search in "## Details" section for URLs
|
||||
- If still not found, prompt: "Test environment link not found in PR. Enter test environment link (e.g., https://pr-4743.dev.comet.com):"
|
||||
- Validate URL format and store
|
||||
|
||||
- **Extract component summaries from PR description**:
|
||||
- **FE summary**:
|
||||
- Look for mentions of "frontend", "FE", "React", "UI", or `[FE]` tag in PR description
|
||||
- Extract relevant one-line summary from "## Details" section or component-specific mentions
|
||||
- If not found, prompt: "Frontend summary not found in PR. Enter frontend summary (one line, optional - press Enter to skip):"
|
||||
|
||||
- **BE summary**:
|
||||
- Look for mentions of "backend", "BE", "Java", "API", or `[BE]` tag in PR description
|
||||
- Extract relevant one-line summary from "## Details" section or component-specific mentions
|
||||
- If not found, prompt: "Backend summary not found in PR. Enter backend summary (one line, optional - press Enter to skip):"
|
||||
|
||||
- **Python summary**:
|
||||
- Look for mentions of "Python", "Python SDK", "SDK", or Python-related changes in PR description
|
||||
- Extract relevant one-line summary from "## Details" section or component-specific mentions
|
||||
- If not found, prompt: "Python summary not found in PR. Enter Python summary (one line, optional - press Enter to skip):"
|
||||
|
||||
- **TypeScript summary**:
|
||||
- Look for mentions of "TypeScript", "TypeScript SDK", "TS", "TS SDK", or `[TS]` tag in PR description
|
||||
- Extract relevant one-line summary from "## Details" section or component-specific mentions
|
||||
- If not found, prompt: "TypeScript summary not found in PR. Enter TypeScript summary (one line, optional - press Enter to skip):"
|
||||
|
||||
- **Store extracted information**: Keep all extracted and prompted values for message formatting
|
||||
|
||||
---
|
||||
|
||||
### 4. Prompt for Message Customization
|
||||
|
||||
- **Allow user to customize message**: After extracting all information, prompt the user:
|
||||
> "Would you like to customize the message? (Enter any additional text to prepend/append, or press Enter to use default message):"
|
||||
- **If user provides customization text**: Store it for inclusion in the final message
|
||||
- **If user presses Enter (empty)**: Proceed with default message format
|
||||
- **Note**: User can add context, emphasize certain points, or include additional information
|
||||
|
||||
---
|
||||
|
||||
### 5. Format Slack Message
|
||||
|
||||
- **Use PR link**: Use the PR URL extracted from Step 2 (or provided manually)
|
||||
|
||||
- **Build message text** according to the template:
|
||||
```
|
||||
Hi team,
|
||||
|
||||
Please review the following PR:
|
||||
|
||||
{{user_customization_text_if_provided}}
|
||||
|
||||
:jira_epic: jira link: {{Jira_URL}}
|
||||
:github: pr link: {{PR_link}}
|
||||
:straight_ruler: pr size: {{pr_size}}
|
||||
:test_tube: test env link: {{test_env}}
|
||||
{{baz_approved_status_if_available}}
|
||||
:react: fe summary (optional): {{description_in_one_line}}
|
||||
:java: be summary (optional): {{description_in_one_line}}
|
||||
:python: python summary (optional): {{description_in_one_line}}
|
||||
:typescript: typescript summary (optional): {{description_in_one_line}}
|
||||
```
|
||||
|
||||
- **Message structure**:
|
||||
- **Start with greeting**: Always begin with "Hi team,\n\nPlease review the following PR:\n"
|
||||
- **User customization**: If user provided customization text in Step 4, include it after the greeting (before the structured fields)
|
||||
- **Jira link**: Include full Jira URL with ticket (e.g., `https://comet-ml.atlassian.net/browse/OPIK-1234`)
|
||||
- **PR link**: Include GitHub PR URL
|
||||
- **PR size**: Include the size bucket extracted in Step 3 (e.g., `🟠 L`). Always included — size is always available from GitHub.
|
||||
- **Test env link**: Include test environment URL
|
||||
- **Baz approved status**: Only include if extracted from PR status checks (optional field)
|
||||
- **Component summaries**: Only include optional fields that were provided (skip empty ones)
|
||||
|
||||
- **Format example**:
|
||||
```
|
||||
Hi team,
|
||||
|
||||
Please review the following PR:
|
||||
|
||||
:jira_epic: jira link: https://comet-ml.atlassian.net/browse/OPIK-1234
|
||||
:github: pr link: https://github.com/comet-ml/opik/pull/1234
|
||||
:straight_ruler: pr size: 🟠 L
|
||||
:test_tube: test env link: https://test.opik.com
|
||||
:react: fe summary (optional): Added new metrics dashboard UI
|
||||
:java: be summary (optional): Implemented metrics aggregation endpoint
|
||||
:typescript: typescript summary (optional): Added TypeScript SDK support for metrics
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 6. Generate Copiable Slack Command
|
||||
|
||||
- **Format as Slack command**: Generate a copiable command that can be pasted directly into Slack
|
||||
- **Command format**: The output should be formatted as a code block that can be easily copied
|
||||
- **Display instructions**: Show clear instructions on how to use the generated command:
|
||||
> "📋 **Copiable Slack Command Generated**\n\nCopy the command below and paste it into the #code-review channel in Slack.\n\nYou can edit it before sending to:\n- Add @ mentions for specific reviewers\n- Add media links or video links\n- Make final proof edits\n- Add any additional context\n\n```\n[FORMATTED_MESSAGE]\n```\n\n**To send in Slack:**\n1. Open Slack and navigate to #code-review channel\n2. Paste the command above\n3. Edit as needed (add @ mentions, media links, etc.)\n4. Send the message"
|
||||
|
||||
- **Alternative format (if using Slack CLI)**: If the user prefers, also provide a Slack CLI command format:
|
||||
```
|
||||
slack chat send --channel "#code-review" --text "[FORMATTED_MESSAGE]"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 7. Display & Summary
|
||||
|
||||
- **Display the generated command**: Show the formatted message in a code block for easy copying
|
||||
- **Provide usage instructions**: Remind user how to use the command
|
||||
- **Note about editing**: Emphasize that the command can be edited before sending to add @ mentions, media links, or make final edits
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### **GitHub MCP Errors**
|
||||
|
||||
- **GitHub MCP unavailable**: Stop immediately after testing and provide setup instructions
|
||||
- **PR not found**: Prompt for manual PR link or stop if user cancels
|
||||
- **PR fetch failure**: Show error details and suggest manual input
|
||||
|
||||
### **Extraction Errors**
|
||||
|
||||
- **Jira ticket not found**: Prompt user for ticket number
|
||||
- **Test env link not found**: Prompt user for test environment URL
|
||||
- **Component summaries not found**: Prompt user for summaries (optional)
|
||||
|
||||
### **Input Validation Errors**
|
||||
|
||||
- **Invalid Jira ticket format**: Show expected format (e.g., `OPIK-1234`) and re-prompt
|
||||
- **Invalid PR URL**: Show expected format and re-prompt
|
||||
- **Invalid test env URL**: Show expected format and re-prompt
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. ✅ GitHub MCP is available and accessible
|
||||
2. ✅ PR is found for current branch (or manually provided)
|
||||
3. ✅ Jira ticket is extracted from PR or provided manually
|
||||
4. ✅ Test environment link is extracted from PR or provided manually
|
||||
5. ✅ Component summaries are extracted from PR or provided manually (optional)
|
||||
6. ✅ Message is formatted according to template (with greeting and Jira link)
|
||||
7. ✅ Copiable Slack command is generated and displayed
|
||||
8. ✅ User receives clear instructions on how to use the command
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- **GitHub MCP Required**: Uses GitHub MCP to fetch PR information automatically
|
||||
- **No Slack MCP Required**: This command does not send messages, so Slack MCP configuration is not needed
|
||||
- **Message format**: Follows the exact template provided with emoji prefixes, includes greeting and Jira link
|
||||
- **Automatic extraction**: Extracts information from PR title and description to minimize manual input
|
||||
- **Fallback to prompts**: Only prompts for information that cannot be extracted from PR
|
||||
- **Optional fields**: Only included in message if extracted from PR or provided by user
|
||||
- **Channel**: Message is intended for `#code-review` channel
|
||||
- **PR detection**: Automatically finds PR for current branch, falls back to manual input if needed
|
||||
- **Smart extraction**: Uses heuristics to find test environment links and component summaries in PR description
|
||||
- **Editing before sending**: The generated command can be edited to add @ mentions, media links, video links, or make final proof edits before sending in Slack
|
||||
- **Use case**: This command is particularly useful when you want to:
|
||||
- Add @ mentions for specific reviewers
|
||||
- Include video links or media that Slack MCP cannot send directly
|
||||
- Make final proof edits before sending
|
||||
- Have more control over the final message format
|
||||
- **Alternative for automatic sending**: If you don't need manual editing, use `cursor send-code-review-slack` to send the message directly to Slack.
|
||||
|
||||
---
|
||||
|
||||
## Example Usage
|
||||
|
||||
### Setup and Usage
|
||||
|
||||
```bash
|
||||
# 1. Ensure GitHub MCP is configured (no Slack MCP needed)
|
||||
|
||||
# 2. Run command (on a branch with an open PR)
|
||||
cursor generate-code-review-slack-command
|
||||
```
|
||||
|
||||
# Command execution flow:
|
||||
# 1. Find PR for current branch: https://github.com/comet-ml/opik/pull/1234
|
||||
# 2. Extract Jira ticket from PR title: [OPIK-1234] [FE] feat(api): add metrics dashboard
|
||||
# 3. Extract test env from PR comments or description: https://pr-1234.dev.comet.com (from PR comments or Testing section)
|
||||
# Extract PR size from the size/* label (or additions+deletions): 🟠 L
|
||||
# 4. Extract summaries from PR description:
|
||||
# - FE: Added new metrics dashboard UI (from Details section)
|
||||
# - BE: Implemented metrics aggregation endpoint (from Details section)
|
||||
# - Python: (not found, prompts user)
|
||||
# - TypeScript: (not found, prompts user)
|
||||
# 5. Prompt for message customization (optional)
|
||||
# 6. Format message according to template (with greeting and Jira link)
|
||||
# 7. Generate and display copiable Slack command
|
||||
|
||||
# Output:
|
||||
# 📋 **Copiable Slack Command Generated**
|
||||
#
|
||||
# Copy the command below and paste it into the #code-review channel in Slack.
|
||||
#
|
||||
# You can edit it before sending to:
|
||||
# - Add @ mentions for specific reviewers
|
||||
# - Add media links or video links
|
||||
# - Make final proof edits
|
||||
# - Add any additional context
|
||||
#
|
||||
# ```
|
||||
# Hi team,
|
||||
#
|
||||
# Please review the following PR:
|
||||
#
|
||||
# :jira_epic: jira link: https://comet-ml.atlassian.net/browse/OPIK-1234
|
||||
# :github: pr link: https://github.com/comet-ml/opik/pull/1234
|
||||
# :straight_ruler: pr size: 🟠 L
|
||||
# :test_tube: test env link: https://test.opik.com
|
||||
# :react: fe summary (optional): Added new metrics dashboard UI
|
||||
# :java: be summary (optional): Implemented metrics aggregation endpoint
|
||||
# :typescript: typescript summary (optional): Added TypeScript SDK support for metrics
|
||||
# ```
|
||||
#
|
||||
# **To send in Slack:**
|
||||
# 1. Open Slack and navigate to #code-review channel
|
||||
# 2. Paste the command above
|
||||
# 3. Edit as needed (add @ mentions, media links, etc.)
|
||||
# 4. Send the message
|
||||
|
||||
### Example with Missing Information
|
||||
|
||||
If some information cannot be extracted from PR, the command will prompt:
|
||||
|
||||
```bash
|
||||
cursor generate-code-review-slack-command
|
||||
|
||||
# Found PR: https://github.com/comet-ml/opik/pull/1234
|
||||
# Extracted Jira ticket: OPIK-1234
|
||||
# Test environment link not found in PR. Enter test environment link (e.g., https://test.opik.com): https://test.opik.com
|
||||
# Frontend summary not found in PR. Enter frontend summary (one line, optional - press Enter to skip): [Enter pressed - skipped]
|
||||
# Backend summary not found in PR. Enter backend summary (one line, optional - press Enter to skip): Implemented metrics endpoint
|
||||
# Python summary not found in PR. Enter Python summary (one line, optional - press Enter to skip): [Enter pressed - skipped]
|
||||
# TypeScript summary not found in PR. Enter TypeScript summary (one line, optional - press Enter to skip): [Enter pressed - skipped]
|
||||
# Would you like to customize the message? (Enter any additional text to prepend/append, or press Enter to use default message): [Enter pressed - using default]
|
||||
```
|
||||
|
||||
### Example with Customization
|
||||
|
||||
```bash
|
||||
cursor generate-code-review-slack-command
|
||||
|
||||
# ... extraction steps ...
|
||||
# Would you like to customize the message? (Enter any additional text to prepend/append, or press Enter to use default message): This PR includes important security updates, please review carefully.
|
||||
#
|
||||
# Generated command includes the customization:
|
||||
# Hi team,
|
||||
#
|
||||
# Please review the following PR:
|
||||
#
|
||||
# This PR includes important security updates, please review carefully.
|
||||
#
|
||||
# :jira_epic: jira link: https://comet-ml.atlassian.net/browse/OPIK-1234
|
||||
# ...
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,31 @@
|
||||
# Investigate E2E Failure
|
||||
|
||||
## Overview
|
||||
|
||||
Investigate a failed Opik E2E test and propose a fix. The command gathers the evidence (trace, error, history), classifies the failure as a real regression, a flake, or environment/selector drift, and proposes a specific fix. **Read-only** — it diagnoses and proposes; it does not edit tests.
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **Failure reference** (required) — any one of:
|
||||
- A GitHub Actions run or a red `Run v2 suite` check (URL or run id).
|
||||
- An Allure TestOps launch.
|
||||
- A test name (e.g. `dataset-crud-smoke`) when you want its history / flake status.
|
||||
- "The failure I just hit locally" — a test that failed in your local run.
|
||||
- **Optional** — the PR or branch involved, so the diagnosis can correlate the failure to recent changes.
|
||||
|
||||
---
|
||||
|
||||
## Instructions
|
||||
|
||||
**Invoke the `debugging-e2e-tests` skill and follow it exactly.** It carries the loop — resolve the entry point → gather the trace + history → classify regression vs. flake → diagnose → report and propose — along with the evidence sources (the `allure-testops` MCP, `gh` artifacts, `npx playwright show-trace`) and the classification heuristics.
|
||||
|
||||
---
|
||||
|
||||
## Success criteria
|
||||
|
||||
1. A **verdict**: regression / flake / environment-or-selector, with a confidence level.
|
||||
2. **Cited evidence**: the failing trace step, the error, the history pattern, and the correlated change (if any).
|
||||
3. A **specific proposed fix** (or, for a known flake, a poll/quarantine/no-fix recommendation).
|
||||
4. **No edits made** — to apply the fix, hand off to the `writing-e2e-tests` skill.
|
||||
@@ -0,0 +1,415 @@
|
||||
# Review GitHub PR
|
||||
|
||||
**Command**: `cursor review-github-pr`
|
||||
|
||||
## Overview
|
||||
|
||||
Given a PR number/URL (or auto-detected from the current branch), fetch the PR diff from `comet-ml/opik`, perform a thorough code review, present findings to the user, and post approved review comments directly on the PR. This is the complement of `/address-github-pr-comments` — that command *responds* to reviewer feedback, this command *generates* it.
|
||||
|
||||
- **Execution model**: Always runs from scratch. Each invocation re-fetches the PR, re-analyzes the diff, and re-generates findings.
|
||||
|
||||
This workflow will:
|
||||
|
||||
- Verify `gh` CLI is available and authenticated (stop if not)
|
||||
- Optionally use GitHub MCP for richer reading (but `gh` CLI alone is sufficient)
|
||||
- Locate the PR (from argument, current branch, or prompt)
|
||||
- Fetch the full PR diff and changed files
|
||||
- Analyze changes using Opik domain knowledge (backend, frontend, SDK rules)
|
||||
- Generate categorized review findings with inline code suggestions
|
||||
- Present findings to the user for approval before posting
|
||||
- Post approved comments on the PR using `gh api` with AI watermark
|
||||
- **Never** submit a formal review approval or "request changes" — only post individual comments
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **PR identifier (optional)**: PR number (e.g., `5395`), full URL (e.g., `https://github.com/comet-ml/opik/pull/5395`), or omitted to auto-detect from current branch
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Check `gh` CLI**: Verify `gh` is installed and authenticated (required for both reading and posting)
|
||||
```bash
|
||||
gh auth status
|
||||
```
|
||||
> If not authenticated, respond with: "Please run `gh auth login` first."
|
||||
> Stop here.
|
||||
- **Check GitHub MCP (optional)**: If GitHub MCP is available, use it for richer PR data. If not, fall back to `gh` CLI for everything — no setup instructions needed.
|
||||
|
||||
---
|
||||
|
||||
### 2. Locate the PR
|
||||
|
||||
- **If argument provided**:
|
||||
- If numeric (e.g., `5395`): Use as PR number directly
|
||||
- If URL (e.g., `https://github.com/comet-ml/opik/pull/5395`): Extract PR number from URL
|
||||
- **If no argument provided**:
|
||||
- Check current Git branch
|
||||
- Search for an open PR for this branch in `comet-ml/opik`:
|
||||
```bash
|
||||
gh pr list --repo comet-ml/opik --head <branch-name> --state open --json number,title,url
|
||||
```
|
||||
- If GitHub MCP is available, use it as an alternative to `gh pr list`
|
||||
- If no PR found, prompt: "No open PR found for this branch. Enter a PR number or URL:"
|
||||
- **Fetch PR metadata**: Get PR title, description, author, base branch, and state
|
||||
```bash
|
||||
gh pr view {pr_number} --repo comet-ml/opik --json title,body,author,baseRefName,state,headRefOid
|
||||
```
|
||||
- **If PR is merged or closed**: Print warning and ask if user wants to continue reviewing anyway
|
||||
|
||||
---
|
||||
|
||||
### 3. Fetch PR Diff, Changed Files, and Existing Reviews
|
||||
|
||||
- **Fetch existing review comments**: Before analyzing, collect any comments already posted by this command on this PR
|
||||
```bash
|
||||
gh api repos/comet-ml/opik/pulls/{pr_number}/comments --paginate
|
||||
```
|
||||
- Filter for comments containing the `🤖 *Review posted via /review-github-pr*` marker
|
||||
- Index them by `path` + `start_line` + `line` for deduplication in step 6 (for single-line comments, `start_line` equals `line`)
|
||||
- This prevents posting duplicate comments when the command is run multiple times on the same PR
|
||||
|
||||
- **Get changed files**: List all files changed in the PR
|
||||
```bash
|
||||
gh pr diff {pr_number} --repo comet-ml/opik --name-only
|
||||
```
|
||||
- **Get full diff**:
|
||||
```bash
|
||||
gh pr diff {pr_number} --repo comet-ml/opik
|
||||
```
|
||||
- **If GitHub MCP is available**: Use `get_pull_request_files` for structured file data (additions, deletions, patch)
|
||||
- **Categorize files by domain**:
|
||||
- **Backend**: `apps/opik-backend/**` → Java, API, services, DAOs, migrations
|
||||
- **Frontend**: `apps/opik-frontend/**` → React, TypeScript, components, hooks
|
||||
- **Python SDK**: `sdks/python/**` → Python SDK patterns
|
||||
- **TypeScript SDK**: `sdks/opik-typescript/**` → TypeScript SDK patterns
|
||||
- **E2E Tests**: `tests_end_to_end/**` → Playwright tests
|
||||
- **Docs**: `*.md`, `docs/**` → Documentation
|
||||
- **Config/Infra**: Docker, CI/CD, configuration files
|
||||
- **Skip binary files and lock files** from review
|
||||
- **Flag sensitive files**: Files matching patterns like `.env`, `.pem`, `.key`, `credentials.*`, `*.secret` should be flagged. Do not include raw content from these files in posted comments — instead, post a high-level note referencing the file without quoting secret values
|
||||
|
||||
---
|
||||
|
||||
### 4. Analyze Changes
|
||||
|
||||
Review the diff using Opik domain knowledge from `.agents/skills/` and `.agents/rules/`. For each domain touched by the PR, apply the relevant review criteria:
|
||||
|
||||
#### General (all files)
|
||||
- Security: SQL injection, XSS, hardcoded secrets, input validation at boundaries
|
||||
- Error handling: Appropriate error propagation, no swallowed exceptions
|
||||
- Naming: Clear, consistent variable/function/class naming
|
||||
- Logic: Off-by-one errors, null/undefined handling, race conditions
|
||||
- Tests: Are new code paths covered by tests?
|
||||
|
||||
#### Backend (`apps/opik-backend/**`)
|
||||
- Architecture: Resources → Services → DAOs → Models pattern
|
||||
- SQL: Parameterized queries, proper transaction handling
|
||||
- API: RESTful conventions, proper HTTP status codes, input validation
|
||||
- Migrations: Backward-compatible schema changes, proper rollback support
|
||||
- Logging: Appropriate log levels, no sensitive data in logs
|
||||
|
||||
#### Frontend (`apps/opik-frontend/**`)
|
||||
- Components: Proper React patterns, hook dependencies, memoization where needed
|
||||
- State: Appropriate state management, no prop drilling
|
||||
- Types: TypeScript type safety, no unnecessary `any`
|
||||
- Performance: Unnecessary re-renders, large bundle imports
|
||||
- Accessibility: Proper ARIA attributes, keyboard navigation
|
||||
|
||||
#### SDKs (`sdks/**`)
|
||||
- API compatibility: Breaking changes flagged
|
||||
- Error handling: Clear error messages, proper exception hierarchies
|
||||
- Documentation: Public API methods have proper docs
|
||||
|
||||
---
|
||||
|
||||
### 5. Generate Review Findings
|
||||
|
||||
Organize findings into categories with severity levels:
|
||||
|
||||
#### Severity Levels
|
||||
- 🚫 **blocker**: Must fix before merge — bugs, security issues, data loss risks
|
||||
- 💡 **suggestion**: Recommended improvement — better patterns, performance, readability
|
||||
- 🧹 **nit**: Minor style/preference — naming, formatting, minor simplifications
|
||||
- ❓ **question**: Needs clarification — unclear intent, missing context, design decisions
|
||||
|
||||
#### Finding Format
|
||||
For each finding, prepare:
|
||||
- **File and line range**: Where in the diff the issue is
|
||||
- **Category**: Security, Logic, Performance, Style, Architecture, Testing, etc.
|
||||
- **Severity**: blocker / suggestion / nit / question
|
||||
- **Description**: Clear explanation of the issue
|
||||
- **Suggestion** (if applicable): Concrete code suggestion using GitHub suggestion syntax
|
||||
|
||||
---
|
||||
|
||||
### 6. Present Findings and Get Approval
|
||||
|
||||
- **Deduplicate**: Compare findings against existing review comments fetched in step 3. If a finding targets the same `path + start_line + line` as an existing `/review-github-pr` comment, mark it as "already posted" and exclude it from the list. If all findings are duplicates, print: "All findings were already posted in a previous run." and stop.
|
||||
- **Display summary**: Show total count by severity (e.g., "Found 2 blockers, 3 suggestions, 1 nit, 1 question — 1 already posted, skipped")
|
||||
- **For each finding, show**:
|
||||
- The relevant code snippet from the diff
|
||||
- The review comment that would be posted
|
||||
- The code suggestion (if applicable)
|
||||
- **Ask user for decisions**: Offer bulk and per-item options:
|
||||
- **Post all**: Post every finding
|
||||
- **Post blockers only**: Post only blocker-severity findings
|
||||
- **Post all except nits**: Post blockers, suggestions, and questions
|
||||
- **Cherry-pick**: Let user select individual findings to post, skip, or edit
|
||||
- **For cherry-pick mode**, per finding:
|
||||
- **Post**: Post this comment as-is
|
||||
- **Edit**: Modify the comment text before posting
|
||||
- **Skip**: Don't post this one
|
||||
|
||||
---
|
||||
|
||||
### 7. Post Review Comments
|
||||
|
||||
Post approved comments as a **single batched review** to minimize notifications to the PR author. All inline comments are grouped into one review submission.
|
||||
|
||||
#### Batched Review (primary approach)
|
||||
|
||||
Collect all approved inline comments into a JSON payload and submit as one review:
|
||||
|
||||
```bash
|
||||
gh api repos/comet-ml/opik/pulls/{pr_number}/reviews \
|
||||
--input - <<'EOF'
|
||||
{
|
||||
"commit_id": "<head_sha>",
|
||||
"event": "COMMENT",
|
||||
"comments": [
|
||||
{
|
||||
"path": "<file_path>",
|
||||
"line": <line_number>,
|
||||
"side": "RIGHT",
|
||||
"body": "<comment text with AI marker>"
|
||||
},
|
||||
{
|
||||
"path": "<file_path>",
|
||||
"start_line": <start_line>,
|
||||
"line": <end_line>,
|
||||
"start_side": "RIGHT",
|
||||
"side": "RIGHT",
|
||||
"body": "<multi-line comment text with AI marker>"
|
||||
}
|
||||
]
|
||||
}
|
||||
EOF
|
||||
```
|
||||
|
||||
This sends **one notification** to the PR author regardless of how many comments are included.
|
||||
|
||||
#### Fallback: Individual Comments
|
||||
|
||||
If the batched review fails (e.g., a comment references a line not in the diff), fall back to posting comments individually. Identify which comments caused the failure and post the valid ones one by one:
|
||||
|
||||
```bash
|
||||
gh api repos/comet-ml/opik/pulls/{pr_number}/comments \
|
||||
-f body="<comment text>" \
|
||||
-f path="<file_path>" \
|
||||
-f commit_id="<head_sha>" \
|
||||
-F line=<line_number> \
|
||||
-f side="RIGHT"
|
||||
```
|
||||
|
||||
Log any comments that still fail individually (e.g., line not in diff) and fall back to posting them as general PR comments instead.
|
||||
|
||||
#### Code Suggestions
|
||||
|
||||
When suggesting specific code changes, use GitHub suggestion syntax in the comment body:
|
||||
|
||||
````
|
||||
<description of the suggestion>
|
||||
|
||||
```suggestion
|
||||
<replacement code>
|
||||
```
|
||||
````
|
||||
|
||||
#### General Comments (PR-level)
|
||||
|
||||
For comments not tied to a specific line (overall architecture, missing tests, etc.), post as issue comments. These cannot be included in the batched review and are posted separately:
|
||||
|
||||
```bash
|
||||
gh api repos/comet-ml/opik/issues/{pr_number}/comments \
|
||||
-f body="<comment text>"
|
||||
```
|
||||
|
||||
#### Secret Sanitization
|
||||
|
||||
Before posting any comment, scan the comment body for common secret patterns (API keys, bearer tokens, private key blocks, long hex/base64 strings, `export VAR=value` assignments). Replace any detected values with `[REDACTED]`. For files flagged as sensitive in Step 3, never include raw file content in comment bodies — reference the file and line range without quoting the actual values.
|
||||
|
||||
#### AI Marker
|
||||
|
||||
All posted comments **must** include a footer marker to distinguish them from human-written comments:
|
||||
|
||||
```
|
||||
🤖 *Review posted via /review-github-pr*
|
||||
```
|
||||
|
||||
#### Example Posted Comments
|
||||
|
||||
````
|
||||
🚫 **blocker** | Security
|
||||
|
||||
This query concatenates user input directly. Use parameterized queries instead.
|
||||
|
||||
```suggestion
|
||||
String query = "SELECT * FROM traces WHERE id = ?";
|
||||
jdbi.withHandle(h -> h.createQuery(query).bind(0, traceId).mapTo(Trace.class).one());
|
||||
```
|
||||
|
||||
🤖 *Review posted via /review-github-pr*
|
||||
````
|
||||
|
||||
````
|
||||
💡 **suggestion** | Performance
|
||||
|
||||
Consider using batch insert here to avoid N+1 queries.
|
||||
|
||||
🤖 *Review posted via /review-github-pr*
|
||||
````
|
||||
|
||||
````
|
||||
🧹 **nit** | Style
|
||||
|
||||
This variable name could be more descriptive.
|
||||
|
||||
🤖 *Review posted via /review-github-pr*
|
||||
````
|
||||
|
||||
````
|
||||
❓ **question** | Architecture
|
||||
|
||||
Is this intentionally bypassing the service layer? The other endpoints go through SpanService first.
|
||||
|
||||
🤖 *Review posted via /review-github-pr*
|
||||
````
|
||||
|
||||
---
|
||||
|
||||
### 8. Generate & Approve Summary Comment
|
||||
|
||||
After inline comments are handled, generate a **general PR summary comment** that gives the author a high-level view of the review.
|
||||
|
||||
#### What to include
|
||||
|
||||
1. **Positive observations** (always lead with these): Call out things the PR does well — clean architecture, good test coverage, clear naming, smart abstractions, thorough error handling, well-structured migrations, etc. Be specific and genuine; don't fabricate praise.
|
||||
2. **Solution assessment**: A brief, honest take on the PR as a whole — does the approach make sense? Is the scope appropriate? Are there any structural concerns not captured by inline comments?
|
||||
3. **Findings recap**: One-line summary of what was flagged inline (e.g., "Left 2 suggestions and 1 nit — nothing blocking").
|
||||
4. **Tone**: Supportive and constructive. The goal is to make the author feel their work is seen and valued, while still being honest about improvements.
|
||||
|
||||
#### Format
|
||||
|
||||
````
|
||||
👋 **Review summary**
|
||||
|
||||
**What looks good**
|
||||
- <specific positive observation>
|
||||
- <another positive observation>
|
||||
|
||||
**Overall**
|
||||
<1–3 sentences on the PR as a solution>
|
||||
|
||||
**Inline comments**: <count by severity, e.g., "2 suggestions, 1 nit"> (or "None — looks clean!" if no findings were posted)
|
||||
|
||||
🤖 *Review posted via /review-github-pr*
|
||||
````
|
||||
|
||||
#### User approval
|
||||
|
||||
- **Show the summary comment** to the user before posting
|
||||
- **Offer options**:
|
||||
- **Post**: Post this summary as-is
|
||||
- **Edit**: Modify the summary text before posting
|
||||
- **Skip**: Don't post a summary comment
|
||||
|
||||
#### Deduplication
|
||||
|
||||
Before generating, check if a summary comment from a previous run already exists (look for comments containing both `👋 **Review summary**` and the AI marker). If found, inform the user: "A summary comment was already posted in a previous run." and skip this step.
|
||||
|
||||
#### Posting (if approved)
|
||||
|
||||
```bash
|
||||
gh api repos/comet-ml/opik/issues/{pr_number}/comments \
|
||||
-f body="<summary comment>"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 9. Local Summary
|
||||
|
||||
After all posting is complete, display to the user (not on GitHub):
|
||||
- Total inline comments posted (by severity)
|
||||
- Total inline comments skipped
|
||||
- Whether the summary comment was posted, edited, or skipped
|
||||
- Link to the PR
|
||||
- Reminder: "This review does not constitute an approval. A human reviewer should still approve the PR."
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### **CLI Authentication Errors**
|
||||
|
||||
- **`gh` not installed**: Stop and provide installation instructions
|
||||
- **`gh` not authenticated**: Stop and instruct to run `gh auth login`
|
||||
|
||||
### **PR Discovery Errors**
|
||||
|
||||
- **PR not found**: Prompt for manual input or stop
|
||||
- **Invalid PR number/URL**: Show expected format and re-prompt
|
||||
- **PR is in draft**: Warn but allow review to proceed
|
||||
|
||||
### **Diff Fetch Errors**
|
||||
|
||||
- **Diff too large**: Review files in batches, warn about potential missed context
|
||||
- **Binary files**: Skip with note
|
||||
- **API rate limiting**: Wait and retry, or stop with guidance
|
||||
|
||||
### **Comment Posting Errors**
|
||||
|
||||
- **Line not in diff**: Fall back to a general PR comment instead of inline
|
||||
- **Permission denied**: Check repository access and `gh` auth
|
||||
- **API errors**: Show error details, skip the failed comment, continue with remaining
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. ✅ `gh` CLI is available and authenticated
|
||||
2. ✅ PR is located and metadata fetched
|
||||
3. ✅ PR diff and changed files are retrieved
|
||||
4. ✅ Changes are analyzed with domain-specific knowledge
|
||||
5. ✅ Findings are presented to user with severity and categories
|
||||
6. ✅ User approves which findings to post
|
||||
7. ✅ Approved inline comments are posted with AI marker
|
||||
8. ✅ Summary comment is presented for approval and posted (if approved)
|
||||
9. ✅ Local summary is displayed with posted/skipped counts
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- **Repository**: Always targets `comet-ml/opik`
|
||||
- **`gh` CLI is the only hard requirement**: GitHub MCP is optional and used for richer reading when available. All operations can be performed with `gh` CLI alone.
|
||||
- **No formal review submission**: This command submits reviews with event `COMMENT` only — NEVER with "approve" or "request changes". Human reviewers must still formally approve.
|
||||
- **Complementary to `/address-github-pr-comments`**: That command responds to existing review feedback. This command generates review feedback. Both post comments via `gh api` with AI markers.
|
||||
- **AI marker required**: Every posted comment must include the `🤖 *Review posted via /review-github-pr*` footer — never omit it
|
||||
- **Domain-aware**: Uses Opik-specific rules from `.agents/skills/` and `.agents/rules/` to provide relevant, project-specific feedback rather than generic code review
|
||||
- **User control**: Every finding is presented for approval before posting. Nothing is posted without explicit user consent.
|
||||
- **Idempotent**: Re-runs full analysis on every invocation but deduplicates against existing `/review-github-pr` comments on the PR (matched by `path + start_line + line`). Safe to run multiple times — only new findings are posted.
|
||||
- **Severity-driven**: Findings are organized by severity to help users prioritize what matters
|
||||
- **Code suggestions**: Uses GitHub's suggestion syntax so PR authors can apply fixes with one click
|
||||
- **No approval by design**: The command is intentionally restricted from submitting formal approvals. Code review approval must remain a human-in-the-loop action.
|
||||
- **No Delete Operations**: This command only creates content; it never deletes files, comments, or other content
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,399 @@
|
||||
# Send Code Review Slack Message
|
||||
|
||||
**Command**: `cursor send-code-review-slack`
|
||||
|
||||
## Overview
|
||||
|
||||
Send a formatted Slack message to the #code-review channel with PR information, Jira ticket, test environment link, and optional component summaries (FE, BE, Python, TypeScript). Automatically extracts information from the GitHub PR for the current branch.
|
||||
|
||||
- **Execution model**: Automatically extracts information from GitHub PR, prompts only for missing information, formats the message according to the template, and sends it via Slack MCP.
|
||||
|
||||
This workflow will:
|
||||
|
||||
- Find the GitHub PR for the current branch
|
||||
- Extract Jira ticket from PR title
|
||||
- Extract test environment link from PR description
|
||||
- Extract component summaries (FE, BE, Python, TypeScript) from PR description
|
||||
- Prompt only for missing information
|
||||
- Allow user to customize the message slightly
|
||||
- Format the message according to the code review template
|
||||
- Send the message to #code-review channel via Slack MCP (`ghcr.io/korotovsky/slack-mcp-server`)
|
||||
- Verify successful delivery
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
### Required Information (auto-extracted from PR, prompted if missing)
|
||||
- **Jira ticket**: Extracted from PR title (format: `[OPIK-1234]`) or prompted if not found
|
||||
- **PR link**: Automatically determined from current branch's PR or prompted if no PR found
|
||||
- **Test env link**: Extracted from PR description (Testing or Details section) or prompted if not found
|
||||
|
||||
### Optional Information (auto-extracted from PR, prompted if missing)
|
||||
- **FE summary**: Extracted from PR description (looks for frontend/React/FE mentions) or prompted if not found
|
||||
- **BE summary**: Extracted from PR description (looks for backend/Java/BE mentions) or prompted if not found
|
||||
- **Python summary**: Extracted from PR description (looks for Python/Python SDK mentions) or prompted if not found
|
||||
- **TypeScript summary**: Extracted from PR description (looks for TypeScript/TypeScript SDK/TS mentions) or prompted if not found
|
||||
- **Baz approved status**: Extracted from PR status checks (optional, may not be available)
|
||||
|
||||
### Configuration
|
||||
- **Slack MCP**: Required - uses custom Slack MCP server (`ghcr.io/korotovsky/slack-mcp-server`)
|
||||
- Uses `SLACK_MCP_XOXP_TOKEN` (User OAuth Token) to post messages as your authenticated user account
|
||||
- See `.agents/docs/SLACK_MCP_SETUP.md` for complete setup instructions
|
||||
- Requires Docker to be installed and running
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Check GitHub MCP**: Test GitHub MCP availability by attempting to fetch repository info using `get_file_contents` for `comet-ml/opik`
|
||||
> If unavailable, respond with: "This command needs GitHub MCP configured. Set MCP config/env, run `make cursor` (Cursor) or `make claude` (Claude CLI), then retry."
|
||||
> Stop here.
|
||||
- **Check Git repository**: Verify we're in a Git repository
|
||||
- **Check current branch**: Get current branch name
|
||||
- **Check Slack MCP availability**:
|
||||
- Use Slack MCP tool `channels_list` to test availability (from `ghcr.io/korotovsky/slack-mcp-server`)
|
||||
- Call with minimal parameters (for example: `limit: 1`) so that both public and private channels can be returned
|
||||
- This custom MCP server uses `SLACK_MCP_XOXP_TOKEN` (User OAuth Token) to post messages as your authenticated user account
|
||||
- The server must be configured with `mcp-server --transport stdio` in the Docker args and have the `conversations_add_message` tool enabled (e.g., via `SLACK_MCP_ADD_MESSAGE_TOOL=true`) so the final send step can succeed
|
||||
- If Slack MCP tool is available and callable: Proceed with message sending via MCP (messages will post as user)
|
||||
- If Slack MCP is not available or tool call fails:
|
||||
> "Slack MCP is not available. Please configure Slack MCP according to `.agents/docs/SLACK_MCP_SETUP.md` and restart Cursor IDE."
|
||||
> Stop here.
|
||||
|
||||
---
|
||||
|
||||
### 2. Find GitHub PR for Current Branch
|
||||
|
||||
- **Search for PR**: Use GitHub MCP to find an open PR for the current branch in `comet-ml/opik`
|
||||
- **If PR found**:
|
||||
- Extract PR number, URL, title, and description
|
||||
- Store PR information for extraction steps
|
||||
- **If no PR found**:
|
||||
> "No open PR found for this branch. Please provide the PR link manually:"
|
||||
- Prompt for PR link and validate URL format
|
||||
- If provided, fetch PR details using GitHub MCP
|
||||
- If PR cannot be fetched, stop with error message
|
||||
- **After successfully fetching PR by manual link**: Extract PR number, URL, title, and description and store PR information for extraction steps (same as "PR found" branch)
|
||||
|
||||
---
|
||||
|
||||
### 3. Extract Information from PR
|
||||
|
||||
- **Extract Jira ticket from PR title**:
|
||||
- Parse PR title for pattern `[OPIK-\d+]` or `[issue-\d+]` or `[NA]`
|
||||
- Extract ticket number (e.g., `OPIK-1234` from `[OPIK-1234] [BE] feat(api): add trace request validation endpoint`)
|
||||
- If not found in title, check PR description "## Issues" section for `OPIK-\d+` pattern
|
||||
- If still not found, prompt: "Jira ticket not found in PR. Enter Jira ticket number (e.g., OPIK-1234):"
|
||||
- Validate format and store
|
||||
- **Build Jira URL**: Construct Jira link as `https://comet-ml.atlassian.net/browse/{TICKET}` (e.g., `https://comet-ml.atlassian.net/browse/OPIK-1234`)
|
||||
|
||||
- **Extract Baz approved status (optional)**:
|
||||
- Try to fetch PR status checks using GitHub MCP
|
||||
- Look for status checks or CI checks that might indicate "Baz approved" or similar approval status
|
||||
- If found, store the status (e.g., "Baz approved: ✅" or "Baz status: pending")
|
||||
- If not found or unavailable, skip this field (it\'s optional)
|
||||
|
||||
- **Extract PR size**: Follow the shared steps in `.agents/docs/PR_SIZE_EXTRACTION.md` (read the `size/*` label; fall back to changed lines using the workflow's own ignore list + thresholds). Store the bucket as `{emoji} {BUCKET}` (e.g. `🟠 L`).
|
||||
|
||||
- **Extract test environment link from PR description and comments**:
|
||||
- First, search PR comments for test environment deployment messages (look for "Test environment is now available!" or similar)
|
||||
- When fetching PR comments via `gh api`, **always** use `--paginate` to ensure all results are fetched:
|
||||
```bash
|
||||
# Issue comments (general PR comments, including deployment bot messages)
|
||||
gh api repos/comet-ml/opik/issues/{pr_number}/comments --paginate
|
||||
|
||||
# Review comments (inline code comments)
|
||||
gh api repos/comet-ml/opik/pulls/{pr_number}/comments --paginate
|
||||
```
|
||||
> **Why pagination is required**: GitHub API returns 30 items per page by default. Opik PRs regularly exceed this — 17 CI test group comments + deployment bot comments + reviewer comments can push past 30 total. Without `--paginate`, the deployment bot comment with the test environment link may end up on page 2 and get silently missed.
|
||||
- Extract URL from comment body (typically in format `https://pr-XXXX.dev.comet.com` or `https://test.opik.com`)
|
||||
- If not found in comments, search PR description for URLs in "## Testing" section
|
||||
- Look for common test environment patterns: `https://pr-*.dev.comet.com`, `https://test.opik.com`, `https://*.opik.com`, or any `https://` URL
|
||||
- If multiple URLs found, prefer the first one that looks like a test environment
|
||||
- If not found, search in "## Details" section for URLs
|
||||
- If still not found, prompt: "Test environment link not found in PR. Enter test environment link (e.g., https://pr-4743.dev.comet.com):"
|
||||
- Validate URL format and store
|
||||
|
||||
- **Extract component summaries from PR description**:
|
||||
- **FE summary**:
|
||||
- Look for mentions of "frontend", "FE", "React", "UI", or `[FE]` tag in PR description
|
||||
- Extract relevant one-line summary from "## Details" section or component-specific mentions
|
||||
- If not found, prompt: "Frontend summary not found in PR. Enter frontend summary (one line, optional - press Enter to skip):"
|
||||
|
||||
- **BE summary**:
|
||||
- Look for mentions of "backend", "BE", "Java", "API", or `[BE]` tag in PR description
|
||||
- Extract relevant one-line summary from "## Details" section or component-specific mentions
|
||||
- If not found, prompt: "Backend summary not found in PR. Enter backend summary (one line, optional - press Enter to skip):"
|
||||
|
||||
- **Python summary**:
|
||||
- Look for mentions of "Python", "Python SDK", "SDK", or Python-related changes in PR description
|
||||
- Extract relevant one-line summary from "## Details" section or component-specific mentions
|
||||
- If not found, prompt: "Python summary not found in PR. Enter Python summary (one line, optional - press Enter to skip):"
|
||||
|
||||
- **TypeScript summary**:
|
||||
- Look for mentions of "TypeScript", "TypeScript SDK", "TS", "TS SDK", or `[TS]` tag in PR description
|
||||
- Extract relevant one-line summary from "## Details" section or component-specific mentions
|
||||
- If not found, prompt: "TypeScript summary not found in PR. Enter TypeScript summary (one line, optional - press Enter to skip):"
|
||||
|
||||
- **Store extracted information**: Keep all extracted and prompted values for message formatting
|
||||
|
||||
---
|
||||
|
||||
### 4. Prompt for Message Customization
|
||||
|
||||
- **Allow user to customize message**: After extracting all information, prompt the user:
|
||||
> "Would you like to customize the message? (Enter any additional text to prepend/append, or press Enter to use default message):"
|
||||
- **If user provides customization text**: Store it for inclusion in the final message
|
||||
- **If user presses Enter (empty)**: Proceed with default message format
|
||||
- **Note**: User can add context, emphasize certain points, or include additional information
|
||||
|
||||
---
|
||||
|
||||
### 5. Format Slack Message
|
||||
|
||||
- **Use PR link**: Use the PR URL extracted from Step 2 (or provided manually)
|
||||
|
||||
- **Build message text** according to the template:
|
||||
```
|
||||
Hi team,
|
||||
|
||||
Please review the following PR:
|
||||
|
||||
{{user_customization_text_if_provided}}
|
||||
|
||||
:jira_epic: jira link: {{Jira_URL}}
|
||||
:github: pr link: {{PR_link}}
|
||||
:straight_ruler: pr size: {{pr_size}}
|
||||
:test_tube: test env link: {{test_env}}
|
||||
{{baz_approved_status_if_available}}
|
||||
:react: fe summary (optional): {{description_in_one_line}}
|
||||
:java: be summary (optional): {{description_in_one_line}}
|
||||
:python: python summary (optional): {{description_in_one_line}}
|
||||
:typescript: typescript summary (optional): {{description_in_one_line}}
|
||||
```
|
||||
|
||||
- **Message structure**:
|
||||
- **Start with greeting**: Always begin with "Hi team,\n\nPlease review the following PR:\n"
|
||||
- **User customization**: If user provided customization text in Step 4, include it after the greeting (before the structured fields)
|
||||
- **Jira link**: Include full Jira URL with ticket (e.g., `https://comet-ml.atlassian.net/browse/OPIK-1234`)
|
||||
- **PR link**: Include GitHub PR URL
|
||||
- **PR size**: Include the size bucket extracted in Step 3 (e.g., `🟠 L`). Always included — size is always available from GitHub.
|
||||
- **Test env link**: Include test environment URL
|
||||
- **Baz approved status**: Only include if extracted from PR status checks (optional field)
|
||||
- **Component summaries**: Only include optional fields that were provided (skip empty ones)
|
||||
|
||||
- **Only include optional fields** that were provided (skip empty ones)
|
||||
- **Format example**:
|
||||
```
|
||||
Hi team,
|
||||
|
||||
Please review the following PR:
|
||||
|
||||
:jira_epic: jira link: https://comet-ml.atlassian.net/browse/OPIK-1234
|
||||
:github: pr link: https://github.com/comet-ml/opik/pull/1234
|
||||
:straight_ruler: pr size: 🟠 L
|
||||
:test_tube: test env link: https://test.opik.com
|
||||
:react: fe summary (optional): Added new metrics dashboard UI
|
||||
:java: be summary (optional): Implemented metrics aggregation endpoint
|
||||
:typescript: typescript summary (optional): Added TypeScript SDK support for metrics
|
||||
```
|
||||
|
||||
- **Note about videos**: Slack has limitations on sending videos directly. Videos should be added to the PR description, and the link can be shared in Slack. For easier communication with the product team, consider using the `cursor generate-code-review-slack-command` command to generate a copiable Slack command that you can edit before sending (allowing you to add @ mentions, media links, and final proof editing).
|
||||
|
||||
---
|
||||
|
||||
### 6. Send Slack Message
|
||||
|
||||
- **Use Slack MCP tool `conversations_add_message`** from `ghcr.io/korotovsky/slack-mcp-server`
|
||||
- This custom MCP server uses `SLACK_MCP_XOXP_TOKEN` (User OAuth Token) configured according to `.agents/docs/SLACK_MCP_SETUP.md`
|
||||
- The server must be configured with:
|
||||
- `mcp-server --transport stdio` in the Docker args (required for MCP protocol)
|
||||
- `SLACK_MCP_ADD_MESSAGE_TOOL=true` environment variable (required to enable the tool - disabled by default for safety)
|
||||
- Call the tool with the following parameters:
|
||||
- `channel_id`: `#code-review` (channel name starting with # or channel ID)
|
||||
- `payload`: The formatted message text (the complete message with all fields)
|
||||
- `content_type`: `text/markdown` (optional, defaults to text/markdown)
|
||||
- **Important**: Messages will be posted as your authenticated user account (not as a bot)
|
||||
- Handle MCP response:
|
||||
- If successful: Show success message with message timestamp/ID if provided
|
||||
- If failed: Show error message and stop
|
||||
- Common errors:
|
||||
- **Tool disabled error**: The `conversations_add_message` tool is disabled by default. Add `SLACK_MCP_ADD_MESSAGE_TOOL=true` to Docker args and restart Cursor (see `.agents/docs/SLACK_MCP_SETUP.md`)
|
||||
- Authentication errors: Check `SLACK_MCP_XOXP_TOKEN` configuration (should start with `xoxp-`) - see `.agents/docs/SLACK_MCP_SETUP.md`
|
||||
- Channel not found: Verify channel name `#code-review` exists and you have access
|
||||
- Permission errors: Verify `chat:write` scope is configured in User Token Scopes (see `.agents/docs/SLACK_MCP_SETUP.md`)
|
||||
- Docker errors: Ensure Docker is running and can pull the image `ghcr.io/korotovsky/slack-mcp-server:latest`
|
||||
- MCP not configured: Verify Slack MCP is properly configured according to `.agents/docs/SLACK_MCP_SETUP.md` and Cursor IDE has been restarted
|
||||
|
||||
---
|
||||
|
||||
### 7. Verification & Summary
|
||||
|
||||
- **Display confirmation**:
|
||||
> "✅ Slack message sent successfully to #code-review channel"
|
||||
|
||||
- **Show message preview**: Display the formatted message that was sent
|
||||
- **Provide next steps**: Remind user to check Slack channel for delivery
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### **Slack MCP Errors**
|
||||
|
||||
- **Slack MCP unavailable**:
|
||||
- Check if Slack MCP server is properly configured according to `.agents/docs/SLACK_MCP_SETUP.md`
|
||||
- Verify Docker is installed and running (`docker --version`)
|
||||
- Verify the Docker image can be pulled: `docker pull ghcr.io/korotovsky/slack-mcp-server:latest`
|
||||
- Restart Cursor IDE after configuring MCP
|
||||
- **Slack MCP tool not found**:
|
||||
- Verify `conversations_add_message` tool is available from `ghcr.io/korotovsky/slack-mcp-server`
|
||||
- Ensure `mcp-server --transport stdio` is included in Docker args (see `.agents/docs/SLACK_MCP_SETUP.md`)
|
||||
- Ensure `SLACK_MCP_ADD_MESSAGE_TOOL=true` is set in Docker args (see `.agents/docs/SLACK_MCP_SETUP.md`)
|
||||
- Check Cursor Settings > Features > MCP for error messages
|
||||
- Ensure MCP server is running and properly configured
|
||||
- Check Docker logs if the container fails to start
|
||||
- **MCP authentication errors**:
|
||||
- Check `SLACK_MCP_XOXP_TOKEN` configuration (should start with `xoxp-` for user token) - see `.agents/docs/SLACK_MCP_SETUP.md`
|
||||
- Verify token is not expired
|
||||
- Reinstall Slack app to workspace if needed
|
||||
- Ensure token has `chat:write` scope in **User Token Scopes** (not Bot Token Scopes) - see `.agents/docs/SLACK_MCP_SETUP.md`
|
||||
- **Docker errors**:
|
||||
- Ensure Docker is running: `docker ps`
|
||||
- Check if Docker can pull the image: `docker pull ghcr.io/korotovsky/slack-mcp-server:latest`
|
||||
- Verify Docker has network access to reach Slack API
|
||||
- **Channel access errors**:
|
||||
- Verify channel name `#code-review` exists
|
||||
- Ensure you have `chat:write` scope in User Token Scopes (see `.agents/docs/SLACK_MCP_SETUP.md`)
|
||||
- Check that you are a member of the channel (join it if needed)
|
||||
- Try using channel ID instead of name if issues persist
|
||||
|
||||
### **GitHub MCP Errors**
|
||||
|
||||
- **GitHub MCP unavailable**: Stop immediately after testing and provide setup instructions
|
||||
- **PR not found**: Prompt for manual PR link or stop if user cancels
|
||||
- **PR fetch failure**: Show error details and suggest manual input
|
||||
|
||||
### **Extraction Errors**
|
||||
|
||||
- **Jira ticket not found**: Prompt user for ticket number
|
||||
- **Test env link not found**: Prompt user for test environment URL
|
||||
- **Component summaries not found**: Prompt user for summaries (optional)
|
||||
|
||||
### **Input Validation Errors**
|
||||
|
||||
- **Invalid Jira ticket format**: Show expected format (e.g., `OPIK-1234`) and re-prompt
|
||||
- **Invalid PR URL**: Show expected format and re-prompt
|
||||
- **Invalid test env URL**: Show expected format and re-prompt
|
||||
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. ✅ GitHub MCP is available and accessible
|
||||
2. ✅ PR is found for current branch (or manually provided)
|
||||
3. ✅ Slack MCP is available and configured with `SLACK_MCP_XOXP_TOKEN` (User OAuth Token)
|
||||
4. ✅ Jira ticket is extracted from PR or provided manually
|
||||
5. ✅ Test environment link is extracted from PR or provided manually
|
||||
6. ✅ Component summaries are extracted from PR or provided manually (optional)
|
||||
7. ✅ Message is formatted according to template (with greeting and Jira link)
|
||||
8. ✅ Slack message is sent successfully via `conversations_add_message` MCP tool (posts as user account)
|
||||
9. ✅ User receives confirmation of successful delivery
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- **GitHub MCP Required**: Uses GitHub MCP to fetch PR information automatically
|
||||
- **Slack MCP Required**: Uses custom Slack MCP server (`ghcr.io/korotovsky/slack-mcp-server`) for sending messages
|
||||
- Configure according to `.agents/docs/SLACK_MCP_SETUP.md`
|
||||
- Uses `SLACK_MCP_XOXP_TOKEN` (User OAuth Token) - messages are posted as your authenticated user account
|
||||
- Uses `conversations_add_message` tool with `channel_id` and `payload` parameters
|
||||
- Uses `channels_list` tool to verify MCP availability
|
||||
- Requires `mcp-server --transport stdio` in Docker args for proper MCP protocol communication
|
||||
- Requires `SLACK_MCP_ADD_MESSAGE_TOOL=true` to enable the tool
|
||||
- Requires Docker to be installed and running
|
||||
- Token starts with `xoxp-` and requires `chat:write` scope in User Token Scopes
|
||||
- **Message format**: Follows the exact template provided with emoji prefixes, includes greeting and Jira link
|
||||
- **Automatic extraction**: Extracts information from PR title and description to minimize manual input
|
||||
- **Fallback to prompts**: Only prompts for information that cannot be extracted from PR
|
||||
- **Optional fields**: Only included in message if extracted from PR or provided by user
|
||||
- **Channel**: Message is always sent to `#code-review` channel
|
||||
- **PR detection**: Automatically finds PR for current branch, falls back to manual input if needed
|
||||
- **Smart extraction**: Uses heuristics to find test environment links and component summaries in PR description
|
||||
- **MCP Configuration**: Slack MCP must be configured before using this command (see `.agents/docs/SLACK_MCP_SETUP.md` for detailed setup)
|
||||
- **Video limitations**: Slack cannot send videos directly. Use `cursor generate-code-review-slack-command` to generate a copiable command that you can edit before sending (allowing you to add @ mentions, media links, and final proof editing)
|
||||
|
||||
---
|
||||
|
||||
## Example Usage
|
||||
|
||||
### Setup and Usage
|
||||
|
||||
```bash
|
||||
# 1. Configure Slack MCP according to .agents/docs/SLACK_MCP_SETUP.md
|
||||
# - Follow the complete setup guide for configuring the Slack MCP server
|
||||
# - Add your User OAuth Token (xoxp-...) to the Docker args
|
||||
# - Ensure 'mcp-server --transport stdio' is included in Docker args
|
||||
# - Ensure 'SLACK_MCP_ADD_MESSAGE_TOOL=true' is set to enable the tool
|
||||
|
||||
# 2. Restart Cursor IDE to load the MCP configuration
|
||||
|
||||
# 3. Run command (on a branch with an open PR)
|
||||
cursor send-code-review-slack
|
||||
```
|
||||
|
||||
|
||||
# Command execution flow:
|
||||
# 1. Find PR for current branch: https://github.com/comet-ml/opik/pull/1234
|
||||
# 2. Extract Jira ticket from PR title: [OPIK-1234] [BE] feat(api): add metrics dashboard
|
||||
# 3. Extract test env from PR comments or description: https://pr-1234.dev.comet.com (from PR comments or Testing section)
|
||||
# Extract PR size from the size/* label (or additions+deletions): 🟠 L
|
||||
# 4. Extract summaries from PR description:
|
||||
# - FE: Added new metrics dashboard UI (from Details section)
|
||||
# - BE: Implemented metrics aggregation endpoint (from Details section)
|
||||
# - Python: (not found, prompts user)
|
||||
# - TypeScript: (not found, prompts user)
|
||||
# 5. Prompt for message customization (optional)
|
||||
# 6. Format message according to template (with greeting and Jira link)
|
||||
# 7. Send formatted message to Slack via conversations_add_message MCP tool (posts as user account)
|
||||
|
||||
# Output:
|
||||
# ✅ Slack message sent successfully to #code-review channel
|
||||
#
|
||||
# Message sent:
|
||||
# :jira_epic: jira link: https://comet-ml.atlassian.net/browse/OPIK-1234
|
||||
# :github: pr link: https://github.com/comet-ml/opik/pull/1234
|
||||
# :straight_ruler: pr size: 🟠 L
|
||||
# :test_tube: test env link: https://test.opik.com
|
||||
# :react: fe summary (optional): Added new metrics dashboard UI
|
||||
# :java: be summary (optional): Implemented metrics aggregation endpoint
|
||||
# :typescript: typescript summary (optional): Added TypeScript SDK support for metrics
|
||||
```
|
||||
|
||||
### Example with Missing Information
|
||||
|
||||
If some information cannot be extracted from PR, the command will prompt:
|
||||
|
||||
```bash
|
||||
cursor send-code-review-slack
|
||||
|
||||
# Found PR: https://github.com/comet-ml/opik/pull/1234
|
||||
# Extracted Jira ticket: OPIK-1234
|
||||
# Test environment link not found in PR. Enter test environment link (e.g., https://test.opik.com): https://test.opik.com
|
||||
# Frontend summary not found in PR. Enter frontend summary (one line, optional - press Enter to skip): [Enter pressed - skipped]
|
||||
# Backend summary not found in PR. Enter backend summary (one line, optional - press Enter to skip): Implemented metrics endpoint
|
||||
# Python summary not found in PR. Enter Python summary (one line, optional - press Enter to skip): [Enter pressed - skipped]
|
||||
# TypeScript summary not found in PR. Enter TypeScript summary (one line, optional - press Enter to skip): [Enter pressed - skipped]
|
||||
# Would you like to customize the message? (Enter any additional text to prepend/append, or press Enter to use default message): [Enter pressed - using default]
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,196 @@
|
||||
# Share Diagram
|
||||
|
||||
**Command**: `cursor share-diagram`
|
||||
|
||||
## Overview
|
||||
|
||||
Analyze the current branch's diff and Jira ticket context, then generate a self-contained HTML architecture diagram at `{MAIN_REPO_ROOT}/diagrams/opik-{TICKET_NUMBER}-diagram.html`.
|
||||
|
||||
- **Execution model**: Stateless. Each invocation re-checks tool availability (`gh` preferred, GitHub MCP fallback), re-analyzes the current branch diff and Jira context, then generates a fresh diagram.
|
||||
|
||||
This workflow will:
|
||||
|
||||
- Extract Jira ticket context from the current branch
|
||||
- Gather the git diff (PR diff if available, otherwise branch diff from main)
|
||||
- Analyze changes to understand data flow, architecture, and design decisions
|
||||
- Generate a self-contained HTML diagram using the `diagram-generation` skill
|
||||
- Save the diagram under the **main repo root**'s `diagrams/` folder (gitignored, local artifact) — even when running inside a worktree, so the diagram survives worktree removal
|
||||
- Provide the file path for the user to open in a browser
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **Ticket number (optional)**: Override auto-detection from branch name. E.g., `OPIK-4501`
|
||||
- **Focus (optional)**: What aspect to emphasize — `flow`, `architecture`, `files`, `decisions`, or `all` (default: `all`)
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Check Jira MCP**: Test availability by attempting to fetch user info
|
||||
> If unavailable: warn but continue — diagram can be generated from code diff alone without Jira context.
|
||||
- **Check GitHub CLI (preferred)**: Ensure `gh` is installed and authenticated (`gh auth status`)
|
||||
- **GitHub fallback path**: If `gh` is unavailable or unauthenticated, test GitHub MCP availability by fetching repository info for `comet-ml/opik`.
|
||||
> If both are unavailable: warn but continue — diagram can be generated from local git diff alone.
|
||||
- **Check git repository**: Verify we're in a git repo with commits.
|
||||
- **Resolve main repo root**: Compute `MAIN_REPO_ROOT="$(dirname "$(git rev-parse --path-format=absolute --git-common-dir)")"`. This returns the main worktree root whether we're currently in the main checkout or in a linked worktree under `.claude/worktrees/`. All diagram output goes under `${MAIN_REPO_ROOT}/diagrams/` so the files outlive any temporary worktree.
|
||||
- **Ensure `${MAIN_REPO_ROOT}/diagrams/` directory exists**: Create if missing.
|
||||
|
||||
---
|
||||
|
||||
### 2. Extract Context
|
||||
|
||||
#### Ticket Context
|
||||
- **Parse branch name**: Extract `OPIK-<number>` from current branch (format: `<username>/OPIK-<number>-<description>`)
|
||||
- **If ticket not found in branch name** (including `main`, `NA` branches like `user/NA-some-task`, or other non-standard formats): Check if a ticket number was provided as input. If not, ask the user for one.
|
||||
- **If no ticket number available**: Skip Jira fetch entirely — generate diagram from code diff alone, using the branch description as the title.
|
||||
- **Fetch Jira ticket** (if ticket found): Get summary, description, issue type, status, priority, labels, comments
|
||||
- **Build ticket context**: Title, type, key requirements, acceptance criteria
|
||||
|
||||
#### Code Changes
|
||||
- **Check for PR**: Use `gh pr view --json number,title,url` (preferred) or GitHub MCP to find an open PR for the current branch on `comet-ml/opik`
|
||||
- If PR exists: use `gh pr diff` or GitHub MCP PR diff (preferred — includes review context)
|
||||
- If no PR or no GitHub tool available: use `git diff main...HEAD` for local changes
|
||||
- **Parse diff**: Identify files changed, grouped by component layer:
|
||||
- **API layer**: Resources, endpoints, request/response models
|
||||
- **Service layer**: Business logic, services
|
||||
- **DAO layer**: Database access, SQL, queries
|
||||
- **Model layer**: Domain models, DTOs, enums
|
||||
- **Migration layer**: Database migrations (Liquibase, ClickHouse)
|
||||
- **Test layer**: Unit tests, integration tests
|
||||
- **Frontend layer**: React components, hooks, stores
|
||||
- **SDK layer**: Python/TypeScript SDK changes
|
||||
- **Identify patterns**: New files vs modified, what data flows through the change
|
||||
|
||||
---
|
||||
|
||||
### 3. Analyze & Plan Diagram Sections
|
||||
|
||||
Based on the changes, determine which sections to include:
|
||||
|
||||
| Change Type | Sections |
|
||||
|------------|----------|
|
||||
| **New API endpoint** | Request flow, files changed, design decisions |
|
||||
| **Bug fix** | Problem/solution banners, before/after flow, safety guards |
|
||||
| **New feature** | Data flow, architecture tree, files changed, design decisions |
|
||||
| **Refactor** | Before/after architecture, files changed |
|
||||
| **Database migration** | Storage diagram, data flow, migration notes |
|
||||
| **Cross-component** | Full flow (API → Service → DAO → Storage), files by layer |
|
||||
|
||||
Always include **Files Changed** section. Include at most 4 sections total.
|
||||
|
||||
---
|
||||
|
||||
### 4. Generate HTML Diagram
|
||||
|
||||
- **Load the `diagram-generation` skill**: Use the style guide and template from `.agents/skills/diagram-generation/`
|
||||
- **Fill template sections**:
|
||||
- Title: `OPIK-{TICKET} — {Jira Title}`
|
||||
- Subtitle: One-line summary of the change
|
||||
- Sections: Based on analysis from step 3
|
||||
- **Apply style guide**: Use semantic colors, flow boxes, section labels with dots, and all CSS patterns from the style guide
|
||||
- **Include "Copy as image" button**: Using the Canvas API + Clipboard API script from the template
|
||||
- **Content safety**:
|
||||
- **NEVER** include raw secrets, API keys, tokens, or passwords in diagram content
|
||||
- **NEVER** include customer names unless specifically requested — they should not be public
|
||||
- Diagram content should be high-level summaries (component names, flow descriptions, file names) — never raw diff hunks or verbatim Jira comments
|
||||
- **Content rules**:
|
||||
- Keep text concise — diagram is visual, not a document
|
||||
- Use `<b>` for component names inside boxes
|
||||
- Use `<br>` for secondary details in boxes
|
||||
- Use `.code` spans for inline code references
|
||||
- Use `.note` for supplementary explanations below flows
|
||||
- Max 4 sections per diagram
|
||||
- Each flow row should have 3-6 boxes max
|
||||
|
||||
---
|
||||
|
||||
### 5. Save Diagram
|
||||
|
||||
- **Output path**: `${MAIN_REPO_ROOT}/diagrams/opik-{TICKET_NUMBER}-diagram.html` — always under the main repo root, even when the session is running inside a worktree. This keeps the diagram accessible after the worktree is removed.
|
||||
- **Create `${MAIN_REPO_ROOT}/diagrams/` directory** if it doesn't exist
|
||||
- **Write the HTML file** using the Write tool (pass the absolute path)
|
||||
- **Verify file was written** by reading the first few lines back
|
||||
|
||||
---
|
||||
|
||||
### 6. Render PNG Preview (requires Playwright MCP)
|
||||
|
||||
Render the diagram to PNG and display it inline in chat.
|
||||
|
||||
**Note**: Playwright MCP blocks `file://` URLs, so serve the HTML over a local HTTP server.
|
||||
|
||||
1. **Start server**: `python3 -m http.server 8787 --bind 127.0.0.1` in the `${MAIN_REPO_ROOT}/diagrams/` directory (run in background)
|
||||
2. **Navigate**: `browser_navigate` to `http://localhost:8787/opik-{TICKET_NUMBER}-diagram.html`
|
||||
3. **Hide button**: `browser_evaluate` with `document.querySelector('.copy-btn').style.display = 'none'` — prevents the fixed-position button from appearing in the screenshot
|
||||
4. **Snapshot**: `browser_snapshot` to get the element ref for `#diagram`
|
||||
5. **Screenshot**: `browser_take_screenshot` targeting the `#diagram` element ref
|
||||
- Use an **absolute path** for the filename: `${MAIN_REPO_ROOT}/diagrams/opik-{TICKET_NUMBER}-diagram.png`
|
||||
- Playwright saves relative to its own CWD, so absolute paths ensure the PNG lands in the main repo's `diagrams/` folder (not the worktree's)
|
||||
6. **Close & cleanup**: `browser_close`, then kill the HTTP server process
|
||||
7. **Display**: Use the `Read` tool on the PNG — it renders as an image inline in the chat
|
||||
|
||||
If Playwright MCP is **not available**, skip this step and fall back to showing the HTML file path.
|
||||
|
||||
---
|
||||
|
||||
### 7. Present Result
|
||||
|
||||
- **Show the PNG inline** if generated (step 6) — this is the primary visual output
|
||||
- **Show the HTML file path** as a markdown link using the **absolute path** (e.g., `[diagrams/opik-{TICKET}-diagram.html](${MAIN_REPO_ROOT}/diagrams/opik-{TICKET}-diagram.html)`). Absolute paths ensure the link is clickable regardless of which folder the IDE has open as its workspace — important when the session is running inside a worktree while the user has the main repo open in VSCode.
|
||||
- **Summarize sections**: Brief description of what each section covers
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### Jira MCP Failures
|
||||
- Connection issues: Check network and authentication
|
||||
- Ticket not found: Ask user to verify ticket number
|
||||
- No description: Generate diagram from code diff alone, note the gap
|
||||
|
||||
### GitHub CLI / MCP Failures
|
||||
- `gh` not installed or unauthenticated: Fall back to GitHub MCP
|
||||
- Both unavailable: Fall back to local `git diff main...HEAD`
|
||||
- No PR found: Fall back to local `git diff main...HEAD`
|
||||
- PR diff too large: Summarize by file, focus on key changes
|
||||
|
||||
### Git Failures
|
||||
- On `main` with no ticket: Prompt for ticket number
|
||||
- No commits ahead of main: Show error — nothing to diagram
|
||||
- Merge conflicts: Warn but generate from available diff
|
||||
|
||||
### Diagram Generation Failures
|
||||
- Changes too small: Generate a minimal diagram with just files changed
|
||||
- Changes too large (>30 files): Group by component, show top-level flow only
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. Jira ticket context was fetched (or gracefully skipped)
|
||||
2. Code diff was analyzed and categorized by layer
|
||||
3. HTML diagram was generated with correct styling
|
||||
4. "Copy as image" button works in the generated HTML
|
||||
5. File was saved to `${MAIN_REPO_ROOT}/diagrams/opik-{TICKET_NUMBER}-diagram.html`
|
||||
6. User was shown the file path (as an absolute-path markdown link) and summary
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- Diagrams are **local artifacts** — the main repo's `/diagrams/` folder is gitignored
|
||||
- Diagrams are always written under the **main repo root**'s `diagrams/` folder. When the session runs inside a worktree under `.claude/worktrees/`, the command resolves `MAIN_REPO_ROOT` via `git rev-parse --git-common-dir` so the file lands in the shared `diagrams/` folder and survives worktree removal.
|
||||
- The diagram-generation logic lives in `.agents/skills/diagram-generation/` (shared skill)
|
||||
- The "Copy as image" button uses the browser Canvas API — requires opening in a modern browser
|
||||
- When run without a PR, the diagram reflects local uncommitted + committed changes vs main
|
||||
- Diagrams should be concise and visual — prefer boxes and flows over paragraphs of text
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,227 @@
|
||||
# Share Progress in Jira
|
||||
|
||||
**Command**: `cursor share-progress-in-jira`
|
||||
|
||||
## Overview
|
||||
|
||||
Extract the Jira ticket number from the current Git branch name, generate a diff of all commits since branching from main (excluding generated files), summarize the changes, and add a progress comment to the corresponding Jira ticket.
|
||||
|
||||
This workflow will:
|
||||
|
||||
- Verify Jira MCP availability
|
||||
- Extract Jira ticket number from current branch name
|
||||
- Generate git diff between current branch and main
|
||||
- Summarize the changes in a human-readable format
|
||||
- Add a progress comment to the Jira ticket
|
||||
- Provide clear success/failure feedback
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **None required**: Automatically uses current Git branch and working directory
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Check Jira MCP**: Test Jira MCP availability by attempting to fetch user info using `atlassianUserInfo`
|
||||
> If unavailable, respond with: "This command needs Jira MCP configured. Set MCP config/env, run `make cursor` (Cursor) or `make claude` (Claude CLI), then retry."
|
||||
> Stop here.
|
||||
- **Check Git repository**: Verify we're in a Git repository
|
||||
- **Check current branch**: Ensure we're not on `main`
|
||||
- **Validate branch format**: Confirm branch follows pattern: `<username>/OPIK-<ticket-number>-<kebab-short-description>` or `<username>/issue-<number>-<description>` or `<username>/NA-<description>`
|
||||
|
||||
---
|
||||
|
||||
### 2. Extract Jira Ticket Information
|
||||
|
||||
- **Parse branch name**: Extract `OPIK-<number>` or `issue-<number>` from current branch
|
||||
- **Validate format**: Ensure ticket number follows expected pattern
|
||||
- **If invalid format**: Show error and stop:
|
||||
> "Branch name doesn't match expected format: `<username>/OPIK-<ticket-number>-<kebab-short-description>` or `<username>/issue-<number>-<description>` or `<username>/NA-<description>`"
|
||||
> Current branch: `<actual-branch-name>`
|
||||
|
||||
---
|
||||
|
||||
### 3. Fetch Jira Ticket
|
||||
|
||||
- **Extract key**: Use the `OPIK-<number>` or `issue-<number>` from branch name
|
||||
- **Fetch ticket**: Use Jira MCP to get ticket details
|
||||
- **If fetch fails**: Show error message and stop
|
||||
- **If not found**: Show error and stop:
|
||||
> "Jira ticket `OPIK-<number>` not found. Please verify the ticket exists and you have access."
|
||||
|
||||
---
|
||||
|
||||
### 4. Generate Git Diff
|
||||
|
||||
- **Get diff base**: Find the commit where current branch diverged from main
|
||||
- **Generate diff**: Create diff between current branch and main (all commits since branching)
|
||||
- **Important**: Use `git diff main...HEAD` (three dots) to compare the tip of current branch with the common ancestor, NOT `git diff main..HEAD` (two dots) which compares the tip of main to the tip of HEAD and shows all changes on HEAD that are not on main
|
||||
- **Exclude files**: Filter out generated files like `package-lock.json`, `target/`, `node_modules/`, etc.
|
||||
- **Commands to run**:
|
||||
|
||||
```bash
|
||||
# Get all file changes since branching from main (excludes generated files)
|
||||
git diff main...HEAD --name-only | grep -v -E "(package-lock\.json|target/|node_modules/|\.git/)"
|
||||
|
||||
# Get detailed diff statistics for file analysis
|
||||
git diff main...HEAD --stat
|
||||
|
||||
# Get commit history for implementation context
|
||||
git log --oneline main...HEAD
|
||||
|
||||
# Optional: Get detailed commit messages for better context
|
||||
git log --pretty=format:"%h - %s (%an, %ar)" main...HEAD
|
||||
```
|
||||
|
||||
- **If no changes**: Show message and stop:
|
||||
> "No changes found between current branch and main. Branch may be up to date or no commits made yet."
|
||||
|
||||
---
|
||||
|
||||
### 5. Analyze and Summarize Changes
|
||||
|
||||
- **Parse diff files**: Analyze all added, modified, and deleted files since branching from main
|
||||
- **Parse commit history**: Review commit messages to understand implementation context and progress
|
||||
- **Extract meaningful context**: Identify what was actually implemented, improved, or fixed
|
||||
- **Categorize changes**: Group by file type and implementation phases based on Opik project structure:
|
||||
- **Backend changes**: Java files, SQL migrations, configuration files
|
||||
- **Frontend changes**: TypeScript/React files, UI components, styles
|
||||
- **SDK changes**: Python/TypeScript SDK files
|
||||
- **Documentation**: README files, API docs, configuration guides
|
||||
- **Testing**: Test files, test configurations
|
||||
- **Infrastructure**: Docker files, deployment configs, CI/CD files
|
||||
- **Generate summary**: Create meaningful context for stakeholders based on the actual changes made
|
||||
- **Summary quality**: Focus on what was accomplished, not technical file details
|
||||
- **Format summary**: Use the standard 2-section format:
|
||||
|
||||
```
|
||||
**Release Notes:**
|
||||
[User-facing changes and features - what Product Managers need to know]
|
||||
• [User-visible feature/improvement 1]
|
||||
• [User-visible feature/improvement 2]
|
||||
• [API changes or new endpoints]
|
||||
• [Configuration or environment changes]
|
||||
|
||||
If no user-facing changes: "No user-facing changes were made in this ticket."
|
||||
|
||||
**Technical Details:**
|
||||
[Developer-focused information - technical details for the development team]
|
||||
• [Implementation detail 1]
|
||||
• [Technical improvement 2]
|
||||
• [Configuration/setup change 3]
|
||||
• [Testing coverage added]
|
||||
• [Dependencies updated]
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 6. Add Comment to Jira Ticket
|
||||
|
||||
- **Create comment**: Use Jira MCP to add the progress summary
|
||||
- **Verify success**: Confirm comment was added successfully
|
||||
- **If comment fails**: Show error details and stop
|
||||
|
||||
---
|
||||
|
||||
### 7. Completion Summary & Validation
|
||||
|
||||
- **Confirm all steps completed**:
|
||||
- ✅ Jira MCP available and working
|
||||
- ✅ Branch name parsed successfully
|
||||
- ✅ Jira ticket found and accessible
|
||||
- ✅ Git diff generated and analyzed
|
||||
- ✅ Progress summary created
|
||||
- ✅ Comment added to Jira ticket
|
||||
- **Show summary**: Display what was shared and where
|
||||
- **Next steps**: Suggest when to use this command again
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### **Branch Name Errors**
|
||||
|
||||
- Invalid format: Show expected pattern and current branch
|
||||
- On main: Explain this command is for feature branches only
|
||||
|
||||
### **Jira MCP Failures**
|
||||
|
||||
- Connection issues: Check network and authentication
|
||||
- Permission errors: Verify user access to the ticket
|
||||
- Rate limiting: Wait and retry
|
||||
|
||||
### **Git Operation Failures**
|
||||
|
||||
- Not in Git repo: Navigate to correct directory
|
||||
- No commits: Explain branch needs commits to generate diff
|
||||
- Diff generation fails: Check Git status and resolve conflicts
|
||||
|
||||
### **Comment Addition Failures**
|
||||
|
||||
- Permission denied: Verify user can comment on ticket
|
||||
- Invalid content: Check comment format and length
|
||||
- Network issues: Verify connectivity to Atlassian services
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. ✅ Jira MCP is available and working
|
||||
2. ✅ Branch name is parsed and ticket number extracted
|
||||
3. ✅ Jira ticket is found and accessible
|
||||
4. ✅ Git diff is generated successfully
|
||||
5. ✅ Changes are summarized using 2-section format: Release Notes (for PMs) and Technical Details (for developers)
|
||||
6. ✅ Progress comment is added to Jira ticket
|
||||
7. ✅ All operations complete without errors
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### **Common Issues**
|
||||
|
||||
- **Jira MCP not available**: Configure MCP and run `make cursor` (Cursor) or `make claude` (Claude CLI)
|
||||
- **Branch name format**: Ensure it follows Opik conventions:
|
||||
- `username/OPIK-<number>-<description>` for Jira tickets
|
||||
- `username/issue-<number>-<description>` for GitHub issues
|
||||
- `username/NA-<description>` for tasks without tickets
|
||||
- **No changes found**: Make commits to your branch first
|
||||
- **Permission errors**: Check Jira access and ticket permissions
|
||||
|
||||
### **Fallback Options**
|
||||
|
||||
- If diff generation fails: Check Git status and resolve conflicts
|
||||
- If comment addition fails: Copy summary manually to Jira
|
||||
- If ticket not found: Verify ticket number and access permissions
|
||||
|
||||
---
|
||||
|
||||
## Integration with Existing Workflow
|
||||
|
||||
This command complements the `work-on-jira-ticket` command by:
|
||||
|
||||
- **Progress tracking**: Document development progress in Jira
|
||||
- **Change visibility**: Keep stakeholders informed of implementation details
|
||||
- **Documentation**: Maintain a record of what was built and when
|
||||
- **Review preparation**: Prepare summaries for code reviews and PRs
|
||||
|
||||
---
|
||||
|
||||
## Notes
|
||||
|
||||
- **MCP Testing**: Actively tests Jira MCP availability using `atlassianUserInfo` before proceeding
|
||||
- **Git Diff Accuracy**: Uses three-dot syntax (`git diff main...HEAD`) for accurate change detection
|
||||
- **Summary Focus**: Emphasizes stakeholder value over technical implementation details
|
||||
- **Error Handling**: Stops immediately on critical failures to prevent incomplete operations
|
||||
- **Format Flexibility**: Adapts summary structure based on the type of work being done
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,257 @@
|
||||
# Work on GitHub Issue
|
||||
|
||||
**Command**: `cursor work-on-github-issue`
|
||||
|
||||
## Overview
|
||||
|
||||
Fetch a GitHub issue by link, build full context (title, description, comments, labels, assignees, milestone), and generate an actionable implementation plan.
|
||||
This workflow will:
|
||||
|
||||
- Verify GitHub MCP availability (or instruct how to install it).
|
||||
- Fetch and parse the issue details.
|
||||
- If not found, list your assigned **open** issues.
|
||||
- Check local git status in the Opik repository and propose a branch if on `main`.
|
||||
- Suggest moving the issue to **In Progress** if it's currently **open**.
|
||||
- Validate all operations and provide clear success/failure feedback.
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **GitHub issue link (required)**: e.g., `https://github.com/comet-ml/opik/issues/1234`
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Check GitHub MCP**: If unavailable, respond with:
|
||||
> "This command needs GitHub MCP configured. Set MCP config/env, run `make cursor` (Cursor) or `make claude` (Claude CLI), then retry."
|
||||
> Stop here.
|
||||
- **Check development environment**: Verify project dependencies, build tools, and project structure are ready.
|
||||
- **Check local git branch** in the Opik repository:
|
||||
- If on `main`, propose a new branch following Opik naming convention:
|
||||
```
|
||||
{USERNAME}/issue-{ISSUE-NUMBER}-{ISSUE-SUMMARY}
|
||||
```
|
||||
Example:
|
||||
- Issue: `https://github.com/comet-ml/opik/issues/1234`
|
||||
- Title: "Add cursor git workflow rule"
|
||||
- Branch: `andrescrz/issue-1234-add-cursor-git-workflow-rule`
|
||||
|
||||
---
|
||||
|
||||
### 2. Fetch Issue
|
||||
|
||||
- Extract issue number from link.
|
||||
- Fetch with GitHub MCP: title, body, state, labels, assignees, milestone, comments, created_at, updated_at.
|
||||
- **If fetch fails**: Show error message and suggest troubleshooting steps.
|
||||
- **If not found**: Search for open issues assigned to current user: `is:open assignee:@me` (max 10).
|
||||
- **Show list**: `#1234 — Title (State)` and stop if issue not found.
|
||||
|
||||
---
|
||||
|
||||
### 3. Build GitHub Context
|
||||
|
||||
- **Number, Title, State, Labels, Assignees, Milestone**
|
||||
- **Description**: verbatim if short, otherwise concise summary + key quotes.
|
||||
- **If no description**: Note this and suggest adding context for better implementation planning.
|
||||
- **Comments**: newest → oldest, `[author @ date] summary` with important snippets.
|
||||
- **Linked PRs**: Include any linked pull requests if available.
|
||||
- **Project context**: Include project board information if available.
|
||||
|
||||
---
|
||||
|
||||
### 4. Determine Implementation Scope
|
||||
|
||||
- **Analyze issue scope**: Determine which Opik components are affected:
|
||||
- **Backend only**: Java API changes, database migrations, services
|
||||
- **Frontend only**: React components, UI changes, state management
|
||||
- **SDK only**: Python or TypeScript SDK changes
|
||||
- **Cross-component**: Changes affecting multiple layers
|
||||
- **Infrastructure**: Docker, deployment, configuration changes
|
||||
- **Identify affected areas**: Map issue requirements to specific Opik modules and files
|
||||
- **Estimate complexity**: Consider if changes require database migrations, API versioning, or breaking changes
|
||||
|
||||
---
|
||||
|
||||
### 5. Task Plan
|
||||
|
||||
- **Bugfix**: repro steps, root cause hypothesis, affected files, fix approach, risks, tests, verification.
|
||||
- **Feature**: user story recap, acceptance criteria, implementation plan, tests, rollout notes.
|
||||
- **Always reference shared and domain guidance in the right place**:
|
||||
- **Global policy**: `.agents/rules/*` (git workflow, security, code style, routing)
|
||||
- **Backend guidance**: `.agents/skills/opik-backend/*`
|
||||
- **Frontend guidance**: `.agents/skills/opik-frontend/*`
|
||||
- **SDK guidance**: `.agents/skills/python-sdk/*` and `.agents/skills/typescript-sdk/*`
|
||||
- **Component-specific guidance**: Use the appropriate rule set based on the implementation scope identified in step 4
|
||||
|
||||
---
|
||||
|
||||
### 6. Git & Branch Setup
|
||||
|
||||
- Repo: Opik repository (current workspace)
|
||||
- **CRITICAL**: Handle working directory state BEFORE branching:
|
||||
- If working directory has changes:
|
||||
- **Option 1**: Stash changes: `git stash push -m "WIP: before issue-{ISSUE-NUMBER}"`
|
||||
- **Option 2**: Ask user what to do with uncommitted changes
|
||||
- **NEVER commit directly to main** (following Opik git workflow)
|
||||
- If on `main`, create branch following Opik conventions:
|
||||
```bash
|
||||
# Ensure you're on main and pull latest
|
||||
git checkout main
|
||||
git pull origin main
|
||||
|
||||
# Create task-specific branch
|
||||
git checkout -b {USERNAME}/issue-{ISSUE-NUMBER}-{ISSUE-SUMMARY}
|
||||
```
|
||||
- **After branch creation**: Apply stashed changes if any: `git stash pop`
|
||||
- **Verify branch creation**: Confirm new branch is active and clean.
|
||||
|
||||
---
|
||||
|
||||
### 7. Status Management (Optional)
|
||||
|
||||
- **Move issue to "In Progress"** if currently **open**:
|
||||
- Use GitHub MCP to add appropriate labels (e.g., "in-progress")
|
||||
- **Verify label addition**: Confirm labels were added successfully
|
||||
- **Handle label failures**: Provide error details and retry options
|
||||
|
||||
---
|
||||
|
||||
### 8. Implementation Suggestion
|
||||
|
||||
- **Based on GitHub context and Opik agent guidance**, suggest implementing the feature/bugfix:
|
||||
- Reference global policy in `.agents/rules/*` and domain guidance in `.agents/skills/*`
|
||||
- Provide specific implementation steps based on the task plan
|
||||
- Include code examples or file paths where appropriate
|
||||
- Suggest testing approaches and quality checks
|
||||
- Follow Opik architecture patterns (Resources → Services → DAOs → Models for backend)
|
||||
- **Commit Message Format**: Follow shared conventions from `.agents/rules/git-workflow.mdc`.
|
||||
|
||||
**First Commit (PR-title source, required):**
|
||||
```
|
||||
[<TICKET-KEY>] [BE/FE/SDK/DOCS] <type>: <description>
|
||||
```
|
||||
where `<TICKET-KEY>` is `OPIK-####`, `issue-####`, or `NA`.
|
||||
|
||||
**Follow-up Commits (preferred):**
|
||||
```
|
||||
<type>(<scope>): <description>
|
||||
```
|
||||
where `<type>` is one of: `feat`, `fix`, `refactor`, `test`, `docs`, `chore`.
|
||||
|
||||
**Last-resort fallback (discouraged):**
|
||||
```
|
||||
Revision N: <description>
|
||||
```
|
||||
|
||||
**Component Types:**
|
||||
- `[BE]` - Backend changes (Java, API endpoints, services)
|
||||
- `[FE]` - Frontend changes (React, TypeScript, UI components)
|
||||
- `[SDK]` - SDK changes (Python, TypeScript SDKs)
|
||||
- `[DOCS]` - Documentation updates, README changes, comments, swagger/OpenAPI documentation
|
||||
|
||||
**Examples:**
|
||||
```
|
||||
[OPIK-1234] [BE] feat: add create trace endpoint
|
||||
[issue-1234] [FE] fix: guard project custom metrics empty state
|
||||
[OPIK-1234] [DOCS] docs: update API documentation
|
||||
[NA] [SDK] chore: align SDK lint configuration
|
||||
feat(experiments): add run-level metadata capture
|
||||
```
|
||||
|
||||
### 9. User Confirmation
|
||||
|
||||
- **Ask for user approval** before proceeding with implementation:
|
||||
- Present the implementation plan clearly
|
||||
- Ask: "Would you like me to proceed with implementing this feature/fix now?"
|
||||
- **Wait for explicit user confirmation** before making any code changes
|
||||
- If user declines: Stop here and provide guidance for manual implementation
|
||||
- If user confirms: Proceed to implementation phase
|
||||
|
||||
### 10. Implementation Phase (Optional)
|
||||
|
||||
- **Only proceed if user confirmed** in previous step
|
||||
- Execute the implementation plan:
|
||||
- Create/modify necessary files following Opik patterns
|
||||
- Apply code changes according to the plan
|
||||
- Run quality checks and tests
|
||||
- Commit changes with proper issue number prefix
|
||||
- **If user declined**: Provide manual implementation guidance and stop
|
||||
|
||||
---
|
||||
|
||||
### 11. Completion Summary & Validation
|
||||
|
||||
- **Confirm all steps completed**:
|
||||
- ✅ GitHub MCP available and working
|
||||
- ✅ Issue fetched and analyzed successfully
|
||||
- ✅ Labels updated (if applicable)
|
||||
- ✅ Feature branch created and active following Opik naming convention
|
||||
- ✅ Development environment ready
|
||||
- ✅ Implementation suggestion provided
|
||||
- **Next steps**: Provide clear guidance on what to do next
|
||||
- **Error summary**: If any steps failed, provide troubleshooting guidance
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### **GitHub MCP Failures**
|
||||
|
||||
- Connection issues: Check network and authentication
|
||||
- Permission errors: Verify user access to the repository
|
||||
- Rate limiting: Wait and retry
|
||||
- API errors: Check GitHub API status and retry
|
||||
|
||||
### **Git Operation Failures**
|
||||
|
||||
- Branch creation fails: Check for conflicts, verify permissions
|
||||
- Pull fails: Resolve merge conflicts, check remote status
|
||||
- Working directory dirty: **NEVER commit to main** - stash changes first
|
||||
- **CRITICAL SAFETY**: Always verify current branch before any commits
|
||||
- Uncommitted changes: Stash before branching, pop after branch creation
|
||||
|
||||
### **Label Management Failures**
|
||||
|
||||
- Invalid label: Check available labels in the repository
|
||||
- Permission denied: Verify user can modify issue labels
|
||||
- Label doesn't exist: Create the label or use existing ones
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. ✅ GitHub issue is successfully fetched and analyzed
|
||||
2. ✅ Feature branch is created and active following Opik naming convention
|
||||
3. ✅ Issue labels are updated (if requested)
|
||||
4. ✅ Implementation suggestion provided based on context and Opik rules
|
||||
5. ✅ User confirmation received (proceed or decline)
|
||||
6. ✅ Implementation executed (if confirmed) or manual guidance provided (if declined)
|
||||
7. ✅ All operations complete without errors
|
||||
8. ✅ Clear next steps are provided to the user
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### **Common Issues**
|
||||
|
||||
- **GitHub MCP not available**: Configure MCP and run `make cursor` (Cursor) or `make claude` (Claude CLI)
|
||||
- **Git branch conflicts**: Resolve conflicts before proceeding
|
||||
- **Permission errors**: Check user access and repository settings
|
||||
- **Network issues**: Verify connectivity to GitHub services
|
||||
|
||||
### **Fallback Options**
|
||||
|
||||
- If issue fetch fails: List user's assigned open issues
|
||||
- If branch creation fails: Provide manual git commands
|
||||
- If label update fails: Continue with development setup
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,291 @@
|
||||
# Work on Jira Ticket
|
||||
|
||||
**Command**: `cursor work-on-jira-ticket`
|
||||
|
||||
## Overview
|
||||
|
||||
Fetch a Jira ticket by link, build full context (title, description, comments, type, status, priority, assignee, labels), and generate an actionable implementation plan.
|
||||
This workflow will:
|
||||
|
||||
- Verify Jira MCP availability (or instruct how to install it).
|
||||
- Fetch and parse the ticket details.
|
||||
- If not found, list your assigned **To Do** issues.
|
||||
- Check local git status in the Opik repository and propose a branch if on `main`.
|
||||
- Suggest moving the ticket to **In Progress** if it's currently in **To Do**.
|
||||
- Validate all operations and provide clear success/failure feedback.
|
||||
|
||||
---
|
||||
|
||||
## Inputs
|
||||
|
||||
- **Jira link (required)**: e.g., `https://comet-ml.atlassian.net/browse/OPIK-1234`
|
||||
- **Worktree (optional)**: Pass `worktree` to work in an isolated git worktree. Defaults to no worktree when the argument is omitted — the command never prompts about worktrees.
|
||||
|
||||
---
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Preflight & Environment Check
|
||||
|
||||
- **Check Jira MCP**: If unavailable, respond with:
|
||||
> "This command needs Jira MCP configured. Set MCP config/env, run `make cursor` (Cursor) or `make claude` (Claude CLI), then retry."
|
||||
> Stop here.
|
||||
- **Check development environment**: Verify project dependencies, build tools, and project structure are ready.
|
||||
- **Check local git branch** in the Opik repository:
|
||||
- If on `main`, propose a new branch following Opik naming convention:
|
||||
```
|
||||
{USERNAME}/OPIK-{TICKET-NUMBER}-{TICKET-SUMMARY}
|
||||
```
|
||||
Example:
|
||||
- Ticket: `https://comet-ml.atlassian.net/browse/OPIK-2180`
|
||||
- Title: "Add cursor git workflow rule"
|
||||
- Branch: `andrescrz/OPIK-2180-add-cursor-git-workflow-rule`
|
||||
|
||||
---
|
||||
|
||||
### 2. Fetch Ticket
|
||||
|
||||
- Extract key from link (`OPIK-<number>`).
|
||||
- Fetch with Jira MCP: summary, description, issue type, status, priority, assignee, labels, reporter, comments.
|
||||
- **If fetch fails**: Show error message and suggest troubleshooting steps.
|
||||
- **If not found**: Search JQL: `assignee = currentUser() AND status = "To Do" ORDER BY updated DESC` (max 10).
|
||||
- **Show list**: `OPIK-#### — Summary (Status)` and stop if ticket not found.
|
||||
|
||||
---
|
||||
|
||||
### 3. Build Jira Context
|
||||
|
||||
- **Key, Title, Type, Status, Priority, Assignee, Labels**
|
||||
- **Description**: verbatim if short, otherwise concise summary + key quotes. The description contains the **WHY** (motivation) and **WHAT** (high-level scope + acceptance criteria); implementation details ("HOW") live in a separate comment — see below.
|
||||
- **If no description**: Note this and suggest adding context for better implementation planning.
|
||||
- **Comments**: newest → oldest, `[author @ date] summary` with important snippets.
|
||||
- **HOW comment**: scan comments for the most recent one whose body matches `^#\s*HOW\b` (case-insensitive). If found, surface it as the **implementation suggestion** — clearly framed as a note from when the ticket was filed, not a plan of record. The current code state is authoritative; the HOW is one input. If no HOW comment exists, proceed without one — many tickets won't have one, and that's fine (older tickets predate the convention; some tickets have nothing worth adding beyond the WHAT).
|
||||
- **Epic/Story context**: Include parent issue information if available.
|
||||
|
||||
---
|
||||
|
||||
### 4. Determine Implementation Scope
|
||||
|
||||
- **Analyze ticket scope**: Determine which Opik components are affected:
|
||||
- **Backend only**: Java API changes, database migrations, services
|
||||
- **Frontend only**: React components, UI changes, state management
|
||||
- **SDK only**: Python or TypeScript SDK changes
|
||||
- **Cross-component**: Changes affecting multiple layers
|
||||
- **Infrastructure**: Docker, deployment, configuration changes
|
||||
- **Identify affected areas**: Map ticket requirements to specific Opik modules and files
|
||||
- **Estimate complexity**: Consider if changes require database migrations, API versioning, or breaking changes
|
||||
|
||||
---
|
||||
|
||||
### 5. Task Plan
|
||||
|
||||
- **Bugfix**: repro steps, root cause hypothesis, affected files, fix approach, risks, tests, verification.
|
||||
- **Feature**: user story recap, acceptance criteria, implementation plan, tests, rollout notes.
|
||||
- **HOW comment, if one exists**: weave its suggestions into the plan **after** independently reading the current code state. If the HOW conflicts with what the code looks like today, trust the code and note the divergence in your plan. The HOW is a hint from when the ticket was filed, not a contract.
|
||||
- **Always reference shared + domain guidance in the right place**:
|
||||
- **Global policy**: `.agents/rules/*` (git workflow, security, code style, routing)
|
||||
- **Backend guidance**: `.agents/skills/opik-backend/*`
|
||||
- **Frontend guidance**: `.agents/skills/opik-frontend/*`
|
||||
- **SDK guidance**: `.agents/skills/python-sdk/*` and `.agents/skills/typescript-sdk/*`
|
||||
- **Component-specific guidance**: Use the appropriate guidance set based on the implementation scope identified in step 4
|
||||
|
||||
---
|
||||
|
||||
### 6. Git & Branch Setup
|
||||
|
||||
- Repo: Opik repository (current workspace)
|
||||
- **NEVER commit directly to main** (following Opik git workflow)
|
||||
|
||||
#### 6a. Worktree Decision
|
||||
|
||||
Worktree usage is strictly opt-in:
|
||||
|
||||
- **If `worktree` was passed as an argument**: Use a worktree (continue to 6b).
|
||||
- **Otherwise (default)**: Skip 6b and go straight to 6c (normal path). Do not prompt the user.
|
||||
|
||||
If the `EnterWorktree` tool is not available (e.g., running in Cursor or another editor) even when `worktree` was passed, fall back to 6c without prompting.
|
||||
|
||||
#### 6b. Worktree Path (if using worktree)
|
||||
|
||||
1. Slugify `{TICKET-SUMMARY}` for the worktree name: replace any character not in `[A-Za-z0-9._-]` with `-`, collapse consecutive `-` into one, and trim leading/trailing `-`. Then call `EnterWorktree` with name `{USERNAME}-OPIK-{TICKET-NUMBER}-{SLUGIFIED-SUMMARY}`.
|
||||
2. Inside the worktree, fetch the latest remote state and create the properly named branch based on `origin/main` (using the same slugified summary):
|
||||
```bash
|
||||
git fetch origin
|
||||
git checkout -b {USERNAME}/OPIK-{TICKET-NUMBER}-{SLUGIFIED-SUMMARY} origin/main
|
||||
```
|
||||
The worktree intentionally branches off `origin/main` regardless of the parent checkout's current branch or local `main` state, and skips the rebase-strategy prompt — isolation is the whole point of the worktree.
|
||||
|
||||
Do not inspect or prompt about the parent checkout's working tree (staged, unstaged, or untracked files; current branch). Worktrees are physically isolated, so parent state has no effect on the worktree and is not the agent's concern in this path.
|
||||
3. Continue with implementation in the worktree directory.
|
||||
|
||||
#### 6c. Normal Path (if not using worktree)
|
||||
|
||||
- **Handle working directory state BEFORE branching**:
|
||||
- If working directory has changes:
|
||||
- **Option 1**: Stash changes: `git stash push -m "WIP: before OPIK-{TICKET-NUMBER}"`
|
||||
- **Option 2**: Ask user what to do with uncommitted changes
|
||||
- Slugify `{TICKET-SUMMARY}` the same way as step 6b: replace any character not in `[A-Za-z0-9._-]` with `-`, collapse consecutive `-` into one, and trim leading/trailing `-`.
|
||||
- If on `main`, create branch following Opik conventions:
|
||||
```bash
|
||||
git checkout main
|
||||
git pull origin main
|
||||
git checkout -b {USERNAME}/OPIK-{TICKET-NUMBER}-{SLUGIFIED-SUMMARY}
|
||||
```
|
||||
- **After branch creation**: Apply stashed changes if any: `git stash pop`
|
||||
- **Verify branch creation**: Confirm new branch is active and clean.
|
||||
|
||||
---
|
||||
|
||||
### 7. Status Management (Optional)
|
||||
|
||||
- **Move ticket to "In Progress"** if currently in "To Do":
|
||||
- Use Jira MCP to transition status
|
||||
- **Verify transition success**: Confirm status actually changed
|
||||
- **Handle transition failures**: Provide error details and retry options
|
||||
|
||||
---
|
||||
|
||||
### 8. Implementation Suggestion
|
||||
|
||||
- **Based on Jira context and Opik agent guidance**, suggest implementing the feature/bugfix:
|
||||
- Reference global policy in `.agents/rules/*` and domain guidance in `.agents/skills/*`
|
||||
- Provide specific implementation steps based on the task plan
|
||||
- Include code examples or file paths where appropriate
|
||||
- Suggest testing approaches and quality checks
|
||||
- Follow Opik architecture patterns (Resources → Services → DAOs → Models for backend)
|
||||
- **Commit Message Format**: Use semantic commits. The first commit on a branch is critical because PR title is derived from it:
|
||||
|
||||
**First Commit (PR-title source, required):**
|
||||
```
|
||||
[OPIK-####] [BE/FE/SDK/DOCS] <type>: <description>
|
||||
```
|
||||
|
||||
**Allowed ticket-key variants (when applicable):**
|
||||
```
|
||||
[issue-####] [BE/FE/SDK/DOCS] <type>: <description>
|
||||
[NA] [BE/FE/SDK/DOCS] <type>: <description>
|
||||
```
|
||||
|
||||
**Follow-up Commits (preferred):**
|
||||
```
|
||||
<type>(<scope>): <description>
|
||||
```
|
||||
where `<type>` is one of: `feat`, `fix`, `refactor`, `test`, `docs`, `chore`.
|
||||
|
||||
**Last-resort fallback (discouraged):**
|
||||
```
|
||||
Revision N: <description>
|
||||
```
|
||||
|
||||
**Component Types:**
|
||||
- `[BE]` - Backend changes (Java, API endpoints, services)
|
||||
- `[FE]` - Frontend changes (React, TypeScript, UI components)
|
||||
- `[SDK]` - SDK changes (Python, TypeScript SDKs)
|
||||
- `[DOCS]` - Documentation updates, README changes, comments, swagger/OpenAPI documentation
|
||||
|
||||
**Examples:**
|
||||
```
|
||||
[OPIK-1234] [BE] feat: add create trace endpoint
|
||||
[OPIK-1234] [FE] feat: add project custom metrics UI dashboard
|
||||
[OPIK-1234] [DOCS] docs: update API documentation
|
||||
[issue-1234] [SDK] feat: add new Python SDK method
|
||||
fix(metrics): handle empty dashboard responses
|
||||
test(api): cover project metrics endpoint pagination
|
||||
Revision 2: small follow-up rename after emergency patch
|
||||
```
|
||||
|
||||
**Jira key convention in commit messages** (see git-workflow rule): the GitHub for Jira app links any `OPIK-<digits>` it finds in a commit message to that ticket's Development panel, and the link can't be removed. The ticket this branch resolves keeps the hyphen (`OPIK-1234`) — that's the prefix. But any **other** ticket a commit message mentions without resolving (an escalation, a reference to an older ticket) must be written with an underscore (`OPIK_7000`) and with no Jira URL, so the scanner doesn't link it.
|
||||
|
||||
### 9. User Confirmation
|
||||
|
||||
- **Ask for user approval** before proceeding with implementation:
|
||||
- Present the implementation plan clearly
|
||||
- Ask: "Would you like me to proceed with implementing this feature/fix now?"
|
||||
- **Wait for explicit user confirmation** before making any code changes
|
||||
- If user declines: Stop here and provide guidance for manual implementation
|
||||
- If user confirms: Proceed to implementation phase
|
||||
|
||||
### 10. Implementation Phase (Optional)
|
||||
|
||||
- **Only proceed if user confirmed** in previous step
|
||||
- Execute the implementation plan:
|
||||
- Create/modify necessary files following Opik patterns
|
||||
- Apply code changes according to the plan
|
||||
- Run quality checks and tests
|
||||
- Commit changes with proper ticket number prefix
|
||||
- **Post-push PR description sync**: Whenever this skill (or any follow-up step the user runs from this conversation) invokes `git push` or `git push --force-with-lease` to a branch with an open PR in `comet-ml/opik`, invoke the `_pr-description-sync` sub-skill (`.agents/commands/comet/_pr-description-sync.md`) immediately after the push completes. The sub-skill is a no-op when no PR exists, when the description is already in sync, or when the user has opted out of refreshes for this repo. This keeps the PR description aligned with what was actually shipped instead of what was claimed when the PR was opened.
|
||||
- **If user declined**: Provide manual implementation guidance and stop
|
||||
|
||||
---
|
||||
|
||||
### 11. Completion Summary & Validation
|
||||
|
||||
- **Confirm all steps completed**:
|
||||
- ✅ Jira MCP available and working
|
||||
- ✅ Ticket fetched and analyzed successfully
|
||||
- ✅ Status updated (if applicable)
|
||||
- ✅ Feature branch created and active following Opik naming convention
|
||||
- ✅ Development environment ready
|
||||
- ✅ Implementation suggestion provided
|
||||
- **Next steps**: Provide clear guidance on what to do next
|
||||
- **Error summary**: If any steps failed, provide troubleshooting guidance
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### **Jira MCP Failures**
|
||||
|
||||
- Connection issues: Check network and authentication
|
||||
- Permission errors: Verify user access to the ticket
|
||||
- Rate limiting: Wait and retry
|
||||
|
||||
### **Git Operation Failures**
|
||||
|
||||
- Branch creation fails: Check for conflicts, verify permissions
|
||||
- Pull fails: Resolve merge conflicts, check remote status
|
||||
- Working directory dirty: **NEVER commit to main** - stash changes first
|
||||
- **CRITICAL SAFETY**: Always verify current branch before any commits
|
||||
- Uncommitted changes: Stash before branching, pop after branch creation
|
||||
|
||||
### **Status Transition Failures**
|
||||
|
||||
- Invalid transition: Check available transitions for current status
|
||||
- Permission denied: Verify user can modify ticket status
|
||||
- Workflow restrictions: Check project workflow configuration
|
||||
|
||||
---
|
||||
|
||||
## Success Criteria
|
||||
|
||||
The command is successful when:
|
||||
|
||||
1. ✅ Jira ticket is successfully fetched and analyzed
|
||||
2. ✅ Feature branch is created and active following Opik naming convention
|
||||
3. ✅ Ticket status is updated (if requested)
|
||||
4. ✅ Implementation suggestion provided based on context and Opik rules
|
||||
5. ✅ User confirmation received (proceed or decline)
|
||||
6. ✅ Implementation executed (if confirmed) or manual guidance provided (if declined)
|
||||
7. ✅ All operations complete without errors
|
||||
8. ✅ Clear next steps are provided to the user
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### **Common Issues**
|
||||
|
||||
- **Jira MCP not available**: Configure MCP and run `make cursor` (Cursor) or `make claude` (Claude CLI)
|
||||
- **Git branch conflicts**: Resolve conflicts before proceeding
|
||||
- **Permission errors**: Check user access and project settings
|
||||
- **Network issues**: Verify connectivity to Atlassian services
|
||||
|
||||
### **Fallback Options**
|
||||
|
||||
- If ticket fetch fails: List user's assigned tickets
|
||||
- If branch creation fails: Provide manual git commands
|
||||
- If status update fails: Continue with development setup
|
||||
|
||||
---
|
||||
|
||||
**End Command**
|
||||
@@ -0,0 +1,36 @@
|
||||
# Extract PR size (shared)
|
||||
|
||||
Shared logic for the code-review Slack commands (`send-code-review-slack`,
|
||||
`generate-code-review-slack-command`) to derive the PR-size field. Kept in one
|
||||
place so the two commands can't drift.
|
||||
|
||||
## How to extract
|
||||
|
||||
1. The `📏 Auto Label PR Size` workflow (`.github/workflows/pr-size-labeler.yml`)
|
||||
applies exactly one size label to every PR. Read the PR labels and look for a
|
||||
`size/*` label: `🔵 size/XS`, `🟢 size/S`, `🟡 size/M`, `🟠 size/L`, or
|
||||
`🔴 size/XL`.
|
||||
2. Store the bucket as `{emoji} {BUCKET}` (e.g. `🟠 L`) — just the bucket, no
|
||||
line counts.
|
||||
3. **Fallback** (only if no `size/*` label is present yet — the workflow may not
|
||||
have run): derive the bucket from the PR's changed lines
|
||||
(`additions + deletions`), applying **the same ignore list and thresholds as
|
||||
the workflow** so the fallback can't land in a different bucket than the
|
||||
labeler. The workflow (`.github/workflows/pr-size-labeler.yml`) is the single
|
||||
source of truth for both:
|
||||
- **Ignore list** — read `IGNORE_GLOBS` in the workflow (lockfiles, generated
|
||||
REST clients under `sdks/*/src/opik/rest_api/**`, and snapshot/image files).
|
||||
Do not re-list the globs here; they would drift.
|
||||
- **Thresholds** — read `BUCKETS` in the workflow: XS `< 20`, S `20–100`,
|
||||
M `101–300`, L `301–600`, XL `> 600`.
|
||||
Store the result as `{emoji} {BUCKET}`.
|
||||
4. GitHub is the source of truth; the size at message-publish time is good enough
|
||||
(PRs rarely change bucket after review starts).
|
||||
|
||||
## Message field
|
||||
|
||||
Include a size line in the message template, bucket only (no `+/-` counts):
|
||||
|
||||
```
|
||||
:straight_ruler: pr size: {{pr_size}}
|
||||
```
|
||||
@@ -0,0 +1,152 @@
|
||||
# Sentry MCP Configuration Guide
|
||||
|
||||
This guide explains how to enable the **Sentry MCP server** for Claude Code, Cursor, and other MCP-compatible clients so you can query Sentry issues, events, and stack traces from your agent without leaving the IDE.
|
||||
|
||||
## Setup
|
||||
|
||||
The Sentry server is wired into `.agents/mcp.json` and uses a User Auth Token loaded from `.env.local` — same pattern as the GitHub, Jira, and Slack MCPs in this repo.
|
||||
|
||||
```json
|
||||
"Sentry": {
|
||||
"command": "npx",
|
||||
"args": ["-y", "@sentry/mcp-server@latest"],
|
||||
"envFile": "${workspaceFolder}/.env.local"
|
||||
}
|
||||
```
|
||||
|
||||
### Step 1: Create a Sentry User Auth Token
|
||||
|
||||
1. Go to <https://sentry.io/settings/account/api/auth-tokens/>.
|
||||
2. Click **Create New Token**, name it something recognizable (e.g., `opik-mcp-local`).
|
||||
3. Select these least-privilege scopes:
|
||||
- `org:read`
|
||||
- `project:read`
|
||||
- `team:read`
|
||||
- `event:write` *(required by some MCP calls; does not allow event deletion)*
|
||||
4. Add `project:write` and `team:write` **only** if you want write actions (resolve issues, comment, assign). Read-only is recommended.
|
||||
5. **Create Token** and copy the value.
|
||||
|
||||
### Step 2: Add the token to `.env.local`
|
||||
|
||||
If you don't have `.env.local` yet, copy `.env.template`:
|
||||
|
||||
```bash
|
||||
cp .env.template .env.local
|
||||
```
|
||||
|
||||
Then set the value:
|
||||
|
||||
```bash
|
||||
SENTRY_ACCESS_TOKEN=your-user-auth-token-here
|
||||
```
|
||||
|
||||
`.env.local` is gitignored — do not commit it.
|
||||
|
||||
### Step 3: Generate the Claude config
|
||||
|
||||
```bash
|
||||
make claude
|
||||
```
|
||||
|
||||
This regenerates `.claude/` and `.mcp.json`. The convert script reads `.env.local` and inlines values into the Sentry block of `.mcp.json` (same as the Jira MCP). Verify:
|
||||
|
||||
```bash
|
||||
jq '.mcpServers.Sentry.env | has("SENTRY_ACCESS_TOKEN")' .mcp.json
|
||||
# → true
|
||||
```
|
||||
|
||||
If you rotate the token later, update `.env.local` and re-run `make claude`.
|
||||
|
||||
### Step 4: Restart your MCP client
|
||||
|
||||
Restart Claude Code (or Cursor, etc.) so it loads the new MCP server.
|
||||
|
||||
### Step 5: Verify
|
||||
|
||||
Ask the agent something like:
|
||||
|
||||
> *List recent Sentry issues for opik-python-sdk in the comet-or org.*
|
||||
|
||||
You should see it call a Sentry tool and return results.
|
||||
|
||||
## Tools Exposed
|
||||
|
||||
The official Sentry MCP exposes (non-exhaustive):
|
||||
|
||||
- **Read**: `find_organizations`, `find_projects`, `find_releases`, `find_teams`, `get_sentry_resource`, `get_issue_tag_values`, `get_replay_details`, `whoami`
|
||||
- **Write (only with `*:write` scopes)**: `update_issue` (resolve / assign / ignore)
|
||||
|
||||
## Avoid the NL-Backed Search Tools
|
||||
|
||||
The Sentry MCP exposes three "search" tools — `search_issues`, `search_events`, `search_issue_events` — and one analysis tool, `analyze_issue_with_seer`. **All of them route through Sentry's own OpenAI account for natural-language → Sentry query translation**, and that account is frequently rate-limited (`You exceeded your current quota`). Treat them as best-effort; do not build workflows around them.
|
||||
|
||||
**Direct, non-LLM tools that always work:**
|
||||
`get_sentry_resource`, `get_issue_tag_values`, `find_organizations`, `find_projects`, `find_releases`, `find_teams`, `whoami`, `update_issue`.
|
||||
|
||||
**When you need to enumerate events inside an issue** (the direct tools fetch a single resource but cannot paginate events), call Sentry's REST API directly using the same `SENTRY_ACCESS_TOKEN`. The `/analyze-sentry-issue` slash command (defined in [../commands/comet/analyze-sentry-issue.md](../commands/comet/analyze-sentry-issue.md)) drives this — paginates `/api/0/issues/<id>/events/` and aggregates by exception message, tags, and users.
|
||||
|
||||
## Self-Hosted Sentry
|
||||
|
||||
If you point at a self-hosted Sentry, add `SENTRY_HOST=sentry.example.com` (no scheme) to `.env.local`. For plain-HTTP self-hosted deployments, append `--insecure-http` to the `args` array in `.agents/mcp.json`.
|
||||
|
||||
## Alternative: OAuth / Remote Server (no token)
|
||||
|
||||
If you don't want to manage a personal token, you can run the official remote MCP via OAuth instead. Replace the Sentry block in `.agents/mcp.json` with:
|
||||
|
||||
```json
|
||||
"Sentry": {
|
||||
"command": "npx",
|
||||
"args": ["-y", "mcp-remote", "https://mcp.sentry.dev/mcp"],
|
||||
"env": {}
|
||||
}
|
||||
```
|
||||
|
||||
Then `make claude` + restart your MCP client. The first Sentry tool call opens a browser for OAuth. Tokens are cached at `~/.mcp-auth/`. Note: this path won't help the `/analyze-sentry-issue` slash command, which still needs `SENTRY_ACCESS_TOKEN` in `.env.local`.
|
||||
|
||||
## Security Notes
|
||||
|
||||
- `.agents/mcp.json` is committed. **Never put a token in it directly** — always go through `.env.local` via `envFile`.
|
||||
- `.mcp.json` is gitignored (generated by `make claude`), but treat it as if it weren't — re-running `make claude` after editing `.env.local` is the only supported way to update it.
|
||||
- The token inherits your existing Sentry permissions; the MCP cannot escalate access.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### `SENTRY_ACCESS_TOKEN` missing from `.mcp.json` after `make claude`
|
||||
|
||||
```bash
|
||||
jq '.mcpServers.Sentry.env | has("SENTRY_ACCESS_TOKEN")' .mcp.json
|
||||
```
|
||||
|
||||
If `false`: confirm `.env.local` has the line `SENTRY_ACCESS_TOKEN=...` (no surrounding quotes, no leading whitespace) and re-run `make claude`.
|
||||
|
||||
### "Server not found" after `make claude`
|
||||
|
||||
Confirm both files are in sync:
|
||||
|
||||
```bash
|
||||
jq '.mcpServers | keys' .agents/mcp.json
|
||||
jq '.mcpServers | keys' .mcp.json
|
||||
```
|
||||
|
||||
Restart Claude Code after `make claude` — it only loads MCP servers at startup.
|
||||
|
||||
### 401 / authentication errors
|
||||
|
||||
- Verify the token is current and has the scopes from Step 1.
|
||||
- Make sure no whitespace was copied with the value.
|
||||
- Re-run `make claude` after editing `.env.local`.
|
||||
|
||||
### Token cached for the wrong org (OAuth alternative only)
|
||||
|
||||
```bash
|
||||
rm -rf ~/.mcp-auth
|
||||
```
|
||||
|
||||
Then trigger a Sentry tool call again to start a fresh OAuth flow.
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [Sentry MCP docs](https://docs.sentry.io/product/sentry-mcp/)
|
||||
- [Sentry MCP source](https://github.com/getsentry/sentry-mcp)
|
||||
- [Sentry User Auth Tokens](https://sentry.io/settings/account/api/auth-tokens/)
|
||||
- [Model Context Protocol](https://modelcontextprotocol.io/)
|
||||
@@ -0,0 +1,204 @@
|
||||
# Slack MCP Configuration Guide
|
||||
|
||||
This guide explains how to configure the **Slack MCP server** for use with Cursor IDE, specifically for the `send-code-review-slack` command.
|
||||
|
||||
## Slack MCP Server Setup
|
||||
|
||||
This setup uses the **custom Slack MCP server** (`ghcr.io/korotovsky/slack-mcp-server`) which supports **User OAuth Tokens** (`SLACK_MCP_XOXP_TOKEN`). Messages will be posted as your authenticated user account, not as a bot.
|
||||
|
||||
### Setup Overview
|
||||
|
||||
**Workspace-Level Setup (One-Time, Done by Admin):**
|
||||
- **Step 1**: Create a Slack App
|
||||
- **Step 2**: Configure User Token Scopes
|
||||
|
||||
**User-Level Setup (Per Developer):**
|
||||
- **Step 3**: Install App to Workspace
|
||||
- **Step 4**: Get Your User OAuth Token
|
||||
- **Step 5**: Environment Variables
|
||||
- **Step 6**: Restart Cursor
|
||||
|
||||
---
|
||||
|
||||
### Step 1: Create a Slack App
|
||||
|
||||
**Note**: This is a one-time workspace setup step, typically done by a workspace admin. Once the app is created and configured, all users in the workspace can use it.
|
||||
|
||||
1. Go to [Slack API Apps](https://api.slack.com/apps)
|
||||
2. Click **"Create New App"** → **"From scratch"**
|
||||
3. Name your app (e.g., "Opik Code Review")
|
||||
4. Select your workspace
|
||||
5. Click **"Create App"**
|
||||
|
||||
### Step 2: Configure User Token Scopes
|
||||
|
||||
**Note**: This is a one-time workspace setup step, typically done by a workspace admin. Once the scopes are configured, all users in the workspace will have access to these scopes when they install the app.
|
||||
|
||||
**CRITICAL**: You must add scopes to **"User Token Scopes"** (NOT "Bot Token Scopes") for the User OAuth Token to appear.
|
||||
|
||||
1. In your app settings, go to **"OAuth & Permissions"**
|
||||
2. Scroll down to the **"Scopes"** section
|
||||
3. Find **"User Token Scopes"** section (this is different from "Bot Token Scopes" which is above it)
|
||||
- You'll see the description: *"Scopes that access user data and act on behalf of users that authorize them."*
|
||||
- This confirms you're in the right section - these scopes allow the app to act as YOU, not as a bot
|
||||
4. Click **"Add an OAuth Scope"** under "User Token Scopes"
|
||||
5. Add the following scopes one by one:
|
||||
- `chat:write` - Send messages as your authenticated user account
|
||||
- `channels:read` - View basic information about public channels
|
||||
- `users:read` - Read user information (required by the MCP server for caching)
|
||||
- `channels:history` - View messages in public channels (required by the MCP server for channel caching)
|
||||
6. Click **"Save Changes"**
|
||||
|
||||
**Note**: The `users:read` and `channels:history` scopes are required by the `ghcr.io/korotovsky/slack-mcp-server` to properly cache and access channel information. Without these, you may see "missing_scope" errors in the logs, though basic message posting may still work.
|
||||
|
||||
### Step 3: Install App to Workspace
|
||||
|
||||
**Note**: This is a per-user step. Each developer needs to install the app to their workspace to authorize it and get their own User OAuth Token.
|
||||
|
||||
1. Scroll to the top of **"OAuth & Permissions"**
|
||||
2. Click **"Install to Workspace"** (or **"Reinstall to Workspace"** if you already installed it)
|
||||
3. Review permissions and click **"Allow"**
|
||||
- **Note**: The permission screen may say "Send messages as [App Name]" - this is just the app requesting permission. When you use the User OAuth Token, messages will be posted as **your personal account**, not as the app.
|
||||
|
||||
### Step 4: Get Your User OAuth Token
|
||||
|
||||
**After installing/reinstalling the app**, you need to find your **User OAuth Token**:
|
||||
|
||||
1. Stay on the **"OAuth & Permissions"** page (or refresh it)
|
||||
2. Scroll down to the **"OAuth Tokens for Your Workspace"** section
|
||||
3. Look for **"User OAuth Token"** (NOT "Bot User OAuth Token")
|
||||
- If you only see "Bot User OAuth Token", you need to:
|
||||
- Go back to Step 2 and make sure you added scopes to **"User Token Scopes"** (not "Bot Token Scopes")
|
||||
- Then come back here and click **"Reinstall to Workspace"** again
|
||||
4. The User OAuth Token should start with `xoxp-` (not `xoxb-`)
|
||||
5. Click **"Show"** or **"Reveal"** to see the full token
|
||||
6. **Copy this token** - this is what will post messages as your authenticated user account
|
||||
|
||||
### Step 5: Environment Variables
|
||||
|
||||
2. **Create or edit `.env.local`** in your project root and add your User OAuth Token:
|
||||
|
||||
```bash
|
||||
SLACK_MCP_XOXP_TOKEN=xoxp-your-user-oauth-token-here
|
||||
```
|
||||
|
||||
**Important Configuration Options:**
|
||||
|
||||
1. **`mcp-server --transport stdio`**: Required for the server to communicate with Cursor via the MCP protocol. Without these, the server will start an SSE server instead, which won't work with Cursor's MCP integration.
|
||||
|
||||
2. **`SLACK_MCP_ADD_MESSAGE_TOOL`**: Required to enable the `conversations_add_message` tool (disabled by default for safety).
|
||||
- `true` or `1`: Enable for all channels and DMs
|
||||
- Channel IDs (comma-separated): Enable only for specific channels (e.g., `XXXXXXXX,YYYYYYYY`)
|
||||
- `!XXXXXXXX`: Enable for all channels except the specified one
|
||||
- **For the code review command**: Use `true` to enable posting to `#code-review`
|
||||
|
||||
3. **`envFile`**: Points to `${workspaceFolder}/.env.local` where your `SLACK_MCP_XOXP_TOKEN` is stored securely.
|
||||
|
||||
**Important**:
|
||||
- Replace `xoxp-your-user-oauth-token-here` in `.env.local` with your **User OAuth Token** from Step 4 (starts with `xoxp-`, NOT `xoxb-`)
|
||||
- The token is loaded from `.env.local` via the `envFile` configuration, keeping it out of `mcp.json`
|
||||
|
||||
**Note**: This configuration uses the custom Slack MCP server that supports User OAuth Tokens. Messages will be posted as your authenticated user account, not as a bot.
|
||||
|
||||
**Requirements**:
|
||||
- Docker must be installed and running on your system
|
||||
- The Docker image `ghcr.io/korotovsky/slack-mcp-server:latest` will be pulled automatically on first use
|
||||
- `.env.local` should be added to `.gitignore` to prevent committing your token
|
||||
|
||||
### Step 6: Restart Cursor
|
||||
|
||||
1. Close and reopen Cursor IDE
|
||||
2. The Slack MCP server should now be available with message posting enabled
|
||||
3. You can verify by checking Cursor Settings > Features > MCP
|
||||
4. Test by running: `cursor send-code-review-slack`
|
||||
|
||||
---
|
||||
|
||||
## Security Notes
|
||||
|
||||
- **Never commit tokens to Git**:
|
||||
- Add `.env.local` to `.gitignore` to prevent committing your token
|
||||
- The `mcp.json` file references the token via environment variable, but it may contain secrets, check it before committing
|
||||
- **Token storage**: The `SLACK_MCP_XOXP_TOKEN` is stored in `.env.local` (which should be in `.gitignore`), not directly in `mcp.json`
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### MCP Server Not Appearing
|
||||
|
||||
1. Check that `mcp.json` is in the correct location (`.cursor/mcp.json` or `~/.cursor/mcp.json`)
|
||||
2. Verify JSON syntax is valid (use a JSON validator)
|
||||
3. **Restart Cursor completely** (quit and reopen, not just reload window)
|
||||
4. Check Cursor Settings > Features > MCP for error messages
|
||||
5. Verify Docker is installed and running: `docker --version` and `docker ps`
|
||||
|
||||
### Docker Command Errors
|
||||
|
||||
If you see errors like "Usage: docker [OPTIONS] COMMAND" in the MCP logs:
|
||||
|
||||
1. **Verify Docker is running**:
|
||||
```bash
|
||||
docker ps
|
||||
```
|
||||
If this fails, start Docker Desktop or Docker daemon
|
||||
|
||||
2. **Test the Docker command manually**:
|
||||
```bash
|
||||
docker run -i --rm -e SLACK_MCP_XOXP_TOKEN=xoxp-your-token -e LOG_LEVEL=error ghcr.io/korotovsky/slack-mcp-server:latest
|
||||
```
|
||||
This should start the MCP server. If it fails, check the error message.
|
||||
|
||||
3. **Pull the Docker image**:
|
||||
```bash
|
||||
docker pull ghcr.io/korotovsky/slack-mcp-server:latest
|
||||
```
|
||||
|
||||
4. **Check mcp.json structure**: Ensure the `args` array is properly formatted with each argument as a separate string element
|
||||
|
||||
### Authentication Errors
|
||||
|
||||
1. **Verify your user token is correct** (should start with `xoxp-`, NOT `xoxb-`)
|
||||
- If you see `xoxb-` in the logs, you're using a bot token instead of a user token
|
||||
- Go back to Step 4 and get the **User OAuth Token** (not Bot User OAuth Token)
|
||||
2. Ensure the app has been installed to your workspace
|
||||
3. Verify user token scopes include all required scopes in **User Token Scopes**:
|
||||
- `chat:write` (required for posting messages)
|
||||
- `channels:read` (required for channel access)
|
||||
- `users:read` (required by MCP server for caching)
|
||||
- `channels:history` (recommended for full functionality)
|
||||
4. Make sure you're using `SLACK_MCP_XOXP_TOKEN` in your Docker args (not `SLACK_BOT_TOKEN`)
|
||||
5. **Check your actual mcp.json file** - the token in the Docker args should be `xoxp-...`
|
||||
6. **Restart Cursor completely** after updating the token in mcp.json
|
||||
7. Verify Docker is running and can pull the image
|
||||
|
||||
### Missing Scope Errors
|
||||
|
||||
If you see "missing_scope" errors in the MCP logs:
|
||||
|
||||
1. **Check which scope is missing** - the error message will indicate the specific scope
|
||||
2. **Add the missing scope** to your Slack app's **User Token Scopes**:
|
||||
- Go to your Slack app's "OAuth & Permissions" page
|
||||
- Scroll to "User Token Scopes"
|
||||
- Click "Add an OAuth Scope" and add the missing scope
|
||||
- Common missing scopes: `users:read`, `channels:history`
|
||||
3. **Reinstall the app** to your workspace after adding scopes
|
||||
4. **Update the token** in `mcp.json` if a new token was generated
|
||||
5. **Restart Cursor** to reload the MCP configuration
|
||||
|
||||
**Note**: Some "missing_scope" errors may be warnings that don't prevent basic functionality (like posting messages), but adding all recommended scopes ensures full MCP server functionality.
|
||||
|
||||
### Permission Errors
|
||||
|
||||
1. Ensure the user token has `chat:write` scope in **User Token Scopes**
|
||||
2. Verify you have access to the `#code-review` channel
|
||||
3. Check that you're a member of the channel
|
||||
4. If using Docker, ensure Docker has network access to reach Slack API
|
||||
|
||||
---
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [Model Context Protocol Documentation](https://modelcontextprotocol.io/)
|
||||
- [Slack API Documentation](https://api.slack.com/)
|
||||
- [Custom Slack MCP Server](https://github.com/korotovsky/slack-mcp-server)
|
||||
@@ -0,0 +1,72 @@
|
||||
{
|
||||
"mcpServers": {
|
||||
"GitHub": {
|
||||
"command": "docker",
|
||||
"args": [
|
||||
"run",
|
||||
"-i",
|
||||
"--rm",
|
||||
"-e",
|
||||
"GITHUB_PERSONAL_ACCESS_TOKEN",
|
||||
"-e",
|
||||
"GITHUB_TOOLSETS=repos,pull_requests,issues",
|
||||
"ghcr.io/github/github-mcp-server"
|
||||
],
|
||||
"envFile": "${workspaceFolder}/.env.local"
|
||||
},
|
||||
"slack": {
|
||||
"command": "docker",
|
||||
"args": [
|
||||
"run",
|
||||
"-i",
|
||||
"--rm",
|
||||
"-e",
|
||||
"SLACK_MCP_XOXP_TOKEN",
|
||||
"-e", "LOG_LEVEL=error",
|
||||
"-e",
|
||||
"SLACK_MCP_ADD_MESSAGE_TOOL=true",
|
||||
"ghcr.io/korotovsky/slack-mcp-server:latest",
|
||||
"mcp-server",
|
||||
"--transport",
|
||||
"stdio"
|
||||
],
|
||||
"envFile": "${workspaceFolder}/.env.local"
|
||||
},
|
||||
"Jira-Headless": {
|
||||
"command": "uvx",
|
||||
"args": ["mcp-atlassian"],
|
||||
"envFile": "${workspaceFolder}/.env.local"
|
||||
},
|
||||
"Notion": {
|
||||
"command": "npx",
|
||||
"args": ["-y", "mcp-remote", "https://mcp.notion.com/mcp"],
|
||||
"env": {}
|
||||
},
|
||||
"Sentry": {
|
||||
"command": "npx",
|
||||
"args": ["-y", "@sentry/mcp-server@latest"],
|
||||
"envFile": "${workspaceFolder}/.env.local"
|
||||
},
|
||||
"Playwright": {
|
||||
"command": "npx",
|
||||
"args": ["@playwright/mcp@latest"],
|
||||
"env": {}
|
||||
},
|
||||
"chrome-devtools": {
|
||||
"command": "npx",
|
||||
"args": ["chrome-devtools-mcp@latest"]
|
||||
},
|
||||
"playwright-test": {
|
||||
"command": "npx",
|
||||
"args": [
|
||||
"playwright",
|
||||
"run-test-mcp-server"
|
||||
],
|
||||
"cwd": "${workspaceFolder}/tests_end_to_end/e2e"
|
||||
},
|
||||
"context7": {
|
||||
"command": "npx",
|
||||
"args": ["-y", "@upstash/context7-mcp@latest"]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
---
|
||||
description: Agent deployment rules
|
||||
alwaysApply: true
|
||||
---
|
||||
# Agent Deployment
|
||||
|
||||
## Domain Routing
|
||||
- `apps/opik-backend/**` → Backend skill
|
||||
- `apps/opik-frontend/**` → Frontend skill
|
||||
- `sdks/opik_optimizer/**` → Opik Optimizer skill
|
||||
- `sdks/python/**` → Python SDK skill
|
||||
- `sdks/typescript/**` → TypeScript SDK skill
|
||||
- `tests_end_to_end/**` → Playwright E2E skill
|
||||
|
||||
## Frontend v1/v2 Structure
|
||||
- `apps/opik-frontend/src/v1/` → current Opik 1 UI (layout, pages, pages-shared)
|
||||
- `apps/opik-frontend/src/v2/` → next Opik 2 UI (layout, pages, pages-shared)
|
||||
- All new FE development targets v2 only
|
||||
- v1 changes only on special request — v1 is maintained but not actively developed
|
||||
- v1 MUST NOT import from v2; v2 MUST NOT import from v1
|
||||
- Both share: ui/, shared/, api/, types/, store/, hooks/, lib/, constants/, contexts/, plugins/
|
||||
- Shared components: backward-compatible changes only; must not be version-aware (no `isV2` props)
|
||||
|
||||
## Task Approach
|
||||
- Complex features → Plan before implementing
|
||||
- Bug fixes → Reproduce with test first
|
||||
- Multi-component → Coordinate across domains
|
||||
|
||||
## Execution
|
||||
- Independent tasks → Run in parallel
|
||||
- Use thinking model (Opus, GPT 5.3, GPT 5.3 High) for: complex planning, security, architecture
|
||||
- Use fast model (Haiku, GPT Nano, GPT Codex Low/Medium) for: simple test runs, formatting
|
||||
|
||||
## Formatting and Linting
|
||||
- For local verification and PR iterations, run `pre-commit` on changed files, not repository-wide file lists.
|
||||
- Avoid `--all-files` unless a cleanup migration explicitly requires repository-wide normalization.
|
||||
@@ -0,0 +1,25 @@
|
||||
---
|
||||
description: Code style conventions
|
||||
alwaysApply: true
|
||||
---
|
||||
|
||||
# Code Style
|
||||
|
||||
## Comments
|
||||
- Don't add comments that restate what code does
|
||||
- Don't add JSDoc/docstrings unless required by the codebase
|
||||
- Only comment **why**, not **what**
|
||||
- No "Added by Claude" or attribution comments
|
||||
- No placeholder TODOs like `// TODO: implement`
|
||||
|
||||
## Don't Over-Engineer
|
||||
- No helper functions for one-time operations
|
||||
- No abstractions for single use cases
|
||||
- No extra error handling "just in case"
|
||||
- No backwards-compatibility shims when changing code directly works
|
||||
|
||||
## Stay Focused
|
||||
- Only modify code relevant to the task
|
||||
- No drive-by refactoring of unrelated code
|
||||
- No reformatting lines you didn't change
|
||||
- No adding type annotations where inference works
|
||||
@@ -0,0 +1,46 @@
|
||||
---
|
||||
description: Git conventions
|
||||
alwaysApply: true
|
||||
---
|
||||
# Git Workflow
|
||||
|
||||
## Branch Naming
|
||||
`{username}/{ticket}-{summary}`
|
||||
- `andrescrz/OPIK-2180-add-feature`
|
||||
- `user/issue-123-fix-bug`
|
||||
- `user/NA-hotfix-description`
|
||||
|
||||
## Commit Format
|
||||
First commit in branch (used as PR title source):
|
||||
`[<TICKET-KEY>] [COMPONENT] <type>: description`
|
||||
- `[OPIK-1234] [BE] feat: add create trace endpoint`
|
||||
- `[issue-123] [FE] fix: guard empty state`
|
||||
- `[NA] [DOCS] docs: update API documentation`
|
||||
|
||||
Follow-up commits (preferred):
|
||||
`<type>(<scope>): description`
|
||||
- `feat(metrics): add project chart filters`
|
||||
- `fix(api): handle null feedback score`
|
||||
- `test(backend): cover trace creation`
|
||||
|
||||
`Revision N: ...` is a last-resort fallback only.
|
||||
|
||||
## PR Title
|
||||
`[<TICKET-KEY>] [COMPONENT] <type>: description`
|
||||
|
||||
## Jira Key References (commit messages & PR body)
|
||||
The GitHub for Jira app links a PR to a ticket's Development panel whenever it finds an issue key matching `[A-Z][A-Z0-9]+-\d+` (project key, literal hyphen, digits) in the branch name, PR title, PR body, or any commit message. It matches on the regex alone — it cannot tell "this PR resolves the ticket" from "this just mentions it" — and the resulting link cannot be removed afterward. So free text must only contain hyphenated keys for tickets the PR actually resolves.
|
||||
|
||||
- Tickets this PR **resolves** → keep the hyphen: `OPIK-1234`. Jira links/URLs are fine and wanted (you want these to link). A PR may resolve more than one ticket — list every resolved key this way.
|
||||
- Tickets **related but not resolved** in this PR (e.g. an escalation, or a fix that references an older ticket — anything not in the PR title or branch):
|
||||
- Replace the hyphen with an underscore: `OPIK_7000`. The underscore breaks the literal `-` the scanner requires, so the key cannot match.
|
||||
- Do **not** paste a Jira URL for them either — `.../browse/OPIK-7000` contains the hyphenated key and would link anyway. Backticks/parentheses do not help; only breaking the hyphen does.
|
||||
|
||||
Branch name, PR title, and the PR template's `## Issues` section are unaffected — resolved keys use normal `OPIK-1234` there.
|
||||
|
||||
## Never
|
||||
- Commit directly to main
|
||||
- Include customer names in commits/PRs
|
||||
- Commit secrets or .env files
|
||||
- Force push to main/master
|
||||
- Commit or raise security issues on a public repository
|
||||
@@ -0,0 +1,18 @@
|
||||
---
|
||||
description: Security requirements
|
||||
alwaysApply: true
|
||||
---
|
||||
# Security
|
||||
|
||||
## Never
|
||||
- Hardcode secrets, API keys, tokens
|
||||
- Commit .env files or credentials
|
||||
- Log sensitive data (passwords, tokens, PII)
|
||||
- Trust user input without validation
|
||||
- Use string concatenation for SQL queries
|
||||
|
||||
## Always
|
||||
- Use parameterized queries for SQL
|
||||
- Validate input at API boundaries
|
||||
- Use environment variables for secrets
|
||||
- Sanitize user input before rendering
|
||||
@@ -0,0 +1,23 @@
|
||||
# Skills Index
|
||||
|
||||
Domain-specific agent skills for the Opik monorepo. Each skill provides patterns, conventions, and guidance for a specific area of the codebase.
|
||||
|
||||
| Skill | Path | Description |
|
||||
|-------|------|-------------|
|
||||
| add-code-quality-hook | `add-code-quality-hook/` | Recipe for wiring a new linter into Opik's unified `🐙 Code Quality` pipeline (pre-commit + CI). Use when adding a pre-commit-driven linter — enumerates every file that must change, the non-obvious gotchas (mandatory explicit `files:`, blank-description trap, `TOOLCHAIN_BY_ID`/`TYPED_IDS`), the fix-vs-suppress policy, and the verification loop. |
|
||||
| analytics-instrumentation | `analytics-instrumentation/` | Add analytics events to Opik features. Use when wiring PostHog events on the frontend or backend for product analytics tracking. |
|
||||
| debugging-e2e-tests | `debugging-e2e-tests/` | Investigate a failed Opik E2E test and propose a fix (read-only). Use when a test goes red in CI, a TestOps launch, or locally — gathers the trace + history, classifies regression vs. flake, proposes a fix. |
|
||||
| diagram-generation | `diagram-generation/` | Generate self-contained HTML architecture diagrams. Use when creating visual diagrams for PRs, task plans, or architectural explanations. |
|
||||
| documentation | `documentation/` | Feature documentation and release notes patterns. Use when documenting changes, writing PR descriptions, or preparing releases. |
|
||||
| local-dev | `local-dev/` | Local development environment setup and commands. Use when helping with dev server, Docker, or local testing. |
|
||||
| metrics-instrumentation | `metrics-instrumentation/` | Instrument an opik-backend workflow with operational OpenTelemetry metrics and a flow-ordered Grafana dashboard. Use when a pipeline is a black box and you need per-stage throughput/latency/error visibility plus a per-customer drill. Distinct from analytics-instrumentation (PostHog product events). |
|
||||
| opik-backend | `opik-backend/` | Java backend patterns for Opik. Use when working in `apps/opik-backend`, designing APIs, database operations, or services. |
|
||||
| opik-external-integrations | `opik-external-integrations/` | Build an Opik integration that lives outside this repo — a standalone `opik-*` package or Opik support contributed into a third-party project (LiteLLM, Dify, …). Activates only for external-repo targets. |
|
||||
| opik-frontend | `opik-frontend/` | React frontend patterns for Opik. Use when working in `apps/opik-frontend`, on components, state, or data fetching. |
|
||||
| opik-integrations | `opik-integrations/` | Build, update, test, and document Opik SDK integrations (Python & TypeScript) that ship under `sdks/`. Runs a questionnaire-first, autonomous workflow: investigate → design → implement → verify via the Opik MCP → test → document → report. |
|
||||
| playwright-pom-discovery | `playwright-pom-discovery/` | Choose stable selectors against the live UI when building a Page Object Model for the E2E suite (`tests_end_to_end/e2e/pom/`). Used as the discovery sub-step by `writing-e2e-tests`. |
|
||||
| python-sdk | `python-sdk/` | Python SDK patterns for Opik. Use when working in `sdks/python`, on SDK APIs, integrations, or message processing. |
|
||||
| typescript-sdk | `typescript-sdk/` | TypeScript SDK patterns for Opik. Use when working in `sdks/typescript`. |
|
||||
| writing-e2e-tests | `writing-e2e-tests/` | Add an end-to-end test for an Opik feature in `tests_end_to_end/e2e/`. Use when a developer wants to write a test for a feature, page, or branch — runs the full loop: analyze, explore the live UI, write the POM + spec, and run it locally until green. |
|
||||
|
||||
Each skill directory contains a `SKILL.md` entry point plus supporting documents (testing, code quality, patterns, etc.).
|
||||
@@ -0,0 +1,109 @@
|
||||
---
|
||||
name: add-code-quality-hook
|
||||
description: Recipe for wiring a new linter into Opik's unified 🐙 Code Quality pipeline (pre-commit + CI). Use when adding a pre-commit-driven linter/formatter to the repo — enumerates every file that must change (`.pre-commit-config.yaml`, `scripts/precommit-hook-descriptions.tsv`, `scripts/precommit-detect-hooks.py`, `CONTRIBUTING.md`), the non-obvious gotchas (mandatory explicit `files:`, the blank-description trap, `TOOLCHAIN_BY_ID`/`TYPED_IDS`), the retroactive fix-vs-suppress policy, and the pass/fail verification loop. Worked examples: actionlint (live) and hadolint (OPIK-6673).
|
||||
---
|
||||
|
||||
# Add a Code Quality Hook
|
||||
|
||||
Opik runs all linters through **one** pipeline: pre-commit locally, and the `🐙 Code Quality` workflow (`.github/workflows/code_quality.yml`) in CI. CI does not run pre-commit wholesale — it derives, per PR, the set of hooks that actually have work using `scripts/precommit-detect-hooks.py`, then runs **one CI job per matched linter**. That design makes adding a linter a small, fixed set of edits — but each edit is load-bearing, and skipping one produces a *silent* gap (the hook runs nowhere, or renders blank in the summary, or provisions the wrong runtime) rather than a loud failure. This skill is the checklist.
|
||||
|
||||
The whole recipe is a generalization of two real PRs: **actionlint** (live in `.pre-commit-config.yaml` today — grep it as you read) and **hadolint** (OPIK-6673, PR #7352 — the first Docker-image hook, `toolchain: none`). Read the actionlint hook alongside this doc; it is the canonical, verifiable example.
|
||||
|
||||
## The wiring files
|
||||
|
||||
Each linter touches these four files. Do all four.
|
||||
|
||||
### 1. `.pre-commit-config.yaml` — add the hook
|
||||
|
||||
Add the upstream hook (repo / rev / id). Pin `rev` to a tag or SHA — never a floating ref.
|
||||
|
||||
**An explicit `files:` regex is mandatory, not optional.** This is the single most common miss. The CI matrix detector (`precommit-detect-hooks.py`) routes work to hooks by **path regex**, not by pre-commit's `types:`. Most upstream hooks (actionlint, hadolint) ship a `types:`-only match with no `files:`. If you copy them verbatim, the detector cannot route any file to your hook and **CI silently never runs it** — pre-commit locally still works, so the gap hides until something slips through. The detector guards against this: it raises loudly if a hook has `types:`/`types_or:` without `files:` (see `precommit-detect-hooks.py` lines ~115). So a missing `files:` fails the detect step rather than regressing silently — but you still must write the regex.
|
||||
|
||||
Write a `files:` regex that captures exactly the paths the linter should gate. Example (actionlint — workflows only):
|
||||
|
||||
```yaml
|
||||
- repo: https://github.com/rhysd/actionlint
|
||||
rev: v1.7.12
|
||||
hooks:
|
||||
- id: actionlint
|
||||
name: ⚙️ actionlint — github workflows
|
||||
files: ^\.github/workflows/.+\.(yml|yaml)$
|
||||
```
|
||||
|
||||
The `name:` is what the reader and the CI summary see — give it a clear, emoji-prefixed display name matching the house style of the other hooks. The **description keyword you add in step 2 is matched against this name**, so pick a name containing a stable, distinctive substring (e.g. `actionlint`, `hadolint`).
|
||||
|
||||
### 2. `scripts/precommit-hook-descriptions.tsv` — add the description row
|
||||
|
||||
This TSV is the single source of truth for the per-hook descriptions shown in the Code Quality timing/skipped tables (the CI summary comment). Format: `<keyword>\t<description>`. The keyword is matched as a **substring of the hook display name** (`name:` from step 1).
|
||||
|
||||
**Miss this and the hook renders with a blank description in the CI summary.** Add a row:
|
||||
|
||||
```
|
||||
hadolint Lint Dockerfiles
|
||||
```
|
||||
|
||||
**Order matters — most-specific first.** Matching is first-substring-wins top-to-bottom, so a more specific keyword must precede any that it contains (the file already does this: `ruff-format` before `ruff`). If your keyword is a substring of an existing one, place it above that line.
|
||||
|
||||
### 3. `scripts/precommit-detect-hooks.py` — toolchain and content-type maps
|
||||
|
||||
Two maps in this file may need an entry. Most new hooks need **neither** (default is `toolchain: none`, no content-type narrowing) — but decide deliberately:
|
||||
|
||||
- **`TOOLCHAIN_BY_ID`** — add your hook id here only if the leg's CI job must provision a **heavy runtime**: `java` (shells out to mvn), `node-fe` / `node-ts` (shells out to npm). Pre-commit's own hooks self-provision in isolated envs and need nothing; a `language: golang` hook (actionlint) self-builds; a Docker-image hook (hadolint) runs the image — all of these are `none`. `code_quality.yml` branches its setup steps on `matrix.leg.toolchain`, so a wrong value means a job either wastes minutes provisioning an unused runtime or lacks the runtime it needs.
|
||||
|
||||
- **`TYPED_IDS`** — add your hook id here only if it carries an upstream `types:` that narrows its `files:` match to a content **suffix**, so the detector doesn't emit a leg that would no-op at runtime. Value is the tuple of suffixes the hook actually acts on (e.g. `(".py", ".pyi")`). Symptom of a missing entry: the CI timing comment reports fewer ran rows than emitted legs ("detect over-emitted a leg"). If your `files:` regex is already suffix-precise (like actionlint's `\.(yml|yaml)$`), you don't need `TYPED_IDS`.
|
||||
|
||||
Rule of thumb by hook type:
|
||||
|
||||
| Hook type | `TOOLCHAIN_BY_ID` | `TYPED_IDS` |
|
||||
|---|---|---|
|
||||
| Docker-image linter (hadolint) | `none` (omit) | usually omit — make `files:` suffix-precise |
|
||||
| `language: golang`/self-built (actionlint) | `none` (omit) | omit if `files:` is suffix-precise |
|
||||
| Python tool (ruff, mypy) | `none` (omit) | add suffixes if `files:` is a broad dir regex |
|
||||
| Shells out to mvn | `java` | as needed |
|
||||
| Shells out to npm (FE/TS) | `node-fe` / `node-ts` | as needed |
|
||||
|
||||
### 4. `CONTRIBUTING.md` — "how to run locally" note
|
||||
|
||||
Add a short section alongside the existing **GitHub Actions workflows** (actionlint) note: what the linter checks, that it runs in the unified `🐙 Code Quality` workflow and locally via pre-commit, and that `make hooks` enables it. If the hook needs a local dependency (a Docker daemon for hadolint, etc.), say so; if pre-commit self-provisions it (actionlint builds from source), say that instead.
|
||||
|
||||
## Retroactive step — fix the existing violations
|
||||
|
||||
Adding a linter to a repo that has never run it will surface pre-existing violations. **The gate must be green on day one.** Policy, in order of preference:
|
||||
|
||||
1. **Fix the violation in place** — this is the default. Most findings are real and worth fixing.
|
||||
2. **Suppress inline, with a reason, next to the code** — only when a fix would be *genuinely undesirable*. The canonical case: exact-pinning rolling-channel OS packages (e.g. `apt-get install foo=1.2.3`), which rots as mirrors move — the honest answer is an inline `# hadolint ignore=DL3008` with a one-line why. Inline suppression is scoped to that one line and visible in review.
|
||||
3. **Never a global config-file ignore.** A repo-wide ignore (a `.hadolint.yaml` `ignored:` list, an eslint config-level disable) **fails open on every future file** — it silently exempts code no one has reviewed. Inline-with-reason keeps the gate strict on everything new.
|
||||
|
||||
See the OPIK-6673 discussion for a worked case where some rules genuinely couldn't be honestly fixed and inline-with-reason was the right call.
|
||||
|
||||
## Verification loop
|
||||
|
||||
Before opening the PR, confirm the wiring end-to-end — don't trust that the four edits compose:
|
||||
|
||||
1. **Detect emits a leg for a target file.** Run the detector against a file the hook should gate and confirm your hook id appears in `legs` with the right `toolchain`:
|
||||
```bash
|
||||
python3 scripts/precommit-detect-hooks.py .pre-commit-config.yaml path/to/target.file
|
||||
```
|
||||
If it lands in `skipped` instead, your `files:` regex doesn't match. If it errors about `types:` without `files:`, add the `files:` regex (step 1).
|
||||
|
||||
2. **Hook passes clean.** Run it on the current tree and confirm green (this is also the retroactive check):
|
||||
```bash
|
||||
pre-commit run <hook-id> --all-files
|
||||
```
|
||||
|
||||
3. **Hook fails on a new violation.** Introduce a deliberate violation in a target file and confirm the hook catches it, then revert. A hook that can't fail isn't gating anything.
|
||||
|
||||
4. **Description resolves.** Confirm the summary won't render blank:
|
||||
```bash
|
||||
echo "<your hook display name>" | python3 scripts/precommit-hook-desc.py
|
||||
```
|
||||
Expect `<name>\t<description>` — an empty second column means the keyword row (step 2) is missing or mis-ordered.
|
||||
|
||||
## Checklist
|
||||
|
||||
- [ ] `.pre-commit-config.yaml`: hook added with pinned `rev` and an **explicit `files:`** regex
|
||||
- [ ] `scripts/precommit-hook-descriptions.tsv`: keyword→description row, most-specific-first ordering
|
||||
- [ ] `scripts/precommit-detect-hooks.py`: `TOOLCHAIN_BY_ID` entry iff heavy runtime needed; `TYPED_IDS` entry iff `types:`-narrowed and `files:` isn't suffix-precise
|
||||
- [ ] `CONTRIBUTING.md`: local "how to run" note added
|
||||
- [ ] Retroactive: existing violations fixed (or inline-suppressed-with-reason); gate green on day one
|
||||
- [ ] Verified: detect emits the leg, hook passes clean, hook fails on a new violation, description resolves non-blank
|
||||
@@ -0,0 +1,147 @@
|
||||
---
|
||||
name: analytics-instrumentation
|
||||
description: Add analytics events to Opik features. Use when wiring PostHog events on the frontend or backend for product analytics tracking.
|
||||
---
|
||||
|
||||
# Analytics Instrumentation
|
||||
|
||||
## Event Naming
|
||||
|
||||
All events MUST be prefixed with `opik_`. Segment routes `opik_*` events to PostHog. The tooling enforces this automatically, but event names defined in code should already include the prefix.
|
||||
|
||||
Examples: `opik_onboarding_agent_name_submitted`, `opik_eval_suite_created`, `opik_optimization_created`
|
||||
|
||||
## Frontend Events
|
||||
|
||||
### Files
|
||||
- **Tracking utility**: `apps/opik-frontend/src/lib/analytics/tracking.ts` (mode-agnostic; safe to import from any project code)
|
||||
- **Segment init**: `apps/opik-frontend/src/plugins/comet/analytics/index.ts` (comet-only)
|
||||
- **Plugin init**: `apps/opik-frontend/src/plugins/comet/init.tsx` (comet-only)
|
||||
|
||||
### Adding a new event
|
||||
|
||||
1. Add the event name to the `OpikEvent` const in `tracking.ts`:
|
||||
```typescript
|
||||
export const OpikEvent = {
|
||||
ONBOARDING_AGENT_NAME_SUBMITTED: "opik_onboarding_agent_name_submitted",
|
||||
} as const;
|
||||
```
|
||||
|
||||
2. Call `trackEvent` from the component or hook where the action happens:
|
||||
```typescript
|
||||
import { trackEvent, OpikEvent } from "@/lib/analytics/tracking";
|
||||
|
||||
trackEvent(OpikEvent.ONBOARDING_AGENT_NAME_SUBMITTED, {
|
||||
agent_name: agentName,
|
||||
});
|
||||
```
|
||||
|
||||
### How it works
|
||||
- `trackEvent()` safely no-ops when Segment isn't loaded (OSS mode)
|
||||
- `opik_` prefix is enforced at runtime as a safety net
|
||||
- `OPIK_ANALYTICS_ENVIRONMENT` is injected into event properties automatically by `trackEvent()`
|
||||
- Frontend custom events flow through Segment (same pipeline as backend): Segment → PostHog
|
||||
- PostHog still handles automatic pageviews, user identification, and feature flags directly
|
||||
|
||||
## Backend Events
|
||||
|
||||
### Files
|
||||
- **Service**: `apps/opik-backend/src/main/java/com/comet/opik/infrastructure/bi/AnalyticsService.java`
|
||||
- **Config**: `apps/opik-backend/src/main/java/com/comet/opik/infrastructure/AnalyticsConfig.java`
|
||||
- **YAML config**: `apps/opik-backend/config.yml` (under `analytics:`)
|
||||
|
||||
### API
|
||||
`AnalyticsService` exposes two overloads:
|
||||
|
||||
```java
|
||||
void trackEvent(String eventType, Map<String, String> properties);
|
||||
void trackEvent(String eventType, Map<String, String> properties, String identity);
|
||||
```
|
||||
|
||||
- 2-arg resolves identity from the current request scope via `RequestContext`.
|
||||
- 3-arg takes an explicit identity — use it any time the call executes outside a request scope (reactive schedulers, background threads, event listeners).
|
||||
|
||||
### How it works
|
||||
- `trackEvent()` no-ops when `OPIK_ANALYTICS_ENABLED` is `false` (default).
|
||||
- `opik_` prefix is auto-prepended if missing — but keep the prefix in code for grep-ability.
|
||||
- `environment` property is auto-injected from `OPIK_ANALYTICS_ENVIRONMENT`.
|
||||
- Events flow: Backend → comet-stats → Segment → PostHog.
|
||||
- `AnalyticsService.sendEvent` wraps the body in `catch (RuntimeException)` — callers must not add their own try/catch.
|
||||
|
||||
### From a synchronous request handler
|
||||
|
||||
Inject and call inline. The 2-arg overload resolves identity from `RequestContext`.
|
||||
|
||||
```java
|
||||
private final @NonNull AnalyticsService analyticsService;
|
||||
|
||||
analyticsService.trackEvent("opik_onboarding_first_trace",
|
||||
Map.of("trace_id", traceId, "project_id", projectId));
|
||||
```
|
||||
|
||||
### From a reactive chain (`doOnSuccess`, `doOnNext`, etc.)
|
||||
|
||||
Two things are required: **offload with `Schedulers.boundedElastic()`** and **pass identity explicitly**.
|
||||
|
||||
**Why offload**: when identity is absent `AnalyticsService.resolveIdentity()` falls back to `UsageReportService.getAnonymousId()`, which is a synchronous JDBC read. Inside a `doOnSuccess` lambda that runs on the reactor event loop, that read blocks a scheduler-critical thread.
|
||||
|
||||
**Why explicit identity**: `RequestContext` is bound to the request thread via a Guice scope — inside the scheduler's lambda it throws `ProvisionException`, and you silently degrade to the anonymous-ID fallback, losing user attribution.
|
||||
|
||||
Capture `userName` up front from the reactor context alongside `workspaceId`, then pass both into the scheduled call:
|
||||
|
||||
```java
|
||||
return Mono.deferContextual(ctx -> {
|
||||
String workspaceId = ctx.get(RequestContext.WORKSPACE_ID);
|
||||
// Use getOrDefault on paths that internal/system callers reach without seeding USER_NAME
|
||||
// (e.g. a self-triggered cancellation written only with WORKSPACE_ID in the context).
|
||||
String userName = ctx.getOrDefault(RequestContext.USER_NAME, null);
|
||||
|
||||
return someDao.write(...)
|
||||
.doOnSuccess(__ -> Schedulers.boundedElastic().schedule(
|
||||
() -> analyticsService.trackEvent("opik_thing_happened",
|
||||
Map.of(
|
||||
"thing_id", thing.id().toString(),
|
||||
"workspace_id", workspaceId),
|
||||
userName)));
|
||||
});
|
||||
```
|
||||
|
||||
If you already depend on a `Schedulers.boundedElastic().schedule(() -> { ... })` block that does other non-reactive work (e.g. a blocking `datasetService.getById` like `ExperimentService.trackEvalSuiteRunIfApplicable`), add the `trackEvent` call inside that existing lambda instead of nesting another.
|
||||
|
||||
### Don'ts
|
||||
|
||||
- **Don't add try/catch around `trackEvent`** — `sendEvent` catches `RuntimeException` internally. Extra catches are noise and diverge from the codebase pattern.
|
||||
- **Don't add helper methods that only delegate to `trackEvent`** — inline the call at the entry point. Wrap in a helper only when it encapsulates real logic (e.g. applicability check + enrichment + tracking).
|
||||
- **Don't re-fetch ClickHouse rows to get "fresh" values for analytics payloads** — a write and a read-after-write can land on different replicas, so you may see a stale snapshot or even a spurious `NotFound`. Use the pre-write snapshot; some analytics drift is acceptable, a failed user-facing request is not.
|
||||
- **Don't add unit tests that `verify(analyticsService)...`** — the codebase convention is for existing integration tests to exercise these paths organically. Sister analytics PRs (#6326 eval suite, #6333 onboarding, #6338 agent config) ship without emission assertions.
|
||||
- **Don't assume `trackEvent` is fully non-blocking** — the Javadoc contract is aspirational; the identity-fallback path is synchronous JDBC today. Offload from reactive chains as shown above.
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Default | Purpose |
|
||||
|---|---|---|
|
||||
| `OPIK_ANALYTICS_ENABLED` | `false` | Backend: controls whether analytics events are sent |
|
||||
| `OPIK_ANALYTICS_ENVIRONMENT` | empty | Both: tags events with deployment name (e.g. `staging`, `production`) |
|
||||
| `OPIK_POSTHOG_KEY` | — | Frontend: PostHog API key (set in `config.js`) |
|
||||
| `OPIK_POSTHOG_HOST` | — | Frontend: PostHog API host (set in `config.js`) |
|
||||
|
||||
Analytics is disabled by default. OSS installations are unaffected.
|
||||
|
||||
## Event Flow
|
||||
|
||||
```
|
||||
Frontend custom events: Browser → Segment → PostHog
|
||||
Backend events: Java → comet-stats → Segment → PostHog
|
||||
PostHog native: Browser → posthog-js → PostHog (pageviews, feature flags, identification)
|
||||
```
|
||||
|
||||
## Event Property Conventions
|
||||
|
||||
- **Consistent typing per property**: A given property key should always carry the same kind of value. Don't pass a UUID in one code path and a human-readable name in another for the same key.
|
||||
- **Separate ID and name properties**: When both a UUID and a display name exist, use distinct keys (e.g. `blueprint_id` for the UUID, `blueprint_name` for the display name). If one is unavailable in a code path, omit the key or send an empty string — don't repurpose the other key.
|
||||
- **Include `workspace_id`**: All backend analytics events should include the workspace ID for segmentation.
|
||||
|
||||
## Deciding Frontend vs Backend
|
||||
|
||||
- **Frontend**: UI interactions (button clicks, wizard steps, form submissions, page visits)
|
||||
- **Backend**: SDK-triggered actions (trace creation, test suite runs), server-side computations, events that happen without the user being on the page
|
||||
@@ -0,0 +1,98 @@
|
||||
---
|
||||
name: debugging-e2e-tests
|
||||
description: Use when an Opik E2E test has failed and a developer wants it investigated — e.g. "why did this e2e test fail?", "investigate the failing run on my PR", "is dataset-crud-smoke flaky?", "the nightly e2e suite went red". Takes a failure from a CI check, a TestOps launch, a test name, or a local run; gathers the trace and history, classifies regression vs. flake, and proposes a fix. Read-only — it diagnoses and proposes, it does not edit tests.
|
||||
---
|
||||
|
||||
# Debugging E2E Tests
|
||||
|
||||
This skill investigates a failed test in the Opik E2E suite (`tests_end_to_end/e2e/`). You give it a failure from wherever you noticed it; it gathers the evidence, decides whether it's a real regression or a flake, and proposes a fix.
|
||||
|
||||
**Announce at start:** "I'm using the debugging-e2e-tests skill to investigate X."
|
||||
|
||||
## What this does — and doesn't
|
||||
|
||||
- It **diagnoses and proposes**, grounded in cited evidence (the trace, the error, the history). It is **read-only**: it does not edit tests, and it does not re-run the suite as part of investigating.
|
||||
- To **apply** a proposed fix, hand off to the `writing-e2e-tests` skill (or just say "apply it") — that's a separate, deliberate act with its own run-until-green loop.
|
||||
|
||||
## Where the evidence lives
|
||||
|
||||
- **Local run** — traces under `tests_end_to_end/e2e/test-results/` (retained on failure), Allure results under `allure-results/`.
|
||||
- **CI run** — the suite uploads three artifacts per run (7-day retention): `test-results-v2` (Playwright traces + videos), `playwright-report-v2` (the HTML report), `allure-results-v2`. Download with `gh run download <run-id> -n test-results-v2 -D <dir>`.
|
||||
- **Allure TestOps** (`comet.testops.cloud`, project id `1`) — results stream live during CI. Launches are named `Opik v2 … <tier> - <run_id>` (the trailing number is the GitHub Actions run id; the env segment varies — `E2E`, `Post-Merge`, `Local`, `staging`, `production`).
|
||||
|
||||
## Tooling
|
||||
|
||||
- **`allure-testops` MCP** (already connected) — the richest source. Validated calls:
|
||||
- `list_launches(projectId: 1, search: "<run_id or name fragment>", sort: ["createdDate,DESC"])` or `search_launches(rql: …)` — find the launch.
|
||||
- `list_test_results(launchId)` — per-test `name`, `fullName` (spec path + line, e.g. `datasets/dataset-crud-smoke.spec.ts:8:7`), `status`, a TestOps-computed **`flaky`** flag, `muted`/`known`, `tags`, `jobRun.url` (the GitHub Actions run), and the result `id`. Use `search` to filter to the failing test.
|
||||
- `get_test_result_history(id)` — the pass/fail timeline for that test across recent launches. This is the flake signal.
|
||||
- **`gh`** — `gh run view <run-id>` to find the failed job; `gh run download <run-id> -n test-results-v2 -D <dir>` for the trace artifact. A launch's `jobRun.url` gives you the run id.
|
||||
- **`npx playwright show-trace <trace.zip>`** (from `tests_end_to_end/e2e/`) — open the trace to see the exact step that failed, the DOM snapshot, and console/network at that moment.
|
||||
- **`git`** — diff the suspected change against the failing test's code path.
|
||||
|
||||
## The loop
|
||||
|
||||
```dot
|
||||
digraph debugging_e2e {
|
||||
rankdir=TB;
|
||||
"1. Resolve entry point" [shape=box];
|
||||
"2. Gather evidence" [shape=box];
|
||||
"3. Classify" [shape=box];
|
||||
"4. Diagnose" [shape=box];
|
||||
"5. Report + propose (no edits)" [shape=box];
|
||||
|
||||
"1. Resolve entry point" -> "2. Gather evidence";
|
||||
"2. Gather evidence" -> "3. Classify";
|
||||
"3. Classify" -> "4. Diagnose";
|
||||
"4. Diagnose" -> "5. Report + propose (no edits)";
|
||||
}
|
||||
```
|
||||
|
||||
### Step 1 — Resolve the entry point
|
||||
|
||||
Normalize whatever you were given into "a failed test + where its evidence lives":
|
||||
|
||||
- **A red CI check / Actions run** — take the run id. `gh run view <run-id>` for the failed job; find the matching launch via `list_launches(projectId: 1, search: "<run-id>")`; the trace is in the `test-results-v2` artifact (`gh run download`).
|
||||
- **A TestOps launch** — query it directly: `list_test_results(launchId)`, filter to the failed results.
|
||||
- **A test name** — `list_test_results` with `search` across a recent launch, or search launches, to find the result `id`; then pull its history.
|
||||
- **A local failure** — use the local `test-results/` trace and `allure-results/` directly; TestOps may have nothing for an uncommitted local run, which is fine.
|
||||
|
||||
### Step 2 — Gather evidence
|
||||
|
||||
- The failed assertion and error message (from the trace, the report, or the TestOps result).
|
||||
- The **trace**: `npx playwright show-trace` on the retained/downloaded `.zip`. Read the failing step, the DOM snapshot at that point, and console/network around it.
|
||||
- Screenshot / video if present (`only-on-failure` / `retain-on-failure`).
|
||||
- The test's **history** via `get_test_result_history(id)`, plus the TestOps `flaky` flag on the result. **Skip history gracefully** when TestOps isn't reachable (e.g. a purely local run) and fall back to trace + diff reasoning.
|
||||
|
||||
### Step 3 — Classify
|
||||
|
||||
Decide: **real regression**, **flake**, or **environment / selector drift**.
|
||||
|
||||
- **History when available:** a clean pass streak that broke right after a related change → lean regression. Intermittent pass/fail with no related change, or a TestOps `flaky: true` → lean flake.
|
||||
- **Diff correlation:** does a recent change touch the code path the failed assertion exercises (the page/component, the POM method, the fixture)? If yes → regression is likely. If the failed area is untouched → flake or environment is likely.
|
||||
- **Default to "flake / uncertain"** when history is intermittent and no related diff exists — don't over-call a regression without evidence.
|
||||
|
||||
### Step 4 — Diagnose
|
||||
|
||||
Root cause, grounded in cited evidence (the specific trace step, the error, the history pattern) — not speculation. Apply the suite's lenses:
|
||||
|
||||
- **Verify the test render before blaming the backend.** A "X didn't appear" failure is often a DOM race (a loading spinner still up, an eventually-consistent write not yet landed), not a backend regression. Check the trace's DOM snapshot at the failing step.
|
||||
- **Selector drift** — the FE changed an accessible name / removed a `data-testid`, so a locator no longer resolves.
|
||||
- **Eventually-consistent state** — async scoring/ingestion that needed a poll, not a fixed wait.
|
||||
- **Fixture seed-shape mismatch** — the page rendered an empty/partial state because the seed didn't match what the assertion expects.
|
||||
|
||||
### Step 5 — Report + propose (no edits)
|
||||
|
||||
Produce:
|
||||
|
||||
- **Verdict** — classification (regression / flake / environment-or-selector) + a confidence level.
|
||||
- **Evidence** — the trace step, the error, the history pattern, and the correlated change (if any), each cited.
|
||||
- **Proposed fix** — specific. For a regression: the code/selector/poll change to make. For a flake: a poll instead of a fixed wait, a quarantine, or "no code fix — known flaky, retry."
|
||||
|
||||
Do **not** edit anything. If the developer wants the fix applied, hand off to `writing-e2e-tests`.
|
||||
|
||||
## Boundaries
|
||||
|
||||
- Read-only: no test edits, no investigation-driven re-runs.
|
||||
- Works from all four entry points; degrades gracefully without TestOps (local failures use the trace + diff alone).
|
||||
- Distinct from authoring: `writing-e2e-tests` makes a new test; this explains a red one.
|
||||
@@ -0,0 +1,52 @@
|
||||
---
|
||||
name: diagram-generation
|
||||
description: Generate self-contained HTML architecture diagrams. Use when creating visual diagrams for PRs, task plans, or architectural explanations.
|
||||
---
|
||||
|
||||
# Diagram Generation
|
||||
|
||||
Generate self-contained HTML diagrams that visualize code changes, data flows, and architecture decisions.
|
||||
|
||||
## When to Use
|
||||
|
||||
- Visualizing PR changes for code review
|
||||
- Explaining architectural decisions
|
||||
- Documenting data/request flows
|
||||
- Illustrating before/after comparisons
|
||||
|
||||
## Output
|
||||
|
||||
- Self-contained HTML file at `{MAIN_REPO_ROOT}/diagrams/opik-{TICKET_NUMBER}-diagram.html` — always resolved against the main repo root (via `git rev-parse --git-common-dir`), even when the session runs inside a worktree, so the file outlives the worktree
|
||||
- Includes "Copy as image" button for sharing in Slack, Jira, PR descriptions
|
||||
- Dark GitHub theme, semantic color coding, responsive layout
|
||||
|
||||
## How to Generate
|
||||
|
||||
Follow the style guide in `style-guide.md` and use the HTML template in `template.md`.
|
||||
|
||||
### Required Sections (pick what applies)
|
||||
|
||||
1. **Request / Data Flow** — how data moves through layers
|
||||
2. **Why This Approach** — problem vs solution comparison
|
||||
3. **Files Changed by Layer** — grid of affected files grouped by component
|
||||
4. **Key Design Decisions** — numbered guards, trade-offs, or constraints
|
||||
|
||||
### Section Selection
|
||||
|
||||
- **Bug fix**: Focus on before/after flow, root cause, safety guards
|
||||
- **New feature**: Focus on data flow, architecture, files changed
|
||||
- **Refactor**: Focus on before/after architecture, files changed
|
||||
- **Cross-component**: Show all layers with connecting flows
|
||||
|
||||
## Reference Files
|
||||
|
||||
- [style-guide.md](style-guide.md) — Semantic colors, box themes, section labels, flow patterns, architecture trees
|
||||
- [template.md](template.md) — Base HTML structure, copy-as-image script, section recipes
|
||||
|
||||
## Common Gotchas
|
||||
|
||||
- **SRI hash on CDN scripts**: The html2canvas `<script>` tag must include `integrity` and `crossorigin` attributes — see template.md for the current hash
|
||||
- **Absolute paths for Playwright screenshots**: Playwright saves relative to its own CWD, not the repo root — always use absolute paths when calling `browser_take_screenshot`
|
||||
- **Max 4 sections**: More than 4 sections makes diagrams too tall for screenshots and hard to scan visually
|
||||
- **No raw diff content**: Diagrams show high-level summaries (component names, file names, flow descriptions) — never embed verbatim diff hunks or Jira comments
|
||||
- **`toBlob` can return null**: The Canvas `toBlob` call in the copy-as-image script needs a null check — see template.md
|
||||
@@ -0,0 +1,127 @@
|
||||
# Diagram Style Guide
|
||||
|
||||
## Theme
|
||||
|
||||
- Background: `#0d1117` (GitHub dark)
|
||||
- Text: `#c9d1d9` (primary), `#8b949e` (subtitle), `#6e7681` (notes)
|
||||
- Font: `-apple-system, BlinkMacSystemFont, 'Segoe UI', sans-serif`
|
||||
- Code font: `'JetBrains Mono', 'Fira Code', monospace`
|
||||
- Container: `max-width: 960px; margin: 0 auto; padding: 40px`
|
||||
|
||||
## Semantic Colors
|
||||
|
||||
| Role | Color | Use for |
|
||||
|------|-------|---------|
|
||||
| **Blue** (API/info) | `#58a6ff` | Titles, API endpoints, request flow, highlights |
|
||||
| **Green** (new/good) | `#3fb950` | New files, success states, solutions, converters |
|
||||
| **Red** (error/bad) | `#f85149` | Errors, problems, breaking changes, deadlocks |
|
||||
| **Purple** (special) | `#bc8cff` | SQL, mappers, enums, special logic |
|
||||
| **Yellow** (highlight) | `#fbbf24` | Key fields, important values |
|
||||
| **Orange** (modified) | `#d29922` | Modified files, DAO changes |
|
||||
|
||||
## Section Labels
|
||||
|
||||
```html
|
||||
<div class="section-label blue"><span class="dot"></span> SECTION TITLE</div>
|
||||
```
|
||||
|
||||
- Font: 11px, 700 weight, 1.5px letter-spacing, uppercase
|
||||
- Colored dot (8px circle) before text
|
||||
- Colors: `.blue`, `.red`, `.green`, `.purple`
|
||||
|
||||
## Flow Rows
|
||||
|
||||
```html
|
||||
<div class="flow">
|
||||
<div class="box client">Source</div>
|
||||
<div class="arrow green">→</div>
|
||||
<div class="box layer"><b>Component</b><br>detail</div>
|
||||
</div>
|
||||
```
|
||||
|
||||
- `display: flex; align-items: center`
|
||||
- Boxes: `padding: 9px 14px; border-radius: 8px; font-size: 12px`
|
||||
- Use `→` entity or SVG arrows between boxes
|
||||
|
||||
## Box Themes
|
||||
|
||||
| Class | Background | Border | Text | Use |
|
||||
|-------|-----------|--------|------|-----|
|
||||
| `.client` | `#1a1f2e` | `#2d3548` | `#93c5fd` | Client/request origin |
|
||||
| `.layer` | `#161b22` | `#21262d` | `#c9d1d9` | Generic layer/component |
|
||||
| `.converter` | `#122117` | `#1e5631` | `#86efac` | Converters, new logic |
|
||||
| `.bad` | `#2d1111` | `#f85149` 2px | `#fca5a5` | Error states, problems |
|
||||
| `.good` | `#0d2818` | `#3fb950` 2px | `#86efac` | Success states, solutions |
|
||||
| `.sql` | `#1b1728` | `#3b2d5e` | `#c4b5fd` | SQL/database queries |
|
||||
|
||||
## Architecture Trees
|
||||
|
||||
- Vertical lines: `width: 1px; height: 14px; background: #30363d`
|
||||
- Horizontal branches: `height: 1px; background: #30363d`
|
||||
- Base class: dashed border (`1px dashed #475569`)
|
||||
- Implementations: solid border, colored by type
|
||||
- NEW badge: `background: #3fb950; color: #0d1117; font-size: 9px; border-radius: 4px`
|
||||
|
||||
## Problem / Solution Banners
|
||||
|
||||
```html
|
||||
<div class="banner problem">
|
||||
<h2>Problem</h2>
|
||||
<p>Description...</p>
|
||||
</div>
|
||||
<div class="banner solution">
|
||||
<h2>Solution</h2>
|
||||
<p>Description...</p>
|
||||
</div>
|
||||
```
|
||||
|
||||
- Problem: `background: linear-gradient(135deg, #2d1215, #1c1012); border: 1px solid #f8514966`
|
||||
- Solution: `background: linear-gradient(135deg, #0c2d1a, #0d1117); border: 1px solid #3fb95066`
|
||||
|
||||
## Files Changed Grid
|
||||
|
||||
```html
|
||||
<div style="display: grid; grid-template-columns: 1fr 1fr; gap: 8px; max-width: 700px;">
|
||||
<div class="box layer" style="font-size: 11px;"><b>FileName</b> — change summary</div>
|
||||
<div class="box converter" style="font-size: 11px;"><b>NewFile</b> — new</div>
|
||||
</div>
|
||||
```
|
||||
|
||||
- 2-column grid for files
|
||||
- Use `.layer` for modified, `.converter` for new files
|
||||
|
||||
## Dividers
|
||||
|
||||
```html
|
||||
<hr class="divider">
|
||||
```
|
||||
`border: none; border-top: 1px solid #21262d; margin: 24px 0`
|
||||
|
||||
## Notes
|
||||
|
||||
```html
|
||||
<div class="note">Explanation with <span class="code">inline code</span></div>
|
||||
```
|
||||
`color: #6e7681; font-size: 11px`
|
||||
|
||||
## Tags
|
||||
|
||||
```html
|
||||
<span class="tag new">new</span>
|
||||
<span class="tag modified">modified</span>
|
||||
```
|
||||
|
||||
- New: `background: #0d419d; color: #79c0ff`
|
||||
- Modified: `background: #3d2e00; color: #d29922`
|
||||
|
||||
## Safety Guards / Numbered Lists
|
||||
|
||||
```html
|
||||
<div class="guard">
|
||||
<div class="guard-icon">1</div>
|
||||
Guard description
|
||||
</div>
|
||||
```
|
||||
|
||||
- Icon: 20px circle, numbered
|
||||
- Green: `background: #064e3b; color: #6ee7b7`
|
||||
@@ -0,0 +1,208 @@
|
||||
# HTML Diagram Template
|
||||
|
||||
Use this as the base structure for generating diagrams. Adapt sections based on what the changes require.
|
||||
|
||||
## Base HTML Structure
|
||||
|
||||
```html
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<title>OPIK-{TICKET} – {Title}</title>
|
||||
<style>
|
||||
* { margin: 0; padding: 0; box-sizing: border-box; }
|
||||
body { font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', sans-serif; background: #0d1117; color: #c9d1d9; padding: 40px; }
|
||||
.container { max-width: 960px; margin: 0 auto; padding: 40px; }
|
||||
h1 { color: #58a6ff; font-size: 24px; margin-bottom: 6px; }
|
||||
.subtitle { color: #8b949e; font-size: 14px; margin-bottom: 36px; }
|
||||
.section { margin-bottom: 32px; }
|
||||
.section-label { font-size: 11px; font-weight: 700; letter-spacing: 1.5px; text-transform: uppercase; margin-bottom: 14px; display: flex; align-items: center; gap: 8px; }
|
||||
.section-label .dot { width: 8px; height: 8px; border-radius: 50%; display: inline-block; }
|
||||
.section-label.blue { color: #58a6ff; }
|
||||
.section-label.blue .dot { background: #58a6ff; }
|
||||
.section-label.red { color: #f85149; }
|
||||
.section-label.red .dot { background: #f85149; }
|
||||
.section-label.green { color: #3fb950; }
|
||||
.section-label.green .dot { background: #3fb950; }
|
||||
.section-label.purple { color: #bc8cff; }
|
||||
.section-label.purple .dot { background: #bc8cff; }
|
||||
|
||||
.flow { display: flex; align-items: center; gap: 0; margin: 10px 0; flex-wrap: wrap; }
|
||||
.box { padding: 9px 14px; border-radius: 8px; font-size: 12px; white-space: nowrap; font-weight: 500; line-height: 1.4; }
|
||||
.arrow { padding: 0 5px; flex-shrink: 0; font-size: 18px; color: #30363d; }
|
||||
.arrow.green { color: #3fb950; }
|
||||
.arrow.red { color: #f85149; }
|
||||
|
||||
.box.client { background: #1a1f2e; border: 1px solid #2d3548; color: #93c5fd; }
|
||||
.box.layer { background: #161b22; border: 1px solid #21262d; color: #c9d1d9; }
|
||||
.box.layer b { color: #58a6ff; }
|
||||
.box.converter { background: #122117; border: 1px solid #1e5631; color: #86efac; }
|
||||
.box.converter b { color: #3fb950; }
|
||||
.box.bad { background: #2d1111; border: 2px solid #f85149; color: #fca5a5; }
|
||||
.box.good { background: #0d2818; border: 2px solid #3fb950; color: #86efac; }
|
||||
.box.sql { background: #1b1728; border: 1px solid #3b2d5e; color: #c4b5fd; font-family: 'JetBrains Mono', 'Fira Code', monospace; font-size: 11px; }
|
||||
|
||||
.code { font-family: 'JetBrains Mono', 'Fira Code', monospace; background: #161b22; padding: 1px 5px; border-radius: 4px; font-size: 11px; color: #e6edf3; }
|
||||
.note { color: #6e7681; font-size: 11px; margin-top: 5px; margin-left: 4px; }
|
||||
.divider { border: none; border-top: 1px solid #21262d; margin: 24px 0; }
|
||||
|
||||
.tag { display: inline-block; font-size: 10px; padding: 2px 8px; border-radius: 20px; font-weight: 600; margin-left: 6px; vertical-align: middle; }
|
||||
.tag.new { background: #0d419d; color: #79c0ff; }
|
||||
.tag.modified { background: #3d2e00; color: #d29922; }
|
||||
|
||||
.banner { width: 100%; border-radius: 10px; padding: 16px 24px; margin-bottom: 16px; }
|
||||
.banner.problem { background: linear-gradient(135deg, #2d1215, #1c1012); border: 1px solid #f8514966; }
|
||||
.banner.solution { background: linear-gradient(135deg, #0c2d1a, #0d1117); border: 1px solid #3fb95066; }
|
||||
.banner h2 { font-size: 14px; margin-bottom: 6px; }
|
||||
.banner.problem h2 { color: #f85149; }
|
||||
.banner.solution h2 { color: #3fb950; }
|
||||
.banner p { font-size: 12.5px; color: #b1bac4; line-height: 1.6; }
|
||||
|
||||
.guard { display: flex; align-items: center; gap: 10px; margin-bottom: 8px; font-size: 13px; }
|
||||
.guard-icon { width: 20px; height: 20px; border-radius: 50%; display: flex; align-items: center; justify-content: center; font-size: 11px; flex-shrink: 0; background: #064e3b; color: #6ee7b7; }
|
||||
|
||||
.copy-btn {
|
||||
position: fixed; top: 16px; right: 16px; z-index: 100;
|
||||
background: #238636; color: #fff; border: none; border-radius: 8px;
|
||||
padding: 10px 20px; font-size: 13px; font-weight: 600; cursor: pointer;
|
||||
font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', sans-serif;
|
||||
transition: background 0.2s;
|
||||
}
|
||||
.copy-btn:hover { background: #2ea043; }
|
||||
.copy-btn.copied { background: #1a7f37; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
|
||||
<button class="copy-btn" onclick="copyAsImage(this)">Copy as image</button>
|
||||
|
||||
<div class="container" id="diagram">
|
||||
|
||||
<h1>OPIK-{TICKET} — {Title}</h1>
|
||||
<div class="subtitle">{One-line summary of what this change does}</div>
|
||||
|
||||
<!-- ─── SECTION: Request / Data Flow ─── -->
|
||||
<!-- Use .flow rows with .box and .arrow elements -->
|
||||
|
||||
<hr class="divider">
|
||||
|
||||
<!-- ─── SECTION: Why This Approach ─── -->
|
||||
<!-- Use .banner.problem and .banner.solution side by side -->
|
||||
<!-- Or use before/after flow comparison -->
|
||||
|
||||
<hr class="divider">
|
||||
|
||||
<!-- ─── SECTION: Files Changed ─── -->
|
||||
<!-- Use grid layout with .box.layer for modified, .box.converter for new -->
|
||||
|
||||
<hr class="divider">
|
||||
|
||||
<!-- ─── SECTION: Key Design Decisions ─── -->
|
||||
<!-- Use .guard elements with numbered icons -->
|
||||
|
||||
</div>
|
||||
|
||||
<script src="https://cdnjs.cloudflare.com/ajax/libs/html2canvas/1.4.1/html2canvas.min.js" integrity="sha384-ZZ1pncU3bQe8y31yfZdMFdSpttDoPmOZg2wguVK9almUodir1PghgT0eY7Mrty8H" crossorigin="anonymous"></script>
|
||||
<script>
|
||||
async function copyAsImage(btn) {
|
||||
btn.textContent = 'Rendering...';
|
||||
try {
|
||||
const el = document.getElementById('diagram');
|
||||
const canvas = await html2canvas(el, {
|
||||
backgroundColor: '#0d1117',
|
||||
scale: 2,
|
||||
useCORS: true,
|
||||
});
|
||||
const blob = await new Promise(r => canvas.toBlob(r, 'image/png'));
|
||||
if (!blob) throw new Error('Canvas toBlob returned null');
|
||||
await navigator.clipboard.write([new ClipboardItem({ 'image/png': blob })]);
|
||||
btn.textContent = 'Copied!';
|
||||
btn.classList.add('copied');
|
||||
setTimeout(() => { btn.textContent = 'Copy as image'; btn.classList.remove('copied'); }, 2000);
|
||||
} catch (e) {
|
||||
console.error(e);
|
||||
btn.textContent = 'Failed — check console';
|
||||
setTimeout(() => { btn.textContent = 'Copy as image'; }, 3000);
|
||||
}
|
||||
}
|
||||
</script>
|
||||
|
||||
</body>
|
||||
</html>
|
||||
```
|
||||
|
||||
## Section Recipes
|
||||
|
||||
### Flow Row (horizontal data flow)
|
||||
```html
|
||||
<div class="section">
|
||||
<div class="section-label blue"><span class="dot"></span> REQUEST FLOW</div>
|
||||
<div class="flow">
|
||||
<div class="box client">GET /endpoint</div>
|
||||
<div class="arrow green">→</div>
|
||||
<div class="box layer"><b>Resource</b><br>validation</div>
|
||||
<div class="arrow green">→</div>
|
||||
<div class="box layer"><b>Service</b><br>business logic</div>
|
||||
<div class="arrow green">→</div>
|
||||
<div class="box sql">WHERE clause</div>
|
||||
</div>
|
||||
<div class="note">Description of the flow</div>
|
||||
</div>
|
||||
```
|
||||
|
||||
### Problem / Solution Side-by-Side
|
||||
```html
|
||||
<div style="display: flex; gap: 16px; flex-wrap: wrap;">
|
||||
<div class="banner problem" style="flex: 1; min-width: 380px;">
|
||||
<h2>Problem</h2>
|
||||
<p>What was wrong...</p>
|
||||
</div>
|
||||
<div class="banner solution" style="flex: 1; min-width: 380px;">
|
||||
<h2>Solution</h2>
|
||||
<p>How we fixed it...</p>
|
||||
</div>
|
||||
</div>
|
||||
```
|
||||
|
||||
### Before / After Flow Comparison
|
||||
```html
|
||||
<div style="display: flex; gap: 24px; flex-wrap: wrap;">
|
||||
<div style="flex: 1; min-width: 380px;">
|
||||
<div style="color: #f85149; font-size: 12px; font-weight: 600; margin-bottom: 8px;">Before</div>
|
||||
<div class="flow">
|
||||
<div class="box client">Input</div>
|
||||
<div class="arrow red">→</div>
|
||||
<div class="box bad">Error</div>
|
||||
</div>
|
||||
</div>
|
||||
<div style="flex: 1; min-width: 380px;">
|
||||
<div style="color: #3fb950; font-size: 12px; font-weight: 600; margin-bottom: 8px;">After</div>
|
||||
<div class="flow">
|
||||
<div class="box client">Input</div>
|
||||
<div class="arrow green">→</div>
|
||||
<div class="box good">Success</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
```
|
||||
|
||||
### Files Changed Grid
|
||||
```html
|
||||
<div class="section">
|
||||
<div class="section-label blue"><span class="dot"></span> FILES CHANGED</div>
|
||||
<div style="display: grid; grid-template-columns: 1fr 1fr; gap: 8px; max-width: 700px;">
|
||||
<div class="box layer" style="font-size: 11px;"><b>ExistingFile</b> — what changed</div>
|
||||
<div class="box converter" style="font-size: 11px;"><b>NewFile</b> — new</div>
|
||||
</div>
|
||||
</div>
|
||||
```
|
||||
|
||||
### Design Decisions / Safety Guards
|
||||
```html
|
||||
<div class="section">
|
||||
<div class="section-label green"><span class="dot"></span> KEY DESIGN DECISIONS</div>
|
||||
<div class="guard"><div class="guard-icon">1</div> Decision or constraint explanation</div>
|
||||
<div class="guard"><div class="guard-icon">2</div> Another decision</div>
|
||||
</div>
|
||||
```
|
||||
@@ -0,0 +1,94 @@
|
||||
---
|
||||
name: documentation
|
||||
description: Feature documentation and release notes patterns. Use when documenting changes, writing PR descriptions, or preparing releases.
|
||||
---
|
||||
|
||||
# Documentation
|
||||
|
||||
## PR Description
|
||||
|
||||
```markdown
|
||||
## Summary
|
||||
- What this PR does (bullet points)
|
||||
|
||||
## Test Plan
|
||||
- How to verify it works
|
||||
|
||||
## Related Issues
|
||||
- Resolves #123
|
||||
```
|
||||
|
||||
## Changelog Entry
|
||||
|
||||
```markdown
|
||||
### [VERSION] - [DATE]
|
||||
|
||||
#### New Features
|
||||
- **Feature Name**: Brief description
|
||||
|
||||
#### Improvements
|
||||
- **Improvement**: What changed and why
|
||||
|
||||
#### Bug Fixes
|
||||
- **Fix**: What was broken (#issue)
|
||||
|
||||
#### Breaking Changes
|
||||
- **Change**: What breaks, migration steps
|
||||
```
|
||||
|
||||
## Feature Documentation
|
||||
|
||||
When documenting a feature, cover:
|
||||
|
||||
**User Impact**
|
||||
- What capability does this add?
|
||||
- How do users access it?
|
||||
|
||||
**Technical Changes**
|
||||
- API changes (endpoints, params)
|
||||
- SDK changes (new methods)
|
||||
- Database migrations
|
||||
- Config changes
|
||||
|
||||
**Breaking Changes** (if any)
|
||||
- What breaks
|
||||
- Migration steps
|
||||
|
||||
## Key Files
|
||||
|
||||
- `CHANGELOG.md` - Self-hosted deployment changelog (breaking/critical changes only)
|
||||
- `apps/opik-documentation/documentation/fern/docs/changelog/` - Main product docs changelog entries (dated `.mdx` files)
|
||||
- `apps/opik-documentation/documentation/fern/docs/agent_optimization/getting_started/changelog.mdx` - Agent Optimizer release changelog
|
||||
- `apps/opik-documentation/documentation/fern/docs.yml` - Docs routing/navigation source of truth for changelog surfaces
|
||||
- `.github/release-drafter.yml` - Release template
|
||||
|
||||
## Changelog Routing Rules
|
||||
|
||||
- Pick the changelog target by scope; do not default everything to root `CHANGELOG.md`.
|
||||
- Use `CHANGELOG.md` only for self-hosted deployment breaking/critical/security-impacting notes.
|
||||
- Use `apps/opik-documentation/documentation/fern/docs/changelog/*.mdx` for general Opik product release notes shown in `/docs/opik/changelog`.
|
||||
- Use `apps/opik-documentation/documentation/fern/docs/agent_optimization/getting_started/changelog.mdx` for Agent Optimizer version updates (for example `sdks/opik_optimizer` releases like `3.1.0`).
|
||||
- Liquibase `changelog.xml` files are migration manifests, not user-facing release-note changelogs.
|
||||
- If unsure where an entry belongs, confirm the surface from `apps/opik-documentation/documentation/fern/docs.yml` before editing.
|
||||
|
||||
## Images in documentation
|
||||
|
||||
- **Use `fern/img`** for documentation images (e.g. `apps/opik-documentation/documentation/fern/img/...`).
|
||||
- **Do not use `static/img`** for new assets; it is a legacy folder used by external integrations and cannot be deleted.
|
||||
- Reference images in docs as `/img/...` (e.g. `/img/tracing/openai_integration.png`).
|
||||
- In repos that define `docs.yaml`/`docs.yml`, treat that file as the routing source of truth; do not assume URLs mirror directory layout.
|
||||
|
||||
## Internationalized READMEs
|
||||
|
||||
Non-English README files (`readme_CN.md`, `readme_JP.md`, `readme_KO.md`, `readme_PT_BR.md`) are AI machine-translated from the English `README.md`.
|
||||
|
||||
- Each non-English README must have a notice at the top (as a blockquote) warning that the file is AI-translated and welcoming improvements.
|
||||
- When the English README is updated with significant content changes, re-translate the affected non-English READMEs using AI and update accordingly.
|
||||
- Do not manually edit translated READMEs for content changes; update the English source and re-translate.
|
||||
|
||||
## Style
|
||||
|
||||
- User perspective, not implementation details
|
||||
- Specific (version numbers, dates)
|
||||
- Code examples for API/SDK changes
|
||||
- Concise - link to docs, don't duplicate
|
||||
@@ -0,0 +1,93 @@
|
||||
---
|
||||
name: local-dev
|
||||
description: Local development environment setup and commands. Use when helping with dev server, Docker, or local testing.
|
||||
---
|
||||
|
||||
# Local Development
|
||||
|
||||
## Quick Start
|
||||
|
||||
```bash
|
||||
./scripts/dev-runner.sh --restart # First time / full rebuild
|
||||
./scripts/dev-runner.sh --start # Daily start (no rebuild)
|
||||
./scripts/dev-runner.sh --stop # Stop everything
|
||||
./scripts/dev-runner.sh --verify # Check status
|
||||
```
|
||||
|
||||
## Modes
|
||||
|
||||
| Mode | Command | Frontend | Use When |
|
||||
|------|---------|----------|----------|
|
||||
| Standard | `--restart` | localhost:5174 | Frontend work, full-stack |
|
||||
| BE-only | `--be-only-restart` | localhost:5173 | Backend-focused, faster rebuilds |
|
||||
| Platform (EM) | `PLATFORM_ENABLED=true ./scripts/dev-runner.sh --restart` | localhost:9100 | Opik-team only: run Opik connected to the Comet Platform |
|
||||
|
||||
### Platform (EM) mode — Opik-team only
|
||||
|
||||
`PLATFORM_ENABLED=true` runs the Comet EM/Platform stack (`comet-backend` +
|
||||
`comet-react`, auto-detected sibling checkouts) alongside Opik behind a
|
||||
single-origin nginx proxy, with Opik in **comet mode** authenticating/resolving
|
||||
workspaces via comet-backend. Off by default — Standard/BE-only dev is
|
||||
unaffected. comet-backend builds under a detected JDK 17/21; Opik still uses JDK 25.
|
||||
|
||||
```bash
|
||||
PLATFORM_ENABLED=true ./scripts/dev-runner.sh --restart # then also --start/--stop/--verify
|
||||
# Integrated UI: http://localhost:9100 (comet-react / Platform)
|
||||
# http://localhost:9100/opik (Opik, comet mode)
|
||||
```
|
||||
|
||||
Env vars (`COMET_BACKEND_PATH`, `COMET_REACT_PATH`, `EM_JAVA_HOME`, `EM_*_PORT`)
|
||||
and `--platform-build` are documented in `./scripts/dev-runner.sh --help`.
|
||||
|
||||
## URLs
|
||||
|
||||
- **Frontend**: localhost:5174 (standard) / localhost:5173 (BE-only)
|
||||
- **Backend API**: localhost:8080
|
||||
- **Health**: http://localhost:8080/health-check?name=all
|
||||
|
||||
## Build Commands
|
||||
|
||||
```bash
|
||||
./scripts/dev-runner.sh --build-be # Backend only
|
||||
./scripts/dev-runner.sh --build-fe # Frontend only
|
||||
./scripts/dev-runner.sh --lint-be # Spotless
|
||||
./scripts/dev-runner.sh --lint-fe # ESLint
|
||||
./scripts/dev-runner.sh --migrate # DB migrations
|
||||
```
|
||||
|
||||
## Logs
|
||||
|
||||
```bash
|
||||
tail -f /tmp/opik-backend.log # Backend
|
||||
tail -f /tmp/opik-frontend.log # Frontend (standard)
|
||||
docker logs -f opik-frontend-1 # Frontend (BE-only)
|
||||
```
|
||||
|
||||
## SDK Config
|
||||
|
||||
```bash
|
||||
export OPIK_URL_OVERRIDE='http://localhost:8080'
|
||||
export OPIK_WORKSPACE='default'
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
**Won't start:**
|
||||
```bash
|
||||
./scripts/dev-runner.sh --verify
|
||||
lsof -i :8080 # Port conflict?
|
||||
./scripts/dev-runner.sh --stop && ./scripts/dev-runner.sh --restart
|
||||
```
|
||||
|
||||
**Build fails:**
|
||||
```bash
|
||||
cd apps/opik-backend && mvn clean install -DskipTests
|
||||
cd apps/opik-frontend && rm -rf node_modules && npm install
|
||||
```
|
||||
|
||||
**Database issues:**
|
||||
```bash
|
||||
./scripts/dev-runner.sh --stop
|
||||
./opik.sh --clean # WARNING: deletes data
|
||||
./scripts/dev-runner.sh --restart
|
||||
```
|
||||
@@ -0,0 +1,124 @@
|
||||
---
|
||||
name: metrics-instrumentation
|
||||
description: Specification for instrumenting an opik-backend workflow with operational OpenTelemetry metrics — per-stage throughput/latency/error counters and native histograms, dimensioned per-customer (workspace). Use when a pipeline (scoring, ingestion, experiments, jobs) needs per-stage visibility. Covers metric emission only; building the Grafana dashboard from these metrics is a separate skill. Distinct from analytics-instrumentation (PostHog product events).
|
||||
---
|
||||
|
||||
# Metrics Instrumentation
|
||||
|
||||
Normative spec for the backend half of operational observability: per-stage OpenTelemetry metrics in `apps/opik-backend`. The metrics are designed so a flow-ordered Grafana dashboard can read top-to-bottom — the failing stage is the one whose numbers break — with a per-workspace drill. Building that dashboard (layout, query contracts, per-customer name resolution, dependency panels, validation) is a separate concern, specified by the dashboard-authoring skill in comet monitoring tooling; this skill covers only what to emit.
|
||||
|
||||
Patterns applied in this implementation (online scoring is the worked example):
|
||||
- **Producer metrics** — every stage that emits work is counted at the source: sampler decisions (`sampler_decisions_total{decision}`) and enqueue-to-Redis (`enqueue_total{result}`). A producer that emits nothing makes downstream starvation explainable rather than mysterious.
|
||||
- **Consumer metrics** — throughput and per-stage timing on the side that drains the queue (`processing_time`, plus per-Redis-op `read/claim/ack_and_remove/list_pending_time`).
|
||||
- **Entrypoint RED** — the workflow's front door (the HTTP ingest route) is measured Rate / Errors / Duration from `http_server_request_duration_seconds`, with 5xx broken down by endpoint × `error_type` × workspace, so an ingestion problem is never mistaken for a scoring problem.
|
||||
- **Errors** — a dedicated error counter per stage, dimensioned by `error_type` (the exception class) and, for shared async plumbing, by component (listener/subscriber): `processing_errors_total`, `unexpected_errors_total`, and `enqueue_total{result="error"}` (a push failure = real loss).
|
||||
- **Success** — success is derived, never double-counted: throughput − errors, surfaced as one "success rate" headline tile.
|
||||
- **Queue time & end-to-end latency** — `queue_delay` (enqueue→pickup) is kept separate from `processing_time` (scorer/LLM work) so a backlog is distinguishable from a slow scorer; end-to-end = queue_delay + processing_time.
|
||||
- **Backpressure** — poll-tick skips are counted but are benign (consumer busy), never lost work.
|
||||
- **Saturation & resource levels** — gauges for in-flight work (max per pod) and JVM heap used-vs-limit per pod expose the pipeline approaching a ceiling *before* it starts failing (the USE method — utilization / saturation / errors — complementing RED).
|
||||
- **Volume & payload size** — byte/char counters (bandwidth, total bytes) and payload-size distributions, broken down by content type and workspace, for cost and impact attribution.
|
||||
- **Per-workspace dimensioning** — the customer drill (§1.3) is a first-class label, not an afterthought.
|
||||
- **Infrastructure dependencies** — the datastores the flow leans on (Redis streams, ClickHouse, locks, MySQL) are surfaced on the dashboard from their exporters and `system.query_log`, so a "slow pipeline" resolves to the dependency causing it.
|
||||
|
||||
To see these conventions already in practice, grep `apps/opik-backend` for the existing metric families rather than specific classes (metric names are the stable contract; class locations move): the online-scoring `*_sampler_decisions_total`, `*_enqueue_total{result}`, `*_processing_time_milliseconds`, `*_queue_delay_milliseconds`, `*_processing_errors_total{error_type}` / `*_unexpected_errors_total`, the per-Redis-op `*_{read,claim,ack_and_remove,list_pending}_time_milliseconds`, and the attachment upload byte/size families.
|
||||
|
||||
Apply alongside the `opik-backend` skill (general conventions, logging rule).
|
||||
|
||||
---
|
||||
|
||||
## 1. Model
|
||||
|
||||
1.1 Decompose the workflow into ordered stages, each defined by: input, output, failure mode, existing instrumentation.
|
||||
|
||||
1.2 Choose the instrument by the question it answers. Most stages need a **counter** (did it happen, and how often?) and a **histogram** (how long did it take?); add a **gauge** only for a level you cannot derive from those.
|
||||
|
||||
- **Counter** — a monotonically increasing total, read as a rate. Use for **events**: throughput, decisions, results, errors, and cumulative volume (messages, bytes). Anything you phrase "per second" or "how many since start". Surfaces with a `_total` suffix. MUST NOT be used for a value that can decrease.
|
||||
- **Native histogram** — a latency/size distribution, read with `histogram_quantile`. Use whenever a **p95/p99 matters, not just an average**: processing time, queue delay, per-dependency op time, end-to-end latency, and payload size (bytes/chars). Prefer native (no explicit buckets, §2.1) over classic `le`-bucketed histograms; reach for classic buckets only when a downstream consumer (e.g. an existing exporter) forces them. Do NOT use a histogram where a counter suffices — a plain success/error tally needs no distribution.
|
||||
- **Gauge** — an instantaneous level, read as-is. Use for a quantity that **rises and falls and can't be reconstructed from a counter**: current queue depth / backlog, in-flight / in-progress count (saturation), batch/read/claim size, lock waiters, heap used. If you actually want a trend, count events with a counter rather than sampling a level with a gauge.
|
||||
- **UpDownCounter** (OTel) — when the level is naturally maintained as signed +1/−1 deltas (in-flight = increment on start, decrement on finish) rather than sampled. Exports like a gauge but is cheaper and less racy than reading a size on every observation.
|
||||
|
||||
Rule of thumb: "how many happened" → counter; "how long / how big" → histogram; "how much is there right now" → gauge (or UpDownCounter). Cover each stage with **RED** (Rate, Errors, Duration) for the work flowing through it and **USE** (Utilization, Saturation, Errors) for the resource it runs on.
|
||||
|
||||
1.3 Define identity dimensions, each bounded in cardinality:
|
||||
- `workspace_id` **and** `workspace_name` — the customer drill. Always paired; `workspace_name` falls back to `workspace_id` when the name is absent (§2.3).
|
||||
- a **stage/type** label — what kind of work this is (`evaluator_type`, `decision`, the Redis op, content/mime type). Lets the dashboard `sum by(...)` per stage (§2.2).
|
||||
- `result` ∈ {`success`, `error`} on outcome counters.
|
||||
- `error_type` on every error counter — the exception class / failure category — plus a component label (listener/subscriber/endpoint) where one counter serves many call sites, so the outcome panel breaks errors down by both cause and origin.
|
||||
|
||||
Cardinality MUST stay bounded by `#workspaces × #types × #error_types`; an unbounded value (trace id, user input, raw message, full URL) MUST NOT be placed on a label.
|
||||
|
||||
---
|
||||
|
||||
## 2. Backend metrics (OpenTelemetry)
|
||||
|
||||
2.1 Meters MUST be created via the OTel API (`GlobalOpenTelemetry.getMeter(namespace)`), one namespace per workflow:
|
||||
```java
|
||||
private static final String METRIC_NAMESPACE = "<workflow>";
|
||||
var meter = GlobalOpenTelemetry.getMeter(METRIC_NAMESPACE);
|
||||
meter.counterBuilder("%s_<stage>_total".formatted(METRIC_NAMESPACE)).setDescription("…").build();
|
||||
meter.histogramBuilder("%s_<stage>_time".formatted(METRIC_NAMESPACE)).setUnit("ms").ofLongs().build(); // native histogram
|
||||
meter.gaugeBuilder("%s_<stage>_size".formatted(METRIC_NAMESPACE)).build();
|
||||
meter.upDownCounterBuilder("%s_<stage>_in_flight".formatted(METRIC_NAMESPACE)).build(); // signed level (inc/dec)
|
||||
```
|
||||
Counters surface in Prometheus with a `_total` suffix. Native histograms MUST NOT define explicit buckets (no `_bucket`/`_sum`/`_count`/`le`).
|
||||
|
||||
2.2 The stage/type dimension MUST be a **label**, not part of the metric name (e.g. `online_scoring_enqueue_total{evaluator_type, result}`), so the dashboard can `sum by(evaluator_type)(rate(...))`. Name-encoding the dimension is permitted ONLY to extend an existing metric family; it complicates dashboard aggregation (a name-encoded dimension cannot be `rate()`d across names in one selector), so prefer a label.
|
||||
|
||||
2.3 `workspace_id` and `workspace_name` MUST be read from the reactive request context (the workspace-id / workspace-name context keys), reusing the shared workspace attribute-key constants rather than redeclaring them per call site:
|
||||
```java
|
||||
var workspaceId = ctx.getOrDefault(WORKSPACE_ID, "");
|
||||
var workspaceName = StringUtils.defaultIfBlank(ctx.getOrDefault(WORKSPACE_NAME, workspaceId), workspaceId);
|
||||
counter.add(1, Attributes.of(TYPE_KEY, type, WORKSPACE_ID_KEY, workspaceId, WORKSPACE_NAME_KEY, workspaceName, RESULT_KEY, "success"));
|
||||
```
|
||||
- `workspace_name` MUST fall back to `workspace_id` when the name is absent.
|
||||
- A name-service lookup MUST NOT be used to resolve the name on a path where the reactive context already carries it.
|
||||
- If an event does not yet carry the name, add a nullable `workspaceName` field and populate it from the workspace-name context key at the publish site (as the entity-created events feeding this workflow already do).
|
||||
|
||||
2.4 The instrumented operation MUST be reactive and read the context with `deferContextual`:
|
||||
```java
|
||||
public Mono<Void> enqueue(List<?> messages, Type type) {
|
||||
return Flux.deferContextual(ctx -> { /* resolve workspace per 2.3 */
|
||||
return Flux.fromIterable(messages).flatMap(m -> redisAdd(m)
|
||||
.doOnNext(id -> counter.add(1, successAttrs))
|
||||
.doOnError(e -> { counter.add(1, errorAttrs); log.error("Error … id='{}'", id, e); }));
|
||||
}).then().subscribeOn(Schedulers.boundedElastic());
|
||||
}
|
||||
```
|
||||
- The method MUST return `Mono<Void>` and MUST NOT self-subscribe. Callers MUST subscribe or compose it.
|
||||
- Request-scoped reactive callers MUST compose it (`.then(...)` / `flatMap`) so it inherits the workspace context.
|
||||
- Event-driven / synchronous callers MUST fire-and-forget via a single shared helper that subscribes with an explicit `.contextWrite(ctx -> ctx.put(WORKSPACE_ID, id).put(WORKSPACE_NAME, defaultIfBlank(name, id)))` and an error-logging consumer (one helper, not duplicated per caller).
|
||||
- Blocking lookups (JDBC/`findById`) MUST run via `Mono.fromCallable(...).subscribeOn(Schedulers.boundedElastic())`. The enqueue/IO work MUST run on a bounded scheduler, not the caller's (e.g. EventBus) thread. Where a caller already holds a resolved object, provide an overload that skips the lookup.
|
||||
|
||||
2.5 Log statements MUST single-quote placeholders (`evaluator='{}' workspaceId='{}'`) per `.agents/skills/opik-backend/SKILL.md`, and SHOULD include the batch size on enqueue logs.
|
||||
|
||||
2.6 `mvn -o compile` MUST succeed (spotless clean) before delivery.
|
||||
|
||||
---
|
||||
|
||||
## 3. Tests
|
||||
|
||||
3.1 Introducing a `Mono<Void>` (lazy) return breaks tests that called the method for its side effect. Restore green by stubbing the reactive method (`lenient().when(pub.enqueue(any(),any())).thenReturn(Mono.empty())`) where production composes it, `.block()`-ing the returned `Mono` in unit tests that assert downstream effects, and replacing `verifyNoInteractions(mock)` with `verify(mock, never()).method(...)` where a lenient stub now exists.
|
||||
|
||||
---
|
||||
|
||||
## 4. Constraints (normative)
|
||||
|
||||
4.1 A returned `Mono` is inert until subscribed; an unsubscribed enqueue is a silent no-op. Every call site MUST compose or subscribe it.
|
||||
|
||||
4.2 The workspace MUST be sourced from the reactive context, not a name-service lookup, where the context carries it (§2.3).
|
||||
|
||||
4.3 IO/enqueue work MUST run on a bounded scheduler, off the caller's thread (§2.4).
|
||||
|
||||
4.4 A backpressure / poll-tick counter (e.g. `backpressure_drops_total`) counts skipped scheduler ticks while a consumer is busy; it is NOT lost work and MUST NOT be alerted on alone. Emit it, but document it as benign for whoever builds the dashboard.
|
||||
|
||||
4.5 Work MUST be done against `origin/main` (create the worktree from it), not a possibly-stale local checkout.
|
||||
|
||||
---
|
||||
|
||||
## 5. Delivery
|
||||
|
||||
5.1 The change is delivered as two PRs on their own branches `<user>/<TASK>-<name>`: this metrics PR (opik repo) and a companion dashboard PR built per the dashboard-authoring skill and delivered to comet monitoring.
|
||||
|
||||
5.2 The metrics PR (this repo) MUST contain no customer, cluster, or infrastructure identifiers, MUST follow `.github/pull_request_template.md` including a `## Documentation` section (the PR linter fails without it), `## Issues` (`Resolves OPIK-XXXX`), and `## AI-WATERMARK: yes` with tools/model/scope/human-verification. Title `[OPIK-XXXX] [BE] …`. Announce per `.agents/commands/comet/send-code-review-slack.md`.
|
||||
|
||||
5.3 The dashboard panel for a new metric stays empty until this backend PR deploys — so the two PRs are independent and can land in either order.
|
||||
@@ -0,0 +1,210 @@
|
||||
---
|
||||
name: opik-backend
|
||||
description: Java backend patterns for Opik. Use when working in apps/opik-backend, designing APIs, database operations, or services.
|
||||
---
|
||||
|
||||
# Opik Backend
|
||||
|
||||
## Architecture
|
||||
- **Layered**: Resource → Service → DAO (never skip layers)
|
||||
- **DI**: Guice modules, constructor injection with `@Inject`
|
||||
- **Databases**: MySQL (metadata, transactional) + ClickHouse (analytics, append-only)
|
||||
|
||||
## Naming Conventions
|
||||
|
||||
### Plural Names (Resources, Tests, URLs, DB Tables)
|
||||
- **Resource classes**: `TracesResource`, `SpansResource`, `DatasetsResource` (not `TraceResource`)
|
||||
- **Resource test classes**: `TracesResourceTest`, `SpansResourceTest`, `DatasetsResourceTest` (not `TraceResourceTest`)
|
||||
- **URL paths**: `/v1/private/traces`, `/v1/private/spans` (not `/v1/private/trace`)
|
||||
- **DB table names**: `traces`, `spans`, `feedback_scores` (not `trace`, `span`, `feedback_score`)
|
||||
|
||||
### Singular Names (DAO, Service)
|
||||
- **DAO classes**: `TraceDAO`, `SpanDAO`, `DatasetDAO` (not `TracesDAO`)
|
||||
- **Service classes**: `TraceService`, `SpanService`, `DatasetService` (not `TracesService`)
|
||||
|
||||
```java
|
||||
// ✅ GOOD
|
||||
@Path("/v1/private/traces")
|
||||
public class TracesResource { }
|
||||
|
||||
// ✅ GOOD - DAO and Service use singular
|
||||
public class TraceDAO { }
|
||||
public class TraceService { }
|
||||
|
||||
// ✅ GOOD - test classes match plural resource name
|
||||
public class TracesResourceTest { }
|
||||
|
||||
// ❌ BAD - singular test class
|
||||
public class TraceResourceTest { }
|
||||
|
||||
// ❌ BAD - singular resource/URL
|
||||
@Path("/v1/private/trace")
|
||||
public class TraceResource { }
|
||||
|
||||
// ❌ BAD - plural DAO/Service
|
||||
public class TracesDAO { }
|
||||
public class TracesService { }
|
||||
```
|
||||
|
||||
## Lombok Conventions
|
||||
|
||||
### Records and DTOs
|
||||
- Always annotate records/DTOs with `@Builder(toBuilder = true)`
|
||||
- Use builders (not constructors) when instantiating records
|
||||
- For **internal records** (built programmatically, never validated by Bean Validation), use Lombok `@NonNull` on required fields — it generates a runtime null check at construction
|
||||
- For **request-body DTOs** validated via `@Valid` cascade (Jakarta validators like `@NotNull`/`@NotBlank`/`@Size`), use Jakarta annotations only — do **not** stack `@NonNull` on top. Bean Validation already enforces the contract at the API boundary; doubling up is redundant noise
|
||||
|
||||
```java
|
||||
// ✅ GOOD - internal record, Lombok @NonNull
|
||||
@Builder(toBuilder = true)
|
||||
record MyData(@NonNull UUID id, @NonNull String name, String description) {}
|
||||
|
||||
MyData data = MyData.builder()
|
||||
.id(id)
|
||||
.name(name)
|
||||
.build();
|
||||
|
||||
// ✅ GOOD - request-body DTO, Jakarta validators only
|
||||
@Builder(toBuilder = true)
|
||||
public record MyRequest(
|
||||
@NotNull UUID id,
|
||||
@NotBlank String name,
|
||||
@NotNull @Size(min = 1, max = 1000) @Valid List<MyItem> items) {}
|
||||
|
||||
// ❌ BAD - plain constructor (positional mistakes, less readable)
|
||||
new MyData(id, name, null);
|
||||
|
||||
// ❌ BAD - @Builder without toBuilder
|
||||
@Builder
|
||||
record MyData(UUID id, String name) {}
|
||||
|
||||
// ❌ BAD - stacking @NonNull and @NotNull on the same field
|
||||
public record MyRequest(@NonNull @NotNull UUID id) {}
|
||||
```
|
||||
|
||||
### Dependency Injection
|
||||
- Use `@RequiredArgsConstructor(onConstructor_ = @Inject)` instead of manual constructors
|
||||
|
||||
```java
|
||||
// ✅ GOOD
|
||||
@RequiredArgsConstructor(onConstructor_ = @Inject)
|
||||
public class MyService {
|
||||
private final @NonNull DependencyA depA;
|
||||
private final @NonNull DependencyB depB;
|
||||
}
|
||||
|
||||
// ❌ BAD - boilerplate constructor
|
||||
public class MyService {
|
||||
private final DependencyA depA;
|
||||
@Inject
|
||||
public MyService(DependencyA depA) {
|
||||
this.depA = depA;
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Interfaces
|
||||
- Don't put validation annotations (`@NonNull`) on interface method parameters
|
||||
- Keep interfaces free of implementation details
|
||||
|
||||
```java
|
||||
// ✅ GOOD
|
||||
interface MyService {
|
||||
void process(String workspaceId, UUID promptId);
|
||||
}
|
||||
|
||||
// ❌ BAD - validation on interface
|
||||
interface MyService {
|
||||
void process(@NonNull String workspaceId, @NonNull UUID promptId);
|
||||
}
|
||||
```
|
||||
|
||||
## Critical Gotchas
|
||||
|
||||
### StringTemplate Memory Leak
|
||||
```java
|
||||
// ✅ GOOD
|
||||
var template = TemplateUtils.newST(QUERY);
|
||||
|
||||
// ❌ BAD - causes memory leak via STGroup singleton
|
||||
var template = new ST(QUERY);
|
||||
```
|
||||
|
||||
### List Access
|
||||
```java
|
||||
// ✅ GOOD
|
||||
users.getFirst()
|
||||
users.getLast()
|
||||
|
||||
// ❌ BAD
|
||||
users.get(0)
|
||||
users.get(users.size() - 1)
|
||||
```
|
||||
|
||||
### SQL Text Blocks
|
||||
```java
|
||||
// ✅ GOOD - text blocks for multi-line SQL
|
||||
@SqlQuery("""
|
||||
SELECT * FROM datasets
|
||||
WHERE workspace_id = :workspace_id
|
||||
<if(name)> AND name like concat('%', :name, '%') <endif>
|
||||
""")
|
||||
|
||||
// ❌ BAD - string concatenation
|
||||
@SqlQuery("SELECT * FROM datasets " +
|
||||
"WHERE workspace_id = :workspace_id " +
|
||||
"<if(name)> AND name like concat('%', :name, '%') <endif> ")
|
||||
```
|
||||
|
||||
### Immutable Collections
|
||||
```java
|
||||
// ✅ GOOD
|
||||
Set.of("A", "B", "C")
|
||||
List.of(1, 2, 3)
|
||||
Map.of("key", "value")
|
||||
|
||||
// ❌ BAD
|
||||
Arrays.asList("A", "B", "C")
|
||||
```
|
||||
|
||||
## API Design
|
||||
- **Query parameters that accept lists**: Use plural names from the start (e.g., `exclude_category_names` not `exclude_category_name`). Starting with a singular name and later adding a plural variant results in two redundant query params on the same endpoint. Plural names are backward-compatible since they work for both single and multiple values.
|
||||
|
||||
## Error Handling
|
||||
|
||||
### Use Jakarta Exceptions
|
||||
```java
|
||||
throw new BadRequestException("Invalid input");
|
||||
throw new NotFoundException("User not found: '%s'".formatted(id));
|
||||
throw new ConflictException("Already exists");
|
||||
throw new InternalServerErrorException("System error", cause);
|
||||
```
|
||||
|
||||
### Error Response Classes
|
||||
- Simple: `io.dropwizard.jersey.errors.ErrorMessage`
|
||||
- Complex: `com.comet.opik.api.error.ErrorMessage`
|
||||
- **Never create new error message classes**
|
||||
|
||||
## Logging
|
||||
|
||||
### Format Convention
|
||||
```java
|
||||
// ✅ GOOD - values in single quotes
|
||||
log.info("Created user: '{}'", userId);
|
||||
log.error("Failed for workspace: '{}'", workspaceId, exception);
|
||||
|
||||
// ❌ BAD - no quotes
|
||||
log.info("Created user: {}", userId);
|
||||
```
|
||||
|
||||
### Never Log
|
||||
- Emails, passwords, tokens, API keys
|
||||
- PII, personal identifiers
|
||||
- Database credentials
|
||||
|
||||
## Reference Files
|
||||
- [clickhouse.md](clickhouse.md) - ClickHouse query patterns
|
||||
- [mysql.md](mysql.md) - TransactionTemplate patterns
|
||||
- [testing.md](testing.md) - PODAM, naming, assertion patterns
|
||||
- [migrations.md](migrations.md) - Liquibase format for MySQL/ClickHouse
|
||||
- [permissions.md](permissions.md) - `@RequiredPermissions` annotation guidance for endpoints
|
||||
@@ -0,0 +1,177 @@
|
||||
# ClickHouse Patterns
|
||||
|
||||
## Schema First
|
||||
- **Never invent schema** - check migrations: `src/main/resources/liquibase/db-app-analytics/migrations/`
|
||||
- Find similar queries before writing new ones
|
||||
|
||||
## Deduplication (ReplacingMergeTree)
|
||||
|
||||
Most tables need deduplication. **Exception**: audit/log tables (ReplicatedMergeTree) keep full history.
|
||||
|
||||
Updates are modeled as inserts — multiple row versions coexist until background merge.
|
||||
|
||||
```sql
|
||||
-- Pattern 1: LIMIT 1 BY (reads only referenced columns, requires explicit sort)
|
||||
SELECT * FROM traces
|
||||
WHERE workspace_id = :workspace_id
|
||||
ORDER BY (workspace_id, project_id, id) DESC, last_updated_at DESC
|
||||
LIMIT 1 BY id
|
||||
|
||||
-- Pattern 2: FINAL (query-time merge, streams through sorted parts)
|
||||
SELECT * FROM traces FINAL
|
||||
WHERE workspace_id = :workspace_id
|
||||
|
||||
-- ❌ BAD - missing deduplication, returns duplicates
|
||||
SELECT * FROM traces WHERE workspace_id = :workspace_id
|
||||
```
|
||||
|
||||
### FINAL vs LIMIT 1 BY
|
||||
|
||||
There's no general rule for which is faster. It depends on the query and table state:
|
||||
|
||||
- **LIMIT 1 BY** tends to win when selecting a small subset of columns (columnar advantage)
|
||||
or on tables with many unmerged updates (FINAL must merge all versions).
|
||||
- **FINAL** tends to win on well-merged tables (streams sorted parts without re-sorting)
|
||||
or on large result sets (LIMIT 1 BY must explicitly sort all matching rows).
|
||||
|
||||
### Mutable column filtering with LIMIT 1 BY
|
||||
|
||||
With `LIMIT 1 BY`, filters on mutable columns (`status`, `name`, `tags`, etc.) **MUST go
|
||||
AFTER dedup** in an outer query. Filtering before dedup can return stale/phantom rows.
|
||||
|
||||
```sql
|
||||
-- ✅ CORRECT -- ❌ WRONG
|
||||
SELECT * FROM ( SELECT * FROM traces
|
||||
SELECT * FROM traces WHERE status = :status -- before dedup!
|
||||
WHERE workspace_id = :wid ORDER BY (...) DESC, last_updated_at DESC
|
||||
ORDER BY (...) DESC LIMIT 1 BY id
|
||||
LIMIT 1 BY id
|
||||
) WHERE status = :status
|
||||
```
|
||||
|
||||
Safe to filter before dedup: immutable columns (`workspace_id`, `project_id`, `id`, `created_at`)
|
||||
and monotonically-changing columns (`last_updated_at` — only increases, so a lower-bound cutoff
|
||||
can't exclude the latest version while keeping an older one).
|
||||
With `FINAL`, this doesn't apply — dedup happens before WHERE.
|
||||
|
||||
## Skip Indexes in ReplacingMergeTree
|
||||
|
||||
**Never index fields that flip back and forth** (e.g., `status`). The index sees old row
|
||||
versions and can't filter reliably before dedup.
|
||||
|
||||
**Monotonically-changing fields are safe** (e.g., `last_updated_at` only increases). Use
|
||||
`minmax GRANULARITY 1` — ClickHouse skips granules entirely outside the filter range.
|
||||
|
||||
Add a migration comment explaining why the indexed field is safe.
|
||||
|
||||
### Skip indexes + FINAL
|
||||
|
||||
Skip indexes are **ignored** with `FINAL` by default (up to CH 25.3). Enable with:
|
||||
```sql
|
||||
SETTINGS use_skip_indexes_if_final=1
|
||||
```
|
||||
|
||||
### Time-bounded FINAL
|
||||
|
||||
Avoid bare `FINAL` on large tables. Scope it with a monotonic field + minmax index:
|
||||
|
||||
```sql
|
||||
-- Migration (comment why it's safe):
|
||||
ALTER TABLE t ADD INDEX idx_last_updated_at last_updated_at TYPE minmax GRANULARITY 1;
|
||||
ALTER TABLE t MATERIALIZE INDEX idx_last_updated_at;
|
||||
|
||||
-- Query: FINAL only considers recent granules
|
||||
SELECT * FROM (
|
||||
SELECT * FROM t FINAL WHERE last_updated_at > now() - INTERVAL 1 DAY
|
||||
) WHERE status = 'active'
|
||||
SETTINGS use_skip_indexes_if_final=1
|
||||
```
|
||||
|
||||
## StringTemplate Gotchas
|
||||
|
||||
```sql
|
||||
-- Escape < operator
|
||||
WHERE id \\<= :uuid_to_time
|
||||
|
||||
-- Conditional blocks
|
||||
<if(project_id)> AND project_id = :project_id <endif>
|
||||
```
|
||||
|
||||
## Parameter Binding
|
||||
|
||||
```java
|
||||
// ✅ GOOD - snake_case
|
||||
.bind("workspace_id", workspaceId)
|
||||
.bind("project_id", projectId.toString())
|
||||
|
||||
// ❌ BAD - camelCase
|
||||
.bind("workspaceId", workspaceId)
|
||||
```
|
||||
|
||||
## `FORMAT Values` cells must be literals (OPIK-5694)
|
||||
|
||||
Every cell in an `INSERT ... FORMAT Values` per-row tuple must be a plain `:placeholder`
|
||||
bound to a value the driver serialises as a literal. Function calls, `if(...)`, `::`
|
||||
casts, `parseDateTime64BestEffort(...)`, `now64(...)`, `mapFromArrays(...)`,
|
||||
`toDecimal128(...)` etc. trip the fast-path parser. The insert still succeeds, but every
|
||||
row silently bumps `system.errors` codes 26 / 27 / 43 / 70 and writes to pod stderr —
|
||||
flooding logs.
|
||||
|
||||
Bind the right shape:
|
||||
- `DateTime64(P)` → `String` formatted via `ClickHouseDateTimeFormat.formatNanos/formatMicros`
|
||||
(not `Instant.toString()` — its `T`/`Z` form trips the fast-path).
|
||||
- `Map(K,V)` → bind a Java `Map`.
|
||||
- `Decimal128(S)` → bind a `BigDecimal` (`.toString()` would emit a quoted string which
|
||||
Decimal cells also reject).
|
||||
- Non-nullable cell, value null → substitute the column DEFAULT in Java (e.g.
|
||||
`Instant.now()` for `now64()`, `'default'` for an Enum8 default).
|
||||
|
||||
## Numeric Gotchas
|
||||
|
||||
### Handle NaN/Infinity
|
||||
```sql
|
||||
toDecimal64(
|
||||
greatest(least(if(isFinite(v), v, 0), 999999999.999999999), -999999999.999999999),
|
||||
9
|
||||
)
|
||||
```
|
||||
|
||||
### Correct Decimal Precision
|
||||
- Feedback scores: `Decimal64(9)`
|
||||
- Cost fields: `Decimal(38, 12)` → use `toDecimal128(..., 12)`
|
||||
|
||||
## Sorting / Pagination / Field Exclusion (two-phase + deferred wide columns)
|
||||
|
||||
Large `SELECT ... BY project` queries (traces/spans) use a two-phase shape: a light
|
||||
`page_ids` CTE paginates on id + sort key only, then `page_wide` re-reads the full rows
|
||||
(including wide text columns: `input`/`output`/`metadata`) for just that page. Wide columns
|
||||
are deferred (dropped via `EXCEPT`) unless needed.
|
||||
|
||||
Invariants to preserve when editing these queries:
|
||||
- **The pagination pre-filter (`page_ids`) must carry the sort key.** Render `<sort_fields>`
|
||||
into `page_ids`' `ORDER BY` (and the final `ORDER BY`), or pagination returns the wrong page
|
||||
for custom sorting. `page_wide` is id-bounded + `LIMIT 1 BY id`, so its own order is
|
||||
immaterial — the final `SELECT` re-sorts.
|
||||
- **Keep the sort column available.** When sorting by a wide column, `sort_needs_wide` must keep
|
||||
`input`/`output`/`metadata` in the deduped CTE; when excluding a field that is also the sort
|
||||
key, the column must still be selectable for the `ORDER BY`. Excluding the sort field must not
|
||||
drop or break the sort.
|
||||
- **Spans and traces share this shape** — change both together.
|
||||
|
||||
⚠️ Any change here MUST be covered by full-page-content tests (not id-only) across sortable
|
||||
fields, custom/dynamic `sort_fields`, and the sort × field-exclusion combination, for BOTH
|
||||
spans and traces. See `testing.md` → "Sorting / Pagination / Field-Exclusion SQL Changes".
|
||||
|
||||
## Performance
|
||||
|
||||
- **Prefer subqueries over JOINs** for filtering
|
||||
- Filter early in CTEs — but only on immutable columns (see above)
|
||||
- Use `IN (subquery)` pattern
|
||||
- Batch inserts (1000+ rows)
|
||||
- Use `LEFT ANY JOIN` when the right table has at most one match per key
|
||||
|
||||
## Query Logging
|
||||
Always add log comment:
|
||||
```sql
|
||||
SETTINGS log_comment = '<log_comment>'
|
||||
```
|
||||
@@ -0,0 +1,125 @@
|
||||
# Database Migration Patterns
|
||||
|
||||
## Migration Locations
|
||||
- **MySQL**: `apps/opik-backend/src/main/resources/liquibase/db-app-state/migrations/`
|
||||
- **ClickHouse**: `apps/opik-backend/src/main/resources/liquibase/db-app-analytics/migrations/`
|
||||
|
||||
## Table Naming
|
||||
|
||||
Always use **plural** names for database tables: `traces`, `spans`, `feedback_scores`, `datasets`, `experiments` (not `trace`, `span`, `feedback_score`).
|
||||
|
||||
## Liquibase Format
|
||||
|
||||
```sql
|
||||
--liquibase formatted sql
|
||||
--changeset author:000001_description
|
||||
--comment: Brief description of the migration
|
||||
|
||||
-- Your SQL here
|
||||
|
||||
```
|
||||
**Always end with empty line.**
|
||||
|
||||
## MySQL Migration Example
|
||||
|
||||
```sql
|
||||
--liquibase formatted sql
|
||||
--changeset john.doe:000001_add_user_table
|
||||
--comment: Create users table with authentication fields
|
||||
|
||||
CREATE TABLE users (
|
||||
id VARCHAR(36) PRIMARY KEY,
|
||||
name VARCHAR(255) NOT NULL,
|
||||
email VARCHAR(255) NOT NULL,
|
||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP
|
||||
);
|
||||
|
||||
-- Index for email lookups (frequent login queries)
|
||||
CREATE INDEX idx_users_email ON users(email);
|
||||
|
||||
```
|
||||
|
||||
## ClickHouse Migration Example
|
||||
|
||||
```sql
|
||||
--liquibase formatted sql
|
||||
--changeset john.doe:000001_add_analytics_table
|
||||
--comment: Create analytics events table
|
||||
|
||||
CREATE TABLE IF NOT EXISTS analytics_events ON CLUSTER '{cluster}' (
|
||||
id FixedString(36),
|
||||
workspace_id String,
|
||||
event_type Enum8('unknown' = 0, 'view' = 1, 'click' = 2),
|
||||
created_at DateTime64(9, 'UTC') DEFAULT now64(9),
|
||||
last_updated_at DateTime64(6, 'UTC') DEFAULT now64(6)
|
||||
)
|
||||
ENGINE = ReplicatedReplacingMergeTree(
|
||||
'/clickhouse/tables/{shard}/${ANALYTICS_DB_DATABASE_NAME}/analytics_events',
|
||||
'{replica}',
|
||||
last_updated_at
|
||||
)
|
||||
ORDER BY (workspace_id, event_type, id);
|
||||
|
||||
```
|
||||
|
||||
## Rollback: Additive vs In-Place Changes
|
||||
|
||||
The kind of change decides whether a rollback is allowed:
|
||||
|
||||
- **Additive / structural changes → provide a rollback** that undoes them (new table → `DROP TABLE`, new column → `DROP COLUMN`, new index → `DROP INDEX`). Dropping what was just added is safe.
|
||||
- **In-place changes to an existing column → rollback MUST be empty (`--rollback empty`)**: enum modify/rename, type or precision changes, default/codec changes. Reverting them (1) often reintroduces the bug the migration fixed, and (2) is lossy or fails once new data is written under the new definition — they are effectively irreversible.
|
||||
|
||||
```sql
|
||||
-- ✅ Additive change → real rollback
|
||||
ALTER TABLE events ON CLUSTER '{cluster}' ADD COLUMN IF NOT EXISTS source String DEFAULT '';
|
||||
--rollback ALTER TABLE events ON CLUSTER '{cluster}' DROP COLUMN IF EXISTS source;
|
||||
|
||||
-- ✅ In-place enum change → empty rollback
|
||||
ALTER TABLE events ON CLUSTER '{cluster}' MODIFY COLUMN level Enum8('TRACE' = 0, 'INFO' = 2, 'WARN' = 3, 'ERROR' = 4);
|
||||
--rollback empty
|
||||
|
||||
-- ❌ In-place enum change → reverting reintroduces the old (buggy) label and breaks rows already written
|
||||
--rollback ALTER TABLE events ON CLUSTER '{cluster}' MODIFY COLUMN level Enum8('TRACE' = 0, 'INFO' = 2, 'WARM' = 3, 'ERROR' = 4);
|
||||
```
|
||||
|
||||
Always declare an intentionally empty rollback explicitly as `--rollback empty` (see `000029_add_thread_enum_value_to_feedback_scores.sql`) — never just omit it.
|
||||
|
||||
## Rollback Dependency Order
|
||||
|
||||
Rollback statements run in the order they appear. When columns have dependencies (e.g. a `MATERIALIZED` column that references another column), **drop dependent columns first** — the reverse of creation order.
|
||||
|
||||
```sql
|
||||
-- Forward: create base column, then materialized column
|
||||
ALTER TABLE t ADD COLUMN IF NOT EXISTS description String DEFAULT '';
|
||||
ALTER TABLE t ADD COLUMN IF NOT EXISTS description_hash UInt64 MATERIALIZED xxHash64(description);
|
||||
|
||||
-- ❌ BAD - drops base column before its dependent materialized column
|
||||
--rollback ALTER TABLE t DROP COLUMN IF EXISTS description;
|
||||
--rollback ALTER TABLE t DROP COLUMN IF EXISTS description_hash;
|
||||
|
||||
-- ✅ GOOD - drops materialized (dependent) column first, then base column
|
||||
--rollback ALTER TABLE t DROP COLUMN IF EXISTS description_hash;
|
||||
--rollback ALTER TABLE t DROP COLUMN IF EXISTS description;
|
||||
```
|
||||
|
||||
This applies to any dependency chain: `MATERIALIZED`, `ALIAS`, indexes referencing columns, etc.
|
||||
|
||||
## ClickHouse Gotchas
|
||||
|
||||
- **Always use** `ON CLUSTER '{cluster}'` for distributed operations
|
||||
- **Engine**: Use `ReplicatedReplacingMergeTree` for deduplication, `ReplicatedMergeTree` for audit/logs
|
||||
- **ORDER BY**: Include workspace_id first, then logical groupings
|
||||
|
||||
## Index Comments
|
||||
|
||||
Always explain why an index exists:
|
||||
|
||||
```sql
|
||||
-- ❌ BAD - No explanation
|
||||
CREATE INDEX idx_users_created_at ON users(created_at);
|
||||
|
||||
-- ✅ GOOD - Explains purpose
|
||||
-- Index for user registration analytics (used in monthly reports)
|
||||
CREATE INDEX idx_users_created_at ON users(created_at);
|
||||
```
|
||||
@@ -0,0 +1,72 @@
|
||||
# MySQL Transaction Patterns
|
||||
|
||||
## Always Use TransactionTemplate
|
||||
|
||||
```java
|
||||
import static com.comet.opik.infrastructure.db.TransactionTemplateAsync.READ_ONLY;
|
||||
import static com.comet.opik.infrastructure.db.TransactionTemplateAsync.WRITE;
|
||||
|
||||
@Singleton
|
||||
@RequiredArgsConstructor(onConstructor_ = @Inject)
|
||||
public class UserService {
|
||||
|
||||
private final @NonNull TransactionTemplate transactionTemplate;
|
||||
|
||||
// Read operations
|
||||
public UserResponse getUser(String id) {
|
||||
return transactionTemplate.inTransaction(READ_ONLY, handle -> {
|
||||
var repository = handle.attach(UserDao.class);
|
||||
return repository.findById(id)
|
||||
.orElseThrow(() -> new NotFoundException("User not found: '%s'".formatted(id)));
|
||||
});
|
||||
}
|
||||
|
||||
// Write operations
|
||||
public UserResponse createUser(UserCreateRequest request) {
|
||||
return transactionTemplate.inTransaction(WRITE, handle -> {
|
||||
var repository = handle.attach(UserDao.class);
|
||||
var user = buildUser(request);
|
||||
return repository.create(user);
|
||||
});
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Transaction Block Best Practices
|
||||
|
||||
```java
|
||||
// ✅ GOOD - Only database operations inside transaction
|
||||
public UserResponse createUser(UserCreateRequest request) {
|
||||
return transactionTemplate.inTransaction(WRITE, handle -> {
|
||||
var repository = handle.attach(UserDao.class);
|
||||
var user = buildUser(request);
|
||||
return repository.create(user);
|
||||
});
|
||||
}
|
||||
|
||||
// ❌ BAD - Unrelated logic in transaction block
|
||||
public UserResponse createUser(UserCreateRequest request) {
|
||||
return transactionTemplate.inTransaction(WRITE, handle -> {
|
||||
sendEmailNotification(request.getEmail()); // Don't!
|
||||
updateCache(); // Don't!
|
||||
|
||||
var repository = handle.attach(UserDao.class);
|
||||
return repository.create(buildUser(request));
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
## DAO Interface Pattern
|
||||
|
||||
```java
|
||||
@RegisterRowMapper(UserRowMapper.class)
|
||||
public interface UserDao {
|
||||
|
||||
@SqlQuery("SELECT * FROM users WHERE id = :id")
|
||||
Optional<User> findById(@Bind("id") String id);
|
||||
|
||||
@SqlUpdate("INSERT INTO users (id, name, email, created_at) VALUES (:id, :name, :email, :createdAt)")
|
||||
@GetGeneratedKeys
|
||||
User create(@BindBean User user);
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,42 @@
|
||||
# Endpoint Permissions
|
||||
|
||||
## When Adding or Modifying Resource Endpoints
|
||||
|
||||
Every new JAX-RS endpoint method in `apps/opik-backend/src/main/java/com/comet/opik/api/resources/v1/priv/` should be evaluated for a `@RequiredPermissions` annotation.
|
||||
|
||||
### Check Process
|
||||
|
||||
1. **Read the current permissions enum** at `apps/opik-backend/src/main/java/com/comet/opik/infrastructure/auth/WorkspaceUserPermission.java`.
|
||||
|
||||
2. **Check if the endpoint's resource already uses `@RequiredPermissions`** on other methods. If sibling methods have permissions, the new endpoint likely needs one too.
|
||||
|
||||
3. **Logically match the endpoint's operation to a permission**. Do not rely on naming patterns — reason about what the endpoint does and which permission governs that action:
|
||||
- A read/list/get endpoint likely maps to a **view** permission for that domain entity
|
||||
- A create/write/log endpoint likely maps to a **create**, **write**, or **log** permission
|
||||
- An update/edit/modify endpoint likely maps to an **edit** or **update** permission
|
||||
- A delete/remove/purge endpoint likely maps to a **delete** permission
|
||||
- An endpoint may map to a permission from a different entity group if it crosses domain boundaries (e.g., annotating from a queue context uses a trace-level annotation permission)
|
||||
|
||||
4. **If a logically matching permission exists**, do NOT add it automatically. Instead, inform the user which permission you believe matches and why, then ask for confirmation before adding the `@RequiredPermissions` annotation.
|
||||
|
||||
5. **If no matching permission exists**, inform the user:
|
||||
- State which resource/operation has no matching permission
|
||||
- Note that the endpoint will fall back to team-membership authentication
|
||||
- Ask whether a new permission should be added to the enum, or if the fallback is acceptable
|
||||
|
||||
### Example
|
||||
|
||||
```java
|
||||
@GET
|
||||
@Path("/{id}")
|
||||
@RequiredPermissions(WorkspaceUserPermission.DATASET_VIEW)
|
||||
public Response getDatasetById(@PathParam("id") UUID id) { ... }
|
||||
```
|
||||
|
||||
### Current Coverage
|
||||
|
||||
Not all resources have permissions defined yet. The remaining endpoints rely on team-membership authentication. This is expected. Do not add permissions speculatively; only add them when a logically matching `WorkspaceUserPermission` value exists or the user confirms a new one should be created.
|
||||
|
||||
### Reference
|
||||
|
||||
Full permissions spec: https://www.notion.so/cometml/Workspace-permissions-and-user-roles-management-2b77124010a380f8b526e7ecb235c419
|
||||
@@ -0,0 +1,183 @@
|
||||
# Backend Testing Patterns
|
||||
|
||||
## Test Data with PODAM
|
||||
|
||||
```java
|
||||
import com.comet.opik.podam.PodamFactoryUtils;
|
||||
|
||||
private final PodamFactory podamFactory = PodamFactoryUtils.newPodamFactory();
|
||||
|
||||
@Test
|
||||
void createUser() {
|
||||
var request = podamFactory.manufacturePojo(UserCreateRequest.class)
|
||||
.toBuilder()
|
||||
.name("John Doe") // Override only what matters for test
|
||||
.build();
|
||||
|
||||
// ...
|
||||
}
|
||||
```
|
||||
|
||||
**Utility methods:**
|
||||
- `PodamFactoryUtils.manufacturePojoList(factory, Class)` - Generate List
|
||||
- `PodamFactoryUtils.manufacturePojoSet(factory, Class)` - Generate Set
|
||||
|
||||
## Test Naming
|
||||
|
||||
```java
|
||||
// ✅ Happy path - same as method name
|
||||
void createUser() { }
|
||||
|
||||
// ✅ Specific scenarios
|
||||
void createUserWhenValidRequestReturnsUser() { }
|
||||
void createUserWhenUserExistsReturnsConflict() { }
|
||||
|
||||
// ✅ Error paths
|
||||
void createUserWhenInvalidEmailThrowsBadRequestException() { }
|
||||
|
||||
// ❌ Bad
|
||||
void testCreateUser() { }
|
||||
void should_create_user() { }
|
||||
```
|
||||
|
||||
## Sorting Test Anti-Pattern
|
||||
|
||||
```java
|
||||
// ❌ BAD - Self-fulfilling prophecy (always passes!)
|
||||
var actualValues = api.findSorted("name", "ASC");
|
||||
var expectedValues = new ArrayList<>(actualValues);
|
||||
expectedValues.sort(Comparator.naturalOrder());
|
||||
assertThat(actualValues).isEqualTo(expectedValues);
|
||||
|
||||
// ✅ GOOD - Test against known data
|
||||
var page = api.findSorted("name", "ASC");
|
||||
assertThat(page.content())
|
||||
.extracting(Entity::getName)
|
||||
.containsExactly("Alice", "Bob", "Charlie");
|
||||
|
||||
// ✅ GOOD - Use AssertJ sorting assertions
|
||||
assertThat(page.content())
|
||||
.extracting(Entity::getName)
|
||||
.isSorted();
|
||||
|
||||
// ✅ GOOD - Compare against independently sorted original
|
||||
var expectedOrder = originalEntities.stream()
|
||||
.sorted(comparator)
|
||||
.map(Entity::getId)
|
||||
.toList();
|
||||
assertThat(actualOrder).isEqualTo(expectedOrder);
|
||||
```
|
||||
|
||||
## Sorting / Pagination / Field-Exclusion SQL Changes — Coverage Bar
|
||||
|
||||
When you change query SQL that backs **sorting, pagination, or field exclusion** (e.g. the
|
||||
two-phase `page_ids`/`page_wide` CTEs, deferred wide columns, `EXCEPT`/`exclude_fields`,
|
||||
`sort_needs_wide`, dynamic `sort_fields`), the test MUST:
|
||||
|
||||
- **Assert the whole page content, not just IDs.** Reuse the existing full-page assertion
|
||||
helpers (the per-test-class `getAndAssertPage` → `TraceAssertions.assertTraces` /
|
||||
`SpanAssertions.assertSpan`) so every field is verified. ID-only assertions are too weak — they can't catch a row that
|
||||
returns the right id with wrong/empty data.
|
||||
- **Cover custom/dynamic `sort_fields`**, not only static columns — sort by a wide text column
|
||||
(`input`/`output`/`metadata`) AND by a regular column, in both directions.
|
||||
- **Cover the sort × field-exclusion combination.** Sorting by a field while excluding that
|
||||
same field (and while excluding a *different* wide field) is the case that regresses when the
|
||||
deferred-wide-column pre-filter doesn't carry the sort key. Build expected via
|
||||
`EXCLUDE_FUNCTIONS.get(field)` and pass the `exclude` set to `getAndAssertPage`.
|
||||
- **Exercise both spans and traces** — they share the same query shape; a fix on one usually
|
||||
needs the mirror test on the other.
|
||||
|
||||
```java
|
||||
// ✅ GOOD - sort × exclude, full-page assertion (deferred-wide path)
|
||||
var expected = traces.stream().sorted(comparator)
|
||||
.map(t -> TraceAssertions.EXCLUDE_FUNCTIONS.get(excludeField).apply(t))
|
||||
.toList();
|
||||
getAndAssertPage(workspaceName, projectName, null, List.of(), traces, expected, List.of(),
|
||||
apiKey, List.of(sortingField), Set.of(excludeField));
|
||||
```
|
||||
|
||||
## Parameterized Tests
|
||||
|
||||
```java
|
||||
// ❌ BAD - Duplicate methods
|
||||
void testSortByNameAsc() { }
|
||||
void testSortByNameDesc() { }
|
||||
void testSortByTypeAsc() { }
|
||||
|
||||
// ✅ GOOD - Single parameterized test
|
||||
@ParameterizedTest(name = "Sort by {0} {1}")
|
||||
@MethodSource("sortingTestCases")
|
||||
void sortEntities(String field, String direction, Comparator<Entity> comparator) {
|
||||
// Single test handles all scenarios
|
||||
}
|
||||
|
||||
static Stream<Arguments> sortingTestCases() {
|
||||
return Stream.of(
|
||||
Arguments.of("name", "ASC", Comparator.comparing(Entity::getName)),
|
||||
Arguments.of("name", "DESC", Comparator.comparing(Entity::getName).reversed())
|
||||
);
|
||||
}
|
||||
```
|
||||
|
||||
## Awaitility - When to Use
|
||||
|
||||
```java
|
||||
// ❌ BAD - MySQL operations are synchronous
|
||||
Awaitility.await().untilAsserted(() -> {
|
||||
var page = client.findAll();
|
||||
assertThat(page).hasSize(5);
|
||||
});
|
||||
|
||||
// ✅ GOOD - Direct assertion for sync operations
|
||||
var page = client.findAll();
|
||||
assertThat(page).hasSize(5);
|
||||
|
||||
// ✅ GOOD - Awaitility only for truly async (Kafka, background jobs)
|
||||
kafkaProducer.send(message);
|
||||
Awaitility.await()
|
||||
.atMost(5, TimeUnit.SECONDS)
|
||||
.untilAsserted(() -> {
|
||||
var processed = repository.find(message.getId());
|
||||
assertThat(processed).isNotNull();
|
||||
});
|
||||
```
|
||||
|
||||
## Assertion Patterns
|
||||
|
||||
```java
|
||||
// Object equality when you have expected object
|
||||
assertThat(actualUser).isEqualTo(expectedUser);
|
||||
|
||||
// Field assertions for specific checks
|
||||
assertThat(result.getName()).isEqualTo("John Doe");
|
||||
assertThat(result.getId()).isNotBlank();
|
||||
|
||||
// Exception assertions
|
||||
assertThatThrownBy(() -> service.create(invalid))
|
||||
.isInstanceOf(BadRequestException.class)
|
||||
.hasMessageContaining("Name is required");
|
||||
```
|
||||
|
||||
## Don't Run Two ClickHouse-Migrating Test Classes in One `mvn` Reactor
|
||||
|
||||
Each resource test class that touches ClickHouse runs its own Liquibase migration against the
|
||||
Testcontainers instance. Running two such classes in a single `mvn` invocation (e.g. spans + traces
|
||||
together, or a wildcard that matches both) makes the second migration fail with
|
||||
`REPLICA_ALREADY_EXISTS` (the replicated table from migration `000017` already exists) — a confusing
|
||||
failure that looks like a product bug but is purely a test-harness collision.
|
||||
|
||||
When a change spans both spans and traces (the usual case for shared query SQL), run each class in a
|
||||
**separate** `mvn` invocation:
|
||||
|
||||
```bash
|
||||
# ✅ GOOD - separate invocations
|
||||
mvn test -o -Dtest='FindSpansResourceTest$FindSpans#whenFilterSortExcludeAcrossPages*'
|
||||
mvn test -o -Dtest='GetTracesByProjectResourceTest$FindTraces#getTracesByProject__whenFilterSortExcludeAcrossPages*'
|
||||
|
||||
# ❌ BAD - one reactor migrates ClickHouse twice -> REPLICA_ALREADY_EXISTS
|
||||
mvn test -o -Dtest='FindSpansResourceTest,GetTracesByProjectResourceTest'
|
||||
```
|
||||
|
||||
Surefire selectors for `@Nested` + parameterized tests: use `OuterClass$NestedClass#methodPattern`,
|
||||
and prefer a `*wildcard*` over the exact (long) method name — exact long names silently match 0 tests.
|
||||
Combine methods within a class with `+`, classes with `,` (but see the ClickHouse caveat above).
|
||||
@@ -0,0 +1,48 @@
|
||||
---
|
||||
name: opik-external-integrations
|
||||
description: Build or update an Opik integration that lives OUTSIDE this repo — a standalone opik-* package (e.g. opik-openclaw, opik-claude-code-plugin) or Opik support contributed into a third-party project (e.g. LiteLLM, Dify). Use ONLY when the user names an external repository or external package as the target; for integrations under sdks/ use the opik-integrations skill instead.
|
||||
---
|
||||
|
||||
# Opik External Integrations
|
||||
|
||||
This skill builds Opik integrations whose code **does not live in this repository**. Two shapes:
|
||||
|
||||
- **Standalone `opik-*` package / plugin** — its own repo or package that depends on the *published* Opik SDK. Examples: `opik-openclaw`, `opik-claude-code-plugin`.
|
||||
- **Upstream contribution** — Opik support added into a third-party project that already has its own logging/callback/plugin system. Examples: an Opik callback in **LiteLLM**, an Opik integration in **Dify**.
|
||||
|
||||
> **Activation gate — read this first.** Use this skill **only** when the target lives in an external repo or external package. If the user wants an integration that ships inside `sdks/python/src/opik/integrations/` or `sdks/typescript/src/opik/integrations/`, this is the wrong skill — use **`opik-integrations`**. When in doubt, ask the "where does it live" question (below) and route accordingly.
|
||||
|
||||
## Two principles that make external work different
|
||||
|
||||
1. **Follow the HOST repo's conventions, not this repo's.** Structure, dependency management, lint/format, test framework, docs, and the contribution guide are dictated by the target project. Do not impose `sdks/` layout, `fake_backend`, or Fern docs on an external repo. Read its `CONTRIBUTING`/`AGENTS.md` and mirror its closest existing integration/plugin.
|
||||
2. **Consume Opik through its PUBLISHED public API.** Use the released `opik` (Python) / `opik` (TypeScript) SDK or the REST API — the same surface end users get. Never import this repo's internal modules (`opik.decorator.*`, `rest_api`, message-processing internals); they are not available outside this repo and are not a stable contract.
|
||||
|
||||
## Start with the questionnaire
|
||||
|
||||
Collect everything up front, free-form — do not propose a target. See [workflow.md](workflow.md) Phase 0 for the full list. The essentials: **which external repo/package** (URL + links), **integration shape** (standalone package vs. upstream contribution vs. plugin), **host language/stack**, **where the Opik code goes**, **how the host tests and documents**, and **how to run it to verify**.
|
||||
|
||||
Once you know the target, check **[references.md](references.md)** — a curated list of known external integrations (standalone `opik-*` plugins incl. the cookiecutter template, and third-party host repos like LiteLLM / Dify / n8n) with their repo and Opik-docs URLs. If the target is listed or resembles one, start from its links and clone the closest sibling instead of researching blind.
|
||||
|
||||
## The workflow
|
||||
|
||||
Multi-step and autonomous by default (front-load preparation, run through, self-verify, end with a report); ask for the interactive variant to approve the design first. Full playbook in **[workflow.md](workflow.md)**. At a glance:
|
||||
|
||||
0. **Questionnaire + acquire** the external repo (clone/checkout into the scratchpad).
|
||||
1. **Investigate the host** — its integration/plugin conventions, dep management, test framework, docs, contribution guide. Find the closest existing integration to clone.
|
||||
2. **Investigate the Opik surface** — pick the public API to use: `@opik.track` / `track_*` wrappers, the low-level client, or REST. Lean on the `opik` and `instrument` skills for the public API.
|
||||
3. **Design** how Opik plugs into the host's extension point (callback hook, plugin entrypoint, middleware). Map host data → Opik trace/span fields.
|
||||
4. **Implement** in the host checkout, mirroring its closest sibling and its style.
|
||||
5. **Verify via the Opik MCP** — run the host with the integration active, log to the workspace the MCP reads, and read the trace/spans back. Loop until correct.
|
||||
6. **Test** using the host's test framework and conventions.
|
||||
7. **Document** following the host's docs conventions; optionally add/point an Opik docs page (`write-docs`).
|
||||
8. **Report** — what was done, verification evidence, a per-flow supported/tested table, limitations, and how to open the upstream PR.
|
||||
|
||||
## How this maps to the internal skill
|
||||
|
||||
The *mechanism* knowledge is shared — how a provider/framework exposes hooks, streaming, and usage is the same problem as in `opik-integrations`. Read that skill's [python.md](../opik-integrations/python.md) / [typescript.md](../opik-integrations/typescript.md) for the patterns (method patching, callback handler, OTel), but apply them against the **published** SDK surface and inside the **host repo's** structure.
|
||||
|
||||
## Skills this one builds on (do not duplicate them)
|
||||
|
||||
- `opik` / `instrument` — the user-facing public API surface (the right way to consume Opik from outside).
|
||||
- `opik-integrations` — the integration mechanism patterns (patching / callback / OTel) and field mapping.
|
||||
- `write-docs` — only if you also add an Opik-side docs page pointing at the external integration.
|
||||
@@ -0,0 +1,35 @@
|
||||
# External Integration References
|
||||
|
||||
Known external Opik integrations, to seed the research phase. When the target is (or resembles) one of these, start from its repo and docs instead of investigating from scratch — and use the closest one as the "clone the sibling" source.
|
||||
|
||||
> Verify before relying on specifics: repos change. Treat the **paths** below as starting hints, not guarantees, and confirm against the live repo. Add new entries here whenever you build or discover one.
|
||||
|
||||
## Standalone `opik-*` packages & plugins (comet-ml org)
|
||||
|
||||
For the **standalone package** shape — its own repo depending on the published Opik SDK.
|
||||
|
||||
| Repo | What it is | URL |
|
||||
|---|---|---|
|
||||
| `opik-project-template` | Cookiecutter scaffold for a new Opik project — **the starting point for a brand-new standalone package** | https://github.com/comet-ml/opik-project-template |
|
||||
| `opik-openclaw` | Official OpenClaw plugin exporting agent traces to Opik (traces, cost, tokens, errors) | https://github.com/comet-ml/opik-openclaw |
|
||||
| `opik-claude-code-plugin` | Opik plugin for Claude Code | https://github.com/comet-ml/opik-claude-code-plugin |
|
||||
| `opik-codex-plugin` | Opik plugin for Codex | https://github.com/comet-ml/opik-codex-plugin |
|
||||
| `opik-kong-plugin` | Opik plugin for the Kong AI Gateway | https://github.com/comet-ml/opik-kong-plugin |
|
||||
|
||||
## Upstream contributions (Opik support inside a third-party repo)
|
||||
|
||||
For the **upstream contribution** shape — Opik logging/callback added into a third-party project that owns its own plugin/observability system. Host repo (where the Opik code lives) + the Opik-side docs page.
|
||||
|
||||
| Host | Host repo | Opik docs |
|
||||
|---|---|---|
|
||||
| LiteLLM | https://github.com/BerriAI/litellm (Opik logger under `litellm/integrations/opik/`) | https://www.comet.com/docs/opik/integrations/litellm |
|
||||
| Dify | https://github.com/langgenius/dify | https://www.comet.com/docs/opik/integrations/dify |
|
||||
| n8n | https://github.com/n8n-io/n8n | https://www.comet.com/docs/opik/integrations/n8n |
|
||||
| Flowise | https://github.com/FlowiseAI/Flowise | https://www.comet.com/docs/opik/integrations/flowise |
|
||||
| Langflow | https://github.com/langflow-ai/langflow | https://www.comet.com/docs/opik/integrations/langflow |
|
||||
|
||||
## Discovering others
|
||||
|
||||
- **Opik docs integrations index** — the canonical list of published integrations (many are external/host-side): https://www.comet.com/docs/opik/integrations/overview
|
||||
- **comet-ml GitHub org** — search for `opik-` repos (plugins, templates): https://github.com/orgs/comet-ml/repositories?q=opik-
|
||||
- In this repo, the docs sources under `apps/opik-documentation/documentation/fern/docs-v2/integrations/` list every integration slug; host-side ones (e.g. `litellm`, `dify`, `n8n`, `flowise`, `langflow`) map to the docs URLs above.
|
||||
@@ -0,0 +1,81 @@
|
||||
# External Integration Workflow
|
||||
|
||||
End-to-end playbook for building an Opik integration that lives in an external repo or package. Runs autonomously by default; pause at the design gate only if the user asks to review first. **Always end with the Phase 8 report**, and back every "supported" claim with a passing test or an MCP-verified trace.
|
||||
|
||||
## Execution modes
|
||||
|
||||
- **Autonomous (default)** — make every preparation yourself (acquire the repo, find credentials, pick the backend), run all phases, self-verify, and report. Stop early only on a true blocker (no repo access, missing credential with no fallback, unreachable backend, an ambiguous product decision), reporting what you completed.
|
||||
- **Interactive** — pause at Phase 3 for design approval before writing code.
|
||||
|
||||
## Phase 0 — Questionnaire + acquire
|
||||
|
||||
Ask the user, in plain language, and wait for answers. Do **not** propose a target.
|
||||
|
||||
1. **External target** — the repo URL / package name, plus reference links (docs, the host project's contribution guide). If it's a standalone package, the intended Opik artifact name (e.g. `opik-openclaw`).
|
||||
2. **Integration shape** — standalone `opik-*` package/plugin · upstream contribution into a third-party project (LiteLLM, Dify, …) · tool/IDE plugin.
|
||||
3. **Host language / stack** — Python, TypeScript/Node, other; package manager and runtime.
|
||||
4. **Where the Opik code goes** — path inside the host repo (which plugin/integration dir), or a new standalone package layout.
|
||||
5. **Host conventions** — link to its `CONTRIBUTING`/`AGENTS.md`; its test framework; its docs location.
|
||||
6. **Verify + credentials** — how to run the host locally to exercise the integration, and whether the needed API keys (provider + Opik) are available.
|
||||
|
||||
Before acquiring, check **[references.md](references.md)** — known external integrations (standalone `opik-*` plugins + the cookiecutter template, and third-party hosts like LiteLLM / Dify / n8n / Flowise / Langflow) with their repo and Opik-docs URLs. If the target is listed or resembles one, start from those links and pick the closest as the clone source; a brand-new standalone package starts from `opik-project-template`.
|
||||
|
||||
Then **acquire the repo**: clone/checkout into the scratchpad (or add it to the session if supported). Confirm it builds/installs before changing anything. Restate the answers in one line and proceed.
|
||||
|
||||
## Phase 1 — Investigate the host
|
||||
|
||||
- The host's **extension point** for observability: a callback/hook interface, a plugin/entrypoint registry, middleware, or an env-driven logger. This is where Opik attaches.
|
||||
- The **closest existing integration** in the host (e.g. how it integrates another observability/logging vendor) — clone its structure, registration, and style.
|
||||
- The host's **dependency rules** (how it declares optional/extra deps), **test framework**, **lint/format**, and **docs** conventions.
|
||||
- What host data is available at the hook (inputs, outputs, model, usage, errors, timing) and its shape.
|
||||
|
||||
## Phase 2 — Investigate the Opik surface
|
||||
|
||||
Pick the **public** Opik API to use — never this repo's internals:
|
||||
|
||||
- **Python**: `import opik`, `@opik.track`, `track_*` wrappers, or `opik.Opik()` client; `opik.opik_context` for span data; `client.flush()`.
|
||||
- **TypeScript**: the `opik` package — `Opik` client, `track`, domain objects; `flushAll()`.
|
||||
- **REST** when no SDK fits the host runtime.
|
||||
|
||||
Confirm the published SDK version to depend on, and how the host will configure Opik (env vars: `OPIK_API_KEY`, base URL, workspace, project).
|
||||
|
||||
## Phase 3 — Design (gate)
|
||||
|
||||
Write down: the host extension point used; how the Opik client is created/configured and flushed; the mapping from host data → Opik trace/span fields (input/output/model/provider/usage/error/span-type); the file layout inside the host (mirroring its sibling); and the public registration the host user performs to enable it. In autonomous mode, record this in the report; in interactive mode, present and wait for approval.
|
||||
|
||||
## Phase 4 — Implement
|
||||
|
||||
Mirror the host's closest sibling and its style. Depend on the **published** Opik SDK. Keep the integration transparent (wrap, capture, re-raise; no behavior change when Opik is unconfigured). Flush appropriately for the host's lifecycle (request end, process exit, explicit hook).
|
||||
|
||||
## Phase 5 — Verify via Opik MCP
|
||||
|
||||
The proof the integration logs correctly — do not rely on reading code.
|
||||
|
||||
1. Configure Opik env to log into the workspace the connected Opik MCP reads.
|
||||
2. Run the host with the integration enabled, exercising the main flow(s) — non-streaming, streaming, and any extra path — then flush.
|
||||
3. Read the trace/spans back through the MCP (`list` then `read`).
|
||||
4. Check the tree against the Phase 3 mapping: one trace per top-level call; correct span hierarchy and `type`; input/output well-shaped; usage with token counts; model/provider set; errors recorded and re-raised.
|
||||
5. Loop until it matches.
|
||||
|
||||
## Phase 6 — Test
|
||||
|
||||
Use the **host's** test framework and conventions (not `fake_backend`/`testlib`, which are this repo's). Mirror how the host tests its other integrations — mock the Opik client/HTTP at the host's boundary where the host does, or assert against captured calls. Cover the same flows you verified in Phase 5.
|
||||
|
||||
## Phase 7 — Document
|
||||
|
||||
Follow the **host's** docs conventions (its README / docs site / examples dir). Optionally add or update an Opik-side docs page that points to the external integration, using the `write-docs` skill. Credential placeholders only.
|
||||
|
||||
## Phase 8 — Report
|
||||
|
||||
Produce a high-level report:
|
||||
|
||||
- **Integration** — external target, shape, host extension point, Opik public API used, published SDK version depended on.
|
||||
- **What was done** — repo acquired, files added/changed (with paths in the host), how a host user enables it.
|
||||
- **Verification** — MCP trace ids + project, and host test results.
|
||||
- **Flows supported & test coverage** — a table: every user-facing flow × implemented? × host test (by name) × MCP-verified? Flag any flow implemented but untested, and any not implemented.
|
||||
- **What's NOT supported / limitations** — out-of-scope hooks, host-version constraints, env/backend blockers.
|
||||
- **Follow-ups & PR** — how to open the upstream PR (branch/commit per the host's contribution guide), and any maintainer review notes.
|
||||
|
||||
## Definition of done
|
||||
|
||||
Implementation merged-quality in the host checkout, MCP verification passed, tests added per host conventions and passing, host docs updated, report produced. If you changed files under `.agents/` in *this* repo, run `make claude`.
|
||||
@@ -0,0 +1,119 @@
|
||||
---
|
||||
name: opik-frontend
|
||||
description: React frontend patterns for Opik. Use when working in apps/opik-frontend, on components, state, or data fetching.
|
||||
---
|
||||
|
||||
# Opik Frontend
|
||||
|
||||
## Architecture Decisions
|
||||
- **Routing**: TanStack Router (file-based)
|
||||
- **Data fetching**: TanStack Query (never raw fetch/useEffect)
|
||||
- **State**: Zustand for global, React state for local
|
||||
- **Components**: shadcn/ui + Radix UI base
|
||||
- **Forms**: React Hook Form + Zod validation
|
||||
|
||||
## Critical Gotchas
|
||||
|
||||
### Never useEffect for Data Fetching
|
||||
```typescript
|
||||
// ❌ BAD
|
||||
useEffect(() => {
|
||||
fetch('/api/data').then(setData);
|
||||
}, []);
|
||||
|
||||
// ✅ GOOD
|
||||
const { data } = useQuery({
|
||||
queryKey: ['data'],
|
||||
queryFn: fetchData,
|
||||
});
|
||||
```
|
||||
|
||||
### Selective Memoization
|
||||
```typescript
|
||||
// ✅ USE useMemo for: complex computations, large data transforms
|
||||
const filtered = useMemo(() =>
|
||||
data.filter(x => x.status === 'active').map(transform),
|
||||
[data]
|
||||
);
|
||||
|
||||
// ✅ USE useCallback for: functions passed to children
|
||||
const handleClick = useCallback(() => doSomething(id), [id]);
|
||||
|
||||
// ❌ DON'T memoize: simple values, primitives, local functions
|
||||
const name = data?.name ?? ''; // No useMemo needed
|
||||
```
|
||||
|
||||
### Zustand Selectors
|
||||
```typescript
|
||||
// ✅ GOOD - specific selector
|
||||
const selectedEntity = useEntityStore(state => state.selectedEntity);
|
||||
|
||||
// ❌ BAD - selecting entire store causes re-renders
|
||||
const { selectedEntity, filters } = useEntityStore();
|
||||
```
|
||||
|
||||
## Layer Architecture
|
||||
|
||||
### Shared layers (used by all versions)
|
||||
`ui → shared` (one-way only)
|
||||
|
||||
### Per-version layers
|
||||
`ui → shared → v1/pages-shared → v1/pages` (one-way only)
|
||||
`ui → shared → v2/pages-shared → v2/pages` (one-way only)
|
||||
|
||||
### Module boundaries
|
||||
- v1/ CANNOT import from v2/
|
||||
- v2/ CANNOT import from v1/
|
||||
- `src/components/` is BLOCKED (old structure, no longer exists)
|
||||
- After modifying imports: `npm run deps:validate`
|
||||
|
||||
### Shared component rules
|
||||
- Backward-compatible changes only
|
||||
- Must not be version-aware (use `showProjectSelector={true}` not `isV2={true}`)
|
||||
- If behavior needs to change, create a new component instead
|
||||
|
||||
## State Location Decisions
|
||||
- **URL state**: filters, pagination, selected items
|
||||
- **Zustand**: user preferences, cross-component state
|
||||
- **React state**: form inputs, UI toggles
|
||||
|
||||
## Component Structure
|
||||
```typescript
|
||||
const Component: React.FC<Props> = ({ prop }) => {
|
||||
// 1. State hooks
|
||||
// 2. Queries/mutations
|
||||
// 3. Memoization (only when needed)
|
||||
// 4. Event handlers
|
||||
|
||||
if (isLoading) return <Loader />;
|
||||
if (error) return <ErrorComponent />;
|
||||
|
||||
return <div>...</div>;
|
||||
};
|
||||
```
|
||||
|
||||
## Query Patterns
|
||||
```typescript
|
||||
// Query with params
|
||||
const { data } = useQuery({
|
||||
queryKey: [ENTITY_KEY, params],
|
||||
queryFn: (context) => fetchEntity(context, params),
|
||||
});
|
||||
|
||||
// Mutation with invalidation
|
||||
const mutation = useMutation({
|
||||
mutationFn: updateEntity,
|
||||
onSuccess: () => {
|
||||
queryClient.invalidateQueries({ queryKey: [ENTITY_KEY] });
|
||||
},
|
||||
});
|
||||
```
|
||||
|
||||
## Reference Files
|
||||
- [forms.md](forms.md) - React Hook Form + Zod patterns
|
||||
- [ui-components.md](ui-components.md) - Button variants, typography, dark theme
|
||||
- [responsive-design.md](responsive-design.md) - Tailwind breakpoints vs useIsPhone
|
||||
- [testing.md](testing.md) - When to test, Vitest patterns
|
||||
- [code-quality.md](code-quality.md) - Lodash imports, naming, deps:validate
|
||||
- [performance.md](performance.md) - Bundle optimization, rendering, memoization
|
||||
- [permissions.md](permissions.md) - `usePermissions()` guard guidance for UI actions
|
||||
@@ -0,0 +1,166 @@
|
||||
# Code Quality Standards
|
||||
|
||||
## Lodash Import Pattern
|
||||
|
||||
Always import individually for tree-shaking:
|
||||
|
||||
```typescript
|
||||
// ✅ GOOD - Individual imports
|
||||
import pick from "lodash/pick";
|
||||
import merge from "lodash/merge";
|
||||
import cloneDeep from "lodash/cloneDeep";
|
||||
import uniqBy from "lodash/uniqBy";
|
||||
import groupBy from "lodash/groupBy";
|
||||
import isString from "lodash/isString";
|
||||
import isArray from "lodash/isArray";
|
||||
import isEmpty from "lodash/isEmpty";
|
||||
|
||||
// ❌ BAD - Includes entire library
|
||||
import _ from "lodash";
|
||||
|
||||
// ❌ BAD - Less efficient than individual imports
|
||||
import { pick, merge } from "lodash";
|
||||
```
|
||||
|
||||
## Type Checking with Lodash
|
||||
|
||||
```typescript
|
||||
// ✅ Prefer Lodash type methods
|
||||
import isString from "lodash/isString";
|
||||
import isNumber from "lodash/isNumber";
|
||||
import isArray from "lodash/isArray";
|
||||
import isObject from "lodash/isObject";
|
||||
import isNil from "lodash/isNil";
|
||||
import isEmpty from "lodash/isEmpty";
|
||||
|
||||
if (isString(value)) { /* handle string */ }
|
||||
if (isArray(value)) { /* handle array */ }
|
||||
```
|
||||
|
||||
## Naming Conventions
|
||||
|
||||
```typescript
|
||||
// Components: PascalCase
|
||||
DataTable.tsx
|
||||
UserProfile.tsx
|
||||
|
||||
// Utilities: camelCase
|
||||
dateUtils.ts
|
||||
apiClient.ts
|
||||
|
||||
// Hooks: camelCase starting with 'use'
|
||||
useEntityList.ts
|
||||
useDebounce.ts
|
||||
|
||||
// Constants: SCREAMING_SNAKE_CASE
|
||||
const COLUMN_TYPE = { STRING: "string" } as const;
|
||||
const API_ENDPOINTS = { USERS: "/api/v1/users" } as const;
|
||||
|
||||
// Event handlers: Descriptive
|
||||
const handleDeleteClick = useCallback(() => {}, []);
|
||||
const handleUserSelect = useCallback((user: UserData) => {}, []);
|
||||
|
||||
// ❌ Avoid generic names
|
||||
const handleClick = () => {}; // Too vague
|
||||
```
|
||||
|
||||
## Import Order
|
||||
|
||||
```typescript
|
||||
// 1. React
|
||||
import React, { useState, useCallback } from "react";
|
||||
|
||||
// 2. External libraries (tanstack, zod, axios, lucide-react, etc.)
|
||||
import { useQuery } from "@tanstack/react-query";
|
||||
import { SquareArrowOutUpRight } from "lucide-react";
|
||||
import get from "lodash/get";
|
||||
|
||||
// 3. Internal imports (@/) — no strict ordering required
|
||||
import { Button } from "@/ui/button";
|
||||
import DataTable from "@/shared/DataTable/DataTable";
|
||||
import useAppStore from "@/store/AppStore";
|
||||
import useDatasetCreateMutation from "@/api/datasets/useDatasetCreateMutation";
|
||||
import { cn } from "@/lib/utils";
|
||||
import { COLUMN_TYPE } from "@/types/shared";
|
||||
```
|
||||
|
||||
Internal `@/` imports have no strict order. Do not reorder existing imports for style.
|
||||
|
||||
## Dependency Architecture
|
||||
|
||||
**CRITICAL: After modifying imports, run:**
|
||||
```bash
|
||||
cd apps/opik-frontend && npm run deps:validate
|
||||
```
|
||||
|
||||
### Layer Hierarchy (one-way only)
|
||||
`ui → shared → pages-shared → pages`
|
||||
|
||||
### Rules
|
||||
- No circular dependencies
|
||||
- No importing from higher layers
|
||||
- No cross-page imports
|
||||
- API layer cannot import components (except use-toast.ts)
|
||||
|
||||
## Extract Self-Contained Sub-Components
|
||||
|
||||
When a page-level component crosses ~500–600 lines and contains a sub-region with its own state, queries, mutations, and handlers — peel that region into a sibling component. Don't pull a region out just because it's visually distinct; extract when it has *independent data dependencies*.
|
||||
|
||||
### Good extraction candidates
|
||||
- A dropdown menu that owns its own `useQuery`/`useMutation` and a derived map/list
|
||||
- A dialog that manages its own form state and validation
|
||||
- A toolbar section that uses 3+ hooks the rest of the page doesn't need
|
||||
|
||||
### Anti-pattern
|
||||
Pulling out a small JSX block that only takes props — that's "needless component", not cleanup.
|
||||
|
||||
```tsx
|
||||
// PROMPT TAB BEFORE (728 lines) — environment deploy logic inlined
|
||||
const PromptTab = ({ prompt }) => {
|
||||
// ...100+ lines of unrelated state/queries
|
||||
const { data: envs } = useEnvironmentsList();
|
||||
const environments = useMemo(/* sort */, [envs]);
|
||||
const environmentOwners = useMemo(/* map of env→version */, [versions]);
|
||||
const { mutate, isPending } = useSetPromptVersionEnvironmentMutation();
|
||||
const handleDeploy = useCallback(/* ... */, [/* 6 deps */]);
|
||||
const handleClear = useCallback(/* ... */, [/* 5 deps */]);
|
||||
// ...
|
||||
return (
|
||||
<DropdownMenu>{/* 70 lines of menu items */}</DropdownMenu>
|
||||
);
|
||||
};
|
||||
|
||||
// AFTER (584 lines) — sub-component owns its own data dependencies
|
||||
const PromptTab = ({ prompt }) => {
|
||||
// ...no env state in this scope anymore
|
||||
return (
|
||||
<DeployToEnvironmentMenu
|
||||
promptId={prompt.id}
|
||||
versionId={effectiveVersionId}
|
||||
versionLabel={activeVersionLabel}
|
||||
versions={versions}
|
||||
totalVersions={total}
|
||||
activeEnvironment={activeVersionEnvironment}
|
||||
/>
|
||||
);
|
||||
};
|
||||
```
|
||||
|
||||
**Why:** Sub-component owns the environments query, the owner-map memo, the mutation, both handlers, and the toasts. Parent shrinks by ~120 lines and no longer holds workspace/configuration/env imports. The handlers' dependency arrays also shrink.
|
||||
|
||||
**How to apply:** When you see ≥3 hooks (`useQuery`, `useMutation`, `useMemo`, `useCallback`) all feeding one JSX region, that region wants to be its own component. Pass only the upstream props it needs — never pass refs into the child to "share state up." See [PromptTab.tsx](../../../apps/opik-frontend/src/v2/pages/PromptPage/PromptTab/PromptTab.tsx) and [DeployToEnvironmentMenu.tsx](../../../apps/opik-frontend/src/v2/pages/PromptPage/PromptTab/DeployToEnvironmentMenu.tsx) for a worked example.
|
||||
|
||||
## Co-locate Helpers with Their Component Family
|
||||
|
||||
When two sibling components share validation/formatting helpers, put the helpers in a `helpers.ts` next to them — not in `lib/`. `lib/` is for repo-wide utilities. Co-located helpers keep the surface area small and signal the helper is scoped to the family.
|
||||
|
||||
```
|
||||
shared/EnvironmentLabel/
|
||||
EnvironmentLabel.tsx // imports from ./helpers
|
||||
EnvironmentBadge.tsx // imports from ./helpers
|
||||
helpers.ts // resolveEnvironmentColor, getContrastingTextColor
|
||||
```
|
||||
|
||||
**Why:** Before extraction, `EnvironmentLabel.tsx` and `EnvironmentBadge.tsx` each defined their own `resolveColor` (same body, same constants). Promoting to `lib/colorVariants.ts` would collide with an existing palette-based `resolveColor`; co-locating avoided the naming conflict and kept the surface narrow.
|
||||
|
||||
**How to apply:** If a helper is only meaningful within one component family, never make it the codebase's problem. Move it to `lib/` only when a third unrelated consumer appears.
|
||||
@@ -0,0 +1,110 @@
|
||||
# Form Handling Patterns
|
||||
|
||||
## React Hook Form + Zod Setup
|
||||
|
||||
```typescript
|
||||
import { useForm } from "react-hook-form";
|
||||
import { zodResolver } from "@hookform/resolvers/zod";
|
||||
import { z } from "zod";
|
||||
|
||||
// Define schema
|
||||
const formSchema = z.object({
|
||||
name: z.string().min(1, "Name is required"),
|
||||
email: z.string().email("Invalid email"),
|
||||
description: z.string().optional(),
|
||||
});
|
||||
|
||||
type FormData = z.infer<typeof formSchema>;
|
||||
|
||||
// Use in component
|
||||
const form = useForm<FormData>({
|
||||
resolver: zodResolver(formSchema),
|
||||
defaultValues: {
|
||||
name: "",
|
||||
email: "",
|
||||
description: "",
|
||||
},
|
||||
});
|
||||
```
|
||||
|
||||
## Form JSX Structure
|
||||
|
||||
```typescript
|
||||
<Form {...form}>
|
||||
<form onSubmit={form.handleSubmit(onSubmit)} className="space-y-6">
|
||||
<FormField
|
||||
control={form.control}
|
||||
name="name"
|
||||
render={({ field }) => (
|
||||
<FormItem>
|
||||
<FormLabel>Name</FormLabel>
|
||||
<FormControl>
|
||||
<Input {...field} />
|
||||
</FormControl>
|
||||
<FormMessage />
|
||||
</FormItem>
|
||||
)}
|
||||
/>
|
||||
|
||||
<Button type="submit" disabled={form.formState.isSubmitting}>
|
||||
{form.formState.isSubmitting && <Spinner className="mr-2" />}
|
||||
Submit
|
||||
</Button>
|
||||
</form>
|
||||
</Form>
|
||||
```
|
||||
|
||||
## Dynamic Form Fields
|
||||
|
||||
```typescript
|
||||
const { fields, append, remove } = useFieldArray({
|
||||
control: form.control,
|
||||
name: "items",
|
||||
});
|
||||
|
||||
{fields.map((field, index) => (
|
||||
<div key={field.id} className="flex gap-2">
|
||||
<FormField
|
||||
control={form.control}
|
||||
name={`items.${index}.name`}
|
||||
render={({ field }) => (
|
||||
<FormItem>
|
||||
<FormControl>
|
||||
<Input {...field} placeholder="Name" />
|
||||
</FormControl>
|
||||
<FormMessage />
|
||||
</FormItem>
|
||||
)}
|
||||
/>
|
||||
<Button type="button" variant="outline" onClick={() => remove(index)}>
|
||||
Remove
|
||||
</Button>
|
||||
</div>
|
||||
))}
|
||||
|
||||
<Button type="button" onClick={() => append({ name: "", value: "" })}>
|
||||
Add Item
|
||||
</Button>
|
||||
```
|
||||
|
||||
## Conditional Validation
|
||||
|
||||
```typescript
|
||||
const formSchema = z
|
||||
.object({
|
||||
type: z.enum(["user", "admin"]),
|
||||
permissions: z.array(z.string()).optional(),
|
||||
})
|
||||
.refine(
|
||||
(data) => {
|
||||
if (data.type === "admin" && (!data.permissions || data.permissions.length === 0)) {
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
},
|
||||
{
|
||||
message: "Admin users must have at least one permission",
|
||||
path: ["permissions"],
|
||||
},
|
||||
);
|
||||
```
|
||||
@@ -0,0 +1,138 @@
|
||||
# Frontend Performance
|
||||
|
||||
## Bundle Optimization
|
||||
|
||||
### Lazy Load Heavy Components
|
||||
Use `React.lazy` for components not needed on initial render.
|
||||
|
||||
```tsx
|
||||
// BAD - Monaco bundles with main chunk (~300KB)
|
||||
import { MonacoEditor } from './monaco-editor';
|
||||
|
||||
// GOOD - Monaco loads on demand
|
||||
const MonacoEditor = lazy(() =>
|
||||
import('./monaco-editor').then(m => ({ default: m.MonacoEditor }))
|
||||
);
|
||||
|
||||
function CodePanel({ code }: { code: string }) {
|
||||
return (
|
||||
<Suspense fallback={<Skeleton className="h-96 w-full" />}>
|
||||
<MonacoEditor value={code} />
|
||||
</Suspense>
|
||||
);
|
||||
}
|
||||
```
|
||||
|
||||
## Rendering Performance
|
||||
|
||||
### Animate SVG Wrappers
|
||||
Browsers don't hardware-accelerate CSS animations on SVG elements.
|
||||
|
||||
```tsx
|
||||
// BAD - no hardware acceleration
|
||||
<svg className="animate-spin">...</svg>
|
||||
|
||||
// GOOD - GPU accelerated
|
||||
<div className="animate-spin">
|
||||
<svg>...</svg>
|
||||
</div>
|
||||
```
|
||||
|
||||
## Re-render Optimization
|
||||
|
||||
### Defer State Reads
|
||||
Don't subscribe to state only used in callbacks.
|
||||
|
||||
```tsx
|
||||
// BAD - re-renders on every searchParams change
|
||||
function ShareButton({ id }: Props) {
|
||||
const searchParams = useSearchParams();
|
||||
const handleShare = () => {
|
||||
const ref = searchParams.get('ref');
|
||||
share(id, { ref });
|
||||
};
|
||||
return <button onClick={handleShare}>Share</button>;
|
||||
}
|
||||
|
||||
// GOOD - reads on demand, no subscription
|
||||
function ShareButton({ id }: Props) {
|
||||
const handleShare = () => {
|
||||
const params = new URLSearchParams(window.location.search);
|
||||
share(id, { ref: params.get('ref') });
|
||||
};
|
||||
return <button onClick={handleShare}>Share</button>;
|
||||
}
|
||||
```
|
||||
|
||||
### Extract Memoized Components
|
||||
Move expensive work after early returns.
|
||||
|
||||
```tsx
|
||||
// BAD - computes avatar even when loading
|
||||
function Profile({ user, loading }: Props) {
|
||||
const avatar = useMemo(() => computeAvatarId(user), [user]);
|
||||
if (loading) return <Skeleton />;
|
||||
return <Avatar id={avatar} />;
|
||||
}
|
||||
|
||||
// GOOD - skips computation when loading
|
||||
const UserAvatar = memo(function({ user }: { user: User }) {
|
||||
const id = useMemo(() => computeAvatarId(user), [user]);
|
||||
return <Avatar id={id} />;
|
||||
});
|
||||
|
||||
function Profile({ user, loading }: Props) {
|
||||
if (loading) return <Skeleton />;
|
||||
return <UserAvatar user={user} />;
|
||||
}
|
||||
```
|
||||
|
||||
Note: If React Compiler is enabled, manual memoization isn't necessary.
|
||||
|
||||
## Data Fetching
|
||||
|
||||
### Don't refetch what the parent already has
|
||||
If a parent already holds an entity (typically from a list query), pass it down instead of having the child refetch by id. Keep the per-id fetch as a fallback for when the entity isn't in the list.
|
||||
|
||||
```tsx
|
||||
// BAD - parent has the prompt from useProjectPromptsList,
|
||||
// but child fires a second request for the same one
|
||||
const CompactLoadedPrompt = ({ promptId }: Props) => {
|
||||
const { data } = usePromptById({ promptId });
|
||||
return <LoadedPromptDisplay {...derive(data)} />;
|
||||
};
|
||||
|
||||
// GOOD - accept the entity as a prop; only fetch when not provided
|
||||
type Props = { promptId: string; prompt?: Prompt };
|
||||
const CompactLoadedPrompt = ({ promptId, prompt }: Props) => {
|
||||
const { data: fetched } = usePromptById(
|
||||
{ promptId },
|
||||
{ enabled: !!promptId && !prompt },
|
||||
);
|
||||
const data = prompt ?? fetched;
|
||||
return <LoadedPromptDisplay {...derive(data)} />;
|
||||
};
|
||||
```
|
||||
|
||||
**Why:** A list query already returns full `Prompt` objects with `latest_version`, `template_structure`, and `version_count`. Refetching by id costs an extra round-trip per loaded prompt and creates a cache slot duplicating the list's data. The `enabled` guard lets the child still fetch when used in a context that doesn't have the entity in scope.
|
||||
|
||||
### Don't narrow query invalidation past correctness
|
||||
After a mutation that has cross-entity side effects, invalidating the whole keyspace is sometimes the only correct option — narrowing it is a regression, not a cleanup.
|
||||
|
||||
```ts
|
||||
// CORRECT — env can be transferred from version B to version A,
|
||||
// so version B's cache becomes stale too. We don't have B's id here.
|
||||
onSuccess: (_data, { promptId }) => {
|
||||
queryClient.invalidateQueries({ queryKey: ["prompt", { promptId }] });
|
||||
queryClient.invalidateQueries({
|
||||
predicate: (q) =>
|
||||
q.queryKey[0] === "prompt-versions" &&
|
||||
(q.queryKey[1] as { promptId?: string })?.promptId === promptId,
|
||||
});
|
||||
queryClient.invalidateQueries({ queryKey: ["prompt-version"] }); // broad, on purpose
|
||||
}
|
||||
```
|
||||
|
||||
**Why:** `prompt-version` cache keys are by `versionId` only — no `promptId`. When env moves from B→A, only the page state knows about B. The broad invalidate is the safest signal; individual version caches are small. See [useSetPromptVersionEnvironmentMutation.ts](../../../apps/opik-frontend/src/api/prompts/useSetPromptVersionEnvironmentMutation.ts).
|
||||
|
||||
**How to apply:** Before tightening an invalidate, list every entity whose cache could go stale after the mutation succeeds. If the mutation has transfer/move semantics (env-pin, default-version, primary-tag), the "off-target" entity loses state too. Keep the broad invalidate unless every affected entity is reachable from the mutation handler.
|
||||
@@ -0,0 +1,90 @@
|
||||
# UI Permissions
|
||||
|
||||
## When Adding or Modifying Components
|
||||
|
||||
Every new or modified component in `apps/opik-frontend/src/` that performs a guarded action (create, delete, edit, configure, or view gated content) should be evaluated for permission guards.
|
||||
|
||||
### Check Process
|
||||
|
||||
1. **Read the current permissions** in `apps/opik-frontend/src/plugins/comet/PermissionsProvider.tsx`. This file lists all available permission flags (e.g., `canCreateDatasets`, `canDeleteTraces`, `canViewDashboards`).
|
||||
|
||||
2. **Check if sibling or similar components use `usePermissions()`**. If nearby components in the same page or feature already consume permissions, the new component likely needs them too.
|
||||
|
||||
3. **Logically match the UI action to a permission flag**. Do not rely on naming patterns — reason about what the component does and which permission governs that action:
|
||||
- A page or section displaying gated content likely needs a **view** permission (e.g., `canViewExperiments`, `canViewDashboards`, `canViewDatasets`)
|
||||
- A "Create" / "Add" / "New" button likely needs a **create** permission (e.g., `canCreateDatasets`, `canCreateProjects`)
|
||||
- An "Edit" / "Update" / "Save" action likely needs an **edit** permission (e.g., `canEditDashboards`, `canEditDatasets`)
|
||||
- A "Delete" / "Remove" button or menu item likely needs a **delete** permission (e.g., `canDeleteTraces`, `canDeletePrompts`)
|
||||
- A configuration or settings action likely needs a **configure/update** permission (e.g., `canConfigureWorkspaceSettings`, `canUpdateAIProviders`)
|
||||
- Annotation, tagging, and comment actions have their own permissions (`canAnnotateTraceSpanThread`, `canTagTrace`, `canWriteComments`)
|
||||
|
||||
4. **If a logically matching permission exists**, do NOT add it automatically. Instead, inform the user which permission you believe matches and why, then ask for confirmation before adding the guard.
|
||||
|
||||
5. **If no matching permission exists**, inform the user:
|
||||
- State which UI action has no matching permission
|
||||
- Note that without a guard, the action will be available to all authenticated users
|
||||
- Ask whether a new permission should be added to the system, or if the current behavior is acceptable
|
||||
|
||||
### Guard Patterns
|
||||
|
||||
When applying a permission, use the pattern appropriate to the context:
|
||||
|
||||
**Page-level access control** — wrap entire pages with a guard component:
|
||||
```tsx
|
||||
const { permissions: { canViewDashboards } } = usePermissions();
|
||||
|
||||
return (
|
||||
<NoAccessPageGuard resourceName="dashboards" canViewPage={canViewDashboards} />
|
||||
);
|
||||
```
|
||||
|
||||
**Conditional rendering of actions** — hide buttons or menu items:
|
||||
```tsx
|
||||
const { permissions: { canCreateDatasets } } = usePermissions();
|
||||
|
||||
return (
|
||||
<>
|
||||
{canCreateDatasets && (
|
||||
<Button onClick={handleCreate}>Create dataset</Button>
|
||||
)}
|
||||
</>
|
||||
);
|
||||
```
|
||||
|
||||
**Disabling queries** — prevent unnecessary API calls:
|
||||
```tsx
|
||||
const { permissions: { canViewDatasets } } = usePermissions();
|
||||
|
||||
const { data } = useDatasetsList(params, {
|
||||
enabled: canViewDatasets,
|
||||
});
|
||||
```
|
||||
|
||||
**Read-only component states** — render immutable versions instead of hiding:
|
||||
```tsx
|
||||
const { permissions: { canTagTrace } } = usePermissions();
|
||||
|
||||
const tagsProps = canTagTrace
|
||||
? { tags }
|
||||
: { tags: [], immutableTags: tags };
|
||||
```
|
||||
|
||||
### Current Coverage
|
||||
|
||||
Not all UI actions have permission guards yet. This is expected. Do not add guards speculatively; only add them when a logically matching permission flag exists in `PermissionsProvider.tsx` or the user confirms a new one should be created.
|
||||
|
||||
### Common Mistakes
|
||||
|
||||
- **Forgetting to guard delete actions** — delete buttons in row action menus and bulk action panels are easy to miss
|
||||
- **Forgetting navigation visibility** — sidebar menu items should be hidden or disabled based on view permissions (see `SideBarMenuItems.tsx`)
|
||||
- **Forgetting to disable queries** — even if a button is hidden, the underlying query may still fire; use `enabled` to prevent this
|
||||
- **Inconsistent v1/v2 coverage** — if a permission guard is added in `v1/`, the equivalent `v2/` component should also be guarded, and vice versa
|
||||
|
||||
### Reference
|
||||
|
||||
- Permissions type definition: `apps/opik-frontend/src/types/permissions.ts`
|
||||
- Permission context hook: `apps/opik-frontend/src/contexts/PermissionsContext.tsx`
|
||||
- Permission provider (Comet plugin): `apps/opik-frontend/src/plugins/comet/PermissionsProvider.tsx`
|
||||
- Permission computation hook: `apps/opik-frontend/src/plugins/comet/useUserPermission.ts`
|
||||
- Backend permission names enum: `apps/opik-frontend/src/plugins/comet/types.ts` (`ManagementPermissionsNames`)
|
||||
- Full permissions spec: https://www.notion.so/cometml/Workspace-permissions-and-user-roles-management-2b77124010a380f8b526e7ecb235c419
|
||||
@@ -0,0 +1,76 @@
|
||||
# Responsive Design
|
||||
|
||||
## When to Add Phone Support
|
||||
|
||||
Phone support is **NOT required by default**. Only add when:
|
||||
- Explicitly requested in Jira ticket
|
||||
- Working on onboarding features
|
||||
- Component already has phone support
|
||||
|
||||
## Decision Framework
|
||||
|
||||
| Scenario | Use |
|
||||
|----------|-----|
|
||||
| Styling (padding, margins, colors) | Tailwind `md:` |
|
||||
| Layout direction | Tailwind `md:flex-row` |
|
||||
| Show/hide element | `hidden md:block` |
|
||||
| Different component | `useIsPhone` |
|
||||
| Different prop values | `useIsPhone` |
|
||||
| Structural DOM changes | `useIsPhone` |
|
||||
|
||||
## Tailwind (Preferred)
|
||||
|
||||
Mobile-first: base classes = phone, `md:` = tablet+
|
||||
|
||||
```tsx
|
||||
// Styling changes
|
||||
<div className="w-full px-4 md:w-[468px] md:px-0">
|
||||
|
||||
// Layout direction
|
||||
<div className="flex flex-col gap-4 md:flex-row md:gap-6">
|
||||
|
||||
// Visibility
|
||||
<div className="hidden md:block">Desktop only</div>
|
||||
```
|
||||
|
||||
## useIsPhone (When CSS Can't)
|
||||
|
||||
```tsx
|
||||
import { useIsPhone } from "@/hooks/useIsPhone";
|
||||
|
||||
const { isPhonePortrait } = useIsPhone();
|
||||
|
||||
// Different component
|
||||
if (isPhonePortrait) {
|
||||
return <BottomSheet>{content}</BottomSheet>;
|
||||
}
|
||||
return <SideDialog>{content}</SideDialog>;
|
||||
|
||||
// Different props
|
||||
<DialogContent
|
||||
side={isPhonePortrait ? "bottom" : "right"}
|
||||
size={isPhonePortrait ? "full" : "md"}
|
||||
/>
|
||||
```
|
||||
|
||||
## Hooks Reference
|
||||
|
||||
```tsx
|
||||
// Device detection
|
||||
const { isPhone, isPhonePortrait, isPhoneLandscape } = useIsPhone();
|
||||
|
||||
// Custom queries
|
||||
const isTablet = useMediaQuery("(min-width: 768px) and (max-width: 1023px)");
|
||||
|
||||
// Constants
|
||||
import { QUERY_IS_PHONE_PORTRAIT } from "@/constants/responsiveness";
|
||||
```
|
||||
|
||||
## Breakpoints
|
||||
|
||||
| Prefix | Min Width |
|
||||
|--------|-----------|
|
||||
| (none) | 0px (mobile) |
|
||||
| `md:` | 768px |
|
||||
| `lg:` | 1024px |
|
||||
| `xl:` | 1280px |
|
||||
@@ -0,0 +1,88 @@
|
||||
# Frontend Testing Patterns
|
||||
|
||||
## When to Write Tests
|
||||
|
||||
### ALWAYS Test:
|
||||
- Utility functions with multiple scenarios
|
||||
- Data processing logic with edge cases
|
||||
- Parsing/transformation functions
|
||||
- Filter/search logic
|
||||
- Business logic with multiple branches
|
||||
|
||||
### DON'T Test:
|
||||
- Simple UI components without logic
|
||||
- Third-party library integrations
|
||||
- Trivial getters/setters
|
||||
- Pure presentational components
|
||||
|
||||
## Test Structure (AAA Pattern)
|
||||
|
||||
```typescript
|
||||
import { describe, it, expect } from "vitest";
|
||||
|
||||
describe("functionName", () => {
|
||||
describe("feature group", () => {
|
||||
it("should do something when condition", () => {
|
||||
// Arrange
|
||||
const input = createTestData();
|
||||
|
||||
// Act
|
||||
const result = functionUnderTest(input);
|
||||
|
||||
// Assert
|
||||
expect(result).toEqual(expectedOutput);
|
||||
});
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
## Edge Cases to Test
|
||||
|
||||
```typescript
|
||||
describe("edge cases", () => {
|
||||
it("should handle empty input", () => {
|
||||
expect(processData([])).toEqual([]);
|
||||
});
|
||||
|
||||
it("should handle null/undefined input", () => {
|
||||
expect(processData(null)).toEqual(null);
|
||||
expect(processData(undefined)).toEqual(undefined);
|
||||
});
|
||||
|
||||
it("should handle large datasets", () => {
|
||||
const largeArray = Array(1000).fill().map((_, i) => ({ id: i }));
|
||||
expect(processData(largeArray)).toHaveLength(1000);
|
||||
});
|
||||
|
||||
it("should handle malformed data", () => {
|
||||
expect(() => processData({ invalid: "data" })).toThrowError();
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
## Mock Data Pattern
|
||||
|
||||
```typescript
|
||||
const createMockUser = (overrides = {}) => ({
|
||||
id: "user-1",
|
||||
name: "John Doe",
|
||||
email: "john@example.com",
|
||||
role: "admin",
|
||||
createdAt: "2024-01-01T00:00:00Z",
|
||||
...overrides,
|
||||
});
|
||||
|
||||
it("should process user data", () => {
|
||||
const user = createMockUser({ role: "guest" });
|
||||
expect(processUser(user).hasAdminAccess).toBe(false);
|
||||
});
|
||||
```
|
||||
|
||||
## Running Tests
|
||||
|
||||
```bash
|
||||
npm test # Run all tests
|
||||
npm run test:ui # Watch mode
|
||||
npm test -- utils.test.ts # Specific file
|
||||
npm test -- --coverage # With coverage
|
||||
```
|
||||
@@ -0,0 +1,155 @@
|
||||
# UI Components & Design System
|
||||
|
||||
## Button Variants
|
||||
|
||||
```typescript
|
||||
// Primary actions
|
||||
<Button variant="default">Save</Button>
|
||||
<Button variant="special">Special Action</Button>
|
||||
|
||||
// Secondary actions
|
||||
<Button variant="secondary">Cancel</Button>
|
||||
<Button variant="outline">Edit</Button>
|
||||
|
||||
// Destructive actions
|
||||
<Button variant="destructive">Delete</Button>
|
||||
|
||||
// Minimal/Ghost actions
|
||||
<Button variant="ghost">Link Action</Button>
|
||||
<Button variant="ghostInverted">Action on Dark BG</Button>
|
||||
<Button variant="minimal">Subtle Action</Button>
|
||||
|
||||
// Link actions
|
||||
<Button variant="link">Link</Button>
|
||||
<Button variant="tableLink">Table Link</Button>
|
||||
|
||||
// Icon buttons
|
||||
<Button variant="default" size="icon"><Icon /></Button>
|
||||
<Button variant="ghost" size="icon-sm"><Icon /></Button>
|
||||
|
||||
// Sizes: "3xs" | "2xs" | "xs" | "sm" | "default" | "lg"
|
||||
// Icon sizes: "icon-3xs" | "icon-2xs" | "icon-xs" | "icon-sm" | "icon" | "icon-lg"
|
||||
```
|
||||
|
||||
## DataTable Column Types
|
||||
|
||||
```typescript
|
||||
import { COLUMN_TYPE, ROW_HEIGHT } from "@/types/shared";
|
||||
|
||||
// Available types
|
||||
COLUMN_TYPE.string // Text data
|
||||
COLUMN_TYPE.number // Numeric data
|
||||
COLUMN_TYPE.time // Date/time data
|
||||
COLUMN_TYPE.duration // Duration data
|
||||
COLUMN_TYPE.cost // Cost data
|
||||
COLUMN_TYPE.list // Array data
|
||||
COLUMN_TYPE.dictionary // Object data
|
||||
COLUMN_TYPE.numberDictionary // Feedback scores
|
||||
COLUMN_TYPE.category // Category/tag data
|
||||
```
|
||||
|
||||
## Typography Classes
|
||||
|
||||
```css
|
||||
/* Titles */
|
||||
.comet-title-xl /* 3xl font-medium */
|
||||
.comet-title-l /* 2xl font-medium */
|
||||
.comet-title-m /* xl font-medium */
|
||||
.comet-title-s /* lg font-medium */
|
||||
.comet-title-xs /* sm font-medium */
|
||||
|
||||
/* Body text */
|
||||
.comet-body /* base font-normal */
|
||||
.comet-body-accented /* base font-medium */
|
||||
.comet-body-s /* sm font-normal */
|
||||
.comet-body-s-accented /* sm font-medium */
|
||||
.comet-body-xs /* xs font-normal */
|
||||
|
||||
/* Code */
|
||||
.comet-code /* monospace font */
|
||||
```
|
||||
|
||||
## Layout Classes
|
||||
|
||||
```css
|
||||
.comet-header-height /* 64px header */
|
||||
.comet-sidebar-width /* sidebar width */
|
||||
.comet-content-inset /* content padding */
|
||||
.comet-custom-scrollbar /* custom scrollbar */
|
||||
.comet-no-scrollbar /* hide scrollbar */
|
||||
```
|
||||
|
||||
## Custom CSS Class Prefix
|
||||
|
||||
Always use `comet-` prefix for custom classes:
|
||||
```typescript
|
||||
className="comet-table-row-active"
|
||||
className="comet-sidebar-collapsed"
|
||||
```
|
||||
|
||||
## Dark Theme Support
|
||||
|
||||
```typescript
|
||||
// ✅ GOOD - Use theme-aware classes
|
||||
<div className="bg-card text-card-foreground border border-border">
|
||||
<h2 className="text-primary">Title</h2>
|
||||
<p className="text-muted-foreground">Description</p>
|
||||
</div>
|
||||
|
||||
// ❌ BAD - Hard-coded colors
|
||||
<div className="bg-white text-black border-gray-200">
|
||||
```
|
||||
|
||||
Add new colors to `main.css` with dark alternatives:
|
||||
```css
|
||||
:root {
|
||||
--my-custom-color: 210 100% 50%;
|
||||
}
|
||||
.dark {
|
||||
--my-custom-color: 220 100% 60%;
|
||||
}
|
||||
```
|
||||
|
||||
## Color System
|
||||
|
||||
```css
|
||||
/* Primary */
|
||||
bg-primary text-primary-foreground hover:bg-primary-hover
|
||||
|
||||
/* Secondary */
|
||||
bg-secondary text-secondary-foreground
|
||||
|
||||
/* Muted */
|
||||
bg-muted text-muted-foreground text-muted-gray
|
||||
|
||||
/* Destructive */
|
||||
bg-destructive text-destructive-foreground border-destructive
|
||||
```
|
||||
|
||||
## State Classes
|
||||
|
||||
```tsx
|
||||
// Loading
|
||||
<Skeleton className="h-4 w-full" />
|
||||
|
||||
// Error
|
||||
className={cn("border", { "border-destructive": hasError })}
|
||||
|
||||
// Active
|
||||
"comet-table-row-active"
|
||||
|
||||
// Disabled
|
||||
"disabled:opacity-50 disabled:pointer-events-none"
|
||||
```
|
||||
|
||||
## Spacing
|
||||
|
||||
- Gaps: `gap-2`, `gap-4`, `gap-6`, `gap-8`
|
||||
- Padding: `p-2`, `p-4`, `p-6`
|
||||
- Border radius: `rounded-md` (default), `rounded-lg`
|
||||
|
||||
## Component Placement
|
||||
|
||||
- **Reusable**: `shared/`
|
||||
- **Page-specific**: Same folder as parent component
|
||||
- **Low-level UI**: `ui/`
|
||||
@@ -0,0 +1,62 @@
|
||||
---
|
||||
name: opik-integrations
|
||||
description: Build, update, test, and document Opik SDK integrations (Python & TypeScript). Use when adding a new framework/provider integration under sdks/python/src/opik/integrations or sdks/typescript/src/opik/integrations, updating an existing one, or verifying that an integration logs traces correctly.
|
||||
---
|
||||
|
||||
# Opik SDK Integrations
|
||||
|
||||
This skill is for **building integrations into the Opik SDK itself** — the code that ships inside `opik` / `opik-*` packages so that *users* can trace a framework (OpenAI, LangChain, Mistral, …) with one call.
|
||||
|
||||
> Do not confuse this with the user-facing `instrument` / `opik` skills, which add Opik tracing to *someone else's* application. This skill is for SDK contributors editing `sdks/python` and `sdks/typescript`.
|
||||
>
|
||||
> If the integration lives **outside this repo** — a standalone `opik-*` package, or Opik support contributed into a third-party project (LiteLLM, Dify, a plugin, …) — use the **`opik-external-integrations`** skill instead. This skill assumes the code ships inside `sdks/`.
|
||||
|
||||
## Start with the questionnaire
|
||||
|
||||
Never assume or suggest a target. Collect, from the user, before doing anything: **what** to integrate (name + reference links), **where** it lives (this repo vs. external — route external requests to `opik-external-integrations`), **language** (python/typescript/both), **mode** (new/update/maintain), and any **specific flows** to cover. Do not present a menu of candidate libraries — the user names the target.
|
||||
|
||||
## When to use
|
||||
|
||||
- **New integration** — a framework/provider has no dedicated integration yet (today it's reachable only via LiteLLM, the OpenAI-compatible shim, or OpenTelemetry, or not at all).
|
||||
- **Update** — an integration must track new methods, capture new fields, or follow an upstream SDK change.
|
||||
- **Maintain / verify** — confirm an existing integration still logs the correct trace/span tree after a dependency bump or refactor.
|
||||
|
||||
## The workflow
|
||||
|
||||
Integration work is multi-step. By default this skill runs **autonomously**: it makes its own preparations (deps, credentials, backend), runs every phase, self-verifies, and ends with a high-level report — only stopping early on a true blocker. Ask for the interactive variant if you want to approve the design before any code is written. The full playbook — phases, execution modes, the Opik-MCP verification loop, and the report template — lives in **[workflow.md](workflow.md)**. At a glance:
|
||||
|
||||
0. **Prepare** — install/resolve the target library, locate credentials (without printing them), pick a backend the MCP can read.
|
||||
1. **Investigate** the target library (API surface, hooks/callbacks, streaming shape, usage/token format, errors).
|
||||
2. **Collect** findings + a minimal runnable example script.
|
||||
3. **Design** — pick the pattern, file layout, entrypoint. (Interactive mode pauses for approval here; autonomous mode records it in the report.)
|
||||
4. **Implement** by cloning the closest existing same-pattern integration.
|
||||
5. **Verify** the logged data through the Opik MCP (`read`/`list` the trace & spans).
|
||||
6. **Test** with the language's integration-test harness.
|
||||
7. **Document** the Fern page and wire its routing.
|
||||
8. **Report** — a high-level summary: what was done, what's supported (with evidence), what's not, and how to use it.
|
||||
|
||||
## Golden rule: clone the closest sibling
|
||||
|
||||
Never build an integration from a blank file. Identify the existing integration that shares the target's mechanism, copy its structure, and adapt. The decision tree:
|
||||
|
||||
| Target shape | Python pattern | TS pattern | Clone from |
|
||||
|---|---|---|---|
|
||||
| SDK client with methods to wrap (most providers) | Method patching (`BaseTrackDecorator` subclass) | Proxy wrapper | `openai/` · `opik-openai` |
|
||||
| Framework with a callback/tracer interface | Pure callback (`BaseTracer`) | Callback handler | `langchain/` · `opik-langchain` |
|
||||
| Framework already emitting OpenTelemetry spans | OTel | OTel exporter | `otel/` · `opik-vercel` |
|
||||
| Callbacks exist but are unreliable / need method hooks too | Hybrid | (rare) | `adk/` |
|
||||
|
||||
If the target exposes an **OpenAI-compatible endpoint**, first check whether `track_openai(..., provider=...)` already covers the need before building a dedicated integration — sometimes the right answer is a docs page, not new code.
|
||||
|
||||
**OpenTelemetry is backend-first.** If the target already emits OpenTelemetry spans, the heavy lifting is done by Opik's OTLP ingestion endpoint on the backend — many such integrations are *docs-only* (point the framework's OTLP exporter at Opik with auth headers; no SDK code). Build a client-side piece only when you must shape what the backend receives — set Opik semantics, remap attributes, or bridge a framework that won't export raw OTLP. The client-side building block is a `SpanProcessor` in Python (`integrations/otel/`) or a `SpanExporter` in TypeScript (`opik-vercel`); a framework-specific OTel tracer wrapper (`adk/patchers/adk_otel_tracer/`) is the heavier variant. See the OTel sections in [python.md](python.md) / [typescript.md](typescript.md).
|
||||
|
||||
## Language references
|
||||
|
||||
- **Python** → [python.md](python.md) — integration anatomy, shared core modules, mechanism templates, dependency/import rules, test specifics.
|
||||
- **TypeScript** → [typescript.md](typescript.md) — package anatomy, patterns, build/peer-dep rules. Delegates to the canonical `sdks/typescript/design/INTEGRATIONS.md`.
|
||||
|
||||
## Skills this one builds on (do not duplicate them)
|
||||
|
||||
- `python-sdk` — three-layer architecture, batching, `fake_backend`, `testlib` verifiers, error handling.
|
||||
- `typescript-sdk` — layered client, flush semantics, testing with vitest.
|
||||
- `write-docs` — Fern MDX authoring, routing YAML, callouts, images.
|
||||
@@ -0,0 +1,120 @@
|
||||
# Python Integration Anatomy
|
||||
|
||||
Reference for building integrations under `sdks/python/src/opik/integrations/<name>/`. Read this alongside the `python-sdk` skill (architecture, batching, testing fixtures).
|
||||
|
||||
## Directory skeleton
|
||||
|
||||
```
|
||||
sdks/python/src/opik/integrations/<name>/
|
||||
├── __init__.py # exports the public entrypoint, nothing else
|
||||
├── opik_tracker.py # the entrypoint: track_<name>() (patching) or the tracer class
|
||||
├── <name>_decorator.py # BaseTrackDecorator subclass(es) — patching integrations
|
||||
└── <name>_chunks_aggregator.py # merges streamed chunks into one response — if streaming
|
||||
```
|
||||
|
||||
Callback integrations (LangChain-style) replace the decorator file with an `opik_tracer.py` holding a `BaseTracer` subclass plus helpers (`run_parse_helpers.py`, provider usage extractors). Hybrid integrations (ADK) add a `patchers/` package and callback injectors.
|
||||
|
||||
## Mechanism templates — clone, don't invent
|
||||
|
||||
| Mechanism | When | Canonical template |
|
||||
|---|---|---|
|
||||
| **Method patching** | SDK client with methods to wrap (most providers) | `integrations/openai/` |
|
||||
| **Pure callback** | framework exposes a callback/tracer interface | `integrations/langchain/` |
|
||||
| **Hybrid** | callbacks unreliable / also need method hooks | `integrations/adk/` |
|
||||
| **OTel** | framework already emits OpenTelemetry spans | `integrations/otel/` |
|
||||
|
||||
### Method patching (the common case)
|
||||
|
||||
The entrypoint mutates the client in place and returns it. Study `openai/opik_tracker.py` — the shape is:
|
||||
|
||||
```python
|
||||
def track_<name>(client, project_name=None, provider=None):
|
||||
if hasattr(client, "opik_tracked"): # idempotency guard
|
||||
return client
|
||||
client.opik_tracked = True
|
||||
# resolve provider, then patch each method via a decorator factory
|
||||
_patch_<name>(client, provider, project_name)
|
||||
return client
|
||||
```
|
||||
|
||||
Each method is wrapped by a `BaseTrackDecorator` subclass (`<name>_decorator.py`). You implement two preprocessors; everything else (span/trace creation, context nesting, error capture, generator handling) comes from the base class. Read `openai/openai_chat_completions_decorator.py` as the template:
|
||||
|
||||
- `_start_span_inputs_preprocessor(...)` → returns `StartSpanParameters` (type, name, input, metadata, tags, model, provider).
|
||||
- `_end_span_inputs_preprocessor(...)` → returns `EndSpanParameters` (output, usage, model, provider, metadata).
|
||||
- Streaming: pass a `generations_aggregator` to `.track(...)` and override the stream handler so chunks are accumulated and the span finalizes after iteration. Sync iterators, async iterators, and stream context managers each need patching — see `openai/stream_patchers.py`.
|
||||
|
||||
Wrapped calls check `opik.is_tracing_active()` at call time and no-op the telemetry if tracing is off, while still running the underlying call.
|
||||
|
||||
**Delegating methods (patch only the primitive).** When a higher-level method calls a lower-level one you patch (Mistral's `chat.parse` → `chat.complete`, `parse_stream` → `stream`; openai's older `beta…stream` → `create`), do **not** patch both — that produces two spans and double-counts cost. Patch **only the primitive**; the delegating call is then traced through it as one span. Name the span after the primitive (`chat.complete` → `chat_completion_create`, `chat.stream` → `chat_completion_stream`) via `track_options.name` / `func.__name__` — this is unambiguous and always correct. The delegating call (e.g. `parse`) shares that name; the structured JSON is still in the output content, only the deserialized `.parsed` field is absent. Uses only the existing `track()` API — no reentrancy flags, no `contextvars`, no `set_tracing_active` (process-wide and racy). Verify in Phase 5 that the delegating call yields a single span with un-doubled cost.
|
||||
|
||||
> **Do not rename the span from a kwarg unless that kwarg faithfully identifies the mode for *every* caller of the primitive.** openai's `if kwargs.get("stream") is True: name = "chat_completion_stream"` is safe because `stream=True` on `create` always means streaming. The same trick with Mistral's `response_format` is **wrong**: a direct `chat.complete(response_format=…)` is a legitimate structured-output call, not a `parse`, so keying the name off `response_format` misclassifies it (real review finding on the Mistral PR). When the discriminator can't distinguish the delegating call from a direct call to the same primitive, don't rename — keep the primitive's name.
|
||||
|
||||
**Env var mismatch.** A provider SDK's client may not auto-read its own API-key env var (e.g. `mistralai.Mistral()` ignores `MISTRAL_API_KEY`). In tests and examples, pass the key explicitly — `Mistral(api_key=os.environ["MISTRAL_API_KEY"])` — rather than relying on the bare constructor.
|
||||
|
||||
### OpenTelemetry (backend-first)
|
||||
|
||||
When the target already emits OpenTelemetry, most of the work lives on the **backend**: Opik exposes an OTLP ingestion endpoint that maps OTel spans to Opik traces. Many OTel "integrations" are therefore *docs-only* — the user points their framework's OTLP exporter at Opik with auth headers and writes no SDK code. Always check whether that covers the need before writing code.
|
||||
|
||||
Add client-side code only when you must shape what the backend receives (set Opik semantics, remap attributes, bridge a framework that won't emit raw OTLP). The building block is `integrations/otel/`:
|
||||
|
||||
- `OpikSpanProcessor` — an OTel `SpanProcessor` the user registers on their `TracerProvider` to forward/annotate spans.
|
||||
- distributed-trace helpers — `attach_to_parent`, `extract_opik_distributed_trace_attributes`.
|
||||
|
||||
A framework-specific OTel **tracer wrapper** (see `adk/patchers/adk_otel_tracer/`) is the heavier variant, for intercepting a framework's own tracer rather than a generic processor.
|
||||
|
||||
## Shared core — never re-derive these
|
||||
|
||||
| Concern | Module |
|
||||
|---|---|
|
||||
| Decorator base class | `opik/decorator/base_track_decorator.py` |
|
||||
| Span/trace creation respecting context | `opik/decorator/span_creation_handler.py` |
|
||||
| Start/End span dataclasses | `opik/decorator/arguments_helpers.py` |
|
||||
| Error capture (`exception_type`, `traceback`) | `opik/decorator/error_info_collector.py` |
|
||||
| Generator/stream wrapping | `opik/decorator/generator_wrappers.py` |
|
||||
| Token usage normalization | `opik/llm_usage.py` → `try_build_opik_usage_or_log_error(provider=..., usage=...)` |
|
||||
| Recognized providers (cost tracking) | `opik/types.py` → `LLMProvider` |
|
||||
| Global client | `opik/api_objects/opik_client.py` → `get_global_client()` |
|
||||
| Context stack | `opik/context_storage.py` |
|
||||
|
||||
For callback integrations, span/trace creation goes through `span_creation_handler` too; map each framework run id → `SpanData`/`TraceData` and set `metadata["created_from"] = "<name>"`.
|
||||
|
||||
**Add a dedicated usage format — don't piggyback on another provider's parser.** Even when a provider's token-usage payload looks OpenAI-shaped, give it its own format in the `llm_usage` namespace rather than passing `provider=LLMProvider.OPENAI` to `try_build_opik_usage_or_log_error`. Reusing another provider's converter couples you to *its* schema changes and misattributes the parsed shape. The steps (see the Mistral change for the worked example): (1) add `llm_usage/<name>_usage.py` with a `class <Name>Usage(BaseOriginalProviderUsage)` declaring the token fields (+ nested details classes) and `from_original_usage_dict`; (2) add it to the `ProviderUsage` union and a `OpikUsage.from_<name>_dict` classmethod in `llm_usage/opik_usage.py`; (3) register `LLMProvider.<NAME>: [OpikUsage.from_<name>_dict]` in `llm_usage/opik_usage_factory.py`; (4) in the decorator, parse with `provider=LLMProvider.<NAME>`. Non-int fields (e.g. `prompt_audio_seconds`) are dropped from the backend flat dict automatically. Add unit tests under `tests/unit/llm_usage/test_<name>_usage.py` (parser + `build_opik_usage` factory path).
|
||||
|
||||
## Dependencies & imports
|
||||
|
||||
- The framework library is **imported inside the integration module** (`import mistralai` at the top of `opik_tracker.py`). That is safe because the module is only reached when a user does `from opik.integrations.<name> import ...`. Never import an integration from the `opik` package top level.
|
||||
- Do **not** add the framework to `install_requires` unless it is already a core dependency (openai and litellm are; most are not). Users install the framework themselves.
|
||||
- `integrations/<name>/__init__.py` exports only the public entrypoint.
|
||||
- If the integration must support multiple incompatible framework versions, branch at import time on `opik.semantic_version` (see `adk/__init__.py`).
|
||||
- **Import only the framework's public API — never reach into internal modules.** Private paths (`mistralai.utils.eventstreaming.EventStream`, `mistralai.models.chatcompletionresponse.…`) move between releases and break the integration. Prefer top-level exports (`from mistralai import Mistral, ChatCompletionResponse`); check what's public with `hasattr(pkg, name)`. When you need a class that *isn't* exported (e.g. the stream type), don't import it — detect the returned object by its **protocol** (`hasattr(output, "__anext__")` → async stream, `hasattr(output, "__next__")` → sync stream; a pydantic response model has `__iter__` but neither) and operate on `type(output)` when you must patch its dunder methods. Real review finding on the Mistral PR.
|
||||
- **Guard the minimum framework version.** Pin the floor in the test `requirements.txt` (`<lib>>=X,<next-major`), and add a runtime check in `track_<name>()` that raises a clear error — `f"Opik supports <lib>>=X, but {installed} is installed…"` — so users on an incompatible version get a message, not a cryptic `AttributeError`/`ImportError`. Read the installed version with `importlib.metadata.version("<lib>")` (robust: some versions don't expose `<lib>.__version__`) and compare via `opik.semantic_version.SemanticVersion.parse(...)`. Pick the floor empirically — bisect installs to the release where the public API you depend on first appears (for Mistral, `EventStream` landed in 1.3.0).
|
||||
|
||||
## Tests
|
||||
|
||||
Location: `sdks/python/tests/library_integration/<name>/`. Follow the `python-sdk` testing skill for fixtures; the integration-specific shape:
|
||||
|
||||
```python
|
||||
def test_<name>_<method>__happyflow(fake_backend):
|
||||
client = track_<name>(<lib>.Client())
|
||||
response = client.<method>(...) # real call, gated by env fixture
|
||||
opik.flush_tracker()
|
||||
|
||||
EXPECTED = TraceModel(
|
||||
id=ANY_BUT_NONE, name="...", input=ANY_DICT.containing({...}),
|
||||
output=ANY_BUT_NONE, tags=["<name>"], spans=[
|
||||
SpanModel(
|
||||
id=ANY_BUT_NONE, type="llm", name="...",
|
||||
usage=..., model=ANY_STRING.starting_with("..."),
|
||||
provider="<name>", spans=[],
|
||||
)
|
||||
],
|
||||
)
|
||||
assert len(fake_backend.trace_trees) == 1
|
||||
assert_equal(EXPECTED, fake_backend.trace_trees[0])
|
||||
```
|
||||
|
||||
- Real API calls are gated by an `ensure_<name>_configured` fixture (skip if the key is missing) — add one to `conftest.py` mirroring `ensure_openai_configured`.
|
||||
- Cover: happy flow, streaming, `parse`/structured-output (+ its single-span/no-double-cost assertion), custom `provider`, nested-under-`@track`, and an error case.
|
||||
- Centralize the model id in `tests/llm_constants.py`; add a `requirements.txt` in the test dir with the framework package.
|
||||
- Imports: `from tests.testlib import TraceModel, SpanModel, ANY_BUT_NONE, ANY_DICT, ANY_STRING, assert_equal`.
|
||||
- **Wire the tests into CI** — create `.github/workflows/lib-<name>-tests.yml` (single Python version unless asked otherwise) and register it in `lib-integration-tests-runner.yml`. See Phase 6 of [workflow.md](workflow.md); unregistered tests never run.
|
||||
@@ -0,0 +1,52 @@
|
||||
# TypeScript Integration Anatomy
|
||||
|
||||
Reference for building integrations under `sdks/typescript/src/opik/integrations/opik-<name>/`. Read this alongside the `typescript-sdk` skill (layered client, flush semantics, testing).
|
||||
|
||||
## The canonical guide comes first
|
||||
|
||||
`sdks/typescript/design/INTEGRATIONS.md` is the authoritative, maintained guide: it has the three patterns with full code, the package structure, a step-by-step "Creating New Integrations" walkthrough, and streaming helpers. **Read it before writing anything.** This file only records the conventions that guide does not stress and that bite newcomers.
|
||||
|
||||
Companion design docs: `design/API_AND_DATA_FLOW.md` (client/batching/context), `design/TESTING.md` (vitest + MSW), `design/README.md` (navigation).
|
||||
|
||||
## Each integration is its own npm package
|
||||
|
||||
Unlike Python (one package, many integration modules), every TS integration is a **separate published package** (`opik-openai`, `opik-langchain`, `opik-vercel`, `opik-gemini`). That means:
|
||||
|
||||
- Its own `package.json`, `tsconfig.json`, `tsup.config.ts`, `vitest.config.ts`, `README.md`.
|
||||
- `opik` and the wrapped framework are **`peerDependencies`**, not `dependencies` — pin sane ranges and keep the `opik` range current.
|
||||
- Dual **CJS + ESM** build via `tsup`, emitting `dist/index.js` (ESM), `dist/index.cjs` (CJS), `dist/index.d.ts`.
|
||||
- A lazy `OpikClient` singleton (`singleton.ts`) so the integration doesn't spin up multiple clients.
|
||||
|
||||
Clone the closest sibling package wholesale (its configs are correct) and rename, rather than hand-writing build config.
|
||||
|
||||
## Patterns
|
||||
|
||||
| Pattern | When | Template package |
|
||||
|---|---|---|
|
||||
| **Proxy wrapper** | SDK client with methods (most providers) | `opik-openai`, `opik-gemini` |
|
||||
| **Callback handler** | framework with a callback interface | `opik-langchain` |
|
||||
| **OTel exporter** | framework emitting OpenTelemetry spans | `opik-vercel` |
|
||||
|
||||
See `design/INTEGRATIONS.md` for each pattern's code. Shared imports come from `opik`: `Opik`, `Trace`, `Span`, `OpikSpanType`, `generateId`, `logger`.
|
||||
|
||||
**OpenTelemetry is backend-first.** Opik's backend ingests OTLP directly and maps spans to traces, so an OTel-instrumented framework can often be integrated with **docs alone** (point its exporter at Opik). Write a client-side `SpanExporter` — the `opik-vercel` pattern — only when you need to remap attributes/semantics, accumulate spans into trace trees, or bridge a framework that won't emit raw OTLP. Check the docs-only path first.
|
||||
|
||||
## Conventions that bite
|
||||
|
||||
- **Never leak `rest_api`.** Integrations wrap the public `opik` API only — never import generated clients from `opik/rest_api`.
|
||||
- **Provide a `flush()` escape hatch.** Proxy wrappers expose `flush` on the returned object; handlers/exporters expose a `flush()` method. Required for CLIs/tests where the process exits before the async queue drains.
|
||||
- **Keep adapters thin and non-blocking** — domain objects enqueue, they don't do HTTP.
|
||||
- **Version references in prose.** When you change a peer-dep range or minimum version, update the integration's `README.md` and the root `README.md` in the same change — the `typescript-sdk` skill calls this out as a recurring miss.
|
||||
|
||||
## Tests
|
||||
|
||||
`tests/*.test.ts` with **vitest**, per `design/TESTING.md` and the `typescript-sdk` testing skill. Integration-specific shape:
|
||||
|
||||
- Mock the **API layer** (`vi.spyOn(client.api.spans, "createSpans")`) or use **MSW** to intercept HTTP — mirror the sibling package's `mockUtils.ts`.
|
||||
- `await client.flush()` (or advance fake timers) **before** asserting; data only persists after the queue drains.
|
||||
- Assert on captured spans with `toMatchObject({...})` — name, input, output, parentSpanId hierarchy, usage, model.
|
||||
- Keep `tests/setup.ts` (it disables the logger). Don't let ERROR logs leak in tests.
|
||||
|
||||
## Examples
|
||||
|
||||
Add a runnable example to `sdks/typescript/examples/` mirroring `track-decorator.ts` / `langchain-with-thread-id.ts`. This doubles as the script you run during Phase 5 MCP verification.
|
||||
@@ -0,0 +1,183 @@
|
||||
# Integration Workflow
|
||||
|
||||
The end-to-end playbook for building or updating an Opik SDK integration. Run the phases in order.
|
||||
|
||||
Pick the language reference up front — [python.md](python.md) or [typescript.md](typescript.md) — and keep it open throughout.
|
||||
|
||||
## Execution modes
|
||||
|
||||
- **Autonomous (default for this command)** — make every preparation yourself (install deps, find credentials, pick the backend), run all phases without stopping, self-verify, and finish with the **Phase 8 report**. Do not pause at the design gate; record the design in the report instead. Only stop early if you hit a true blocker you cannot resolve (missing API key with no fallback, an ambiguous product decision, a backend you cannot reach) — report what you completed and what's blocked.
|
||||
- **Interactive** — same phases, but pause at Phase 3 to get the design approved before writing code. Use this when the user asks to review the plan first.
|
||||
|
||||
Either way, **always end with the high-level report** (Phase 8). Never invent results — every "supported" claim must be backed by a passing test or an MCP-verified trace; everything unverified goes under "not supported / not verified".
|
||||
|
||||
## Phase 0 — Classify
|
||||
|
||||
- **Mode**: `new` (no integration exists), `update` (extend an existing one), or `maintain` (verify/repair an existing one). Maintain mode skips Phases 2–4 and runs Investigate → Verify → Test against current code to catch drift.
|
||||
- **Language**: Python, TypeScript, or both. They are independent code paths; if both are requested, run the whole workflow once per language.
|
||||
- **Closest sibling**: name the existing integration you will clone (see the decision tree in [SKILL.md](SKILL.md)).
|
||||
|
||||
## Phase 0.5 — Prepare (autonomous)
|
||||
|
||||
Get the environment ready yourself before investigating. Record what you did for the report; never print secret values.
|
||||
|
||||
- **Install / resolve the target library** into the SDK venv (`sdks/python/.venv` for Python; `npm i` in the integration package for TS). Capture the resolved version. If the install is broken or pulls an incompatible core dep (e.g. it bumps `pydantic` and breaks `litellm`), pin to a known-good version and restore the disturbed dep.
|
||||
- **Locate credentials** without printing them: Python tests read the env block in `sdks/python/tests/pytest.ini` (e.g. `MISTRALAI_API_KEY`); also check the shell env. If the library's own default env var differs from the test var (e.g. SDK wants `MISTRAL_API_KEY` but the test sets `MISTRALAI_API_KEY`), read the test var explicitly and pass it to the client.
|
||||
- **Pick the verification backend**: the connected Opik MCP reads one backend — confirm which (list its projects) and that it's reachable. Log the verification run there so the MCP can read it back. Use a dedicated scratch `OPIK_PROJECT_NAME` (e.g. `<name>-integration-demo`).
|
||||
- **Note blockers**: if a credential or backend is missing and has no fallback, record it — the run can still produce code + offline (`fake_backend`) tests, with live MCP verification marked "not verified".
|
||||
|
||||
## Phase 1 — Investigate
|
||||
|
||||
Understand the target library before touching Opik. Prefer fanning out parallel explore agents over the library's installed source / docs.
|
||||
|
||||
Answer all of these:
|
||||
|
||||
- **Entrypoint shape** — Is it a client object with methods (→ patching/proxy), a callback/tracer interface (→ callback), or already OpenTelemetry-instrumented? If it emits OTel, first decide whether a **docs-only OTLP-to-Opik** setup suffices (backend does the mapping) before committing to a client-side processor/exporter — see the OTel notes in the language reference.
|
||||
- **Methods to trace** — the specific calls users invoke (e.g. `chat.complete`, `embed`, `rerank`). List sync *and* async variants, and **treat structured-output methods (`parse`, `response_format=…`) and their streaming variants as in-scope by default** — don't defer them to "follow-ups" unless the user says so. Enumerate the full cross-product up front (complete/parse × sync/async × stream/non-stream).
|
||||
- **Streaming** — does a streamed call return an iterator/async-iterator, a context manager, or emit events? How are chunks shaped, and where does usage land (final chunk? separate event?)?
|
||||
- **Input** — the request fields worth logging (messages, prompt, tools/functions, model params).
|
||||
- **Output** — where the completion text / choices / tool calls live in the response object.
|
||||
- **Usage** — the token-count field names, and whether the provider is one Opik recognizes for cost tracking (`opik.types.LLMProvider`).
|
||||
- **Errors** — what exceptions the library raises.
|
||||
|
||||
Read the closest sibling integration in full — it is the template, and most decisions are already made there.
|
||||
|
||||
**Clone-ability checkpoint (do this before designing).** "Clone the closest sibling" only holds if the target is actually a clean analog. Before committing, confirm it — and if any of these are true, surface it as a design decision **even in autonomous mode** instead of silently picking:
|
||||
|
||||
- The library exposes **multiple incompatible client classes or versions** (e.g. a v1 `Client` and a v2 `ClientV2` with different method signatures and response shapes). Decide which to support, and whether both are in scope.
|
||||
- The **usage/token shape differs** from the sibling's (so the sibling's usage parser won't just work and you need custom mapping).
|
||||
- The **response/streaming shape** is materially different from the sibling's.
|
||||
- **Methods delegate to each other.** Check whether a higher-level method calls a lower-level one you also patch (e.g. Mistral's `chat.parse` calls `chat.complete`; some SDKs' `stream` calls `create`). Patching both naively **double-logs the call and double-counts cost**. Read the delegating method's source to find this. The idiomatic fix is to patch **only the primitive** and name the span after that primitive (see [python.md](python.md)) — verify the span count and cost in Phase 5.
|
||||
|
||||
A target that looks like "just another provider" but has any of the above is *not* a clean clone — say so before writing code.
|
||||
|
||||
## Phase 2 — Collect
|
||||
|
||||
Produce two artifacts in the scratchpad before designing:
|
||||
|
||||
1. **A minimal runnable example script** that exercises the target library directly (no Opik yet) — one non-streaming call, one streaming call, and any second method you plan to trace. This is what you'll later run in Phase 5 to verify logging.
|
||||
2. **A findings note** — a short mapping table:
|
||||
|
||||
| Opik span field | Source in the library's request/response |
|
||||
|---|---|
|
||||
| input | … |
|
||||
| output | … |
|
||||
| usage | … |
|
||||
| model | … |
|
||||
| provider | … |
|
||||
| span type | `llm` / `tool` / `general` |
|
||||
|
||||
## Phase 3 — Design
|
||||
|
||||
Decide and write down:
|
||||
|
||||
- The **pattern** (from the decision tree) and **why** it fits.
|
||||
- The **file layout** (use the skeleton in the language reference).
|
||||
- The **public entrypoint** name and signature — match sibling conventions (`track_<name>` for Python patching; `trackXxx` / `XxxCallbackHandler` / `XxxExporter` for TS).
|
||||
- The methods to patch/handle and the field mapping from Phase 2.
|
||||
|
||||
In **interactive mode**, present this and wait for approval before writing code. In **autonomous mode**, proceed — but capture the design and any open questions (e.g. dedicated integration vs. an OpenAI-compatible docs page) in the Phase 8 report so the reviewer sees the decisions.
|
||||
|
||||
## Phase 4 — Implement
|
||||
|
||||
- Copy the closest sibling's structure and adapt it. Reuse the shared core utilities listed in the language reference — never re-derive span/trace creation, error collection, or usage parsing.
|
||||
- Keep the framework import inside the integration module (Python) or as a `peerDependency` (TS); never import it at SDK package top level.
|
||||
- Preserve the library's original behavior: wrap, capture, re-raise. The integration must be transparent when tracing is disabled.
|
||||
|
||||
## Phase 5 — Verify via Opik MCP
|
||||
|
||||
This is the proof that the integration actually logs correctly. Do not rely on reading code.
|
||||
|
||||
**Verifiability is a hard gate for `new`/`update`.** If you cannot run this phase — no API credential for the target, or no reachable backend the MCP can read — **stop and surface it before writing integration code**, in autonomous mode too. Do not produce an integration and then present it as done with verification "skipped": unverified integration code is the exact failure this skill exists to prevent. Offer the user the choice to supply a credential, proceed explicitly-unverified (clearly labelled, tests key-gated and skipped), or pick a different target. Only `maintain` mode on already-passing code may relax this.
|
||||
|
||||
1. **Point at a backend the MCP can read — and confirm it.** The connected Opik MCP reads one specific backend/workspace, which is often **not** the one in `~/.opik.config` (e.g. MCP → a hosted `*.dev.comet.com`, local config → `localhost`). Before relying on the MCP, confirm they match: log a trace, then try to `read` its project through the MCP. If the MCP can't see it, the backends differ. Either reconfigure the script's env (`OPIK_API_KEY`, `OPIK_URL_OVERRIDE`/base URL, `OPIK_WORKSPACE`, `OPIK_PROJECT_NAME`) to log into the MCP's backend, **or** fall back to **SDK read-back** (next note). Don't silently assume the MCP sees your trace.
|
||||
2. **Run the Phase 2 example** through the new integration (wrap the client / attach the handler), exercising the non-streaming call, the streaming call, and each extra method. Call `flush()` before exit.
|
||||
3. **Read it back through the MCP** — use the opik-mcp `list` (entity_type `trace`, filtered by the project) then `read` (entity_type `trace`, which inlines spans; `read` `span` for detail).
|
||||
4. **Check the trace/span tree against the Phase 2 mapping:**
|
||||
- one trace per top-level call; span hierarchy matches the call structure
|
||||
- span `type` is correct (`llm` for model calls)
|
||||
- `input` / `output` captured and well-shaped (not empty, not the raw object dump)
|
||||
- `usage` present with prompt/completion/total tokens
|
||||
- `model` and `provider` set correctly
|
||||
- streamed calls produce the same shape as non-streamed (aggregated output + usage)
|
||||
- an induced error records `error_info` and still re-raises
|
||||
- **no duplicate spans / double cost** — a call to a delegating method (e.g. `parse`) produces exactly one span, and its `total_estimated_cost` is not doubled
|
||||
- **nesting** — a traced call made inside an `@track` function attaches as a child span of that function's span
|
||||
5. **Loop** until the logged data matches. Fix the integration, re-run, re-read.
|
||||
|
||||
**SDK read-back fallback (equivalent evidence).** When the MCP can't read the backend you can write to, verify against that backend over REST instead: `client = opik.Opik(); client.search_traces(project_name=...)` then `client.search_spans(trace_id=...)`, and assert the same checklist (type, input/output, usage, model, provider). This is a real backend round-trip — note in the report that read-back was via the SDK, not the MCP tool, and why.
|
||||
|
||||
## Phase 6 — Test
|
||||
|
||||
Add coverage with the language's harness — see the test section of the language reference, which delegates to the `python-sdk` / `typescript-sdk` testing skills.
|
||||
|
||||
- **Python**: `sdks/python/tests/library_integration/<name>/`, using `fake_backend` and `testlib` `TraceModel`/`SpanModel` trees with `ANY_*` matchers. Assert input/output/usage/model/provider. Gate real API calls behind an `ensure_<name>_configured` fixture. Name tests `test_<what>__<case>__<expected>`. Cover every enumerated flow — including `parse`/structured-output variants, the delegating-method single-span case, and one **nested-under-`@track`** case (asserts the LLM span attaches as a child).
|
||||
- **TypeScript**: `*.test.ts` with vitest, mocking the API layer (or MSW), `await flush()` before asserting, fake timers for batching. Mirror the sibling integration's test file.
|
||||
|
||||
**Register the tests in CI (Python) — this is part of "done", not optional.** The `tests/library_integration/<name>/` files run only if wired into GitHub Actions:
|
||||
|
||||
1. Create `.github/workflows/lib-<name>-tests.yml` by cloning the closest sibling (e.g. `lib-anthropic-tests.yml`): set the provider's API-key env from `secrets.<PROVIDER>_API_KEY`, install `library_integration/<name>/requirements.txt`, run `pytest -vv .` in the test dir. **Unless asked otherwise, pin a single Python version** (`matrix.python_version: ["3.12"]`) instead of the full `PYTHON_VERSIONS` matrix.
|
||||
2. Register it in `.github/workflows/lib-integration-tests-runner.yml`: add `<name>` to the `libs` `workflow_dispatch` choices, add a `<name>_tests` job (`if: contains(fromJSON('["<name>", "all"]'), …)` + `uses:` the new file + `secrets: inherit`), and add it to the `notify-slack` job's `needs` list and `SUITE_RESULTS` payload.
|
||||
3. Flag that the run needs a `<PROVIDER>_API_KEY` repository secret to exist, and validate both YAML files parse.
|
||||
|
||||
## Phase 7 — Document
|
||||
|
||||
Author the Fern page following the `write-docs` skill for MDX/components, plus these integration-specific conventions:
|
||||
|
||||
- **Check for an existing page first.** A provider often already has a docs page describing a *workaround* (OpenAI-compatibility endpoint via `track_openai`, or LiteLLM) and an entry already in `fern/versions/latest.yml`. If so, this is an **update**: lead the page with the new native integration and demote the workaround to an "Alternative" section (see how `mistral.mdx` keeps LiteLLM). Don't create a duplicate page or a second routing entry.
|
||||
- **File**: `apps/opik-documentation/documentation/fern/docs-v2/integrations/<name>.mdx` for Python, `<name>-typescript.mdx` for TypeScript.
|
||||
- **Title frontmatter** distinguishes language: `Observability for <Lib> (Python) with Opik` vs `(TypeScript)`.
|
||||
- **Page shape** (follow `openai.mdx` / `langchain.mdx`): intro/tips → account setup → getting started (install, configure Opik, configure the library) → basic usage (the wrap/handler call + a screenshot) → advanced usage → cost tracking → supported methods.
|
||||
- **Routing**: add a `- page:` entry under the right language → category section (`Frameworks`, `Model Providers`, …) in `fern/versions/latest.yml`. Do not edit `docs.yml`.
|
||||
- **Overview grid**: add a `<Card>` to `docs-v2/integrations/overview.mdx` under the matching section. Cards are title + href only; section icons are Font Awesome — there is no per-integration icon to create.
|
||||
- Use credential placeholders only (`<API_KEY>`), never real keys.
|
||||
|
||||
## Phase 8 — Report (always)
|
||||
|
||||
End every run with a high-level report. Keep it scannable — it's for a reviewer deciding whether to ship, not a changelog. Use this template:
|
||||
|
||||
```markdown
|
||||
## <Library> integration — report
|
||||
|
||||
**Mode:** new | update | maintain **Language:** Python | TypeScript
|
||||
**Pattern:** method-patching | proxy | callback | OTel exporter **Entrypoint:** `track_<name>(...)`
|
||||
**Library version prepared:** <name>==<version>
|
||||
|
||||
### What was done
|
||||
- <files created/changed — bullets, grouped by integration / tests / docs>
|
||||
- <prep actions: deps installed, version pins, fixtures/env added>
|
||||
|
||||
### Verification
|
||||
- **MCP:** <project name + trace ids read back, or "not verified — <reason>">
|
||||
- **Tests:** <N passing — list cases>; ruff/mypy <clean | issues>
|
||||
|
||||
### Flows supported & test coverage
|
||||
|
||||
Always enumerate **every user-facing flow** the integration handles and map each to its coverage — don't collapse this into a one-line "supported". A flow is a distinct way a user invokes the library; enumerate the cross-product that applies: each traced method, sync vs async, streaming vs non-streaming, nested under `@track`, the error path, and any option that changes behavior (custom `project_name`, `provider` override, tool/function calls, structured output). For each flow state whether it's implemented, which test covers it (by name), and whether it was MCP-verified.
|
||||
|
||||
| Flow | Implemented | Test | MCP-verified |
|
||||
|---|---|---|---|
|
||||
| `chat.complete` (sync, non-stream) | ✅ | `test_<name>_complete__happyflow` | ✅ trace `<id>` |
|
||||
| `chat.complete_async` | ✅ | `test_<name>_complete_async__happyflow` | — |
|
||||
| `chat.stream` (sync) | ✅ | `test_<name>_stream__happyflow` | ✅ trace `<id>` |
|
||||
| `chat.stream_async` | ✅/❌ | … | … |
|
||||
| nested under `@track` | ✅/❌ | … | … |
|
||||
| error → `error_info` logged | ✅ | `test_<name>__error...` | — |
|
||||
| token usage captured | ✅ | asserted in above | ✅ |
|
||||
| custom `project_name` | ✅/❌ | param case | — |
|
||||
|
||||
Explicitly flag the gaps: any flow **implemented but not covered by a test**, and any flow **not implemented at all** (list it in the next section). The goal is that a reviewer can see, per flow, exactly what was proven.
|
||||
|
||||
### What's NOT supported / limitations
|
||||
- <methods intentionally not patched, flows implemented-but-untested, known gaps, provider-not-in-LLMProvider caveats, env/backend blockers>
|
||||
|
||||
### Follow-ups
|
||||
- <suggested next steps: more methods, TS counterpart, PR split, etc.>
|
||||
|
||||
### How to use
|
||||
<minimal code snippet>
|
||||
```
|
||||
|
||||
## Definition of done
|
||||
|
||||
Implementation merged-quality, MCP verification passed (Phase 5 checklist) or its absence reported, tests added and passing **and registered in CI** (Python: `lib-<name>-tests.yml` + runner wiring), docs page authored and routed, and the **Phase 8 report delivered**. Then run `make claude` if you added/edited files under `.agents/`.
|
||||
@@ -0,0 +1,281 @@
|
||||
---
|
||||
name: playwright-pom-discovery
|
||||
description: Use when building or extending a Page Object Model (POM) for the Opik E2E suite (under `tests_end_to_end/e2e/pom/`) and you need to choose stable selectors against the live UI. Walks through seeding required state, exploring the running page with the Playwright MCP (accessibility snapshot + data-testid enumeration), picking the most stable locator for each element, and verifying it before committing. Used as the discovery sub-step by the `writing-e2e-tests` skill.
|
||||
---
|
||||
|
||||
# Playwright POM Discovery
|
||||
|
||||
This skill is the **how** of choosing selectors for a POM in the Opik E2E suite. You already know which POM you're building; this skill tells you how to figure out what's on the page, what selectors are stable, and how to verify your method actually works before checking it in.
|
||||
|
||||
**Announce at start:** "I'm using the playwright-pom-discovery skill to build the X page object."
|
||||
|
||||
## When this skill applies
|
||||
|
||||
- You're touching anything under `tests_end_to_end/e2e/pom/`.
|
||||
- You need to write a `data-testid` selector, a `getByRole`, or any other Playwright locator targeting the live Opik UI.
|
||||
- You're adding a method to an existing POM that interacts with a new element you haven't seen before.
|
||||
|
||||
It does NOT apply to:
|
||||
|
||||
- Pure fixture work (no UI interaction).
|
||||
- Matchers / type-only changes.
|
||||
|
||||
## The procedure
|
||||
|
||||
```dot
|
||||
digraph pom_discovery {
|
||||
rankdir=TB;
|
||||
"Identify target page + entity preconditions" [shape=box];
|
||||
"Seed required state (project, dataset, trace, etc.)?" [shape=diamond];
|
||||
"Create state via SDK / bridge before opening UI" [shape=box];
|
||||
"Open page via browser_navigate (with auth)" [shape=box];
|
||||
"browser_snapshot to get accessibility tree" [shape=box];
|
||||
"browser_evaluate to enumerate data-testids" [shape=box];
|
||||
"Pick selector preference (testid > role > label > css)" [shape=box];
|
||||
"Write POM method" [shape=box];
|
||||
"Verify via browser_generate_locator and test_run" [shape=box];
|
||||
"Method works?" [shape=diamond];
|
||||
"Selector unstable or missing?" [shape=diamond];
|
||||
"Flag for FE data-testid addition" [shape=box];
|
||||
"Commit POM method" [shape=box];
|
||||
|
||||
"Identify target page + entity preconditions" -> "Seed required state (project, dataset, trace, etc.)?";
|
||||
"Seed required state (project, dataset, trace, etc.)?" -> "Create state via SDK / bridge before opening UI" [label="yes"];
|
||||
"Create state via SDK / bridge before opening UI" -> "Open page via browser_navigate (with auth)";
|
||||
"Seed required state (project, dataset, trace, etc.)?" -> "Open page via browser_navigate (with auth)" [label="no — stateless page"];
|
||||
"Open page via browser_navigate (with auth)" -> "browser_snapshot to get accessibility tree";
|
||||
"browser_snapshot to get accessibility tree" -> "browser_evaluate to enumerate data-testids";
|
||||
"browser_evaluate to enumerate data-testids" -> "Pick selector preference (testid > role > label > css)";
|
||||
"Pick selector preference (testid > role > label > css)" -> "Write POM method";
|
||||
"Write POM method" -> "Verify via browser_generate_locator and test_run";
|
||||
"Verify via browser_generate_locator and test_run" -> "Method works?";
|
||||
"Method works?" -> "Commit POM method" [label="yes"];
|
||||
"Method works?" -> "Selector unstable or missing?" [label="no"];
|
||||
"Selector unstable or missing?" -> "Flag for FE data-testid addition" [label="yes"];
|
||||
"Selector unstable or missing?" -> "browser_snapshot to get accessibility tree" [label="no, try different selector"];
|
||||
"Flag for FE data-testid addition" -> "Pick selector preference (testid > role > label > css)";
|
||||
}
|
||||
```
|
||||
|
||||
## Step-by-step
|
||||
|
||||
### 1. Identify target page + preconditions
|
||||
|
||||
Before opening the browser, answer in writing:
|
||||
|
||||
- **What page** does the POM model? E.g., `LogsPage` models `/<workspace>/projects/<projectId>/logs`. Get the route from the FE source under `apps/opik-frontend/src/v2/pages/` — each page has a directory matching its name.
|
||||
- **What entities must exist** for the page to show real data? Examples:
|
||||
- `LogsPage` — needs a project AND at least one trace under it. An empty project shows only the empty state.
|
||||
- `DatasetItemsPage` — needs a dataset AND at least one item. Empty datasets show only the "Create item" CTA.
|
||||
- `TestSuitesPage` — needs at least one test suite to show row interactions.
|
||||
- `OnlineEvaluationPage` — works empty, but creating a rule shows the rule list.
|
||||
- **What auth state** does the page need? The browser session needs the workspace authenticated. See the auth setup below.
|
||||
|
||||
If the page genuinely is stateless (e.g., an empty list page), skip seeding and go straight to step 3. If it's not, seed first — exploring an empty-state UI will lead you to write a POM that only ever sees the empty state.
|
||||
|
||||
### 2. Seed required state via SDK / bridge
|
||||
|
||||
**Never** click-create state through the UI just to populate the page you're exploring. Two reasons:
|
||||
|
||||
1. The page you're exploring may itself BE the UI-create flow you're trying to model — you'd be using the UI to test the UI.
|
||||
2. UI-create is slower and flakier than SDK-create.
|
||||
|
||||
Instead, use the bridge or the TS SDK to seed before opening the browser. For the discovery phase, a one-off seed script is fine — you don't need to wire it into the test suite yet (the test's fixture handles that later).
|
||||
|
||||
**Example seed for `LogsPage` discovery:**
|
||||
|
||||
```ts
|
||||
// scratch script, not committed
|
||||
import { Opik } from 'opik';
|
||||
const opik = new Opik({
|
||||
apiKey: process.env.OPIK_API_KEY,
|
||||
workspaceName: process.env.OPIK_WORKSPACE,
|
||||
apiUrl: process.env.OPIK_BASE_URL + '/api',
|
||||
});
|
||||
|
||||
// project + a few traces with structure variety
|
||||
const projectName = `discovery-logs-${Date.now()}`;
|
||||
await opik.api.projects.createProject({ name: projectName });
|
||||
|
||||
import { track } from 'opik';
|
||||
const decoratedFn = track({ name: 'discovery-trace', projectName }, async (input: string) => {
|
||||
return `output for ${input}`;
|
||||
});
|
||||
await decoratedFn('hello');
|
||||
await decoratedFn('world');
|
||||
await opik.flush();
|
||||
|
||||
console.log(`Discovery project: ${projectName}`);
|
||||
```
|
||||
|
||||
Run it locally with staging or local-dev creds, note the project name, then navigate the UI to that project's logs page in the next step. Tear it down by hand at the end of the discovery session (`backendClient.deleteProject` if it's wired up, or `curl` otherwise) — discovery state should never accumulate.
|
||||
|
||||
**Reusable seed patterns by page family:**
|
||||
|
||||
| Page being built | Seed via | Why |
|
||||
|---|---|---|
|
||||
| `LogsPage`, `TracePanelPage` | `opik.track` decorator + at least one call | Empty projects show only the empty state |
|
||||
| `DatasetsPage` | `opik.api.datasets.createDataset({...})` for ~3 datasets with different shapes | List interactions need multiple rows |
|
||||
| `DatasetItemsPage` | `opik.Dataset(name).insert([...])` with 3+ items | Item-table interactions need data |
|
||||
| `TestSuitesPage`, `TestSuitePage` | `opik.TestSuite(...)` with items + evaluator + run | Suite states (running/completed/failed) need a real run |
|
||||
| `ExperimentsPage` | `experiment.evaluate(...)` against a dataset | Experiment rows need a completed run |
|
||||
| `OnlineEvaluationPage` | (works empty for rule list); seed a rule via UI once during discovery | Rule list shows after at least one rule exists |
|
||||
| `AnnotationQueuesPage`, `AnnotationQueuePage` | `opik.TracesAnnotationQueue(...)` with 3+ traces | Reviewer flow needs items to score |
|
||||
|
||||
Common gotchas:
|
||||
|
||||
- **Trace ingestion is eventually consistent.** After `opik.flush()`, the trace may take 1–3 seconds to appear in the Logs page. If you snapshot too fast, you'll see the empty state. Wait or refresh.
|
||||
- **Online scoring rules apply asynchronously** — if you seed a trace and then snapshot the Online Evaluation page, scores may not have landed yet.
|
||||
- **Dataset items have a slight delay** between insert and table-render — usually ms, but watch for it.
|
||||
|
||||
### 3. Open the page with an authed browser session
|
||||
|
||||
**Auth setup once per discovery session:**
|
||||
|
||||
The browser MCP needs an authenticated page. Against local OSS (`http://localhost:5173`, workspace `default`) there's no login wall — navigate straight to the page. Against an authenticated deployment, reuse the suite's storage state: `global-setup` mints `.auth/user.json` the first time you run a Playwright test, and the browser MCP can load it as its `storageState`. Don't script a full login flow during discovery — it's a distraction.
|
||||
|
||||
**Arm the dialog handler before your first navigation (below).** Several Opik pages guard against data loss with a native `beforeunload` "Leave site?" confirm — it fires the moment you navigate (or reload) with unsaved state: a staged draft on the Dataset Items page (`useNavigationBlocker`), a dirty form, an open editor. A native dialog is **not** in the accessibility snapshot, so `browser_navigate` just silently blocks on it until it times out — you won't see why. Do this before the navigation step below, not after. Guard against it two ways, both cheap:
|
||||
|
||||
1. **Register an auto-accept handler at the start of the session**, before your first navigation:
|
||||
|
||||
```
|
||||
mcp__Playwright__browser_handle_dialog(accept=true)
|
||||
```
|
||||
|
||||
This dismisses any `beforeunload`/`confirm` that appears so navigation never hangs. (If one has *already* blocked you, call the same tool to clear it, then carry on.)
|
||||
|
||||
2. **Leave the page clean.** Before navigating away, clear unsaved state through the UI the way a user would — commit or **discard** the draft, close the editor, reset the form. This is also what the POM method itself must do, so doing it in discovery validates that path. Don't rely on the auto-handler alone: a discarded draft leaves the page in a real, testable state; a force-dismissed dialog leaves stale draft state behind.
|
||||
|
||||
**Standard discovery navigation:**
|
||||
|
||||
```
|
||||
mcp__Playwright__browser_navigate(url="http://localhost:5173/...")
|
||||
```
|
||||
|
||||
Get the exact route from the FE source — `apps/opik-frontend/src/v2/router.tsx` for the route table, or the page directory under `apps/opik-frontend/src/v2/pages/<PageName>/`. Most data pages are project-scoped, e.g. `/{workspace}/projects/{projectId}/datasets/` or `/{workspace}/projects/{projectId}/logs`. Confirm against the router rather than guessing — routes change.
|
||||
|
||||
### 4. Snapshot the accessibility tree
|
||||
|
||||
```
|
||||
mcp__Playwright__browser_snapshot()
|
||||
```
|
||||
|
||||
This returns the **structured accessibility tree**, not pixels. Each interactive element has:
|
||||
|
||||
- A role (`button`, `textbox`, `link`, `combobox`, `row`, etc.)
|
||||
- An accessible name (the visible text or `aria-label`)
|
||||
- A stable `ref` ID you can use with `browser_click(ref="...")` to interact
|
||||
|
||||
Why this is the right primitive: Playwright's `getByRole(...)` selectors map 1:1 to what the snapshot shows. If you see `button "Create suite"` in the snapshot, you can confidently write `page.getByRole('button', { name: 'Create suite' })`.
|
||||
|
||||
**Read the snapshot before writing any selector.** Don't guess. Don't grep the FE source for what you think the button is called — the rendered DOM is the only source of truth that matters.
|
||||
|
||||
### 5. Enumerate `data-testid`s
|
||||
|
||||
The snapshot tells you what's interactive but not what's been explicitly marked stable by the FE team. For that:
|
||||
|
||||
```
|
||||
mcp__Playwright__browser_evaluate(function="""() => {
|
||||
return Array.from(document.querySelectorAll('[data-testid]'))
|
||||
.map(e => ({
|
||||
testid: e.getAttribute('data-testid'),
|
||||
tag: e.tagName.toLowerCase(),
|
||||
text: (e.textContent || '').slice(0, 60).trim(),
|
||||
visible: e.offsetParent !== null,
|
||||
}))
|
||||
.filter(e => e.visible);
|
||||
}""")
|
||||
```
|
||||
|
||||
This returns every test id currently rendered on the page, with enough context to know what each one is. **Test ids are the preferred selector** — they're the FE team's contract for "this won't change." Use them when they exist.
|
||||
|
||||
### 6. Pick the right selector
|
||||
|
||||
In priority order:
|
||||
|
||||
1. **`data-testid`** — most stable. `page.getByTestId('create-suite-button')`. If a test id exists, use it.
|
||||
2. **`getByRole(name)`** — stable across most refactors as long as the accessible name doesn't change. `page.getByRole('button', { name: 'Create suite' })`.
|
||||
3. **`getByLabel`** for form inputs that have a label. `page.getByLabel('Dataset name')`.
|
||||
4. **`getByText`** — fragile if the text is dynamic or i18n'd. Use only for truly static labels.
|
||||
5. **CSS / XPath selectors** — **last resort**. Use only when no test id and no accessible name exists, and leave a comment explaining why: `// no test id; FE team to add — link to ticket`.
|
||||
|
||||
**The decision is binary at write time, not runtime.** Pick one selector and commit to it — write deterministic selectors, not runtime-healing ones.
|
||||
|
||||
### 7. Use `browser_generate_locator` when you're unsure
|
||||
|
||||
If the accessibility tree shows three buttons with similar names, or the element has both a test id and a role and you're not sure which is canonical:
|
||||
|
||||
```
|
||||
mcp__Playwright__browser_snapshot() # to find the ref of the element
|
||||
mcp__playwright-test__browser_generate_locator(ref="<the-ref>")
|
||||
```
|
||||
|
||||
This returns the locator code **Playwright itself would generate** if you used `codegen` against this element. Trust that output — Playwright's locator selection logic is well-tuned for resilience.
|
||||
|
||||
If you don't have access to `mcp__playwright-test__` tools, fall back to writing the selector manually based on the snapshot + test id list, then verify in step 8.
|
||||
|
||||
### 8. Verify by running
|
||||
|
||||
After writing the POM method, the verification loop is:
|
||||
|
||||
```ts
|
||||
// scratch test, not committed
|
||||
import { test, expect } from '@playwright/test';
|
||||
import { LogsPage } from '../pom/logs.page';
|
||||
|
||||
test('discovery: LogsPage.filterByProject works', async ({ page }) => {
|
||||
await page.goto('http://localhost:5173/<workspace>/projects/<seed-project-id>/logs');
|
||||
const logs = new LogsPage(page);
|
||||
await logs.filterByProject('discovery-logs-...');
|
||||
expect(await logs.countTraces()).toBeGreaterThan(0);
|
||||
});
|
||||
```
|
||||
|
||||
Run it through:
|
||||
|
||||
```
|
||||
mcp__playwright-test__test_run(testPath="path/to/scratch.spec.ts")
|
||||
```
|
||||
|
||||
If it passes, the POM method works against the live page. If it fails, **read the failure trace** (Playwright's artifacts), don't just adjust selectors blindly. Common failures:
|
||||
|
||||
- **Timeout waiting for element** — selector is wrong. Re-snapshot and pick a different one.
|
||||
- **Element resolved but assertion failed** — the POM method's logic is wrong, not the selector. Fix the method, not the selector.
|
||||
- **Multiple elements matched** — selector isn't specific enough. Add a parent scope or use a more precise role.
|
||||
|
||||
### 9. When a stable selector doesn't exist
|
||||
|
||||
If after all of the above the only working selector is a CSS path like `.MuiTable-root > tbody > tr:nth-child(2)`, **stop and flag it**. The missing-data-testid protocol:
|
||||
|
||||
- **Default:** add a `data-testid` to the FE component in the **same change** as your POM. Find the component under `apps/opik-frontend/src/v2/pages/<Page>/...` or its shared-component dependency, add `data-testid="<descriptive-name>"`, then use that in the POM.
|
||||
- **Fallback:** if blocked from touching the FE, use `getByRole` with explicit accessible name (survives most refactors).
|
||||
- **Last resort:** structural CSS selector with a comment explaining why it's necessary and that a `data-testid` should be added.
|
||||
|
||||
The `data-testid` naming convention is kebab-case, descriptive, scoped to the page/component: `dataset-items-table`, `create-suite-button`, `trace-row-{traceId}`. Avoid generic names like `submit-button` that conflict across pages.
|
||||
|
||||
### 10. Commit and move on
|
||||
|
||||
Once the POM method works against the live UI:
|
||||
|
||||
- The POM file change goes in with the test that uses it.
|
||||
- Any FE `data-testid` additions go in the **same** change (cross-package is fine; reviewers expect it for test-enablement work).
|
||||
- Any scratch seed script and scratch test file are **NOT** committed. Delete them first.
|
||||
- Tear down any state you created during discovery (`backendClient.deleteProject(...)` or `curl -X DELETE ...`).
|
||||
|
||||
## Anti-patterns
|
||||
|
||||
These are red flags that mean you skipped a step:
|
||||
|
||||
| Symptom | What you skipped |
|
||||
|---|---|
|
||||
| "Let me look at the FE source to find the right selector" | Step 4 — snapshot the rendered DOM, not the source. Components compose; what renders is what matters. |
|
||||
| "I'll write the POM method and check if it works when the test runs" | Step 8 — verify in isolation before committing. Iterating inside a full run is slower. |
|
||||
| "The page is empty, so I'll just check the empty state" | Step 2 — seed real data. An empty-state-only POM never exercises the row template, the open-detail action, etc. |
|
||||
| "I'll use `page.locator('.button:nth-child(3)')` — it's fine" | Step 9 — flag missing testids and add them to the FE. Brittle selectors are the #1 source of E2E flake. |
|
||||
| "The test id I see is generic, like `button-1` — I'll use that" | Step 5 + 9 — generic test ids are nearly as bad as no test id. Rename it to something descriptive in the same change. |
|
||||
| "I'll add the POM but skip the seed because the test will create state" | Step 2 — even during the test, you're now writing untested POM code against a page state you've never seen. |
|
||||
| "`browser_navigate` is hanging / timing out for no reason" | Step 3 — a native "Leave site?" dialog is blocking (unsaved draft/form). It's not in the snapshot. Arm `browser_handle_dialog(accept=true)` up front, and discard the draft in-app before leaving. |
|
||||
|
||||
## Where this fits
|
||||
|
||||
This is the discovery sub-step of writing an E2E test. The `writing-e2e-tests` skill invokes it once it has scoped the test and analyzed the feature: you seed the page's state, explore the live UI, and come back with the selectors each POM method will use plus any `data-testid`s the FE needs. The test and POM get written and run from there.
|
||||
@@ -0,0 +1,108 @@
|
||||
---
|
||||
name: python-sdk
|
||||
description: Python SDK patterns for Opik. Use when working in sdks/python, on SDK APIs, integrations, or message processing.
|
||||
---
|
||||
|
||||
# Python SDK
|
||||
|
||||
## Three-Layer Architecture
|
||||
```
|
||||
Layer 1: Public API (opik.Opik, @opik.track)
|
||||
↓
|
||||
Layer 2: Message Processing (queue, batching, retry)
|
||||
↓
|
||||
Layer 3: REST Client (OpikApi, HTTP)
|
||||
```
|
||||
|
||||
## Critical Gotchas
|
||||
|
||||
### Flush Before Exit
|
||||
```python
|
||||
# ✅ REQUIRED for async operations
|
||||
client = opik.Opik()
|
||||
# ... tracing operations ...
|
||||
client.flush() # Must call before exit!
|
||||
```
|
||||
|
||||
### Async vs Sync Operations
|
||||
|
||||
**Async (via message queue)** - fire-and-forget:
|
||||
- `trace()`, `span()`
|
||||
- `log_traces_feedback_scores()`
|
||||
- `experiment.insert()`
|
||||
|
||||
**Sync (blocking, returns data)**:
|
||||
- `create_dataset()`, `get_dataset()`
|
||||
- `create_prompt()`, `get_prompt()`
|
||||
- `search_traces()`, `search_spans()`
|
||||
|
||||
### Lazy Imports for Integrations
|
||||
```python
|
||||
# ✅ GOOD - integration files assume dependency exists
|
||||
import anthropic # Only imported when user uses integration
|
||||
|
||||
# ❌ BAD - importing at package level
|
||||
from opik.integrations import anthropic # Would fail if not installed
|
||||
```
|
||||
|
||||
## Integration Patterns
|
||||
|
||||
### Pattern Selection
|
||||
```
|
||||
Library has callbacks? → Pure Callback (LangChain, LlamaIndex)
|
||||
No callbacks? → Method Patching (OpenAI, Anthropic)
|
||||
Callbacks unreliable? → Hybrid (ADK)
|
||||
```
|
||||
|
||||
### Method Patching (OpenAI, Anthropic)
|
||||
```python
|
||||
from opik.integrations.anthropic import track_anthropic
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
tracked_client = track_anthropic(client) # Wraps methods
|
||||
```
|
||||
|
||||
### Callback-Based (LangChain)
|
||||
```python
|
||||
from opik.integrations.langchain import OpikTracer
|
||||
|
||||
tracer = OpikTracer()
|
||||
chain.invoke(input, config={"callbacks": [tracer]})
|
||||
```
|
||||
|
||||
### Decorator-Based
|
||||
```python
|
||||
@opik.track
|
||||
def my_function(input: str) -> str:
|
||||
# Auto-creates span, captures input/output
|
||||
return process(input)
|
||||
```
|
||||
|
||||
## Dependency Policy
|
||||
- Avoid adding new dependencies
|
||||
- Use conditional imports for integrations
|
||||
- Keep version bounds flexible: `>=2.0.0,<3.0.0`
|
||||
|
||||
## Batching System
|
||||
Messages batch together for efficiency:
|
||||
- Flush triggers: time (1s), size (100), memory (50MB), manual
|
||||
- Reduces HTTP overhead significantly
|
||||
|
||||
## API Method Naming
|
||||
```python
|
||||
# CRUD: create/get/list/update/delete
|
||||
client.create_experiment(name="exp")
|
||||
client.get_dataset(name="ds")
|
||||
|
||||
# Search for complex queries
|
||||
client.search_spans(project_name="proj")
|
||||
client.search_traces(project_name="proj")
|
||||
|
||||
# Batch for bulk operations
|
||||
client.batch_create_items(...)
|
||||
```
|
||||
|
||||
## Reference Files
|
||||
- [testing.md](testing.md) - fake_backend, verifiers, test naming
|
||||
- [error-handling.md](error-handling.md) - Exception hierarchy, MetricComputationError
|
||||
- [good-code.md](good-code.md) - Access control, imports, factories, DI
|
||||
@@ -0,0 +1,91 @@
|
||||
---
|
||||
name: refactor-helper
|
||||
description: |
|
||||
Python code refactoring assistant. Triggers on:
|
||||
<example>
|
||||
User: "make this better"
|
||||
</example>
|
||||
<example>
|
||||
User: "refactor this method"
|
||||
</example>
|
||||
<example>
|
||||
User: "clean up this code"
|
||||
</example>
|
||||
model: sonnet
|
||||
color: blue
|
||||
tools:
|
||||
- Read
|
||||
- Grep
|
||||
- Glob
|
||||
- Edit
|
||||
---
|
||||
|
||||
# Python Refactoring Helper
|
||||
|
||||
You help refactor Python code in the Opik SDK.
|
||||
|
||||
## Refactoring Checklist
|
||||
|
||||
When asked to improve code:
|
||||
|
||||
1. **Access Control**: Should any public methods be private?
|
||||
- If only used inside the class → prefix with `_`
|
||||
|
||||
2. **Redundant Parameters**: Is the method receiving data it already has?
|
||||
- Use `self._stored_data` instead of passing `data` parameter
|
||||
|
||||
3. **Logic Duplication**: Similar code blocks with minor differences?
|
||||
- Extract helper with parameterized differences
|
||||
|
||||
4. **Module Organization**: Is this file doing too many things?
|
||||
- Split by responsibility
|
||||
|
||||
5. **Method Naming**: Does the name describe what, not how?
|
||||
- `validate_span()` not `check_span_data_for_id()`
|
||||
|
||||
## Common Patterns
|
||||
|
||||
### Extract Helper
|
||||
|
||||
```python
|
||||
# Before: duplicated validation
|
||||
def process_trace(self, trace):
|
||||
if not trace.get("id"):
|
||||
raise ValueError("Missing id")
|
||||
# process...
|
||||
|
||||
def process_span(self, span):
|
||||
if not span.get("id"):
|
||||
raise ValueError("Missing id")
|
||||
# process...
|
||||
|
||||
# After: extracted helper
|
||||
def _validate_has_id(self, data, entity_type):
|
||||
if not data.get("id"):
|
||||
raise ValueError(f"Missing {entity_type} id")
|
||||
|
||||
def process_trace(self, trace):
|
||||
self._validate_has_id(trace, "trace")
|
||||
# process...
|
||||
```
|
||||
|
||||
### Privatize Internal Methods
|
||||
|
||||
```python
|
||||
# Before
|
||||
def process(self, data):
|
||||
cleaned = self.clean_data(data) # Should be private
|
||||
return self.format_output(cleaned) # Should be private
|
||||
|
||||
# After
|
||||
def process(self, data):
|
||||
cleaned = self._clean_data(data)
|
||||
return self._format_output(cleaned)
|
||||
```
|
||||
|
||||
## Guidelines
|
||||
|
||||
- Make minimal changes that address the specific issue
|
||||
- Don't refactor unrelated code
|
||||
- Preserve existing tests
|
||||
- Keep changes reviewable (small PRs)
|
||||
@@ -0,0 +1,85 @@
|
||||
# Python SDK Error Handling
|
||||
|
||||
## Exception Hierarchy
|
||||
|
||||
All custom exceptions inherit from `OpikException`:
|
||||
|
||||
```python
|
||||
from opik.exceptions import (
|
||||
OpikException, # Base for all Opik errors
|
||||
ConfigurationError, # Invalid configuration
|
||||
MetricComputationError, # Metric computation failed
|
||||
GuardrailValidationFailed, # Guardrail check failed
|
||||
ScoreMethodMissingArguments, # Score method missing args
|
||||
OpikCloudRequestsRateLimited, # Rate limited
|
||||
)
|
||||
```
|
||||
|
||||
## MetricComputationError
|
||||
|
||||
**Always raise** `MetricComputationError` from `BaseMetric` subclasses instead of hiding errors:
|
||||
|
||||
```python
|
||||
from opik.evaluation.metrics import BaseMetric
|
||||
from opik.exceptions import MetricComputationError
|
||||
|
||||
class MyMetric(BaseMetric):
|
||||
def score(self, text: str) -> ScoreResult:
|
||||
try:
|
||||
# computation logic
|
||||
return ScoreResult(value=result)
|
||||
except Exception as e:
|
||||
# ✅ GOOD - Raise MetricComputationError
|
||||
raise MetricComputationError(f"Failed to compute metric: {e}")
|
||||
|
||||
# ❌ BAD - Don't hide errors
|
||||
# return ScoreResult(value=0.0)
|
||||
```
|
||||
|
||||
## Structured Exception Information
|
||||
|
||||
```python
|
||||
class ScoreMethodMissingArguments(OpikException):
|
||||
def __init__(
|
||||
self,
|
||||
score_name: str,
|
||||
missing_required_arguments: Sequence[str],
|
||||
available_keys: Sequence[str],
|
||||
):
|
||||
self.score_name = score_name
|
||||
self.missing_required_arguments = missing_required_arguments
|
||||
self.available_keys = available_keys
|
||||
super().__init__(self._get_error_message())
|
||||
|
||||
def _get_error_message(self) -> str:
|
||||
return (
|
||||
f"The scoring method {self.score_name} is missing arguments: "
|
||||
f"{self.missing_required_arguments}. Available keys: {self.available_keys}."
|
||||
)
|
||||
```
|
||||
|
||||
## HTTP Error Handling
|
||||
|
||||
```python
|
||||
try:
|
||||
response = self._rest_client.call(...)
|
||||
except rest_api_core.ApiError as e:
|
||||
if e.status_code == 409:
|
||||
# Conflict - duplicate request, ignore
|
||||
return
|
||||
elif e.status_code == 429:
|
||||
# Rate limited - handle retry
|
||||
raise OpikCloudRequestsRateLimited(...)
|
||||
else:
|
||||
LOGGER.error("API call failed: %s", str(e))
|
||||
raise
|
||||
```
|
||||
|
||||
## Best Practices
|
||||
|
||||
- ✅ Use specific exception types
|
||||
- ✅ Inherit from `OpikException`
|
||||
- ✅ Include meaningful error messages with context
|
||||
- ✅ Handle rate limiting with retries
|
||||
- ❌ Don't catch generic `Exception` without re-raising
|
||||
- ❌ Don't hide errors in metric computations
|
||||
@@ -0,0 +1,106 @@
|
||||
# Code Quality Patterns
|
||||
|
||||
## Access Control
|
||||
|
||||
Methods used only inside their class should be private.
|
||||
|
||||
```python
|
||||
# ✅ Good
|
||||
class DataProcessor:
|
||||
def process(self, data): # Public interface
|
||||
cleaned = self._clean(data)
|
||||
return self._format(cleaned)
|
||||
|
||||
def _clean(self, data): # Private - only used internally
|
||||
pass
|
||||
|
||||
def _format(self, data): # Private - only used internally
|
||||
pass
|
||||
```
|
||||
|
||||
## Module Organization
|
||||
|
||||
One module, one responsibility. Avoid monolithic utils.
|
||||
|
||||
```python
|
||||
# ✅ Good: Focused modules
|
||||
# httpx_client.py - Only HTTP client
|
||||
# config.py - Only configuration
|
||||
|
||||
# ❌ Bad: Kitchen sink module
|
||||
# utils.py
|
||||
class HttpClient: ...
|
||||
class ConfigManager: ...
|
||||
def parse_json(): ...
|
||||
def format_date(): ...
|
||||
```
|
||||
|
||||
## Import Organization
|
||||
|
||||
```python
|
||||
# Standard library
|
||||
import logging
|
||||
from typing import Any, Optional
|
||||
|
||||
# Third-party
|
||||
import httpx
|
||||
|
||||
# Local - import modules, not names
|
||||
from opik import config, exceptions
|
||||
from opik.message_processing import messages
|
||||
|
||||
# TYPE_CHECKING for circular imports
|
||||
from typing import TYPE_CHECKING
|
||||
if TYPE_CHECKING:
|
||||
from langchain_core.messages import BaseMessage
|
||||
```
|
||||
|
||||
## Factory Pattern for Extension
|
||||
|
||||
```python
|
||||
# ✅ Good: Easy to add new providers
|
||||
_PROVIDER_BUILDERS = {
|
||||
LLMProvider.OPENAI: [OpikUsage.from_openai_dict],
|
||||
LLMProvider.ANTHROPIC: [OpikUsage.from_anthropic_dict],
|
||||
}
|
||||
|
||||
def build_usage(provider, usage):
|
||||
for builder in _PROVIDER_BUILDERS[provider]:
|
||||
try:
|
||||
return builder(usage)
|
||||
except Exception:
|
||||
continue
|
||||
raise ValueError(f"Failed for {provider}")
|
||||
```
|
||||
|
||||
## Dependency Injection
|
||||
|
||||
```python
|
||||
# ✅ Good: Dependencies injected
|
||||
class Streamer:
|
||||
def __init__(
|
||||
self,
|
||||
queue: MessageQueue, # Injected
|
||||
batch_manager: BatchManager, # Injected
|
||||
):
|
||||
self._queue = queue
|
||||
self._batch_manager = batch_manager
|
||||
|
||||
# ❌ Bad: Dependencies created internally
|
||||
class Streamer:
|
||||
def __init__(self):
|
||||
self._queue = MessageQueue() # Hard to test
|
||||
self._batch_manager = BatchManager() # Hard to test
|
||||
```
|
||||
|
||||
## Avoid Redundant Parameters
|
||||
|
||||
```python
|
||||
# ❌ Bad: Passing data already stored
|
||||
def validate_span(self, data: Dict) -> bool:
|
||||
return data.get("span_id") is not None
|
||||
|
||||
# ✅ Good: Use internal state
|
||||
def validate_span(self) -> bool:
|
||||
return self._span_data.get("span_id") is not None
|
||||
```
|
||||
@@ -0,0 +1,176 @@
|
||||
# Python SDK Testing Patterns
|
||||
|
||||
## Test Naming Convention
|
||||
|
||||
```python
|
||||
# Pattern: test_WHAT__CASE_DESCRIPTION__EXPECTED_RESULT
|
||||
def test_tracked_function__error_inside_inner_function__caught_in_top_level_span():
|
||||
pass
|
||||
|
||||
# Happy path: test_WHAT__happyflow
|
||||
def test_optimization_lifecycle__happyflow():
|
||||
pass
|
||||
```
|
||||
|
||||
## Using fake_backend (Integration Tests)
|
||||
|
||||
For testing integrations that create traces/spans:
|
||||
|
||||
```python
|
||||
from tests.testlib import TraceModel, SpanModel, ANY_BUT_NONE, assert_equal
|
||||
from opik.decorator import tracker
|
||||
|
||||
def test_track__one_nested_function__happyflow(fake_backend):
|
||||
@tracker.track
|
||||
def f_inner(x):
|
||||
return "inner-output"
|
||||
|
||||
@tracker.track
|
||||
def f_outer(x):
|
||||
f_inner("inner-input")
|
||||
return "outer-output"
|
||||
|
||||
f_outer("outer-input")
|
||||
tracker.flush_tracker()
|
||||
|
||||
EXPECTED_TRACE_TREE = TraceModel(
|
||||
id=ANY_BUT_NONE,
|
||||
name="f_outer",
|
||||
input={"x": "outer-input"},
|
||||
output={"output": "outer-output"},
|
||||
start_time=ANY_BUT_NONE,
|
||||
end_time=ANY_BUT_NONE,
|
||||
spans=[
|
||||
SpanModel(
|
||||
id=ANY_BUT_NONE,
|
||||
name="f_outer",
|
||||
input={"x": "outer-input"},
|
||||
output={"output": "outer-output"},
|
||||
spans=[
|
||||
SpanModel(
|
||||
id=ANY_BUT_NONE,
|
||||
name="f_inner",
|
||||
input={"x": "inner-input"},
|
||||
output={"output": "inner-output"},
|
||||
spans=[],
|
||||
)
|
||||
],
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
assert len(fake_backend.trace_trees) == 1
|
||||
assert_equal(EXPECTED_TRACE_TREE, fake_backend.trace_trees[0])
|
||||
```
|
||||
|
||||
## Using Verifiers (E2E Tests)
|
||||
|
||||
For actual API call tests:
|
||||
|
||||
```python
|
||||
from tests.e2e import verifiers
|
||||
|
||||
def test_trace_creation__e2e__happyflow(opik_client: opik.Opik):
|
||||
trace = opik_client.trace(
|
||||
name="test-trace",
|
||||
input={"query": "test"},
|
||||
output={"result": "success"}
|
||||
)
|
||||
|
||||
verifiers.verify_trace(
|
||||
opik_client=opik_client,
|
||||
trace_id=trace.id,
|
||||
name="test-trace",
|
||||
input={"query": "test"},
|
||||
output={"result": "success"},
|
||||
)
|
||||
```
|
||||
|
||||
## Testlib Utilities
|
||||
|
||||
```python
|
||||
from tests.testlib import ANY_BUT_NONE, ANY_STRING, assert_equal
|
||||
|
||||
# ANY_BUT_NONE - matches any value that is not None
|
||||
# ANY_STRING - matches any string
|
||||
# assert_equal - deep comparison with ANY_* support
|
||||
```
|
||||
|
||||
## Parameterized Tests
|
||||
|
||||
```python
|
||||
@pytest.mark.parametrize(
|
||||
"text,expected_sentiment",
|
||||
[
|
||||
("I love this product!", "positive"),
|
||||
("This is terrible.", "negative"),
|
||||
("The sky is blue.", "neutral"),
|
||||
],
|
||||
)
|
||||
def test_sentiment_classification(text, expected_sentiment):
|
||||
metric = Sentiment()
|
||||
result = metric.score(text)
|
||||
assert expected_sentiment in result.reason
|
||||
```
|
||||
|
||||
## Key Rules
|
||||
|
||||
- Always test **public API only**
|
||||
- Use `fake_backend` for integration tests
|
||||
- Use `verifiers` for E2E tests
|
||||
- Study existing similar tests before adding new ones
|
||||
|
||||
## Running E2E Tests Locally
|
||||
|
||||
Pick the backend setup that matches what you're doing. Both need a few env vars the CI workflow sets implicitly.
|
||||
|
||||
### Option A — CI-equivalent (recommended for full-suite runs)
|
||||
|
||||
```bash
|
||||
# backend in docker, matches GitHub Actions:
|
||||
TOGGLE_RUNNERS_ENABLED=true ./opik.sh --backend
|
||||
|
||||
# then run the suite:
|
||||
cd sdks/python
|
||||
OPIK_URL_OVERRIDE=http://localhost:5173/api/ \
|
||||
venv/bin/pytest tests/e2e/ \
|
||||
--ignore=tests/e2e/test_guardrails.py \
|
||||
-vv --durations=20
|
||||
```
|
||||
|
||||
### Option B — dev-runner (iterating on backend code)
|
||||
|
||||
The native Java backend inherits your shell env, so you must export MinIO credentials and the runners flag before starting it — otherwise attachment and runner tests fail on environmental grounds, not real regressions.
|
||||
|
||||
```bash
|
||||
export AWS_ACCESS_KEY_ID=THAAIOSFODNN7EXAMPLE
|
||||
export AWS_SECRET_ACCESS_KEY=LESlrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY
|
||||
export TOGGLE_RUNNERS_ENABLED=true
|
||||
./scripts/dev-runner.sh --restart
|
||||
|
||||
cd sdks/python
|
||||
OPIK_URL_OVERRIDE=http://localhost:8080/ \
|
||||
venv/bin/pytest tests/e2e/ \
|
||||
--ignore=tests/e2e/test_guardrails.py \
|
||||
-vv --durations=20
|
||||
```
|
||||
|
||||
The `AWS_*` values are MinIO's root-user/root-password (from [deployment/docker-compose/docker-compose.yaml](../../../deployment/docker-compose/docker-compose.yaml)). They are the canonical "EXAMPLE" strings — no real AWS account is involved; the Java backend's S3 client reuses the standard AWS env-var names when pointed at MinIO via `S3_URL`.
|
||||
|
||||
### Why the gotchas
|
||||
|
||||
- **`TOGGLE_RUNNERS_ENABLED`** defaults to `false` in docker-compose. Without it, 8 tests in `tests/e2e/runner/` error out on setup because the runners feature is disabled in the backend.
|
||||
- **`--ignore=tests/e2e/test_guardrails.py`** — the guardrails Python service isn't part of the default compose stack. CI ignores these explicitly ([python_sdk_e2e_tests.yml:90](../../../.github/workflows/python_sdk_e2e_tests.yml#L90)).
|
||||
- **MinIO creds** — only relevant in Option B (dev-runner). Option A's backend container already has them baked in.
|
||||
|
||||
### Debugging a failing e2e test
|
||||
|
||||
```bash
|
||||
# Capture SDK DEBUG logs to a file:
|
||||
OPIK_FILE_LOGGING_LEVEL=DEBUG OPIK_LOGGING_FILE=/tmp/opik-sdk.log \
|
||||
venv/bin/pytest tests/e2e/test_tracing.py::test_name -vv
|
||||
|
||||
# And the backend-side error is in:
|
||||
docker logs opik-backend-1 # Option A
|
||||
tail -f /tmp/opik-opik-backend.log # Option B (dev-runner)
|
||||
```
|
||||
@@ -0,0 +1,64 @@
|
||||
---
|
||||
name: typescript-sdk
|
||||
description: TypeScript SDK patterns for Opik. Use when working in sdks/typescript.
|
||||
---
|
||||
|
||||
# TypeScript SDK
|
||||
|
||||
## Architecture
|
||||
- Layered, non-blocking by default
|
||||
- Data buffered and flushed async to backend
|
||||
- Node >= 18, ESM + CJS builds
|
||||
|
||||
## Layer Flow
|
||||
```
|
||||
Public API → OpikClient → Domain (Trace/Span) → BatchQueues → REST Client → Backend
|
||||
```
|
||||
|
||||
## Critical Gotchas
|
||||
- When changing dependencies or minimum versions, update and verify version references in `README.md` and integration README files in the same PR.
|
||||
|
||||
### Flush Before Exit
|
||||
```typescript
|
||||
// ✅ REQUIRED - especially in CLI/tests
|
||||
await client.flush();
|
||||
// or globally:
|
||||
await flushAll();
|
||||
```
|
||||
|
||||
### Domain Objects Don't Do HTTP
|
||||
```typescript
|
||||
// ✅ GOOD - domain objects enqueue, not HTTP
|
||||
trace.update({ metadata: { key: 'value' } }); // Enqueues update
|
||||
trace.end(); // Enqueues update
|
||||
|
||||
// ❌ BAD - don't call REST directly from domain
|
||||
```
|
||||
|
||||
### Never Leak rest_api
|
||||
```typescript
|
||||
// ✅ GOOD - export from public API
|
||||
export { Opik, track, flushAll } from 'opik';
|
||||
|
||||
// ❌ BAD - don't expose generated clients
|
||||
import { TracesApi } from 'opik/rest_api'; // Internal!
|
||||
```
|
||||
|
||||
## Batching Semantics
|
||||
- Updates wait for pending creates
|
||||
- Deletes wait for creates and updates
|
||||
- `flush()` flushes all queues in order
|
||||
- Debounce window configurable via `OpikConfig`
|
||||
|
||||
## Error Handling
|
||||
- HTTP failures: `OpikApiError`, `OpikApiTimeoutError`
|
||||
- 404s translate to domain errors: `DatasetNotFoundError`, `ExperimentNotFoundError`
|
||||
- Never swallow errors, include context in logs
|
||||
|
||||
## Integration Guidelines
|
||||
- Integrations wrap public API only
|
||||
- Keep adapters thin, non-blocking
|
||||
- Provide `flush()` escape hatch if needed
|
||||
|
||||
## Reference Files
|
||||
- [testing.md](testing.md) - Vitest patterns, mocking, flush timing
|
||||
@@ -0,0 +1,67 @@
|
||||
# TypeScript SDK Testing
|
||||
|
||||
## Test Runner
|
||||
Vitest - run with `npm test` in `sdks/typescript`
|
||||
|
||||
## Key Patterns
|
||||
|
||||
### Always Flush Before Assertions
|
||||
```typescript
|
||||
// Data only persists after flush
|
||||
const client = new Opik();
|
||||
client.trace({ name: "test" });
|
||||
await client.flush(); // Required!
|
||||
|
||||
// Then assert
|
||||
expect(mockFetch).toHaveBeenCalledWith(/* ... */);
|
||||
```
|
||||
|
||||
### Mock Network and Timers
|
||||
```typescript
|
||||
import { vi } from 'vitest';
|
||||
|
||||
// Mock fetch for deterministic tests
|
||||
vi.stubGlobal('fetch', vi.fn().mockResolvedValue(/* ... */));
|
||||
|
||||
// Use fake timers for batching tests
|
||||
vi.useFakeTimers();
|
||||
// ... do work ...
|
||||
vi.advanceTimersByTime(1000); // Trigger batch flush
|
||||
```
|
||||
|
||||
### Control Batching in Tests
|
||||
```typescript
|
||||
// Option 1: Small delays
|
||||
const client = new Opik({ batchDelay: 10 });
|
||||
|
||||
// Option 2: Advance fake timers
|
||||
vi.advanceTimersByTime(1000);
|
||||
|
||||
// Option 3: Explicit flush
|
||||
await client.flush();
|
||||
```
|
||||
|
||||
## Integration Testing
|
||||
|
||||
```typescript
|
||||
// Mock provider clients
|
||||
const mockOpenAI = {
|
||||
chat: {
|
||||
completions: {
|
||||
create: vi.fn().mockResolvedValue({ /* response */ }),
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
// Assert spans recorded
|
||||
expect(mockFetch).toHaveBeenCalledWith(
|
||||
expect.stringContaining('/spans'),
|
||||
expect.objectContaining({ method: 'POST' })
|
||||
);
|
||||
```
|
||||
|
||||
## Best Practices
|
||||
- Test public API only, avoid `rest_api` internals
|
||||
- Mock network to keep tests deterministic
|
||||
- Assert no `ERROR` level logs unintentionally
|
||||
- Always `flush()` before checking persisted data
|
||||
@@ -0,0 +1,296 @@
|
||||
---
|
||||
name: write-docs
|
||||
description: Authoring Fern MDX documentation pages for the Opik docs site, plus release-note and changelog routing. Use when writing or updating pages under apps/opik-documentation/documentation/fern/, drafting PR descriptions, or picking the right changelog surface.
|
||||
---
|
||||
|
||||
# Write Docs
|
||||
|
||||
The Opik docs site is built with [Fern](https://buildwithfern.com/) from MDX sources under `apps/opik-documentation/documentation/fern/`. Two content surfaces coexist:
|
||||
|
||||
- `fern/docs/` — **v1** (the established surface, source of style truth)
|
||||
- `fern/docs-v2/` — **latest** (where new pages go)
|
||||
|
||||
New pages should land in `docs-v2/`. v1 is the reference for writing style and component usage because it is far richer.
|
||||
|
||||
## Where new pages live
|
||||
|
||||
- Create the file at `apps/opik-documentation/documentation/fern/docs-v2/<section>/<page-name>.mdx`.
|
||||
- Register it in `fern/versions/latest.yml` under the right `section:` block.
|
||||
- Only touch `fern/versions/v1.yml` if the page must also ship in the v1 build (rare).
|
||||
- **Do not edit** `fern/docs.yml` when adding a page — that file is the global site config (tabs, redirects), not per-version routing.
|
||||
- Routing is not implied by folder layout. Always check the version YAML.
|
||||
|
||||
## Frontmatter template
|
||||
|
||||
Every page uses YAML frontmatter. `title` and `headline` are required; the `og:*` fields are strongly recommended for SEO/social sharing and are present on every page. Do not repeat `title` as an inline `# H1` in the body — Fern renders it from frontmatter.
|
||||
|
||||
```yaml
|
||||
---
|
||||
title: Page Title
|
||||
headline: Page Title | Opik Documentation
|
||||
og:title: Page Title — Opik
|
||||
og:description: One-line summary used for social sharing and previews
|
||||
og:site_name: Opik Documentation
|
||||
---
|
||||
```
|
||||
|
||||
`subtitle: ...` is an optional field used on concept/overview pages to add a secondary line. Landing/overview pages may also set `layout: overview`.
|
||||
|
||||
## Style and voice
|
||||
|
||||
Pull examples from v1 pages when unsure — `fern/docs/tracing/log_traces.mdx`, `fern/docs/tracing/concepts.mdx`, and `fern/docs/quickstart.mdx` are good anchors.
|
||||
|
||||
- **Person:** "you" and imperative voice. Professional but approachable.
|
||||
- **Opening:** one or two intro sentences before the first `##` heading. No inline H1.
|
||||
- **Headings:** `##` for top-level sections, `###` for subsections. Never introduce an inline `#` — that collides with the frontmatter title.
|
||||
- **Paragraphs:** keep them short (2–4 sentences). Mix prose with bullet lists for features, options, and prerequisites.
|
||||
- **Page shape:**
|
||||
- *Concept pages* start with the *why*, then definitions.
|
||||
- *How-to pages* start with brief context, then the task steps.
|
||||
- *Overview/landing pages* lead with a short pitch and a `<CardGroup>` of links.
|
||||
- **End with "## Next steps"** linking to 2–4 related pages when useful.
|
||||
|
||||
## Fern MDX components
|
||||
|
||||
All examples below are taken from real pages in the repo.
|
||||
|
||||
### `<Tabs>` / `<Tab>` — SDK, language, or environment choice
|
||||
|
||||
Use when the whole section varies (not just a code block). Attach `language="..."` so Fern groups tabs across the site by the reader's last choice.
|
||||
|
||||
```mdx
|
||||
<Tabs>
|
||||
<Tab value="Python SDK" title="Python SDK" language="python">
|
||||
```bash
|
||||
pip install opik
|
||||
```
|
||||
</Tab>
|
||||
<Tab value="Typescript SDK" title="Typescript SDK" language="typescript">
|
||||
```bash
|
||||
npm install opik
|
||||
```
|
||||
</Tab>
|
||||
<Tab value="OpenTelemetry" title="OpenTelemetry">
|
||||
...
|
||||
</Tab>
|
||||
</Tabs>
|
||||
```
|
||||
|
||||
### `<Steps>` / `<Step>` — walkthroughs
|
||||
|
||||
For quickstarts, installs, and any sequential procedure. `title` on each `<Step>` is optional.
|
||||
|
||||
```mdx
|
||||
<Steps>
|
||||
<Step title="Install the Opik skill">
|
||||
```bash
|
||||
npx skills add comet-ml/opik-skills
|
||||
```
|
||||
</Step>
|
||||
<Step title="Run the integration">
|
||||
Once the skill is installed, you can add tracing using the following prompt:
|
||||
```
|
||||
Instrument my agent with Opik using the /instrument command.
|
||||
```
|
||||
</Step>
|
||||
</Steps>
|
||||
```
|
||||
|
||||
### `<CodeBlocks>` — multi-language code, identical surrounding prose
|
||||
|
||||
Prefer this over `<Tabs>` when only the code varies.
|
||||
|
||||
```mdx
|
||||
<CodeBlocks>
|
||||
```python title="Python"
|
||||
import opik
|
||||
opik.configure()
|
||||
```
|
||||
```ts title="Typescript"
|
||||
import Opik from "opik";
|
||||
const client = new Opik();
|
||||
```
|
||||
</CodeBlocks>
|
||||
```
|
||||
|
||||
### `<CardGroup>` / `<Card>` — landing and integration grids
|
||||
|
||||
```mdx
|
||||
<CardGroup cols={3}>
|
||||
<Card title="LangChain" href="/integrations/langchain" icon={<img src="/img/tracing/langchain.svg" />} iconPosition="left"/>
|
||||
<Card title="LlamaIndex" href="/integrations/llama_index" icon={<img src="/img/tracing/llamaindex.svg" />} iconPosition="left"/>
|
||||
<Card title="Anthropic" href="/integrations/anthropic" icon={<img src="/img/tracing/anthropic.svg" />} iconPosition="left"/>
|
||||
</CardGroup>
|
||||
```
|
||||
|
||||
### `<AccordionGroup>` / `<Accordion>` — FAQs and expandable advanced topics
|
||||
|
||||
```mdx
|
||||
<AccordionGroup>
|
||||
<Accordion title="Why use the optimizer?">
|
||||
The Agent Optimizer provides a unified interface...
|
||||
</Accordion>
|
||||
</AccordionGroup>
|
||||
```
|
||||
|
||||
### `<Frame>` — image wrapper (always wrap images)
|
||||
|
||||
```mdx
|
||||
<Frame>
|
||||
<img src="/img/tracing/introduction.png" />
|
||||
</Frame>
|
||||
```
|
||||
|
||||
### Callouts: `<Tip>`, `<Note>`, `<Warning>`, `<Info>`, `<Callout>`
|
||||
|
||||
Pick by intent, not aesthetics:
|
||||
|
||||
- **`<Tip>`** — cross-references, shortcuts, "If you're just getting started, see..."
|
||||
- **`<Note>`** — clarifications and recommendations that aren't risky
|
||||
- **`<Warning>`** — breaking changes, footguns, prerequisites that will break things
|
||||
- **`<Info>`** — informational, interchangeable with `<Note>` in practice
|
||||
- **`<Callout>`** — catch-all when none of the above fits
|
||||
|
||||
```mdx
|
||||
<Tip>
|
||||
If you are just getting started with Opik, we recommend first checking out the [Quickstart](/quickstart) guide.
|
||||
</Tip>
|
||||
|
||||
<Warning>
|
||||
Note that the authorization header value does not include the `Bearer ` prefix.
|
||||
</Warning>
|
||||
```
|
||||
|
||||
## Code examples
|
||||
|
||||
- Use `<CodeBlocks>` for multi-language blocks; use `<Tabs>` when surrounding prose also varies.
|
||||
- Install commands are inline bash blocks (`pip install opik`, `npm install opik`).
|
||||
- There is no snippet-include system. All code is written inline in MDX.
|
||||
- **Use placeholders for credentials:** `<API_KEY>`, `<TOKEN>`, `<your-api-key>`. Never commit real keys.
|
||||
|
||||
## Images
|
||||
|
||||
- Store under `apps/opik-documentation/documentation/fern/img/<section>/...`.
|
||||
- Reference from MDX as `/img/<section>/<file>.png` (path is rooted at the docs base).
|
||||
- **Never** put new assets in `static/img/` — that folder is legacy and only kept for external integrations.
|
||||
- Always wrap with `<Frame>`. Captions are not a repo convention.
|
||||
|
||||
## Cross-links
|
||||
|
||||
Absolute, slug-based paths only. No relative paths (`../foo`).
|
||||
|
||||
```mdx
|
||||
[Python SDK](/reference/python-sdk/overview)
|
||||
[Log traces](/tracing/log_traces)
|
||||
[Integrations overview](/integrations/overview)
|
||||
```
|
||||
|
||||
In-page anchors use the heading slug: `[Concepts](#concepts)`.
|
||||
|
||||
## Routing: adding a page to `latest.yml`
|
||||
|
||||
Add a page entry under the correct `section:` (keep the YAML at 2-space indent):
|
||||
|
||||
```yaml
|
||||
- page: Page Title
|
||||
path: ../docs-v2/section/page-name.mdx
|
||||
slug: page-name
|
||||
```
|
||||
|
||||
If the page also needs to ship in v1, mirror the entry in `fern/versions/v1.yml`. Leave `fern/docs.yml` alone.
|
||||
|
||||
## File naming
|
||||
|
||||
- Kebab-case for new files: `getting-started.mdx`, `log-traces.mdx`.
|
||||
- When editing an existing section that uses snake_case (common in v1), match neighbors rather than renaming. Renames require redirect entries in `docs.yml`.
|
||||
|
||||
## Local verification
|
||||
|
||||
```bash
|
||||
cd apps/opik-documentation/documentation
|
||||
npm install # first time only
|
||||
npm run dev # live-reload preview
|
||||
```
|
||||
|
||||
Open the rendered page and confirm:
|
||||
- Frontmatter renders (title shows, no stray H1 in body).
|
||||
- Every MDX component resolves (no raw `<Tabs>` tags visible).
|
||||
- Every link works (no 404s, no `Broken link` warnings in the terminal).
|
||||
- Images load.
|
||||
|
||||
## Changelog routing
|
||||
|
||||
Pick the changelog target by scope — do not default everything to the root `CHANGELOG.md`.
|
||||
|
||||
- `CHANGELOG.md` (repo root) — self-hosted deployment changelog. Breaking, critical, or security-impacting changes only.
|
||||
- `apps/opik-documentation/documentation/fern/docs/changelog/*.mdx` — general product release notes shown at `/docs/opik/changelog`. One dated `.mdx` per entry.
|
||||
- `apps/opik-documentation/documentation/fern/docs/agent_optimization/getting_started/changelog.mdx` — Agent Optimizer version updates (e.g. `sdks/opik_optimizer` releases like `3.1.0`).
|
||||
- Liquibase `changelog.xml` files are migration manifests, not user-facing release notes. Do not put prose there.
|
||||
- When unsure, confirm the surface from `fern/docs.yml` before editing.
|
||||
|
||||
### Changelog entry template
|
||||
|
||||
```markdown
|
||||
### [VERSION] - [DATE]
|
||||
|
||||
#### New Features
|
||||
- **Feature Name**: Brief description
|
||||
|
||||
#### Improvements
|
||||
- **Improvement**: What changed and why
|
||||
|
||||
#### Bug Fixes
|
||||
- **Fix**: What was broken (#issue)
|
||||
|
||||
#### Breaking Changes
|
||||
- **Change**: What breaks, migration steps
|
||||
```
|
||||
|
||||
## Feature documentation checklist
|
||||
|
||||
When documenting a new feature, cover:
|
||||
|
||||
- **User impact** — What capability does this add? How do users access it?
|
||||
- **Technical changes** — API endpoints and params, SDK methods, config or env vars, migrations.
|
||||
- **Breaking changes** — What breaks and the migration path, if any.
|
||||
|
||||
Keep it user-facing: avoid implementation detail unless it affects how someone uses the feature.
|
||||
|
||||
## PR description template
|
||||
|
||||
```markdown
|
||||
## Summary
|
||||
- What this PR does (bullet points)
|
||||
|
||||
## Test Plan
|
||||
- How to verify it works
|
||||
|
||||
## Related Issues
|
||||
- Resolves #123
|
||||
```
|
||||
|
||||
## Internationalized READMEs
|
||||
|
||||
`readme_CN.md`, `readme_JP.md`, `readme_KO.md`, `readme_PT_BR.md`, `readme_AR.md`, `readme_DE.md`, `readme_ES.md`, `readme_FR.md` are AI machine-translated from the English `README.md`.
|
||||
|
||||
- Each non-English README has a blockquote notice at the top warning that it is AI-translated and welcoming improvements. Keep it.
|
||||
- When the English README changes meaningfully, re-translate the affected files. Do not hand-edit translated READMEs for content changes — update the English source and re-translate.
|
||||
|
||||
## Forbidden and discouraged
|
||||
|
||||
- No real API keys, tokens, or workspace IDs in examples — always placeholders.
|
||||
- Do not edit `fern/docs/cookbook/*.mdx` by hand. Those files are regenerated from `docs/cookbook/*.ipynb` by `update_cookbooks.sh`.
|
||||
- Do not put new images outside `fern/img/`. `static/img/` is legacy-only and cannot be deleted because of external integrations.
|
||||
- Do not infer URL paths from folder layout — always consult the version YAML.
|
||||
- Do not add an inline `# H1` inside the body — the frontmatter `title` already provides it.
|
||||
|
||||
## Key files
|
||||
|
||||
- `apps/opik-documentation/documentation/fern/versions/latest.yml` — routing for the latest version (edit when adding pages).
|
||||
- `apps/opik-documentation/documentation/fern/versions/v1.yml` — routing for v1 (edit only if the page ships in v1 too).
|
||||
- `apps/opik-documentation/documentation/fern/docs.yml` — global site config (tabs, redirects). Treat as read-only for page adds.
|
||||
- `apps/opik-documentation/documentation/fern/docs-v2/` — target directory for new pages.
|
||||
- `apps/opik-documentation/documentation/fern/img/` — image storage.
|
||||
- `apps/opik-documentation/documentation/update_cookbooks.sh` — regenerates cookbook MDX from notebooks.
|
||||
- `apps/opik-documentation/AGENTS.md` — docs-module contribution rules.
|
||||
- `.github/release-drafter.yml` — release notes template.
|
||||
@@ -0,0 +1,153 @@
|
||||
---
|
||||
name: writing-e2e-tests
|
||||
description: Use when a developer wants to add, write, or create an end-to-end test for an Opik feature, page, or branch — e.g. "add an e2e test for the experiments comparison page", "write a test for the feature I just built", "e2e test for this branch", "cover the dataset items flow with a test". Runs the full loop in tests_end_to_end/e2e/ — analyze the feature and frontend code, explore the live UI with the Playwright MCP, write the Page Object Model + spec, and run it locally until green.
|
||||
---
|
||||
|
||||
# Writing E2E Tests
|
||||
|
||||
This skill is how we add an end-to-end test to the Opik E2E suite. You give it a feature, page, or branch; it runs a proven loop end-to-end and leaves you with a working, locally-verified Playwright test.
|
||||
|
||||
**Announce at start:** "I'm using the writing-e2e-tests skill to add an E2E test for X."
|
||||
|
||||
## Where tests live
|
||||
|
||||
The suite is at `tests_end_to_end/e2e/`. Inside it:
|
||||
|
||||
- **Specs:** `tests/<feature>/<name>.spec.ts` — one feature directory per page family (`datasets`, `trace-explore`, `experiments`, `test-suites`, `online-evaluation`, …).
|
||||
- **Page Object Models:** `pom/<name>.page.ts` — one class per page, methods for the interactions a test needs.
|
||||
- **Fixtures:** `fixtures/<name>.fixture.ts` — seed entities (project, dataset, trace, experiment, testSuite) and tear them down. Composed in a chain; re-exported from `fixtures/index.ts`.
|
||||
- **SDK clients:** `core/sdk/` — `sdkClient.python` (HTTP wrapper over the bridge) and `sdkClient.typescript` (direct `new Opik({...})`) for seeding. `core/backend/` holds the typed REST client for inspection + teardown.
|
||||
- **Bridge:** `services/opik-sdk-driver/` — a FastAPI app (run with `uv`) wrapping the Python SDK, exposing routes the TS clients call. Playwright's `webServer` directive auto-spawns it during a test run; you don't start it by hand.
|
||||
|
||||
Specs and POMs import through path aliases: `import { test, expect } from '@e2e/fixtures'` and `import { LogsPage } from '@e2e/pom/logs.page'`.
|
||||
|
||||
## Tooling — already set up
|
||||
|
||||
- The **`Playwright` MCP** (live-UI exploration) and the **`playwright-test` MCP** (`browser_generate_locator`) are already configured in the repo's `.mcp.json`. No setup step.
|
||||
- Tests run via the plain Playwright CLI from `tests_end_to_end/e2e/`. The `webServer` directive spawns the bridge automatically.
|
||||
|
||||
## Conventions
|
||||
|
||||
Read [conventions.md](conventions.md) before writing any POM or spec. It carries the rules that keep tests legible and stable: mandatory `test.step()` wrapping, UI-first assertions, selector preference, public-SDK-only seeding, fixture seed shapes, and the tag taxonomy. They aren't optional polish — each prevents a class of failure.
|
||||
|
||||
## The loop
|
||||
|
||||
```dot
|
||||
digraph writing_e2e {
|
||||
rankdir=TB;
|
||||
"1. Scope (GATE)" [shape=box];
|
||||
"2. Analyze feature + FE code" [shape=box];
|
||||
"3. Discover live UI (GATE)" [shape=box];
|
||||
"4. Write POM + spec" [shape=box];
|
||||
"5. Run until green" [shape=box];
|
||||
"Green?" [shape=diamond];
|
||||
|
||||
"1. Scope (GATE)" -> "2. Analyze feature + FE code";
|
||||
"2. Analyze feature + FE code" -> "3. Discover live UI (GATE)";
|
||||
"3. Discover live UI (GATE)" -> "4. Write POM + spec";
|
||||
"4. Write POM + spec" -> "5. Run until green";
|
||||
"5. Run until green" -> "Green?";
|
||||
"Green?" -> "4. Write POM + spec" [label="no — fix"];
|
||||
"Green?" -> "done" [label="yes"];
|
||||
}
|
||||
```
|
||||
|
||||
### Step 1 — Scope (gate, lightweight)
|
||||
|
||||
Work out, from the request:
|
||||
|
||||
- **What flow / feature** the test covers, and **which page** it lives on. If the dev pointed at a branch or PR, read the diff to find what changed.
|
||||
- **The target.** Default is local OSS at `http://localhost:5173` (`OPIK_DEPLOYMENT=oss`, workspace `default`) — the natural target for "test the feature I just built." Only use another target if the dev asks.
|
||||
- **Tags** — pick a tier (`@t1-smoke` / `@t2-cuj` / `@t3-nightly`) and a feature tag, per [conventions.md](conventions.md).
|
||||
|
||||
Run the **safety check** (below) before any seeding. Then confirm the scope in one short message — feature, page, target, tags — and proceed. Don't write a formal spec document.
|
||||
|
||||
### Step 2 — Analyze the feature and frontend code
|
||||
|
||||
Before touching the browser:
|
||||
|
||||
- Read the page's FE source under `apps/opik-frontend/src/v2/pages/<Page>/` — the route it renders at, the components it composes, and any `data-testid` attributes already present. The route shape is what your POM's `goto()` will use.
|
||||
- Identify the **entity preconditions**: what must exist for the page to render real data (an empty project shows only the empty state). Decide how to seed it — which fixture fits, or which bridge/SDK call. Seed via the SDK/bridge, never by click-creating through the UI.
|
||||
- Check `fixtures/` for an existing fixture that already seeds the shape you need; reuse it before writing a new one.
|
||||
|
||||
### Step 3 — Discover the live UI (gate, lightweight)
|
||||
|
||||
**Invoke the `playwright-pom-discovery` skill** (via the Skill tool). It walks the live page with the Playwright MCP: seed state, navigate authed, snapshot the accessibility tree, enumerate `data-testid`s, pick the most stable selector for each element you'll target, and flag any element that has no stable selector (needs a FE `data-testid` added in this change).
|
||||
|
||||
When discovery is done, report a short summary — the selectors you'll use per element, and any missing testids you'll add — and confirm before writing code. Don't write anything under `pom/` before this step.
|
||||
|
||||
### Step 4 — Write the POM + spec
|
||||
|
||||
- Write or extend the POM in `pom/<name>.page.ts` using the selectors from discovery. Each method wraps its body in `test.step()` and returns through the callback (see [conventions.md](conventions.md)).
|
||||
- Write the spec in `tests/<feature>/<name>.spec.ts`: tier + feature tag on the describe block, coarse `test.step()` phases, UI-first assertions.
|
||||
- If discovery flagged a missing/brittle selector, add a descriptive `data-testid` to the FE component in the **same change**.
|
||||
|
||||
#### Rebuilding the FE after adding a `data-testid`
|
||||
|
||||
The local OSS deployment serves the frontend from a **Docker image** — file changes to `apps/opik-frontend/` are not picked up automatically. After adding a `data-testid`, you must rebuild and restart the container before the test can find it.
|
||||
|
||||
From `deployment/docker-compose/`:
|
||||
|
||||
```bash
|
||||
# 1. Build a new image from the updated source
|
||||
docker compose --profile opik build frontend
|
||||
|
||||
# 2. Recreate the container using the locally built image
|
||||
# (pull_policy defaults to "always" — override it so Docker uses the local build)
|
||||
docker stop opik-frontend-1 && docker rm opik-frontend-1
|
||||
OPIK_FRONTEND_PULL_POLICY=never docker compose --profile opik up -d --no-deps frontend
|
||||
```
|
||||
|
||||
Verify the new `data-testid` is live before running the test:
|
||||
|
||||
```bash
|
||||
docker exec opik-frontend-1 sh -c 'grep -r "your-testid" /usr/share/nginx/html/ | wc -l'
|
||||
# should print a non-zero number
|
||||
```
|
||||
|
||||
> **Network note:** if the rebuilt container loses connectivity to the backend (502 errors), the container may have ended up on the wrong Docker network. Fix it:
|
||||
> ```bash
|
||||
> docker network disconnect opik-opik_default opik-frontend-1
|
||||
> docker network connect opik-opik_default opik-frontend-1
|
||||
> ```
|
||||
|
||||
### Step 5 — Run until green
|
||||
|
||||
From `tests_end_to_end/e2e/`:
|
||||
|
||||
```bash
|
||||
npx playwright test tests/<feature>/<name>.spec.ts --reporter=list
|
||||
```
|
||||
|
||||
The bridge auto-spawns (you'll see its startup line in the output). If a test fails, **read the failure trace** (`npx playwright show-trace`) rather than adjusting selectors blindly — see "verify the test render before blaming the backend" in [conventions.md](conventions.md). Fix and re-run until green. Report the actual run output.
|
||||
|
||||
## Safety: verify local config before seeding
|
||||
|
||||
The Python SDK behind the bridge reads `~/.opik.config`. If it points at a cloud environment, seeding would create real data there. Before any seed against a local target:
|
||||
|
||||
```bash
|
||||
cat ~/.opik.config
|
||||
```
|
||||
|
||||
If `url_override` is anything other than `http://localhost:5173/api`, back it up and point it local:
|
||||
|
||||
```bash
|
||||
cp ~/.opik.config ~/.opik.config.bak 2>/dev/null || true
|
||||
cat > ~/.opik.config << 'EOF'
|
||||
[opik]
|
||||
url_override = http://localhost:5173/api
|
||||
workspace = default
|
||||
EOF
|
||||
```
|
||||
|
||||
When the work is done, remind the dev to restore: `cp ~/.opik.config.bak ~/.opik.config`. If it already points local, skip this.
|
||||
|
||||
## Anti-patterns
|
||||
|
||||
| Symptom | What you skipped |
|
||||
|---|---|
|
||||
| "Let me read the FE source to find the selector" | Discovery — snapshot the rendered DOM. What renders is the only source of truth for selectors. |
|
||||
| "I'll explore the empty page and figure out the rows later" | Seeding — an empty-state-only POM never exercises the row template or open-detail actions. |
|
||||
| "I'll write the POM and find out if it works when the whole suite runs" | Run-until-green in isolation — iterate on the one spec, don't debug it inside a full suite run. |
|
||||
| "`page.locator('tbody tr:nth-child(3)')` is fine" | Flagging the missing testid — brittle structural selectors are the top source of flake; add a `data-testid`. |
|
||||
| "I'll create the dataset through the UI so the page has data" | SDK/bridge seeding — UI-create is what the test exercises, not how you set up. |
|
||||
@@ -0,0 +1,102 @@
|
||||
# E2E Test Conventions
|
||||
|
||||
These are the durable conventions for the Opik E2E suite (`tests_end_to_end/e2e/`). Read this before writing any Page Object Model or spec. They aren't style preferences — each one prevents a class of failure or makes failures legible.
|
||||
|
||||
## `test.step()` wrapping is mandatory
|
||||
|
||||
Wrap logical phases at the test level, and wrap each POM method body in a `test.step()` that returns through the callback. This is what makes the Playwright trace viewer and the Allure timeline readable — without it, a failure is a flat wall of actions with no narrative.
|
||||
|
||||
Granularity: a "phase" is something you'd describe in a complete sentence ("seed three traces", "open the trace and verify the panel").
|
||||
|
||||
In the test:
|
||||
|
||||
```ts
|
||||
import { test, expect } from '@e2e/fixtures';
|
||||
import { LogsPage } from '@e2e/pom/logs.page';
|
||||
|
||||
test('Logs view shows seeded traces in order', async ({ project, sdkClient, page }) => {
|
||||
await test.step('Seed traces via the Python SDK', async () => {
|
||||
await sdkClient.python.createTrace({ project_name: project.name, name: 'a', input: 'i', output: 'o' });
|
||||
});
|
||||
|
||||
await test.step('Open Logs and verify', async () => {
|
||||
const logs = new LogsPage(page);
|
||||
await logs.goto(project.id);
|
||||
await logs.waitForReady();
|
||||
expect(await logs.countTraces()).toBe(1);
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
In a POM method — wrap the body and return through the step callback:
|
||||
|
||||
```ts
|
||||
async openDatasetByName(name: string): Promise<DatasetItemsPage> {
|
||||
return test.step(`open dataset "${name}"`, async () => {
|
||||
const row = this.datasetRow(name);
|
||||
await row.waitFor({ state: 'visible' });
|
||||
await row.getByRole('cell', { name, exact: true }).click();
|
||||
return new DatasetItemsPage(this.page, this.projectId, datasetId);
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
## UI-first assertions by default
|
||||
|
||||
Assert on what the user sees, using Playwright's built-in locator assertions:
|
||||
|
||||
```ts
|
||||
await expect(panel.traceNameInHeader(trace.name)).toBeVisible();
|
||||
await expect(logs.row(name)).toHaveCount(1);
|
||||
await expect(panel.errorBadge).toBeHidden();
|
||||
```
|
||||
|
||||
Don't register custom matchers. Touch `matchers/register.ts` only if you've identified a specific assertion the built-in locator assertions genuinely can't express — and if you do, the test you ship must actually use it. A registered-but-unused matcher is dead code.
|
||||
|
||||
When you need to confirm something the UI doesn't surface (e.g. a feedback score landed on a source trace after an async rule), read it back through the suite's SDK client — but prefer a UI assertion whenever the UI shows the fact.
|
||||
|
||||
## Selector preference
|
||||
|
||||
Pick the most stable locator available, in this order:
|
||||
|
||||
1. **`getByTestId('descriptive-name')`** — the FE team's stability contract. Use it whenever it exists.
|
||||
2. **`getByRole('button', { name: 'Create dataset' })`** — survives most refactors as long as the accessible name holds.
|
||||
3. **`getByLabel('Dataset name')`** — for labelled form inputs.
|
||||
4. **`getByText(...)`** — only for truly static, non-i18n text.
|
||||
5. **CSS / XPath** — last resort.
|
||||
|
||||
If the only working selector is a structural CSS path (`tbody > tr:nth-child(2)`), stop: add a descriptive kebab-case `data-testid` to the FE component (under `apps/opik-frontend/src/v2/pages/<Page>/...` or its shared dependency) in the **same change** as the POM. Name it for the page/element (`create-dataset-sidebar`, `dataset-items-table`), never generically (`button-1`, `submit`). If you genuinely can't touch the FE, leave a comment explaining why the CSS selector is necessary and that a `data-testid` should be added.
|
||||
|
||||
## Public SDK surface only
|
||||
|
||||
Seed and inspect through the suite's SDK clients and the public `Opik` class. Never deep-import REST internals (`opik/rest_api/*`). The public surface is the contract; internals move.
|
||||
|
||||
## Seed state via the SDK/bridge, not the UI
|
||||
|
||||
Create the state a page needs through the bridge or SDK before you open the browser. UI-create is what the *test* exercises — it's not how you set up for exploration or for a precondition. UI-create is also slower and flakier as a setup step.
|
||||
|
||||
## Fixture seed shapes must match what the page renders
|
||||
|
||||
An empty project shows only the empty state; a dataset with no items shows only the "create item" CTA. If your page needs rows, sort order, or a pass/fail mix to be meaningful, the fixture must seed that shape. Reuse existing fixtures (`project`, `dataset`, `trace`, `experiment`, `testSuite`) where they fit; add a new one only when the shape genuinely differs. Verify teardown: some entities cascade with the project, some need explicit deletion — check and clean up what you create.
|
||||
|
||||
## Verify the test render before blaming the backend
|
||||
|
||||
When something "isn't appearing," the usual cause is a DOM race (a loading spinner still up, an eventually-consistent write not yet landed), not a backend regression. Read the failure trace artifact first — `npx playwright show-trace` — and confirm the page actually finished rendering before concluding the data is missing. For genuinely async state (online scoring rules, ingestion lag), poll with `expect.poll(...)` rather than a fixed `setTimeout`.
|
||||
|
||||
## Tags
|
||||
|
||||
Every test carries a tier tag and a feature tag. Tiers are inclusive: `test:t2` runs `@t1-smoke|@t2-cuj`, `test:t3` runs all three.
|
||||
|
||||
- `@t1-smoke` — fast, deterministic, always-on core checks.
|
||||
- `@t2-cuj` — core user journeys; multi-step flows.
|
||||
- `@t3-nightly` — broader / slower coverage.
|
||||
|
||||
Apply them on the describe block alongside the feature tag:
|
||||
|
||||
```ts
|
||||
test.describe('Dataset CRUD — smoke', { tag: ['@t1-smoke', '@datasets'] }, () => {
|
||||
// ...
|
||||
});
|
||||
```
|
||||
|
||||
Pick the tier by what the test costs and how core it is; pick the feature tag to match the page family (`@datasets`, `@trace-explore`, `@experiments`, …).
|
||||
@@ -0,0 +1,53 @@
|
||||
# Cursor-specific ignore file
|
||||
# Prevents Cursor from indexing/scanning these directories
|
||||
|
||||
# Dependencies
|
||||
**/node_modules/
|
||||
**/target/
|
||||
**/.venv/
|
||||
**/.venv*/
|
||||
**/venv/
|
||||
**/__pycache__/
|
||||
|
||||
# Build outputs
|
||||
**/dist/
|
||||
**/build/
|
||||
**/.next/
|
||||
**/.nuxt/
|
||||
**/.output/
|
||||
|
||||
# Caches
|
||||
**/.cache/
|
||||
**/.pytest_cache/
|
||||
**/.ruff_cache/
|
||||
**/.mypy_cache/
|
||||
**/.eslintcache
|
||||
**/.tsbuildinfo
|
||||
**/.turbo/
|
||||
|
||||
# IDE
|
||||
**/.idea/
|
||||
**/.vscode/
|
||||
**/.DS_Store
|
||||
|
||||
# Test outputs
|
||||
**/coverage/
|
||||
**/htmlcov/
|
||||
**/.nyc_output/
|
||||
|
||||
# Generated files
|
||||
**/*.egg-info/
|
||||
**/*.pyc
|
||||
**/*.pyo
|
||||
**/*.pyd
|
||||
**/.Python
|
||||
|
||||
# Large data directories
|
||||
**/artifacts/
|
||||
**/benchmark_results/
|
||||
|
||||
# Temporary files
|
||||
**/temp/
|
||||
**/tmp/
|
||||
**/*.log
|
||||
**/*.tmp
|
||||
@@ -0,0 +1,16 @@
|
||||
root = true
|
||||
|
||||
[*]
|
||||
charset = utf-8
|
||||
end_of_line = lf
|
||||
insert_final_newline = true
|
||||
trim_trailing_whitespace = true
|
||||
indent_style = space
|
||||
indent_size = 2
|
||||
|
||||
[*.py]
|
||||
indent_size = 4
|
||||
|
||||
[Makefile]
|
||||
indent_style = tab
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
# Environment Variables Template
|
||||
# Copy this file to .env.local and fill in your actual values
|
||||
# DO NOT COMMIT .env.local to version control
|
||||
|
||||
# GitHub Personal Access Token
|
||||
# Get from: https://github.com/settings/tokens
|
||||
GITHUB_PERSONAL_ACCESS_TOKEN=your_github_token_here
|
||||
|
||||
|
||||
# Jira Headless MCP Configuration
|
||||
# Get API token from: https://id.atlassian.com/manage-profile/security/api-tokens
|
||||
JIRA_URL=https://comet-ml.atlassian.net
|
||||
JIRA_USERNAME=your.email@comet.com
|
||||
JIRA_API_TOKEN=your_jira_api_token_here
|
||||
|
||||
|
||||
# Sentry MCP Configuration
|
||||
# Create a User Auth Token at: https://sentry.io/settings/account/api/auth-tokens/
|
||||
# Required scopes: org:read, project:read, team:read, event:write
|
||||
# Optional (for write actions): project:write, team:write
|
||||
# See: .agents/docs/SENTRY_MCP_SETUP.md
|
||||
SENTRY_ACCESS_TOKEN=your_sentry_access_token_here
|
||||
@@ -0,0 +1,5 @@
|
||||
# OPIK-5097: FE v1/v2 structural reorganization (file moves + import rewrites)
|
||||
118933eb7554bf8c7ebe632afda5af692d0e24a7
|
||||
|
||||
# NA: Backend Spotless reformat sweep (whitespace-only, ~53 files under apps/opik-backend)
|
||||
6fcfa6f8a8c03dd5ccab6614b7b2a38395127b8d
|
||||
@@ -0,0 +1,8 @@
|
||||
*.java linguist-detectable=false
|
||||
*.mdx linguist-detectable=false
|
||||
*.sh text eol=lf
|
||||
|
||||
# Prevent EOL normalization drift in symlink blobs.
|
||||
.cursor -text
|
||||
.codex -text
|
||||
CLAUDE.md -text
|
||||
@@ -0,0 +1,23 @@
|
||||
# This is a comment.
|
||||
# Each line is a file pattern followed by one or more owners.
|
||||
|
||||
# These owners will be the default owners for everything in
|
||||
# the repo. Unless a later match takes precedence,
|
||||
# @comet-ml/comet-opik-devs will be requested for
|
||||
# review when someone opens a pull request.
|
||||
* @comet-ml/comet-opik-devs # This is an inline comment.
|
||||
|
||||
*.md @comet-ml/product @comet-ml/comet-opik-devs
|
||||
*.mdx @comet-ml/product @comet-ml/comet-opik-devs
|
||||
|
||||
/.github/ISSUE_TEMPLATE/ @comet-ml/product @comet-ml/comet-opik-devs
|
||||
/.github/actions/ @comet-ml/comet-devops @comet-ml/comet-qa @comet-ml/comet-opik-devs
|
||||
/.github/workflows/ @comet-ml/comet-devops @comet-ml/comet-qa @comet-ml/comet-opik-devs
|
||||
|
||||
/apps/opik-documentation/ @comet-ml/product @comet-ml/comet-opik-devs
|
||||
|
||||
/deployment/ @comet-ml/comet-devops @comet-ml/deployment-team @comet-ml/comet-opik-devs
|
||||
|
||||
/sdks/opik_optimizer @comet-ml/comet-opik-optimizer-devs @comet-ml/comet-opik-devs
|
||||
|
||||
/tests_end_to_end/ @comet-ml/comet-qa @comet-ml/comet-opik-devs
|
||||
@@ -0,0 +1,56 @@
|
||||
name: Bug Report
|
||||
description: File a bug report.
|
||||
title: "[Bug]: "
|
||||
labels: ["bug", "triage"]
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Thank you for submitting a bug report.
|
||||
#### To help us resolve your issue, please provide fill in this bug report template.
|
||||
- type: checkboxes
|
||||
attributes:
|
||||
label: What component(s) are affected?
|
||||
options:
|
||||
- label: Opik Python SDK
|
||||
required: false
|
||||
- label: Opik Typescript SDK
|
||||
required: false
|
||||
- label: Opik Agent Optimizer SDK
|
||||
required: false
|
||||
- label: Opik UI
|
||||
required: false
|
||||
- label: Opik Server
|
||||
required: false
|
||||
- label: Documentation
|
||||
required: false
|
||||
validations:
|
||||
required: false
|
||||
- type: textarea
|
||||
validations:
|
||||
required: true
|
||||
attributes:
|
||||
label: Opik version
|
||||
placeholder: The Opik version, you can find the Opik version by running `opik.__version__` in Python.
|
||||
value: |
|
||||
- Opik version: x.x.x
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Describe the problem
|
||||
description: |
|
||||
Describe the problem clearly here, you should include both the expected behavior and the actual behavior.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Reproduction steps and code snippets
|
||||
description: Describe how to reproduce your bug. Please provide detailed steps and include code snippets if possible.
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Error logs or stack trace
|
||||
description: Please provide the error messages, logs, or stack trace that appear when your bug is reproduced, if possible.
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Healthcheck results
|
||||
description: Run `opik healthcheck` to perform some basic checks which might be useful for troubleshooting your issue. Paste your results as a text or a screenshot.
|
||||
placeholder: CLI output from 'opik healthcheck'...
|
||||
@@ -0,0 +1,8 @@
|
||||
blank_issues_enabled: false
|
||||
contact_links:
|
||||
- name: Explore Opik Documentation
|
||||
url: https://www.comet.com/docs/opik
|
||||
about: New to Opik? read our documentation to get setup.
|
||||
- name: Ask Opik Community Support
|
||||
url: https://chat.comet.com
|
||||
about: Ask usage questions in the community before opening an issue.
|
||||
@@ -0,0 +1,27 @@
|
||||
name: Feature Request
|
||||
description: File a feature request.
|
||||
title: "[FR]: "
|
||||
labels: ["enhancement"]
|
||||
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Thank you for submitting a feature request.
|
||||
#### To help us prioritize your request, please provide fill in this feature request template.
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Proposal summary
|
||||
description: Provide a brief summary of your feature request. Adding a code (or pseudo-code) snippet that describes your use-case or the way you would like the feature to work is highly appreciated. The details help the team to process your request faster.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
validations:
|
||||
required: false
|
||||
attributes:
|
||||
label: Motivation
|
||||
description: |
|
||||
Describe the motivation for your feature request, this can include:
|
||||
- What problem are you trying to solve?
|
||||
- How are you currently solving this problem?
|
||||
- What are the benefits of this feature?
|
||||
@@ -0,0 +1,5 @@
|
||||
self-hosted-runner:
|
||||
labels:
|
||||
# Org-hosted larger runner. Not a public GitHub label, so actionlint
|
||||
# needs it declared here to avoid a false "unknown label" error.
|
||||
- ubuntu-latest-m
|
||||
@@ -0,0 +1,39 @@
|
||||
---
|
||||
name: Install OPIK SDK and run E2E lib tests
|
||||
description: Install OPIK SDK with test dependencies and executes E2E integration tests for particular library
|
||||
inputs:
|
||||
library_name:
|
||||
required: true
|
||||
description: "The library name"
|
||||
python_version:
|
||||
required: true
|
||||
description: "Python version"
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Install OPIK, test requirements, and run tests for ${{ inputs.library_name }}
|
||||
shell: bash
|
||||
run: |
|
||||
# create virtual environment
|
||||
uv venv ${{ inputs.library_name }}-env --python ${{ inputs.python_version }}
|
||||
source ${{ inputs.library_name }}-env/bin/activate
|
||||
|
||||
cd ${{ github.workspace }}/sdks/python
|
||||
|
||||
echo "Installing OPIK"
|
||||
uv pip install .
|
||||
|
||||
echo "Installing test dependencies"
|
||||
cd ./tests
|
||||
uv pip install --no-cache-dir --disable-pip-version-check -r test_requirements.txt
|
||||
uv pip install --no-cache-dir --disable-pip-version-check -r e2e_library_integration/${{ inputs.library_name }}/requirements.txt
|
||||
uv pip list
|
||||
|
||||
echo "Running tests"
|
||||
export OPIK_CONSOLE_LOGGING_LEVEL=DEBUG
|
||||
cd ./e2e_library_integration/${{ inputs.library_name }}/
|
||||
python -m pytest -vv .
|
||||
|
||||
# Deactivate virtual environment
|
||||
deactivate
|
||||
@@ -0,0 +1,79 @@
|
||||
---
|
||||
name: NPM Token Preflight
|
||||
description: |
|
||||
Validate NPM_TOKEN before publish: confirms the token authenticates against
|
||||
the registry and is authorized to publish the named package. Fails fast with
|
||||
an actionable message so we never reach `npm publish` with a bad token.
|
||||
inputs:
|
||||
npm_token:
|
||||
required: true
|
||||
description: NPM_TOKEN secret value (passed via env, never echoed)
|
||||
package_name:
|
||||
required: true
|
||||
description: Package name to verify publish access for (e.g. opik, opik-openai)
|
||||
registry_url:
|
||||
required: false
|
||||
description: NPM registry URL
|
||||
default: "https://registry.npmjs.org"
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: NPM token preflight (${{ inputs.package_name }})
|
||||
shell: bash
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ inputs.npm_token }}
|
||||
PACKAGE_NAME: ${{ inputs.package_name }}
|
||||
REGISTRY_URL: ${{ inputs.registry_url }}
|
||||
run: |
|
||||
set -e
|
||||
|
||||
if [ -z "${NODE_AUTH_TOKEN:-}" ]; then
|
||||
echo "::error title=NPM preflight::NPM_TOKEN is empty. The secret is not configured or not exposed to this job."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
REGISTRY_HOST="${REGISTRY_URL#https://}"
|
||||
REGISTRY_HOST="${REGISTRY_HOST#http://}"
|
||||
REGISTRY_HOST="${REGISTRY_HOST%/}"
|
||||
|
||||
TMP_NPMRC="$(mktemp)"
|
||||
trap 'rm -f "$TMP_NPMRC"' EXIT
|
||||
# shellcheck disable=SC2016
|
||||
# Intentional literal ${NODE_AUTH_TOKEN}: npm expands it at .npmrc read time,
|
||||
# keeping the token out of process args and shell history.
|
||||
printf '//%s/:_authToken=${NODE_AUTH_TOKEN}\nregistry=%s\n' "$REGISTRY_HOST" "$REGISTRY_URL" > "$TMP_NPMRC"
|
||||
|
||||
echo "==> npm whoami"
|
||||
if ! WHOAMI_OUT=$(npm --userconfig "$TMP_NPMRC" whoami 2>&1); then
|
||||
echo "::error title=NPM preflight::npm whoami failed. NPM_TOKEN is invalid, expired, or revoked."
|
||||
echo "Output:"
|
||||
echo "$WHOAMI_OUT"
|
||||
exit 1
|
||||
fi
|
||||
echo "Authenticated as: $WHOAMI_OUT"
|
||||
|
||||
echo "==> npm access list packages (filtering for ${PACKAGE_NAME})"
|
||||
if ! ACCESS_OUT=$(npm --userconfig "$TMP_NPMRC" access list packages 2>&1); then
|
||||
echo "::warning title=NPM preflight::npm access list packages failed; skipping authorization check."
|
||||
echo "Output:"
|
||||
echo "$ACCESS_OUT"
|
||||
echo "Proceeding — publish step will be the source of truth."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Escape ERE metacharacters in PACKAGE_NAME so e.g. a future name containing
|
||||
# `.` doesn't match a different package with any char at that position.
|
||||
PACKAGE_NAME_ERE=$(printf '%s' "$PACKAGE_NAME" | sed -e 's/[][\.*+?(){}|^$\\]/\\&/g')
|
||||
|
||||
if echo "$ACCESS_OUT" | grep -qE "^[[:space:]]*\"?${PACKAGE_NAME_ERE}\"?[[:space:]:]+\"?(read-write|write)\"?"; then
|
||||
echo "Token has write access to ${PACKAGE_NAME}."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Classic (legacy) tokens don't always return per-package scopes via this command.
|
||||
# If the package is missing from the list but whoami succeeded, treat as pass-with-warning
|
||||
# rather than block — publish will still surface a real auth error if it exists.
|
||||
echo "::warning title=NPM preflight::Could not confirm write access to ${PACKAGE_NAME} from \`npm access list packages\` output."
|
||||
echo "This is expected for classic automation tokens; granular tokens should list the package explicitly."
|
||||
echo "Continuing — publish step will surface any real authorization error."
|
||||
@@ -0,0 +1,81 @@
|
||||
---
|
||||
name: PyPI Token Preflight
|
||||
description: |
|
||||
Validate PYPI_API_TOKEN before publish: confirms the token has the expected
|
||||
format and is accepted by the PyPI upload endpoint. Fails fast with an
|
||||
actionable message so we never reach `twine upload` with a bad token.
|
||||
inputs:
|
||||
pypi_token:
|
||||
required: true
|
||||
description: PYPI_API_TOKEN secret value (passed via env, never echoed)
|
||||
package_name:
|
||||
required: true
|
||||
description: Package name being published (for diagnostics only; PyPI does not gate at this layer)
|
||||
upload_url:
|
||||
required: false
|
||||
description: PyPI upload endpoint
|
||||
default: "https://upload.pypi.org/legacy/"
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: PyPI token preflight (${{ inputs.package_name }})
|
||||
shell: bash
|
||||
env:
|
||||
PYPI_TOKEN: ${{ inputs.pypi_token }}
|
||||
PACKAGE_NAME: ${{ inputs.package_name }}
|
||||
UPLOAD_URL: ${{ inputs.upload_url }}
|
||||
run: |
|
||||
set -e
|
||||
|
||||
if [ -z "${PYPI_TOKEN:-}" ]; then
|
||||
echo "::error title=PyPI preflight::PYPI_API_TOKEN is empty. The secret is not configured or not exposed to this job."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ "$PYPI_TOKEN" != pypi-* ]]; then
|
||||
echo "::error title=PyPI preflight::PYPI_API_TOKEN does not start with 'pypi-'. PyPI API tokens must use the 'pypi-' prefix."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Token format OK (pypi-* prefix)."
|
||||
echo "==> Probing PyPI upload endpoint: ${UPLOAD_URL}"
|
||||
|
||||
# POST with no multipart body. PyPI's response distinguishes auth from payload errors:
|
||||
# 401/403 -> token rejected (bad/expired/wrong-scope)
|
||||
# 400/422 -> token accepted, but request body invalid (this is the success case for a preflight)
|
||||
# 200 -> shouldn't happen without a real upload, but also indicates auth OK
|
||||
HTTP_CODE=$(curl -s -o /tmp/pypi_probe_body -w "%{http_code}" \
|
||||
-X POST \
|
||||
-u "__token__:${PYPI_TOKEN}" \
|
||||
-H "User-Agent: opik-release-preflight" \
|
||||
"${UPLOAD_URL}" || echo "000")
|
||||
|
||||
echo "HTTP ${HTTP_CODE}"
|
||||
|
||||
case "$HTTP_CODE" in
|
||||
400|422)
|
||||
echo "Token accepted by PyPI (HTTP ${HTTP_CODE} = empty/invalid payload, but auth passed)."
|
||||
exit 0
|
||||
;;
|
||||
200)
|
||||
echo "::warning title=PyPI preflight::Unexpected 200 OK from an empty POST; treating as auth-success."
|
||||
exit 0
|
||||
;;
|
||||
401|403)
|
||||
echo "::error title=PyPI preflight::PYPI_API_TOKEN was rejected by ${UPLOAD_URL} (HTTP ${HTTP_CODE}). Token is invalid, expired, revoked, or not scoped for ${PACKAGE_NAME}."
|
||||
echo "Response body:"
|
||||
cat /tmp/pypi_probe_body || true
|
||||
exit 1
|
||||
;;
|
||||
000)
|
||||
echo "::error title=PyPI preflight::Could not reach ${UPLOAD_URL} at all. Network or DNS issue on the runner."
|
||||
exit 1
|
||||
;;
|
||||
*)
|
||||
echo "::warning title=PyPI preflight::Unexpected HTTP ${HTTP_CODE} from ${UPLOAD_URL}. Continuing — publish step will be the source of truth."
|
||||
echo "Response body:"
|
||||
cat /tmp/pypi_probe_body || true
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,59 @@
|
||||
name: Select E2E version matrix
|
||||
description: >
|
||||
Emits the full version array or a reduced [min, max] array depending on whether
|
||||
the event's changed files match the given SDK-touching regex. Full matrix on
|
||||
workflow_dispatch and when any changed file matches the regex; min+max otherwise.
|
||||
|
||||
inputs:
|
||||
all-versions:
|
||||
required: true
|
||||
description: JSON array of all supported versions (e.g. '["3.10","3.14"]').
|
||||
match-regex:
|
||||
required: true
|
||||
description: POSIX extended regex; any changed file matching it forces the full matrix.
|
||||
github-token:
|
||||
required: true
|
||||
description: Token used to list PR/commit files via the GitHub API.
|
||||
|
||||
outputs:
|
||||
versions:
|
||||
description: JSON array of versions the caller should feed into its strategy matrix.
|
||||
value: ${{ steps.pick.outputs.versions }}
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- id: pick
|
||||
shell: bash
|
||||
env:
|
||||
GH_TOKEN: ${{ inputs.github-token }}
|
||||
ALL: ${{ inputs.all-versions }}
|
||||
REGEX: ${{ inputs.match-regex }}
|
||||
REPO: ${{ github.repository }}
|
||||
EVENT: ${{ github.event_name }}
|
||||
PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
BEFORE: ${{ github.event.before }}
|
||||
SHA: ${{ github.sha }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ "$EVENT" = "workflow_dispatch" ]; then
|
||||
echo "versions=$ALL" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
if [ "$EVENT" = "pull_request" ]; then
|
||||
FILES=$(gh api --paginate "repos/$REPO/pulls/$PR_NUMBER/files" --jq '.[].filename')
|
||||
elif [ -n "$BEFORE" ] && [ "$BEFORE" != "0000000000000000000000000000000000000000" ]; then
|
||||
# Aggregate across the full push range so multi-commit pushes (rebase-merge, direct pushes)
|
||||
# don't miss SDK-touching files landed in non-tip commits.
|
||||
FILES=$(gh api "repos/$REPO/compare/$BEFORE...$SHA" --jq '.files[].filename')
|
||||
else
|
||||
# Branch creation / force-push: before is zero or missing; fall back to tip.
|
||||
FILES=$(gh api "repos/$REPO/commits/$SHA" --jq '.files[].filename')
|
||||
fi
|
||||
# Default to full matrix on empty file list so a degenerate API response
|
||||
# never silently under-tests.
|
||||
if [ -z "$FILES" ] || echo "$FILES" | grep -Eq "$REGEX"; then
|
||||
echo "versions=$ALL" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "versions=$(echo "$ALL" | jq -c '[.[0], .[-1]]')" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
@@ -0,0 +1,279 @@
|
||||
---
|
||||
name: TypeScript SDK Build & Publish
|
||||
description: Common steps for building and publishing TypeScript SDK packages
|
||||
inputs:
|
||||
working_directory:
|
||||
required: true
|
||||
description: Working directory for the package (e.g., sdks/typescript or sdks/typescript/src/opik/integrations/opik-openai)
|
||||
version:
|
||||
required: false
|
||||
description: Version to use (if not provided, will generate from package.json + SHA)
|
||||
is_release:
|
||||
required: false
|
||||
description: Whether this is a release (publishes to NPM)
|
||||
default: "false"
|
||||
package_name:
|
||||
required: false
|
||||
description: Package name for commit message and summary (defaults to package.json name)
|
||||
npm_access:
|
||||
required: false
|
||||
description: NPM publish access (defaults to empty, use 'public' for scoped packages)
|
||||
default: ""
|
||||
enable_cache:
|
||||
required: false
|
||||
description: Enable npm cache
|
||||
default: "false"
|
||||
cache_dependency_path:
|
||||
required: false
|
||||
description: Path to package-lock.json for cache
|
||||
default: ""
|
||||
test_package_installation:
|
||||
required: false
|
||||
description: Run npm pack --dry-run test
|
||||
default: "false"
|
||||
npm_token:
|
||||
required: false
|
||||
description: NPM token for publishing
|
||||
github_token:
|
||||
required: false
|
||||
description: GitHub token for pushing changes
|
||||
skip_commit_push:
|
||||
required: false
|
||||
description: Skip commit and push (for parallel builds in release)
|
||||
default: "false"
|
||||
outputs:
|
||||
version:
|
||||
description: Generated or provided version
|
||||
value: ${{ steps.set_outputs.outputs.version }}
|
||||
package_name:
|
||||
description: Package name from package.json
|
||||
value: ${{ steps.set_outputs.outputs.package_name }}
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: "20"
|
||||
registry-url: "https://registry.npmjs.org"
|
||||
cache: ${{ inputs.enable_cache == 'true' && 'npm' || '' }}
|
||||
cache-dependency-path: ${{ inputs.cache_dependency_path }}
|
||||
|
||||
- name: Install dependencies
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: npm ci
|
||||
|
||||
- name: Generate version
|
||||
id: version
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
env:
|
||||
# Read inputs via env, not `${{ }}` substitution, so a malicious dispatch
|
||||
# input can't inject shell tokens (OWASP / GHA security hardening).
|
||||
INPUT_VERSION: ${{ inputs.version }}
|
||||
run: |
|
||||
if [ -n "${INPUT_VERSION}" ]; then
|
||||
# Validate against a permissive semver regex (X.Y.Z + optional
|
||||
# prerelease/build suffix). Anything else fails fast.
|
||||
if ! [[ "${INPUT_VERSION}" =~ ^[0-9]+\.[0-9]+\.[0-9]+(-[A-Za-z0-9.-]+)?(\+[A-Za-z0-9.-]+)?$ ]]; then
|
||||
echo "::error title=Generate version::inputs.version='${INPUT_VERSION}' does not match the expected semver shape (X.Y.Z[-pre][+build])."
|
||||
exit 1
|
||||
fi
|
||||
VERSION="${INPUT_VERSION}"
|
||||
else
|
||||
# For push/PR: use package.json version with short SHA (not published).
|
||||
PACKAGE_VERSION=$(node -p "require('./package.json').version")
|
||||
SHORT_SHA=$(git rev-parse --short HEAD)
|
||||
VERSION="${PACKAGE_VERSION}-${SHORT_SHA}"
|
||||
fi
|
||||
echo "VERSION=${VERSION}" >> "$GITHUB_OUTPUT"
|
||||
echo "Generated version: ${VERSION}"
|
||||
|
||||
- name: Update version
|
||||
if: ${{ inputs.is_release == 'true' }}
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
env:
|
||||
# `RELEASE_VERSION` comes via env (not `${{ }}` substitution) so it can't
|
||||
# inject shell tokens into the `npm version` command (OWASP). Even though
|
||||
# the `Generate version` step now regex-validates inputs.version, env+
|
||||
# quoted is the defense-in-depth shape recommended by GHA security docs.
|
||||
RELEASE_VERSION: ${{ steps.version.outputs.VERSION }}
|
||||
# `--allow-same-version` treats "package.json already at this version" as a
|
||||
# successful no-op instead of erroring `Version not changed`. The operator's
|
||||
# intent on dispatch is "publish this version" — package.json already being
|
||||
# at the target value is bookkeeping that npm publish doesn't care about, so
|
||||
# the bump step shouldn't either. Without this, replay/re-dispatch of an
|
||||
# already-released version (where the prior run's commit-version step landed
|
||||
# on the dispatched ref) fails here before reaching the publish step's own
|
||||
# already-published skip.
|
||||
run: npm version "${RELEASE_VERSION}" --no-git-tag-version --allow-same-version
|
||||
|
||||
- name: Build package
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: npm run build
|
||||
|
||||
- name: Run tests
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: npm test
|
||||
|
||||
- name: Run linting
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: npm run lint
|
||||
|
||||
- name: Run type checking
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: npm run typecheck
|
||||
|
||||
- name: Test package installation (dry-run)
|
||||
if: ${{ inputs.test_package_installation == 'true' }}
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: |
|
||||
echo "Testing package installation..."
|
||||
npm pack --dry-run
|
||||
echo "Package can be installed successfully"
|
||||
|
||||
- name: Resolve npm package name from package.json
|
||||
id: npm_name
|
||||
if: ${{ inputs.is_release == 'true' }}
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: |
|
||||
NPM_PACKAGE_NAME=$(node -p "require('./package.json').name")
|
||||
echo "name=${NPM_PACKAGE_NAME}" >> "$GITHUB_OUTPUT"
|
||||
echo "Resolved npm package name: ${NPM_PACKAGE_NAME}"
|
||||
|
||||
- name: NPM token preflight
|
||||
if: ${{ inputs.is_release == 'true' }}
|
||||
uses: ./.github/actions/npm-token-preflight
|
||||
with:
|
||||
npm_token: ${{ inputs.npm_token }}
|
||||
package_name: ${{ steps.npm_name.outputs.name }}
|
||||
|
||||
- name: Publish to NPM (with retry)
|
||||
id: npm_publish
|
||||
if: ${{ inputs.is_release == 'true' }}
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: |
|
||||
NEW_VERSION="${{ steps.version.outputs.VERSION }}"
|
||||
PACKAGE_NAME=$(node -p "require('./package.json').name")
|
||||
|
||||
echo "Publishing to NPM: $PACKAGE_NAME@$NEW_VERSION"
|
||||
|
||||
# Retry logic: 3 attempts with 10 second delays and 3-minute timeout per attempt
|
||||
MAX_ATTEMPTS=3
|
||||
ATTEMPT=1
|
||||
SUCCESS=false
|
||||
TIMEOUT_SECONDS=180 # 3 minutes per attempt
|
||||
|
||||
PUBLISH_CMD="npm publish"
|
||||
if [ -n "${{ inputs.npm_access }}" ]; then
|
||||
PUBLISH_CMD="npm publish --access ${{ inputs.npm_access }}"
|
||||
fi
|
||||
|
||||
ALREADY_PUBLISHED=false
|
||||
PUBLISH_LOG=$(mktemp)
|
||||
trap 'rm -f "$PUBLISH_LOG"' EXIT
|
||||
|
||||
while [ $ATTEMPT -le $MAX_ATTEMPTS ]; do
|
||||
echo "Publish attempt $ATTEMPT of $MAX_ATTEMPTS (timeout: ${TIMEOUT_SECONDS}s)..."
|
||||
|
||||
# Run the publish via `if pipeline; then ... else ...; fi`. This shape is
|
||||
# load-bearing for three reasons; do not rewrite to a bare pipeline:
|
||||
# 1. The `if` condition shields the pipeline from `set -e` (GitHub Actions
|
||||
# `shell: bash` uses `bash --noprofile --norc -eo pipefail`; without
|
||||
# this shield, a non-zero pipeline kills the step before we evaluate
|
||||
# the 403-skip below).
|
||||
# 2. PIPESTATUS is preserved inside the `else` branch (a trailing
|
||||
# `|| true` would clobber PIPESTATUS to reflect only `true`, masking
|
||||
# every failure as exit-0 and silently turning real publish failures
|
||||
# into false-green CI).
|
||||
# 3. `tee` streams output live AND captures it to PUBLISH_LOG for the
|
||||
# grep-based 403-already-published check.
|
||||
if timeout $TIMEOUT_SECONDS $PUBLISH_CMD 2>&1 | tee "$PUBLISH_LOG"; then
|
||||
SUCCESS=true
|
||||
break
|
||||
else
|
||||
EXIT_CODE=${PIPESTATUS[0]}
|
||||
|
||||
if [ "$EXIT_CODE" -eq 124 ]; then
|
||||
echo "Attempt $ATTEMPT timed out after ${TIMEOUT_SECONDS}s"
|
||||
else
|
||||
echo "Attempt $ATTEMPT failed with exit code $EXIT_CODE"
|
||||
fi
|
||||
|
||||
# Non-retryable: version already published. The desired end state is
|
||||
# achieved; treat as a skip-with-warning so concurrent / replay
|
||||
# dispatches don't fail CI. Match BOTH `E403` and the specific phrase
|
||||
# `cannot publish over` so we don't accidentally treat other 403s
|
||||
# (e.g. `You may not perform that action with these credentials`, which
|
||||
# means the token lacks publish scope on this package) as already-
|
||||
# published — those must still fail loudly.
|
||||
if grep -q "E403" "$PUBLISH_LOG" && grep -qi "cannot publish over" "$PUBLISH_LOG"; then
|
||||
ALREADY_PUBLISHED=true
|
||||
echo "::warning title=NPM publish skipped::${PACKAGE_NAME}@${NEW_VERSION} is already published; treating as success."
|
||||
break
|
||||
fi
|
||||
|
||||
if [ $ATTEMPT -lt $MAX_ATTEMPTS ]; then
|
||||
echo "Retrying in 10 seconds..."
|
||||
sleep 10
|
||||
fi
|
||||
ATTEMPT=$((ATTEMPT + 1))
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$SUCCESS" = false ] && [ "$ALREADY_PUBLISHED" = false ]; then
|
||||
echo "Failed to publish after $MAX_ATTEMPTS attempts"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Save package name and version for summary
|
||||
echo "PACKAGE_NAME=$PACKAGE_NAME" >> $GITHUB_ENV
|
||||
echo "PUBLISHED_VERSION=$NEW_VERSION" >> $GITHUB_ENV
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ inputs.npm_token }}
|
||||
|
||||
- name: Commit version changes
|
||||
if: ${{ inputs.is_release == 'true' && inputs.skip_commit_push != 'true' }}
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: |
|
||||
git config --local user.email "github-actions@comet.com"
|
||||
git config --local user.name "github-actions"
|
||||
git add package.json package-lock.json
|
||||
|
||||
PACKAGE_NAME="${{ inputs.package_name }}"
|
||||
if [ -z "$PACKAGE_NAME" ]; then
|
||||
PACKAGE_NAME=$(node -p "require('./package.json').name")
|
||||
fi
|
||||
|
||||
git commit -m "Update $PACKAGE_NAME version to ${{ steps.version.outputs.VERSION }}"
|
||||
|
||||
- name: Push changes
|
||||
if: ${{ inputs.is_release == 'true' && inputs.skip_commit_push != 'true' }}
|
||||
uses: ad-m/github-push-action@v0.8.0
|
||||
with:
|
||||
github_token: ${{ inputs.github_token }}
|
||||
branch: ${{ github.ref }}
|
||||
force_with_lease: true
|
||||
|
||||
- name: Set outputs
|
||||
id: set_outputs
|
||||
shell: bash
|
||||
working-directory: ${{ inputs.working_directory }}
|
||||
run: |
|
||||
echo "version=${{ steps.version.outputs.VERSION }}" >> $GITHUB_OUTPUT
|
||||
|
||||
# Get package name from package.json
|
||||
PACKAGE_NAME=$(node -p "require('./package.json').name")
|
||||
echo "package_name=$PACKAGE_NAME" >> $GITHUB_OUTPUT
|
||||
|
||||
@@ -0,0 +1,553 @@
|
||||
# Copilot Code Review Instructions
|
||||
|
||||
> **Scope:** These guidelines apply to all Opik applications including backend, frontend, and SDKs. Use the appropriate sections based on the code being reviewed.
|
||||
|
||||
When Copilot automatically reviews pull requests, use the following guidelines to structure feedback and ensure consistency across the entire Opik project.
|
||||
|
||||
---
|
||||
|
||||
## Project Overview
|
||||
|
||||
Opik is a comprehensive observability and AI evaluation platform with multiple applications:
|
||||
|
||||
- **Backend**: Java-based REST API with MySQL and ClickHouse databases
|
||||
- **Frontend**: React/TypeScript application with modern UI components
|
||||
- **Python SDK**: Client library for Python applications
|
||||
- **TypeScript SDK**: Client library for TypeScript/JavaScript applications
|
||||
- **Documentation**: Comprehensive documentation site
|
||||
- **Testing**: End-to-end and load testing suites
|
||||
|
||||
## 1. Git Workflow & Branch Management
|
||||
|
||||
### Branch Naming Convention
|
||||
```
|
||||
{username}/{ticket}-{summary}
|
||||
```
|
||||
|
||||
**Examples:**
|
||||
```
|
||||
andrescrz/OPIK-2236-add-documentation-and-user-facing-distinction-to-pr-template
|
||||
someuser/issue-1234-some-task
|
||||
someotheruser/NA-some-other-task
|
||||
```
|
||||
|
||||
### Commit Message Standards
|
||||
Use component types to categorize changes (optional but recommended):
|
||||
- `[DOCS]` - Documentation updates, README changes, comments, swagger/OpenAPI documentation
|
||||
- `[FE]` - Frontend changes (React, TypeScript, UI components)
|
||||
- `[BE]` - Backend changes (Java, API endpoints, services)
|
||||
- `[SDK]` - SDK changes (Python, TypeScript SDKs)
|
||||
|
||||
**Examples:**
|
||||
```bash
|
||||
# ✅ Recommended format
|
||||
git commit -m "[OPIK-1234] [FE] Add project custom metrics UI dashboard"
|
||||
git commit -m "[OPIK-1234] [BE] Add create trace endpoint"
|
||||
|
||||
# ✅ Also acceptable
|
||||
git commit -m "[OPIK-1234] Add project custom metrics UI dashboard"
|
||||
```
|
||||
|
||||
### Pull Request Guidelines
|
||||
**Title Format:** `[{ticket}] [{component}] {summary}`
|
||||
|
||||
**Required Sections:**
|
||||
- **Details**: What the change does, why it was made, and any design decisions
|
||||
- **Change checklist**: User facing and Documentation update checkboxes
|
||||
- **Issues**: GitHub issue or Jira ticket references
|
||||
- **Testing**: Scenarios covered by tests and steps to reproduce
|
||||
- **Documentation**: List of docs updated or new configuration introduced
|
||||
|
||||
---
|
||||
|
||||
## 2. Backend (Java) Review Guidelines
|
||||
|
||||
### Technology Stack
|
||||
- **Language**: Java 25
|
||||
- **Framework**: Dropwizard 5.0.0
|
||||
- **Database**: MySQL 9.7.0, ClickHouse 0.9.0
|
||||
- **Build Tool**: Maven with Spotless 3.5.1
|
||||
- **Testing**: JUnit 5, Testcontainers, WireMock
|
||||
|
||||
### Architecture Requirements
|
||||
- **Layered Architecture**: Resources → Services → DAOs → Models
|
||||
- **Separation of Concerns**: Each layer has a single responsibility
|
||||
- **Dependency Injection**: Use Guice with `@Singleton` and `@RequiredArgsConstructor`
|
||||
- **Reactive Design**: Applications must be reactive, non-blocking, and horizontally scalable
|
||||
|
||||
### API Design Standards
|
||||
- **REST Endpoints**: Follow standard HTTP methods and URL patterns
|
||||
- **Validation**: Use `@Valid` and Jakarta validation annotations
|
||||
- **Documentation**: Include `@Operation` with proper `operationId`
|
||||
- **Response Codes**: Use appropriate HTTP status codes (200, 201, 400, 404, 500)
|
||||
|
||||
**Example Controller Pattern:**
|
||||
```java
|
||||
@Path("/api/v1/resources")
|
||||
@Produces(MediaType.APPLICATION_JSON)
|
||||
@Consumes(MediaType.APPLICATION_JSON)
|
||||
@RequiredArgsConstructor(onConstructor_ = @Inject)
|
||||
public class ResourcesResource {
|
||||
|
||||
private final @NonNull ResourceService resourceService;
|
||||
|
||||
@POST
|
||||
@Operation(summary = "Create resource", operationId = "createResource")
|
||||
public Response createResource(@Valid ResourceCreateRequest request) {
|
||||
var resource = resourceService.createResource(request);
|
||||
return Response.status(Response.Status.CREATED)
|
||||
.entity(resource)
|
||||
.build();
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Database Access Patterns
|
||||
- **Always use transactions** for MySQL reads/writes
|
||||
- **Use TransactionTemplate** with READ_ONLY or WRITE types
|
||||
- **JDBI3 interfaces** for DAO implementations
|
||||
- **IdGenerator** for UUID v7 generation
|
||||
|
||||
**Example Service Pattern:**
|
||||
```java
|
||||
@Singleton
|
||||
@RequiredArgsConstructor(onConstructor_ = @Inject)
|
||||
public class ResourceService {
|
||||
|
||||
private final @NonNull ResourceDao resourceDao;
|
||||
private final @NonNull IdGenerator idGenerator;
|
||||
private final @NonNull TransactionTemplate transactionTemplate;
|
||||
|
||||
public ResourceResponse createResource(ResourceCreateRequest request) {
|
||||
return transactionTemplate.inTransaction(WRITE, handle -> {
|
||||
var repository = handle.attach(ResourceDao.class);
|
||||
|
||||
var resource = Resource.builder()
|
||||
.id(idGenerator.generate())
|
||||
.name(request.getName())
|
||||
.createdAt(Instant.now())
|
||||
.build();
|
||||
|
||||
return repository.create(resource);
|
||||
});
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Error Handling
|
||||
- **Specific Exceptions**: Use Jakarta exceptions (BadRequestException, NotFoundException, etc.)
|
||||
- **Graceful Handling**: Always handle exceptions gracefully
|
||||
- **Logging**: Use SLF4J with `@Slf4j` annotation
|
||||
- **Context**: Include relevant context in log messages (surround values with single quotes)
|
||||
|
||||
**Example Error Handling:**
|
||||
```java
|
||||
@Slf4j
|
||||
public class ResourceService {
|
||||
|
||||
public ResourceResponse getResource(String id) {
|
||||
try {
|
||||
return resourceDao.findById(id)
|
||||
.orElseThrow(() -> new NotFoundException("Resource not found: '%s'".formatted(id)));
|
||||
} catch (SQLException exception) {
|
||||
log.error("Database error while retrieving resource: '{}'", id, exception);
|
||||
throw new InternalServerErrorException("Failed to retrieve resource", exception);
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Database Migrations
|
||||
- **MySQL**: Place in `src/main/resources/liquibase/db-app-state/migrations/`
|
||||
- **ClickHouse**: Place in `src/main/resources/liquibase/db-app-analytics/migrations/`
|
||||
- **Format**: Include `--liquibase formatted sql` and proper changeset metadata
|
||||
- **Indexes**: Add only relevant indexes with explanatory comments
|
||||
|
||||
### Testing Requirements
|
||||
- **Unit Tests**: Test business logic with mocks
|
||||
- **Integration Tests**: Test component interactions
|
||||
- **Test Data**: Use PODAM for generating test data
|
||||
- **Naming**: Follow camelCase conventions for test methods
|
||||
|
||||
### Code Quality Standards
|
||||
- **File Formatting**: All files must end with a blank line
|
||||
- **Naming**: Use meaningful variable and method names
|
||||
- **Collections**: Prefer `Map.of()`, `List.of()`, `Set.of()` for immutable collections
|
||||
- **List Access**: Use `getFirst()` or `getLast()` instead of `get(0)` or `get(size() - 1)`
|
||||
- **Constants**: Replace magic numbers with named constants
|
||||
- **Documentation**: Use Javadoc for public methods and classes
|
||||
|
||||
---
|
||||
|
||||
## 3. Frontend (React/TypeScript) Review Guidelines
|
||||
|
||||
### Technology Stack
|
||||
- **Language**: TypeScript 5.4.5
|
||||
- **Framework**: React 18.3.1
|
||||
- **Build Tool**: Vite 5.2.11
|
||||
- **Styling**: Tailwind CSS 3.4.3
|
||||
- **State Management**: Zustand 4.5.2
|
||||
- **Testing**: Vitest 3.0.5, Playwright 1.45.3
|
||||
|
||||
### Component Development Patterns
|
||||
- **Performance Optimization**: Always use `useMemo` for data transformations and `useCallback` for event handlers
|
||||
- **Component Structure**: Follow established patterns with proper TypeScript interfaces
|
||||
- **UI Components**: Use shadcn/ui components with consistent variants
|
||||
- **Styling**: Use Tailwind CSS with custom design system classes
|
||||
|
||||
**Example Component Pattern:**
|
||||
```typescript
|
||||
import React, { useMemo, useCallback } from "react";
|
||||
import { cn } from "@/lib/utils";
|
||||
|
||||
type ComponentProps = {
|
||||
// Props interface
|
||||
};
|
||||
|
||||
const Component: React.FunctionComponent<ComponentProps> = ({
|
||||
prop1,
|
||||
prop2,
|
||||
...props
|
||||
}) => {
|
||||
// 1. State hooks
|
||||
// 2. useMemo for expensive computations
|
||||
// 3. useCallback for event handlers
|
||||
// 4. Other hooks
|
||||
|
||||
const processedData = useMemo(() => transformData(rawData), [rawData]);
|
||||
const handleClick = useCallback(() => {}, [deps]);
|
||||
|
||||
return (
|
||||
<div className="component-container">
|
||||
{/* JSX */}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
```
|
||||
|
||||
### Data Fetching Patterns
|
||||
- **React Query**: Use TanStack Query for data fetching and caching
|
||||
- **Query Keys**: Use descriptive keys with proper parameters
|
||||
- **Error Handling**: Implement proper error states and loading indicators
|
||||
- **Optimistic Updates**: Use mutations for data updates
|
||||
|
||||
### State Management
|
||||
- **Zustand**: Use for global state management
|
||||
- **Local Storage**: Use `use-local-storage-state` for persistence
|
||||
- **Selectors**: Create focused selectors for state access
|
||||
|
||||
### Form Handling
|
||||
- **React Hook Form**: Use with Zod validation
|
||||
- **Validation**: Implement comprehensive form validation
|
||||
- **Error Display**: Show validation errors clearly
|
||||
|
||||
### Testing Patterns
|
||||
- **Unit Tests**: Test individual components and hooks
|
||||
- **Integration Tests**: Test component interactions
|
||||
- **E2E Tests**: Test complete user workflows with Playwright
|
||||
- **Test Data**: Use realistic test data and proper mocking
|
||||
|
||||
### UI Component Patterns
|
||||
- **Button Variants**: Use established variant system (default, secondary, outline, destructive, ghost, minimal)
|
||||
- **Data Tables**: Use DataTable wrapper with proper column definitions
|
||||
- **Loading States**: Use Skeleton components for loading states
|
||||
- **Error States**: Use proper error styling with destructive colors
|
||||
|
||||
### Styling Guidelines
|
||||
- **Design System**: Use custom CSS properties and typography classes
|
||||
- **Color System**: Use semantic color classes (primary, secondary, muted, destructive)
|
||||
- **Layout Classes**: Use consistent spacing and sizing patterns
|
||||
- **Responsive Design**: Use Tailwind responsive prefixes appropriately
|
||||
|
||||
---
|
||||
|
||||
## 4. Python SDK Review Guidelines
|
||||
|
||||
### Technology Stack
|
||||
- **Language**: Python 3.8+
|
||||
- **Package Manager**: setuptools with pyproject.toml
|
||||
- **HTTP Client**: httpx
|
||||
- **Validation**: Pydantic 2.x
|
||||
- **Testing**: pytest
|
||||
|
||||
### API Design Principles
|
||||
- **Main API Class**: `opik.Opik` is the main entry point
|
||||
- **Higher Level APIs**: Provide wrappers for complex REST calls
|
||||
- **Backward Compatibility**: Maintain compatibility for public interfaces
|
||||
- **Consistency**: Follow existing API patterns
|
||||
|
||||
### Architecture Patterns
|
||||
- **Layered Architecture**: API Objects → Message Processing → REST API → Backend
|
||||
- **Non-blocking Operations**: Create spans, traces, and feedback scores as background operations
|
||||
- **Context Management**: Use `opik.opik_context` and `@opik.track` decorator
|
||||
- **Integration Patterns**: Extend base decorator classes for new integrations
|
||||
|
||||
### Code Organization
|
||||
- **Import Organization**: Import modules, not names (except from `typing`)
|
||||
- **Access Control**: Use proper access modifiers (protected methods with underscores)
|
||||
- **Module Structure**: Organize by functionality, avoid generic utility modules
|
||||
- **Naming**: Use meaningful names that reflect purpose
|
||||
|
||||
### Dependency Management
|
||||
- **Existing Dependencies**: Prioritize keeping existing dependencies
|
||||
- **Version Bounds**: Use flexible version bounds with appropriate constraints
|
||||
- **Conditional Imports**: Use for optional dependencies (integrations)
|
||||
- **Python Versions**: Ensure compatibility with specified Python versions
|
||||
|
||||
### Error Handling
|
||||
- **Specific Exceptions**: Use specific exception types for different error categories
|
||||
- **Structured Errors**: Use consistent structured error information
|
||||
- **Recovery Logic**: Implement proper retry logic for transient failures
|
||||
- **Provider Errors**: Handle provider-specific errors in integrations
|
||||
|
||||
### Testing Requirements
|
||||
- **Test Naming**: Use convention `test_WHAT__CASE_DESCRIPTION__EXPECTED_RESULT`
|
||||
- **Test Organization**: Unit tests, library integration tests, end-to-end tests
|
||||
- **Test Data**: Use `fake_backend` fixture for emulating real backend
|
||||
- **Coverage**: Test public API only, never violate privacy
|
||||
|
||||
### Logging Guidelines
|
||||
- **Structured Logging**: Use proper logger hierarchies
|
||||
- **Log Levels**: DEBUG for detailed info, INFO/WARNING for user messages, ERROR for problems
|
||||
- **Context**: Include relevant context without exposing sensitive information
|
||||
- **Timing**: Include timing information for API calls and processing
|
||||
|
||||
---
|
||||
|
||||
## 5. TypeScript SDK Review Guidelines
|
||||
|
||||
### Technology Stack
|
||||
- **Language**: TypeScript 5.7.2
|
||||
- **Runtime**: Node.js 18+
|
||||
- **Build Tool**: tsup 8.3.6
|
||||
- **HTTP Client**: node-fetch 3.3.2
|
||||
- **Validation**: Zod 3.25.55
|
||||
|
||||
### Code Quality Standards
|
||||
- **Type Safety**: Use comprehensive TypeScript types
|
||||
- **ES Modules**: Use modern ES module syntax
|
||||
- **Error Handling**: Implement proper error handling with typed errors
|
||||
- **Documentation**: Include comprehensive JSDoc comments
|
||||
|
||||
### Testing Patterns
|
||||
- **Unit Tests**: Test individual functions and classes
|
||||
- **Integration Tests**: Test API interactions
|
||||
- **Mocking**: Use proper mocking for external dependencies
|
||||
- **Type Testing**: Test TypeScript types and interfaces
|
||||
|
||||
---
|
||||
|
||||
## 6. General Code Quality Guidelines
|
||||
|
||||
### Clean Code Principles
|
||||
- **Constants**: Replace magic numbers with named constants
|
||||
- **Meaningful Names**: Variables, functions, and classes should reveal their purpose
|
||||
- **Single Responsibility**: Each function should do exactly one thing
|
||||
- **DRY**: Don't repeat yourself - extract common logic
|
||||
- **Comments**: Explain why, not what - make code self-documenting
|
||||
|
||||
### Performance Considerations
|
||||
- **Efficient Algorithms**: Use appropriate data structures and algorithms
|
||||
- **Memory Management**: Avoid memory leaks and excessive allocations
|
||||
- **Database Optimization**: Use proper indexes and query optimization
|
||||
- **Caching**: Implement appropriate caching strategies
|
||||
|
||||
### Security Guidelines
|
||||
- **Input Validation**: Validate all external inputs
|
||||
- **Authentication**: Implement proper authentication and authorization
|
||||
- **Data Protection**: Never log sensitive information (PII, credentials)
|
||||
- **Dependency Security**: Keep dependencies updated and scan for vulnerabilities
|
||||
|
||||
### Documentation Standards
|
||||
- **API Documentation**: Use OpenAPI/Swagger for backend APIs
|
||||
- **Code Comments**: Use Javadoc, JSDoc, or docstrings as appropriate
|
||||
- **README Files**: Keep documentation up to date
|
||||
- **Examples**: Provide usage examples for complex functionality
|
||||
|
||||
---
|
||||
|
||||
## 7. Testing Guidelines
|
||||
|
||||
### Test Organization
|
||||
- **Unit Tests**: Fast, isolated, no external dependencies
|
||||
- **Integration Tests**: Test component interactions
|
||||
- **E2E Tests**: Test complete user workflows
|
||||
- **Performance Tests**: Load and stress testing where applicable
|
||||
|
||||
### Test Quality Standards
|
||||
- **Coverage**: Aim for comprehensive test coverage
|
||||
- **Readability**: Tests should be easy to understand and maintain
|
||||
- **Reliability**: Tests should be deterministic and not flaky
|
||||
- **Performance**: Tests should run quickly and efficiently
|
||||
|
||||
### Test Data Management
|
||||
- **Realistic Data**: Use realistic but not sensitive test data
|
||||
- **Fixtures**: Use test fixtures for common setup
|
||||
- **Isolation**: Each test should be independent
|
||||
- **Cleanup**: Properly clean up test data and resources
|
||||
|
||||
### Backend Testing (Java)
|
||||
- **PODAM**: Use for generating test data with `PodamFactoryUtils.newPodamFactory()`
|
||||
- **Naming**: Follow camelCase conventions (`shouldCreateUser_whenValidRequest`)
|
||||
- **Assertions**: Use AssertJ for fluent assertions
|
||||
- **Mocking**: Use Mockito for mocking dependencies
|
||||
|
||||
### Frontend Testing (TypeScript)
|
||||
- **React Testing Library**: Use for component testing
|
||||
- **MSW**: Use for API mocking
|
||||
- **Playwright**: Use for E2E testing
|
||||
- **Vitest**: Use for unit testing
|
||||
|
||||
### Python SDK Testing
|
||||
- **pytest**: Use for all testing
|
||||
- **fake_backend**: Use fixture for backend emulation
|
||||
- **Test Naming**: Use descriptive test names with underscores
|
||||
- **Coverage**: Test public API only
|
||||
|
||||
---
|
||||
|
||||
## 8. Dependency Management
|
||||
|
||||
### Version Strategy
|
||||
- **Pin Major Versions**: For production stability
|
||||
- **Allow Minor Updates**: For security patches and bug fixes
|
||||
- **Security Updates**: Automate security patch updates
|
||||
- **Breaking Changes**: Test thoroughly before major version upgrades
|
||||
|
||||
### Dependency Guidelines
|
||||
- **Existing Dependencies**: Prefer existing dependencies over adding new ones
|
||||
- **Security**: Keep dependencies updated and scan for vulnerabilities
|
||||
- **Licensing**: Ensure all dependencies have acceptable licenses
|
||||
- **Size**: Consider the impact of adding new dependencies
|
||||
|
||||
### Technology-Specific Dependencies
|
||||
|
||||
#### Backend (Java)
|
||||
- **Core**: Dropwizard 5.0.0, JDBI3, MySQL 9.7.0, ClickHouse 0.9.0
|
||||
- **Build**: Maven, Spotless 3.5.1
|
||||
- **Testing**: JUnit 5, Testcontainers, WireMock
|
||||
- **Observability**: OpenTelemetry 2.28.1
|
||||
|
||||
#### Frontend (TypeScript)
|
||||
- **Core**: React 18.3.1, TypeScript 5.4.5, Vite 5.2.11
|
||||
- **UI**: Tailwind CSS 3.4.3, shadcn/ui, Radix UI
|
||||
- **State**: Zustand 4.5.2, TanStack Query 5.45.0
|
||||
- **Testing**: Vitest 3.0.5, Playwright 1.45.3
|
||||
|
||||
#### Python SDK
|
||||
- **Core**: Python 3.8+, httpx, Pydantic 2.x
|
||||
- **Testing**: pytest
|
||||
- **CLI**: Click
|
||||
- **Logging**: Rich, Sentry SDK
|
||||
|
||||
#### TypeScript SDK
|
||||
- **Core**: TypeScript 5.7.2, Node.js 18+, tsup 8.3.6
|
||||
- **HTTP**: node-fetch 3.3.2
|
||||
- **Validation**: Zod 3.25.55
|
||||
- **Logging**: tslog 4.9.3
|
||||
|
||||
---
|
||||
|
||||
## 9. Review Checklist
|
||||
|
||||
### Before Review
|
||||
- [ ] Understand the context and purpose of the changes
|
||||
- [ ] Check if the changes follow established patterns
|
||||
- [ ] Verify that tests are included and appropriate
|
||||
- [ ] Ensure documentation is updated if needed
|
||||
|
||||
### During Review
|
||||
- [ ] Check code quality and adherence to standards
|
||||
- [ ] Verify error handling and edge cases
|
||||
- [ ] Review performance implications
|
||||
- [ ] Check security considerations
|
||||
- [ ] Ensure proper logging and observability
|
||||
- [ ] Verify test coverage and quality
|
||||
|
||||
### After Review
|
||||
- [ ] Provide constructive feedback
|
||||
- [ ] Suggest improvements when appropriate
|
||||
- [ ] Approve only when standards are met
|
||||
- [ ] Follow up on any issues identified
|
||||
|
||||
---
|
||||
|
||||
## 10. Common Issues to Watch For
|
||||
|
||||
### Backend Issues
|
||||
- Missing transaction boundaries
|
||||
- Improper exception handling
|
||||
- Missing validation annotations
|
||||
- Inconsistent logging patterns
|
||||
- Missing or incorrect API documentation
|
||||
- Not using `@Slf4j` annotation
|
||||
- Logging sensitive information
|
||||
- Not surrounding logged values with single quotes
|
||||
|
||||
### Frontend Issues
|
||||
- Missing performance optimizations (useMemo, useCallback)
|
||||
- Improper error handling
|
||||
- Missing loading states
|
||||
- Inconsistent component patterns
|
||||
- Missing accessibility features
|
||||
- Not using proper TypeScript types
|
||||
- Inline functions in JSX props
|
||||
|
||||
### SDK Issues
|
||||
- Breaking API changes without proper deprecation
|
||||
- Missing error handling
|
||||
- Inconsistent naming conventions
|
||||
- Missing documentation
|
||||
- Improper dependency management
|
||||
- Not following import organization rules
|
||||
|
||||
### General Issues
|
||||
- Code duplication
|
||||
- Magic numbers or hardcoded values
|
||||
- Missing tests
|
||||
- Poor error messages
|
||||
- Security vulnerabilities
|
||||
- Performance issues
|
||||
- Files not ending with blank lines
|
||||
- Inconsistent naming conventions
|
||||
|
||||
---
|
||||
|
||||
## 11. Technology-Specific Review Focus Areas
|
||||
|
||||
### Backend (Java) Focus
|
||||
- **Architecture**: Layered architecture compliance
|
||||
- **Transactions**: Proper TransactionTemplate usage
|
||||
- **Validation**: Jakarta validation annotations
|
||||
- **Logging**: SLF4J with proper context
|
||||
- **Testing**: PODAM usage and test naming
|
||||
- **Database**: Migration script quality
|
||||
- **Error Handling**: Specific exception types
|
||||
|
||||
### Frontend (TypeScript) Focus
|
||||
- **Performance**: useMemo and useCallback usage
|
||||
- **TypeScript**: Proper type definitions
|
||||
- **Components**: shadcn/ui patterns
|
||||
- **Styling**: Tailwind CSS conventions
|
||||
- **State Management**: Zustand patterns
|
||||
- **Testing**: Component and E2E test coverage
|
||||
- **Accessibility**: ARIA labels and semantic HTML
|
||||
|
||||
### Python SDK Focus
|
||||
- **API Design**: Main Opik class usage
|
||||
- **Architecture**: Layered patterns
|
||||
- **Testing**: Test naming conventions
|
||||
- **Logging**: Structured logging
|
||||
- **Dependencies**: Minimal dependency addition
|
||||
- **Documentation**: Comprehensive docstrings
|
||||
|
||||
### TypeScript SDK Focus
|
||||
- **Type Safety**: Comprehensive TypeScript usage
|
||||
- **ES Modules**: Modern module syntax
|
||||
- **Error Handling**: Typed error handling
|
||||
- **Documentation**: JSDoc comments
|
||||
- **Testing**: Unit and integration tests
|
||||
|
||||
---
|
||||
|
||||
Use these guidelines to provide comprehensive, consistent, and helpful code review feedback across all Opik applications. Each section provides specific, actionable guidance for the technology stack being reviewed.
|
||||
@@ -0,0 +1,15 @@
|
||||
# To get started with Dependabot version updates, you'll need to specify which
|
||||
# package ecosystems to update and where the package manifests are located.
|
||||
# Please see the documentation for all configuration options:
|
||||
# https://docs.github.com/code-security/dependabot/dependabot-version-updates/configuration-options-for-the-dependabot.yml-file
|
||||
|
||||
version: 2
|
||||
updates:
|
||||
- package-ecosystem: "maven"
|
||||
directory: "apps/opik-backend"
|
||||
schedule:
|
||||
interval: "weekly" # How often to check for updates
|
||||
ignore:
|
||||
- dependency-name: "*"
|
||||
update-types: ["version-update:semver-patch"]
|
||||
open-pull-requests-limit: 5
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,83 @@
|
||||
Frontend:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file: 'apps/opik-frontend/**/*'
|
||||
|
||||
Backend:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file:
|
||||
- 'apps/opik-backend/**/*'
|
||||
- 'apps/opik-guardrails-backend/**/*'
|
||||
- 'apps/opik-python-backend/**/*'
|
||||
- 'apps/opik-sandbox-executor-python/**/*'
|
||||
|
||||
Python SDK:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file: 'sdks/python/**/*'
|
||||
|
||||
TypeScript SDK:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file: 'sdks/typescript/**/*'
|
||||
|
||||
Optimizer SDK:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file: 'sdks/opik_optimizer/**/*'
|
||||
|
||||
python:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file: '**/*.py'
|
||||
|
||||
java:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file: '**/*.java'
|
||||
|
||||
typescript:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file:
|
||||
- '**/*.ts'
|
||||
- '**/*.tsx'
|
||||
|
||||
tests:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file:
|
||||
- 'apps/opik-backend/src/test/**/*'
|
||||
- 'apps/opik-guardrails-backend/tests/**/*'
|
||||
- 'apps/opik-python-backend/tests/**/*'
|
||||
- 'apps/opik-frontend/e2e/**/*'
|
||||
- 'sdks/python/tests/**/*'
|
||||
- 'sdks/opik_optimizer/tests/**/*'
|
||||
- 'sdks/typescript/tests/**/*'
|
||||
- 'tests_end_to_end/**/*'
|
||||
- 'tests_load/**/*'
|
||||
|
||||
dependencies:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file:
|
||||
- '**/pom.xml'
|
||||
- '**/package.json'
|
||||
- '**/package-lock.json'
|
||||
- '**/requirements.txt'
|
||||
- '**/pyproject.toml'
|
||||
- '**/setup.py'
|
||||
- '**/.python-version'
|
||||
- '**/.java-version'
|
||||
|
||||
Infrastructure:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file:
|
||||
- '.github/**/*'
|
||||
- '.hooks/**/*'
|
||||
- 'deployment/**/*'
|
||||
- 'scripts/**/*'
|
||||
- 'extensions/**/*'
|
||||
- '**/Dockerfile'
|
||||
- '**/.gitignore'
|
||||
- '**/*.sh'
|
||||
- '**/*.ps1'
|
||||
|
||||
documentation:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file:
|
||||
- 'apps/opik-documentation/**/*'
|
||||
- '**/*.mdc'
|
||||
- '**/*.md'
|
||||
- '**/*.mdx'
|
||||
@@ -0,0 +1,51 @@
|
||||
## Details
|
||||
<!--
|
||||
REPLACE ME WITH:
|
||||
Short 2-3 lines single paragraph explaining whats changed and the motivation (why).
|
||||
If you need to have changes explained in detail include single line dot point(s).
|
||||
-->
|
||||
|
||||
## Change checklist
|
||||
- [ ] User facing
|
||||
- [ ] Documentation update
|
||||
|
||||
## Issues
|
||||
|
||||
- Resolves #
|
||||
- OPIK-
|
||||
|
||||
<!--
|
||||
Jira linking: tickets this PR RESOLVES keep the hyphen (OPIK-1234) here and anywhere — they link in Jira's Development panel, which is wanted. A PR may resolve more than one ticket; list each resolved key.
|
||||
Tickets RELATED but NOT resolved here (escalations, references to older tickets) must NOT link: in the sections above/below and in commit messages, write them with an underscore (OPIK_7000) and paste no Jira URL (the URL contains the hyphenated key and links anyway). See CONTRIBUTING.md.
|
||||
-->
|
||||
|
||||
|
||||
## AI-WATERMARK
|
||||
|
||||
AI-WATERMARK: [yes|no]
|
||||
|
||||
- If yes:
|
||||
- Tools:
|
||||
- Model(s):
|
||||
- Scope:
|
||||
- Human verification:
|
||||
|
||||
## Testing
|
||||
<!--
|
||||
REPLACE ME WITH:
|
||||
|
||||
How you tested
|
||||
|
||||
Include:
|
||||
- Exact commands run (for example: `mvn test`, `npm run lint && npm run test`, `pytest ...`)
|
||||
- Scenarios validated (happy path, edge cases, regressions)
|
||||
- Environment/context used (local process mode vs Docker, OS/browser when relevant)
|
||||
- Any tests not run, with reason
|
||||
- Relevant logs/artifacts links if available
|
||||
|
||||
Video evidence:
|
||||
- External contributors: include a short recording for all contributions where possible.
|
||||
- Internal contributors: include a short recording for complex changes where possible.
|
||||
-->
|
||||
|
||||
## Documentation
|
||||
@@ -0,0 +1,3 @@
|
||||
template: |
|
||||
## What’s Changed
|
||||
$CHANGES
|
||||
Executable
+80
@@ -0,0 +1,80 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
NUM_GROUPS=16
|
||||
UNIT_TIMEOUT=10
|
||||
INTEGRATION_TIMEOUT=20
|
||||
TEST_DIR="src/test/java"
|
||||
PATTERN="DropwizardAppExtensionProvider\|MySQLContainer\|ClickHouseContainer\|RedisContainer\|MinIOContainer"
|
||||
|
||||
# Classify: integration tests reference container/Dropwizard patterns
|
||||
integration_files=()
|
||||
unit_files=()
|
||||
while IFS= read -r file; do
|
||||
if grep -ql "$PATTERN" "$file" 2>/dev/null; then
|
||||
integration_files+=("$file")
|
||||
else
|
||||
unit_files+=("$file")
|
||||
fi
|
||||
done < <(find "$TEST_DIR" -name "*Test.java" -o -name "*Tests.java" | sort)
|
||||
|
||||
echo "Classified ${#unit_files[@]} unit test classes and ${#integration_files[@]} integration test classes"
|
||||
|
||||
# Convert file path to Maven class pattern: com.foo.BarTest
|
||||
to_class() { echo "$1" | sed "s|^$TEST_DIR/||;s|/|.|g;s|\.java$||"; }
|
||||
|
||||
# Build unit test list
|
||||
unit_list=""
|
||||
for f in "${unit_files[@]}"; do
|
||||
c=$(to_class "$f")
|
||||
unit_list="${unit_list:+$unit_list,}$c"
|
||||
done
|
||||
|
||||
# Balanced split: sort integration files by line count desc,
|
||||
# then greedy-assign each to the smallest group
|
||||
declare -a group_size group_list
|
||||
for ((i=1; i<=NUM_GROUPS; i++)); do
|
||||
group_size[$i]=0
|
||||
group_list[$i]=""
|
||||
done
|
||||
|
||||
while IFS=' ' read -r lines file; do
|
||||
min_g=1
|
||||
for ((i=2; i<=NUM_GROUPS; i++)); do
|
||||
if (( group_size[i] < group_size[min_g] )); then
|
||||
min_g=$i
|
||||
fi
|
||||
done
|
||||
group_size[$min_g]=$(( group_size[min_g] + lines ))
|
||||
c=$(to_class "$file")
|
||||
group_list[$min_g]="${group_list[$min_g]:+${group_list[$min_g]},}$c"
|
||||
done < <(
|
||||
for f in "${integration_files[@]}"; do
|
||||
echo "$(wc -l < "$f") $f"
|
||||
done | sort -rn
|
||||
)
|
||||
|
||||
echo "Balanced ${#integration_files[@]} integration classes into $NUM_GROUPS groups"
|
||||
for ((i=1; i<=NUM_GROUPS; i++)); do
|
||||
count=$(echo "${group_list[$i]}" | tr ',' '\n' | grep -c '.' || true)
|
||||
echo " Group $i: $count classes (~${group_size[$i]} lines)"
|
||||
done
|
||||
|
||||
# Build JSON matrix: unit tests + N integration groups
|
||||
matrix="{\"include\":["
|
||||
matrix+="{\"name\":\"Unit Tests\",\"tests\":\"$unit_list\",\"timeout\":$UNIT_TIMEOUT}"
|
||||
for ((i=1; i<=NUM_GROUPS; i++)); do
|
||||
matrix+=",{\"name\":\"Integration Group $i\",\"tests\":\"${group_list[$i]}\",\"timeout\":$INTEGRATION_TIMEOUT}"
|
||||
done
|
||||
matrix+="]}"
|
||||
echo "matrix=$matrix" >> "$GITHUB_OUTPUT"
|
||||
echo "Matrix written to GITHUB_OUTPUT ($(( NUM_GROUPS + 1 )) jobs: 1 unit + $NUM_GROUPS integration)"
|
||||
|
||||
# Summary
|
||||
echo "### Test Split Summary" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "- **Unit tests**: ${#unit_files[@]} classes" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "- **Integration tests**: ${#integration_files[@]} classes in $NUM_GROUPS groups" >> "$GITHUB_STEP_SUMMARY"
|
||||
for ((i=1; i<=NUM_GROUPS; i++)); do
|
||||
count=$(echo "${group_list[$i]}" | tr ',' '\n' | grep -c '.' || true)
|
||||
echo " - Group $i: $count classes (~${group_size[$i]} lines)" >> "$GITHUB_STEP_SUMMARY"
|
||||
done
|
||||
+99
@@ -0,0 +1,99 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Build and send the E2E post-merge Slack notification.
|
||||
#
|
||||
# Required env vars:
|
||||
# SLACK_WEBHOOK_URL — Slack incoming webhook URL
|
||||
# TEST_RESULT — "success" or "failure"
|
||||
# SUITE_NAME — test suite name (e.g. "happypaths")
|
||||
# PASSED_TESTS — count of passed tests
|
||||
# FAILED_TESTS — count of failed tests
|
||||
# TESTOPS_URL — link to TestOps dashboard
|
||||
# AUTHOR_DISPLAY — GitHub username of the PR author
|
||||
# MENTIONS — pre-resolved Slack mentions (e.g. "<@U123> <@U456>"), may be empty
|
||||
# GITHUB_REF_NAME — branch name (set by GitHub Actions)
|
||||
# GITHUB_SHA — commit SHA (set by GitHub Actions)
|
||||
# GITHUB_REPOSITORY — owner/repo (set by GitHub Actions)
|
||||
# GITHUB_RUN_ID — workflow run ID (set by GitHub Actions)
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
|
||||
SHORT_SHA="${GITHUB_SHA:0:7}"
|
||||
WORKFLOW_URL="https://github.com/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
|
||||
|
||||
if [ "$TEST_RESULT" == "success" ]; then
|
||||
STATUS_EMOJI="\u2705"
|
||||
STATUS_TEXT="Passed"
|
||||
COLOR="good"
|
||||
else
|
||||
STATUS_EMOJI="\u274c"
|
||||
STATUS_TEXT="Failed"
|
||||
COLOR="danger"
|
||||
fi
|
||||
|
||||
MENTION_BLOCK=""
|
||||
if [ "$TEST_RESULT" == "failure" ] && [ -n "${MENTIONS:-}" ]; then
|
||||
MENTION_BLOCK=',
|
||||
{
|
||||
"type": "section",
|
||||
"text": {
|
||||
"type": "mrkdwn",
|
||||
"text": "\ud83d\udea8 '"${MENTIONS}"' - Tests need attention!"
|
||||
}
|
||||
}'
|
||||
fi
|
||||
|
||||
cat << EOF > /tmp/slack-payload.json
|
||||
{
|
||||
"attachments": [
|
||||
{
|
||||
"color": "$COLOR",
|
||||
"blocks": [
|
||||
{
|
||||
"type": "header",
|
||||
"text": {
|
||||
"type": "plain_text",
|
||||
"text": "$STATUS_EMOJI E2E Tests $STATUS_TEXT - Post Merge",
|
||||
"emoji": true
|
||||
}
|
||||
}$MENTION_BLOCK,
|
||||
{
|
||||
"type": "section",
|
||||
"fields": [
|
||||
{"type": "mrkdwn", "text": "*Suite:*\n\`${SUITE_NAME}\`"},
|
||||
{"type": "mrkdwn", "text": "*Branch:*\n\`${GITHUB_REF_NAME}\`"},
|
||||
{"type": "mrkdwn", "text": "*Results:*\n\u2705 ${PASSED_TESTS} passed, \u274c ${FAILED_TESTS} failed"},
|
||||
{"type": "mrkdwn", "text": "*Commit:*\n<https://github.com/${GITHUB_REPOSITORY}/commit/${GITHUB_SHA}|\`${SHORT_SHA}\`>"}
|
||||
]
|
||||
},
|
||||
{
|
||||
"type": "context",
|
||||
"elements": [
|
||||
{"type": "mrkdwn", "text": "\ud83d\udc64 *Author:* ${AUTHOR_DISPLAY}"}
|
||||
]
|
||||
},
|
||||
{
|
||||
"type": "actions",
|
||||
"elements": [
|
||||
{
|
||||
"type": "button",
|
||||
"text": {"type": "plain_text", "text": "\ud83d\udcca TestOps", "emoji": true},
|
||||
"url": "${TESTOPS_URL}",
|
||||
"style": "primary"
|
||||
},
|
||||
{
|
||||
"type": "button",
|
||||
"text": {"type": "plain_text", "text": "\ud83d\udd0d Workflow", "emoji": true},
|
||||
"url": "${WORKFLOW_URL}"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
EOF
|
||||
|
||||
exec "$SCRIPT_DIR/send-slack-message.sh" /tmp/slack-payload.json
|
||||
+112
@@ -0,0 +1,112 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Build and send the library integration tests Slack notification.
|
||||
#
|
||||
# Required env vars:
|
||||
# SLACK_WEBHOOK_URL — Slack incoming webhook URL
|
||||
# SUITE_RESULTS — JSON object mapping suite display names to results
|
||||
# e.g. '{"OpenAI":"success","LangChain":"failure"}'
|
||||
# TRIGGER_TYPE — "Daily Schedule", "Weekly Schedule", or "Manual Dispatch"
|
||||
# GITHUB_REF_NAME — branch name (set by GitHub Actions)
|
||||
# GITHUB_SHA — commit SHA (set by GitHub Actions)
|
||||
# GITHUB_REPOSITORY — owner/repo (set by GitHub Actions)
|
||||
# GITHUB_RUN_ID — workflow run ID (set by GitHub Actions)
|
||||
#
|
||||
# Optional env vars:
|
||||
# SLACK_USER_ID — Slack user ID to cc on the notification
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
|
||||
SHORT_SHA="${GITHUB_SHA:0:7}"
|
||||
WORKFLOW_URL="https://github.com/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
|
||||
COMMIT_URL="https://github.com/${GITHUB_REPOSITORY}/commit/${GITHUB_SHA}"
|
||||
|
||||
# Categorize suites by result
|
||||
SUCCESS_SUITES=""
|
||||
FAILED_SUITES=""
|
||||
SKIPPED_SUITES=""
|
||||
SUCCESS_COUNT=0
|
||||
FAILURE_COUNT=0
|
||||
SKIPPED_COUNT=0
|
||||
|
||||
while IFS=$'\t' read -r name result; do
|
||||
case "$result" in
|
||||
success)
|
||||
[ -n "$SUCCESS_SUITES" ] && SUCCESS_SUITES="${SUCCESS_SUITES}, "
|
||||
SUCCESS_SUITES="${SUCCESS_SUITES}${name}"
|
||||
SUCCESS_COUNT=$((SUCCESS_COUNT + 1))
|
||||
;;
|
||||
failure)
|
||||
[ -n "$FAILED_SUITES" ] && FAILED_SUITES="${FAILED_SUITES}, "
|
||||
FAILED_SUITES="${FAILED_SUITES}${name}"
|
||||
FAILURE_COUNT=$((FAILURE_COUNT + 1))
|
||||
;;
|
||||
*)
|
||||
[ -n "$SKIPPED_SUITES" ] && SKIPPED_SUITES="${SKIPPED_SUITES}, "
|
||||
SKIPPED_SUITES="${SKIPPED_SUITES}${name}"
|
||||
SKIPPED_COUNT=$((SKIPPED_COUNT + 1))
|
||||
;;
|
||||
esac
|
||||
done < <(echo "$SUITE_RESULTS" | jq -r 'to_entries[] | "\(.key)\t\(.value)"')
|
||||
|
||||
if [ "$FAILURE_COUNT" -gt 0 ]; then
|
||||
COLOR="danger"
|
||||
STATUS_TEXT="Failed"
|
||||
else
|
||||
COLOR="good"
|
||||
STATUS_TEXT="Passed"
|
||||
fi
|
||||
|
||||
# Build blocks with jq
|
||||
BLOCKS='[]'
|
||||
|
||||
BLOCKS=$(echo "$BLOCKS" | jq --arg title "Python SDK Integration Tests $STATUS_TEXT" \
|
||||
'. + [{"type": "header", "text": {"type": "plain_text", "text": $title, "emoji": true}}]')
|
||||
|
||||
BLOCKS=$(echo "$BLOCKS" | jq \
|
||||
--arg trigger "$TRIGGER_TYPE" \
|
||||
--arg branch "$GITHUB_REF_NAME" \
|
||||
--arg commit_url "$COMMIT_URL" \
|
||||
--arg short_sha "$SHORT_SHA" \
|
||||
'. + [{"type": "section", "fields": [
|
||||
{"type": "mrkdwn", "text": ("*Trigger:*\n" + $trigger)},
|
||||
{"type": "mrkdwn", "text": ("*Branch:*\n`" + $branch + "`")},
|
||||
{"type": "mrkdwn", "text": ("*Commit:*\n<" + $commit_url + "|`" + $short_sha + "`>")}
|
||||
]}]')
|
||||
|
||||
BLOCKS=$(echo "$BLOCKS" | jq \
|
||||
--arg s "$SUCCESS_COUNT" --arg f "$FAILURE_COUNT" --arg k "$SKIPPED_COUNT" \
|
||||
'. + [{"type": "section", "text": {"type": "mrkdwn",
|
||||
"text": ("\u2705 *Passed:* " + $s + " | \u274c *Failed:* " + $f + " | \u23ed\ufe0f *Skipped:* " + $k)}}]')
|
||||
|
||||
if [ -n "$FAILED_SUITES" ]; then
|
||||
BLOCKS=$(echo "$BLOCKS" | jq --arg suites "$FAILED_SUITES" \
|
||||
'. + [{"type": "section", "text": {"type": "mrkdwn", "text": ("\u274c *Failed:* " + $suites)}}]')
|
||||
fi
|
||||
|
||||
if [ -n "$SUCCESS_SUITES" ]; then
|
||||
BLOCKS=$(echo "$BLOCKS" | jq --arg suites "$SUCCESS_SUITES" \
|
||||
'. + [{"type": "section", "text": {"type": "mrkdwn", "text": ("\u2705 *Passed:* " + $suites)}}]')
|
||||
fi
|
||||
|
||||
if [ -n "$SKIPPED_SUITES" ]; then
|
||||
BLOCKS=$(echo "$BLOCKS" | jq --arg suites "$SKIPPED_SUITES" \
|
||||
'. + [{"type": "section", "text": {"type": "mrkdwn", "text": ("\u23ed\ufe0f *Skipped:* " + $suites)}}]')
|
||||
fi
|
||||
|
||||
BLOCKS=$(echo "$BLOCKS" | jq --arg url "$WORKFLOW_URL" \
|
||||
'. + [{"type": "actions", "elements": [{"type": "button", "text": {"type": "plain_text", "text": "\ud83d\udd0d View Workflow", "emoji": true}, "url": $url}]}]')
|
||||
|
||||
if [ -n "${SLACK_USER_ID:-}" ]; then
|
||||
BLOCKS=$(echo "$BLOCKS" | jq --arg uid "$SLACK_USER_ID" \
|
||||
'. + [{"type": "context", "elements": [{"type": "mrkdwn", "text": ("\ud83d\udc64 cc: <@" + $uid + ">")}]}]')
|
||||
fi
|
||||
|
||||
echo "$BLOCKS" | jq --arg color "$COLOR" '{"attachments": [{"color": $color, "blocks": .}]}' > /tmp/slack-payload.json
|
||||
|
||||
echo "Payload:"
|
||||
cat /tmp/slack-payload.json
|
||||
|
||||
exec "$SCRIPT_DIR/send-slack-message.sh" /tmp/slack-payload.json
|
||||
Executable
+65
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Build and send a simple test-failure Slack notification.
|
||||
#
|
||||
# Required env vars:
|
||||
# SLACK_WEBHOOK_URL — Slack incoming webhook URL
|
||||
# GITHUB_REF_NAME — branch name (set by GitHub Actions)
|
||||
# GITHUB_SHA — commit SHA (set by GitHub Actions)
|
||||
# GITHUB_REPOSITORY — owner/repo (set by GitHub Actions)
|
||||
# GITHUB_RUN_ID — workflow run ID (set by GitHub Actions)
|
||||
# GITHUB_EVENT_NAME — trigger event (set by GitHub Actions)
|
||||
# GITHUB_ACTOR — user who triggered the workflow (set by GitHub Actions)
|
||||
#
|
||||
# Usage:
|
||||
# notify-slack-test-failure.sh "Guardrails E2E Tests"
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
TEST_NAME="${1:?Usage: notify-slack-failure.sh <test-name>}"
|
||||
|
||||
SHORT_SHA="${GITHUB_SHA:0:7}"
|
||||
WORKFLOW_URL="https://github.com/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
|
||||
|
||||
cat << EOF > /tmp/slack-payload.json
|
||||
{
|
||||
"attachments": [
|
||||
{
|
||||
"color": "danger",
|
||||
"blocks": [
|
||||
{
|
||||
"type": "header",
|
||||
"text": {
|
||||
"type": "plain_text",
|
||||
"text": "\u274c ${TEST_NAME} Failed",
|
||||
"emoji": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "section",
|
||||
"fields": [
|
||||
{"type": "mrkdwn", "text": "*Branch:*\n\`${GITHUB_REF_NAME}\`"},
|
||||
{"type": "mrkdwn", "text": "*Trigger:*\n\`${GITHUB_EVENT_NAME}\`"},
|
||||
{"type": "mrkdwn", "text": "*Author:*\n${GITHUB_ACTOR}"},
|
||||
{"type": "mrkdwn", "text": "*Commit:*\n<https://github.com/${GITHUB_REPOSITORY}/commit/${GITHUB_SHA}|\`${SHORT_SHA}\`>"}
|
||||
]
|
||||
},
|
||||
{
|
||||
"type": "actions",
|
||||
"elements": [
|
||||
{
|
||||
"type": "button",
|
||||
"text": {"type": "plain_text", "text": "\ud83d\udd0d View Workflow", "emoji": true},
|
||||
"url": "${WORKFLOW_URL}",
|
||||
"style": "primary"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
EOF
|
||||
|
||||
exec "$SCRIPT_DIR/send-slack-message.sh" /tmp/slack-payload.json
|
||||
Executable
+35
@@ -0,0 +1,35 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Send a JSON payload to a Slack webhook.
|
||||
#
|
||||
# Required env vars:
|
||||
# SLACK_WEBHOOK_URL — Slack incoming webhook URL
|
||||
#
|
||||
# Usage:
|
||||
# send-slack-message.sh payload.json
|
||||
# build-payload | send-slack-message.sh /dev/stdin
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
PAYLOAD_FILE="${1:?Usage: send-slack-message.sh <payload-file>}"
|
||||
|
||||
if [ -z "${SLACK_WEBHOOK_URL:-}" ]; then
|
||||
echo "::notice::SLACK_WEBHOOK_URL not configured - Slack notification will be skipped"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if [ ! -f "$PAYLOAD_FILE" ] && [ "$PAYLOAD_FILE" != "/dev/stdin" ]; then
|
||||
echo "::error::Payload file not found: $PAYLOAD_FILE"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
HTTP_CODE=$(curl -s -o /dev/null -w "%{http_code}" \
|
||||
-X POST -H 'Content-type: application/json' \
|
||||
--data @"$PAYLOAD_FILE" "$SLACK_WEBHOOK_URL")
|
||||
|
||||
if [ "$HTTP_CODE" -eq 200 ]; then
|
||||
echo "Slack notification sent"
|
||||
else
|
||||
echo "::warning::Slack notification failed with HTTP code: $HTTP_CODE"
|
||||
exit 1
|
||||
fi
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user