chore: import upstream snapshot with attribution
Security / Dependency audit (pip-audit) (push) Has been cancelled
Security / CodeQL (javascript-typescript) (push) Has been cancelled
Security / CodeQL (python) (push) Has been cancelled
Security / Secret scan (gitleaks) (push) Has been cancelled
rust / test (ubuntu) (push) Has been cancelled
rust / simulator e2e (macos-latest) (push) Has been cancelled
rust / simulator e2e (ubuntu-latest) (push) Has been cancelled
rust / simulator e2e (windows-latest) (push) Has been cancelled
rust / wheels (aarch64-apple-darwin) (push) Has been cancelled
rust / wheels (x86_64-unknown-linux-gnu) (push) Has been cancelled
rust / wheels (x86_64-apple-darwin) (push) Has been cancelled
rust / audit (push) Has been cancelled
rust / parity (nightly, allowed to fail during Phase 0) (push) Has been cancelled
CI / commitlint (push) Has been skipped
Dev Containers / validate (.devcontainer/devcontainer.json, default) (push) Failing after 0s
Dev Containers / validate (.devcontainer/memory-stack/devcontainer.json, memory-stack) (push) Failing after 0s
Dev Containers / validate-worktree (push) Failing after 0s
CI / changes (push) Failing after 4s
Deploy Documentation / validate (push) Has been skipped
Deploy Documentation / deploy (push) Failing after 1s
Init Native E2E / init-native (ubuntu-latest, claude) (push) Failing after 1s
Init Native E2E / init-native (ubuntu-latest, codex) (push) Failing after 1s
Install Native E2E / install-native (ubuntu-latest) (push) Failing after 1s
OpenCode Plugin / typecheck + build + test (push) Failing after 1s
Init Native E2E / init-native (ubuntu-latest, copilot) (push) Failing after 1s
Release Please / release-please (push) Failing after 1s
Wrap E2E / docker-wrap-e2e (push) Failing after 1s
Wrap Native E2E / wrap-native (ubuntu-latest) (push) Failing after 1s
Init E2E / docker-init-e2e (push) Failing after 4s
Merge Conflicts / merge-conflicts (push) Failing after 4s
CI / lint (push) Has been cancelled
CI / build-wheel (push) Has been cancelled
CI / build-wheel-windows (push) Has been cancelled
CI / prefetch-model (push) Has been cancelled
CI / test-dashboard-ui (push) Has been cancelled
CI / test (1) (push) Has been cancelled
CI / test (2) (push) Has been cancelled
CI / test (3) (push) Has been cancelled
CI / test (4) (push) Has been cancelled
CI / test-extras (push) Has been cancelled
CI / test-agno (push) Has been cancelled
CI / build (push) Has been cancelled
CI / workflow-validation (push) Has been cancelled
CI / docker-native-e2e (push) Has been cancelled
CI / windows-native-wrapper (push) Has been cancelled
CI / macos-native-wrapper (push) Has been cancelled
Docker / docker-build (map[name:arm64 platform:linux/arm64 runs_on:ubuntu-24.04-arm], map[bake_target:runtime-code-nonroot name:code-nonroot]) (push) Has been cancelled
Docker / docker-build (map[name:arm64 platform:linux/arm64 runs_on:ubuntu-24.04-arm], map[bake_target:runtime-code-slim name:code-slim]) (push) Has been cancelled
Docker / docker-build (map[name:arm64 platform:linux/arm64 runs_on:ubuntu-24.04-arm], map[bake_target:runtime-code-slim-nonroot name:code-slim-nonroot]) (push) Has been cancelled
Docker / docker-build (map[name:arm64 platform:linux/arm64 runs_on:ubuntu-24.04-arm], map[bake_target:runtime-nonroot name:nonroot]) (push) Has been cancelled
Docker / docker-build (map[name:arm64 platform:linux/arm64 runs_on:ubuntu-24.04-arm], map[bake_target:runtime-slim name:slim]) (push) Has been cancelled
Docker / docker-build (map[name:arm64 platform:linux/arm64 runs_on:ubuntu-24.04-arm], map[bake_target:runtime-slim-nonroot name:slim-nonroot]) (push) Has been cancelled
Docker / docker-manifest (map[bake_target:runtime name:]) (push) Has been cancelled
Docker / docker-manifest (map[bake_target:runtime-code name:code]) (push) Has been cancelled
Docker / docker-manifest (map[bake_target:runtime-code-nonroot name:code-nonroot]) (push) Has been cancelled
Docker / docker-manifest (map[bake_target:runtime-code-slim name:code-slim]) (push) Has been cancelled
Docker / docker-manifest (map[bake_target:runtime-code-slim-nonroot name:code-slim-nonroot]) (push) Has been cancelled
Docker / docker-manifest (map[bake_target:runtime-nonroot name:nonroot]) (push) Has been cancelled
Docker / docker-manifest (map[bake_target:runtime-slim name:slim]) (push) Has been cancelled
Docker / docker-manifest (map[bake_target:runtime-slim-nonroot name:slim-nonroot]) (push) Has been cancelled
Docker / docker-build (map[name:amd64 platform:linux/amd64 runs_on:ubuntu-24.04], map[bake_target:runtime name:]) (push) Has been cancelled
Docker / docker-build (map[name:amd64 platform:linux/amd64 runs_on:ubuntu-24.04], map[bake_target:runtime-code name:code]) (push) Has been cancelled
Docker / docker-build (map[name:amd64 platform:linux/amd64 runs_on:ubuntu-24.04], map[bake_target:runtime-code-nonroot name:code-nonroot]) (push) Has been cancelled
Docker / docker-build (map[name:amd64 platform:linux/amd64 runs_on:ubuntu-24.04], map[bake_target:runtime-code-slim name:code-slim]) (push) Has been cancelled
Docker / docker-build (map[name:amd64 platform:linux/amd64 runs_on:ubuntu-24.04], map[bake_target:runtime-code-slim-nonroot name:code-slim-nonroot]) (push) Has been cancelled
Docker / docker-build (map[name:amd64 platform:linux/amd64 runs_on:ubuntu-24.04], map[bake_target:runtime-nonroot name:nonroot]) (push) Has been cancelled
Docker / docker-build (map[name:amd64 platform:linux/amd64 runs_on:ubuntu-24.04], map[bake_target:runtime-slim name:slim]) (push) Has been cancelled
Docker / docker-build (map[name:amd64 platform:linux/amd64 runs_on:ubuntu-24.04], map[bake_target:runtime-slim-nonroot name:slim-nonroot]) (push) Has been cancelled
Docker / docker-build (map[name:arm64 platform:linux/arm64 runs_on:ubuntu-24.04-arm], map[bake_target:runtime name:]) (push) Has been cancelled
Docker / docker-build (map[name:arm64 platform:linux/arm64 runs_on:ubuntu-24.04-arm], map[bake_target:runtime-code name:code]) (push) Has been cancelled
Docker / promote-latest (push) Has been cancelled
Init Native E2E / init-native (macos-latest, claude) (push) Has been cancelled
Init Native E2E / init-native (macos-latest, codex) (push) Has been cancelled
Init Native E2E / init-native (macos-latest, copilot) (push) Has been cancelled
Install Native E2E / install-native (macos-latest) (push) Has been cancelled
Wrap Native E2E / wrap-native (macos-latest) (push) Has been cancelled

This commit is contained in:
wehub-resource-sync
2026-07-13 12:03:20 +08:00
commit 0ef5fcb1c5
1951 changed files with 606278 additions and 0 deletions
+26
View File
@@ -0,0 +1,26 @@
# deps
/node_modules
# generated content
.source
# test & build
/coverage
/.next/
/out/
/build
*.tsbuildinfo
# misc
.DS_Store
*.pem
/.pnp
.pnp.js
npm-debug.log*
yarn-debug.log*
yarn-error.log*
# others
.env*.local
.vercel
next-env.d.ts
+45
View File
@@ -0,0 +1,45 @@
# docs
This is a Next.js application generated with
[Create Fumadocs](https://github.com/fuma-nama/fumadocs).
Run development server:
```bash
npm run dev
# or
pnpm dev
# or
yarn dev
```
Open http://localhost:3000 with your browser to see the result.
## Explore
In the project, you can see:
- `lib/source.ts`: Code for content source adapter, [`loader()`](https://fumadocs.dev/docs/headless/source-api) provides the interface to access your content.
- `lib/layout.shared.tsx`: Shared options for layouts, optional but preferred to keep.
| Route | Description |
| ------------------------- | ------------------------------------------------------ |
| `app/(home)` | The route group for your landing page and other pages. |
| `app/docs` | The documentation layout and pages. |
| `app/api/search/route.ts` | The Route Handler for search. |
### Fumadocs MDX
A `source.config.ts` config file has been included, you can customise different options like frontmatter schema.
Read the [Introduction](https://fumadocs.dev/docs/mdx) for further details.
## Learn More
To learn more about Next.js and Fumadocs, take a look at the following
resources:
- [Next.js Documentation](https://nextjs.org/docs) - learn about Next.js
features and API.
- [Learn Next.js](https://nextjs.org/learn) - an interactive Next.js tutorial.
- [Fumadocs](https://fumadocs.dev) - learn about Fumadocs
+6
View File
@@ -0,0 +1,6 @@
import { HomeLayout } from 'fumadocs-ui/layouts/home';
import { baseOptions } from '@/lib/layout.shared';
export default function Layout({ children }: LayoutProps<'/'>) {
return <HomeLayout {...baseOptions()}>{children}</HomeLayout>;
}
+5
View File
@@ -0,0 +1,5 @@
import { redirect } from 'next/navigation';
export default function HomePage() {
redirect('/docs');
}
+7
View File
@@ -0,0 +1,7 @@
import { source } from '@/lib/source';
import { createFromSource } from 'fumadocs-core/search/server';
export const { GET } = createFromSource(source, {
// https://docs.orama.com/docs/orama-js/supported-languages
language: 'english',
});
+63
View File
@@ -0,0 +1,63 @@
import { getPageImage, getPageMarkdownUrl, source } from '@/lib/source';
import {
DocsBody,
DocsDescription,
DocsPage,
DocsTitle,
MarkdownCopyButton,
ViewOptionsPopover,
} from 'fumadocs-ui/layouts/docs/page';
import { notFound } from 'next/navigation';
import { getMDXComponents } from '@/components/mdx';
import type { Metadata } from 'next';
import { createRelativeLink } from 'fumadocs-ui/mdx';
import { gitConfig } from '@/lib/shared';
export default async function Page(props: PageProps<'/docs/[[...slug]]'>) {
const params = await props.params;
const page = source.getPage(params.slug);
if (!page) notFound();
const MDX = page.data.body;
const markdownUrl = getPageMarkdownUrl(page).url;
return (
<DocsPage toc={page.data.toc} full={page.data.full}>
<DocsTitle>{page.data.title}</DocsTitle>
<DocsDescription className="mb-0">{page.data.description}</DocsDescription>
<div className="flex flex-row gap-2 items-center border-b pb-6">
<MarkdownCopyButton markdownUrl={markdownUrl} />
<ViewOptionsPopover
markdownUrl={markdownUrl}
githubUrl={`https://github.com/${gitConfig.user}/${gitConfig.repo}/blob/${gitConfig.branch}/content/docs/${page.path}`}
/>
</div>
<DocsBody>
<MDX
components={getMDXComponents({
// this allows you to link to other pages with relative file paths
a: createRelativeLink(source, page),
})}
/>
</DocsBody>
</DocsPage>
);
}
export async function generateStaticParams() {
return source.generateParams();
}
export async function generateMetadata(props: PageProps<'/docs/[[...slug]]'>): Promise<Metadata> {
const params = await props.params;
const page = source.getPage(params.slug);
if (!page) notFound();
return {
title: page.data.title,
description: page.data.description,
openGraph: {
images: getPageImage(page).url,
},
};
}
+11
View File
@@ -0,0 +1,11 @@
import { source } from '@/lib/source';
import { DocsLayout } from 'fumadocs-ui/layouts/docs';
import { baseOptions } from '@/lib/layout.shared';
export default function Layout({ children }: LayoutProps<'/docs'>) {
return (
<DocsLayout tree={source.getPageTree()} {...baseOptions()}>
{children}
</DocsLayout>
);
}
+53
View File
@@ -0,0 +1,53 @@
@import 'tailwindcss';
@import 'fumadocs-ui/css/neutral.css';
@import 'fumadocs-ui/css/preset.css';
@import 'fumadocs-twoslash/twoslash.css';
@theme {
--color-fd-background: hsl(0, 0%, 98%);
--color-fd-foreground: hsl(0, 0%, 3.9%);
--color-fd-muted: hsl(0, 0%, 96.1%);
--color-fd-muted-foreground: hsl(0, 0%, 45.1%);
--color-fd-popover: hsl(0, 0%, 100%);
--color-fd-popover-foreground: hsl(0, 0%, 15.1%);
--color-fd-card: hsl(0, 0%, 99.7%);
--color-fd-card-foreground: hsl(0, 0%, 3.9%);
--color-fd-border: hsla(0, 0%, 60%, 0.2);
--color-fd-primary: hsl(0, 0%, 9%);
--color-fd-primary-foreground: hsl(0, 0%, 98%);
--color-fd-secondary: hsl(0, 0%, 96.1%);
--color-fd-secondary-foreground: hsl(0, 0%, 9%);
--color-fd-accent: hsl(0, 0%, 94.1%);
--color-fd-accent-foreground: hsl(0, 0%, 9%);
--color-fd-ring: hsl(0, 0%, 63.9%);
--color-purple: hsl(262, 52%, 56%);
}
.dark {
--color-fd-background: hsl(0, 0%, 2%);
--color-fd-foreground: hsl(0, 0%, 98%);
--color-fd-muted: hsl(0, 0%, 8%);
--color-fd-muted-foreground: hsl(0, 0%, 60%);
--color-fd-popover: hsl(0, 0%, 4%);
--color-fd-popover-foreground: hsl(0, 0%, 98%);
--color-fd-card: hsl(0, 0%, 4%);
--color-fd-card-foreground: hsl(0, 0%, 98%);
--color-fd-border: hsl(0, 0%, 50%, 0.2);
--color-fd-primary: hsl(0, 0%, 98%);
--color-fd-primary-foreground: hsl(0, 0%, 9%);
--color-fd-secondary: hsl(0, 0%, 12.9%);
--color-fd-secondary-foreground: hsl(0, 0%, 98%);
--color-fd-accent: hsl(0, 0%, 15%);
--color-fd-accent-foreground: hsl(0, 0%, 100%);
--color-fd-ring: hsl(0, 0%, 34.9%);
--color-purple: hsl(262, 60%, 65%);
}
html {
scrollbar-gutter: stable;
}
html > body[data-scroll-locked] {
margin-right: 0px !important;
--removed-body-scroll-bar-size: 0px !important;
}
+53
View File
@@ -0,0 +1,53 @@
import { RootProvider } from 'fumadocs-ui/provider/next';
import './global.css';
import { Inter } from 'next/font/google';
import type { Metadata } from 'next';
const inter = Inter({
subsets: ['latin'],
});
// Canonical URL for the live docs. ``metadataBase`` resolves the og:url
// and twitter:url for every page; pointing it at the actual live site
// is what lets crawlers (search + LLM) follow the right canonical and
// pick up ``/llms.txt`` / ``/sitemap.xml`` / og images. Override at
// build time via ``NEXT_PUBLIC_SITE_URL`` (e.g. when promoting to a
// custom domain).
const SITE_URL = process.env.NEXT_PUBLIC_SITE_URL ?? 'https://headroom-docs.vercel.app';
export const metadata: Metadata = {
title: {
default: 'Headroom — Context Optimization Layer for AI Agents',
template: '%s | Headroom',
},
description:
'Compress everything your AI agent reads — tool outputs, logs, files, RAG chunks. Same answers, fraction of the tokens. Library, proxy, MCP server. Local-first. Apache 2.0.',
metadataBase: new URL(SITE_URL),
alternates: {
canonical: '/',
},
openGraph: {
type: 'website',
siteName: 'Headroom',
title: 'Headroom — Context Optimization Layer for AI Agents',
description:
'Compress tool outputs, logs, files, and RAG chunks before they reach the LLM. 6095% fewer tokens, same answers.',
url: '/',
},
twitter: {
card: 'summary_large_image',
title: 'Headroom — Context Optimization Layer for AI Agents',
description:
'Compress tool outputs, logs, files, and RAG chunks before they reach the LLM. 6095% fewer tokens, same answers.',
},
};
export default function Layout({ children }: LayoutProps<'/'>) {
return (
<html lang="en" className={inter.className} suppressHydrationWarning>
<body className="flex flex-col min-h-screen">
<RootProvider>{children}</RootProvider>
</body>
</html>
);
}
+10
View File
@@ -0,0 +1,10 @@
import { getLLMText, source } from '@/lib/source';
export const revalidate = false;
export async function GET() {
const scan = source.getPages().map(getLLMText);
const scanned = await Promise.all(scan);
return new Response(scanned.join('\n\n'));
}
@@ -0,0 +1,23 @@
import { getLLMText, getPageMarkdownUrl, source } from '@/lib/source';
import { notFound } from 'next/navigation';
export const revalidate = false;
export async function GET(_req: Request, { params }: RouteContext<'/llms.mdx/docs/[[...slug]]'>) {
const { slug } = await params;
const page = source.getPage(slug?.slice(0, -1));
if (!page) notFound();
return new Response(await getLLMText(page), {
headers: {
'Content-Type': 'text/markdown',
},
});
}
export function generateStaticParams() {
return source.getPages().map((page) => ({
lang: page.locale,
slug: getPageMarkdownUrl(page).segments,
}));
}
+8
View File
@@ -0,0 +1,8 @@
import { source } from '@/lib/source';
import { llms } from 'fumadocs-core/source';
export const revalidate = false;
export function GET() {
return new Response(llms(source).index());
}
+27
View File
@@ -0,0 +1,27 @@
import { getPageImage, source } from '@/lib/source';
import { notFound } from 'next/navigation';
import { ImageResponse } from 'next/og';
import { generate as DefaultImage } from 'fumadocs-ui/og';
export const revalidate = false;
export async function GET(_req: Request, { params }: RouteContext<'/og/docs/[...slug]'>) {
const { slug } = await params;
const page = source.getPage(slug.slice(0, -1));
if (!page) notFound();
return new ImageResponse(
<DefaultImage title={page.data.title} description={page.data.description} site="My App" />,
{
width: 1200,
height: 630,
},
);
}
export function generateStaticParams() {
return source.getPages().map((page) => ({
lang: page.locale,
slug: getPageImage(page).segments,
}));
}
+42
View File
@@ -0,0 +1,42 @@
// Next.js App Router robots convention (Next 13+). The default Next
// behaviour allows everything; this file makes the intent explicit so
// AI-bot operators that read an opt-in list (GPTBot, ClaudeBot,
// PerplexityBot, Google-Extended, etc.) see a clear green light, and
// so the sitemap is discoverable.
//
// Headroom docs are open-source documentation we WANT indexed. If a
// future page should be excluded, add it to the ``disallow`` list of
// the relevant rule.
import type { MetadataRoute } from 'next';
const SITE_URL = process.env.NEXT_PUBLIC_SITE_URL ?? 'https://headroom-docs.vercel.app';
export default function robots(): MetadataRoute.Robots {
return {
rules: [
// Bot-specific allows. These names are the literal user-agent
// strings each operator publishes. Listing them explicitly is
// the documented way to opt INTO AI-training / AI-search
// indexing — silence (no rule) is treated as opt-out by some
// operators (notably Google-Extended).
{ userAgent: 'GPTBot', allow: '/' },
{ userAgent: 'OAI-SearchBot', allow: '/' },
{ userAgent: 'ChatGPT-User', allow: '/' },
{ userAgent: 'ClaudeBot', allow: '/' },
{ userAgent: 'Claude-Web', allow: '/' },
{ userAgent: 'anthropic-ai', allow: '/' },
{ userAgent: 'PerplexityBot', allow: '/' },
{ userAgent: 'Perplexity-User', allow: '/' },
{ userAgent: 'Google-Extended', allow: '/' },
{ userAgent: 'cohere-ai', allow: '/' },
{ userAgent: 'CCBot', allow: '/' },
{ userAgent: 'Applebot-Extended', allow: '/' },
// Catch-all so traditional search crawlers also see an
// explicit allow.
{ userAgent: '*', allow: '/' },
],
sitemap: `${SITE_URL}/sitemap.xml`,
host: SITE_URL,
};
}
+39
View File
@@ -0,0 +1,39 @@
// Next.js App Router sitemap convention (Next 13+). Pulls every page
// out of the Fumadocs ``source`` (same source that backs ``/llms.txt``,
// search, and the OG image generator) and emits a valid sitemap.xml.
//
// Search engines and AI crawlers use this to enumerate every doc page
// without scraping HTML. The ``robots.ts`` route advertises the
// sitemap URL so well-behaved crawlers find it on the first GET.
import type { MetadataRoute } from 'next';
import { source } from '@/lib/source';
const SITE_URL = process.env.NEXT_PUBLIC_SITE_URL ?? 'https://headroom-docs.vercel.app';
export default function sitemap(): MetadataRoute.Sitemap {
const now = new Date();
// Static top-level routes (home page; docs index is covered by the
// page enumeration below).
const staticRoutes: MetadataRoute.Sitemap = [
{
url: `${SITE_URL}/`,
lastModified: now,
changeFrequency: 'weekly',
priority: 1.0,
},
];
// Every Fumadocs page (introduction, quickstart, installation,
// integrations, …). ``page.url`` is the relative URL like
// ``/docs/quickstart``; ``page.data`` carries the front-matter.
const docPages: MetadataRoute.Sitemap = source.getPages().map((page) => ({
url: `${SITE_URL}${page.url}`,
lastModified: now,
changeFrequency: 'weekly' as const,
priority: 0.8,
}));
return [...staticRoutes, ...docPages];
}
+933
View File
@@ -0,0 +1,933 @@
{
"lockfileVersion": 1,
"configVersion": 0,
"workspaces": {
"": {
"name": "headroom-docs",
"dependencies": {
"@radix-ui/react-slot": "1.3.0",
"class-variance-authority": "0.7.1",
"clsx": "2.1.1",
"dotted-map": "^3.1.0",
"fumadocs-core": "16.10.3",
"fumadocs-mdx": "15.0.12",
"fumadocs-twoslash": "^3.1.3",
"fumadocs-typescript": "^4.0.3",
"fumadocs-ui": "16.10.3",
"headroom-ai": "file:../sdk/typescript",
"lucide-react": "^1.7.0",
"next": "16.2.6",
"react": "^19.2.4",
"react-dom": "^19.2.4",
"recharts": "^3.8.1",
"tailwind-merge": "^3.5.0",
},
"devDependencies": {
"@ai-sdk/openai": "^3.0.51",
"@anthropic-ai/sdk": "^0.106.0",
"@tailwindcss/postcss": "^4.2.2",
"@types/mdx": "^2.0.13",
"@types/node": "^25.5.0",
"@types/react": "^19.2.14",
"@types/react-dom": "^19.2.3",
"ai": "^6.0.149",
"openai": "^6.33.0",
"postcss": "^8.5.13",
"tailwindcss": "^4.2.2",
"typescript": "^5.9.3",
},
},
},
"overrides": {
"postcss": "^8.5.13",
},
"packages": {
"@ai-sdk/gateway": ["@ai-sdk/gateway@3.0.91", "", { "dependencies": { "@ai-sdk/provider": "3.0.8", "@ai-sdk/provider-utils": "4.0.23", "@vercel/oidc": "3.1.0" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, ""],
"@ai-sdk/openai": ["@ai-sdk/openai@3.0.51", "", { "dependencies": { "@ai-sdk/provider": "3.0.8", "@ai-sdk/provider-utils": "4.0.23" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, ""],
"@ai-sdk/provider": ["@ai-sdk/provider@3.0.8", "", { "dependencies": { "json-schema": "^0.4.0" } }, ""],
"@ai-sdk/provider-utils": ["@ai-sdk/provider-utils@4.0.23", "", { "dependencies": { "@ai-sdk/provider": "3.0.8", "@standard-schema/spec": "^1.1.0", "eventsource-parser": "^3.0.6" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, ""],
"@alloc/quick-lru": ["@alloc/quick-lru@5.2.0", "", {}, ""],
"@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.106.0", "", { "dependencies": { "json-schema-to-ts": "^3.1.1", "standardwebhooks": "^1.0.0" }, "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" }, "bin": { "anthropic-ai-sdk": "bin/cli" } }, "sha512-ufwVvYNDBj2dzOGupBCTaNzBLxqcTnGOzI4z8Wouxlt+mT3J3HuOmatgCy1VmwCHOUueqZ41ERhm0O99OUcbWA=="],
"@babel/runtime": ["@babel/runtime@7.29.2", "", {}, ""],
"@emnapi/runtime": ["@emnapi/runtime@1.11.0", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-55coeOFKHv1ywEcUXJtWU5f+Jr/W5tZDvZig8DLKSwUN1JpROQ4rk/SNOQiFWmaR/VKF4zuFyW1B8JduOSv6Pg=="],
"@esbuild/aix-ppc64": ["@esbuild/aix-ppc64@0.28.1", "", { "os": "aix", "cpu": "ppc64" }, "sha512-Svl7tq8k/08+p6CXPpRjQ1fKX+1odH/BQbb48fV6fj3CWHhsoIOoY87w1oHXm0qEpkIK3ZfVgp0hed3XBXzXMQ=="],
"@esbuild/android-arm": ["@esbuild/android-arm@0.28.1", "", { "os": "android", "cpu": "arm" }, "sha512-0k2F129Xdio1TdJfzJ8sy1Q47vUD2NnwdhiAf7drUN1EBTfPf4hsFCtmMgu/6m8JSzsBrlmVjudMBQqOfG8usQ=="],
"@esbuild/android-arm64": ["@esbuild/android-arm64@0.28.1", "", { "os": "android", "cpu": "arm64" }, "sha512-34EGEbCIAgosYz6goLcopX6Mo7NyGv9tfwEM2/7Ce2VcVRk568iSvniGWcUXIy7wEDR1wzolcxcriFVrWYcwBg=="],
"@esbuild/android-x64": ["@esbuild/android-x64@0.28.1", "", { "os": "android", "cpu": "x64" }, "sha512-dbwY7ltSMDWsRatcRpCnES4F+im88OCUgGZjy52shC7GqHRE/cYlxNbB4Z4UpJswpcc4Qxd2oE/ufM0p61IKng=="],
"@esbuild/darwin-arm64": ["@esbuild/darwin-arm64@0.28.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-TZbWkQY7kvTAXbXUT7uVACR5cMHsDiSz9z7ZKAX/RTq/WJEk3QyRr0wZpNhBDX+/0CtdqUIJlOiodQcta6tY3Q=="],
"@esbuild/darwin-x64": ["@esbuild/darwin-x64@0.28.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-zfdzgK9ACBNZLI/CyHTOx81SyNbM6YXn7rxSgX97VjyiPl9W1i4Ka4fgKECEoFCKGpvBj5qArWIGgQjOwkgskQ=="],
"@esbuild/freebsd-arm64": ["@esbuild/freebsd-arm64@0.28.1", "", { "os": "freebsd", "cpu": "arm64" }, "sha512-wG2EA8ENdEI0qhkSZMjfqrdY+ziCYCPMmtZjjIwOmXFjmyzEHn+UUxk5of+SYsjtfs3VpnlC7QLzSI5hY/rOAw=="],
"@esbuild/freebsd-x64": ["@esbuild/freebsd-x64@0.28.1", "", { "os": "freebsd", "cpu": "x64" }, "sha512-i7dZ9vQgnvSCzi/rYCXNgtF/U+eKZNJBzu3eTQbRgHnM7tNSizLOkRFAl3qzVc/Op/u5YkHHa4pf/3DOYHthLQ=="],
"@esbuild/linux-arm": ["@esbuild/linux-arm@0.28.1", "", { "os": "linux", "cpu": "arm" }, "sha512-qVXBOHQS+d5Y722GwJzJUtOLlX7km3CraOaGormF1pDtPd2C/l1SHRPgjLunLGe51Sh5YYWKMFDyV4SxgMQYTQ=="],
"@esbuild/linux-arm64": ["@esbuild/linux-arm64@0.28.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-yHs+0uc8+nvEAfAfxrWQKK5peSNzBc4PegcMO0EJ2hT71uA7vB8Ihg2e77R2P7SG5uYjPbHlLLmve4LLLRCf0g=="],
"@esbuild/linux-ia32": ["@esbuild/linux-ia32@0.28.1", "", { "os": "linux", "cpu": "ia32" }, "sha512-d1z4ZuP0ajrfz/FhGT4vv278rX8KnPPJx8i5+AtK7TYbx9Le9F1hyzurZpkEyjkGa9dUGhQow4C1NmeGvqxN2w=="],
"@esbuild/linux-loong64": ["@esbuild/linux-loong64@0.28.1", "", { "os": "linux", "cpu": "none" }, "sha512-M5sRjUVZrkm1OAPR3dlOYzNmN+loZKGVi1VUQGrwuqLcbR6qeAz+famMhjASeH3YVKvZz+zT1jlh/keC3Rj/lg=="],
"@esbuild/linux-mips64el": ["@esbuild/linux-mips64el@0.28.1", "", { "os": "linux", "cpu": "none" }, "sha512-mRObBZeHh2OxcBFPWE/FjylkRgZdYuiTR3vaTozquCGOH14iP9oN4x4Ge81CoIDYQrXmIxpFumJBu5MtZpnQJQ=="],
"@esbuild/linux-ppc64": ["@esbuild/linux-ppc64@0.28.1", "", { "os": "linux", "cpu": "ppc64" }, "sha512-slScBsMAb3GFDcdrCgLwZtPYRoH2H/youv10QiZyRjmsP48fznoveWytSgCI/R0ZcUgpc0ZhIUEx6LHts8yrfQ=="],
"@esbuild/linux-riscv64": ["@esbuild/linux-riscv64@0.28.1", "", { "os": "linux", "cpu": "none" }, "sha512-kw0owk1o0GFETUJyW0jc0G4Yzs0BHZn0JDZ8JRT088vjJYX777BAs1fDGxAC+q831qOs2DTC96mNsG2opdfyyQ=="],
"@esbuild/linux-s390x": ["@esbuild/linux-s390x@0.28.1", "", { "os": "linux", "cpu": "s390x" }, "sha512-/lAIjX8aYFRByhh6L5rYtPEDRqa9de/4V/juOXcta5frjvzXO4/sqEtyytse0g3zZFuWu5cDN0MkLz2qRDD2Ag=="],
"@esbuild/linux-x64": ["@esbuild/linux-x64@0.28.1", "", { "os": "linux", "cpu": "x64" }, "sha512-u/anNYF2mmVOEDwLtnQ1wOr3EZ9sTNGLWrsYGYwHWzGA3Si84IOkHXlbWTD1NB+9/1lcnweYKO54uhxZydNzfA=="],
"@esbuild/netbsd-arm64": ["@esbuild/netbsd-arm64@0.28.1", "", { "os": "none", "cpu": "arm64" }, "sha512-oks0DYbLwWMmaakTsCb+zL4E+aHRVLom9IJZOAthMQEPiQmydXHkziYEsGYRx0uNV/IjEKGAV941JzH02pflqw=="],
"@esbuild/netbsd-x64": ["@esbuild/netbsd-x64@0.28.1", "", { "os": "none", "cpu": "x64" }, "sha512-aeL6lAnN89Hz43Mlh1G8ARasbuoYvSITDEx0tHh5b7jJnHcssqgjy9Yx430GDpmCa6OyrKoS0aNRjKundRizGg=="],
"@esbuild/openbsd-arm64": ["@esbuild/openbsd-arm64@0.28.1", "", { "os": "openbsd", "cpu": "arm64" }, "sha512-MEFJe5C3R8pwXdZ5Y21oo6m7ePiS0d9pWucn99O/wvyJZChoIQKrQDxKrGeW8F5+T0okTHesAmDeiHDTIq0V/Q=="],
"@esbuild/openbsd-x64": ["@esbuild/openbsd-x64@0.28.1", "", { "os": "openbsd", "cpu": "x64" }, "sha512-i/ZLIOafE0Z8cI/XANJAixoJL/uRAoS2xOA3rb0xN+KK0K177cMAsQYkzHtBrtMXAKuAc7HGgcWiZ/sRC1Nxgw=="],
"@esbuild/openharmony-arm64": ["@esbuild/openharmony-arm64@0.28.1", "", { "os": "none", "cpu": "arm64" }, "sha512-ge+Z7EXFNt2BO1oAMsVpiQ8EwndV9i1xXerAeTIK7AtPs3bKFXQM7nlRxDSIUIMeueR1CNXxqztLzdNeReKBJg=="],
"@esbuild/sunos-x64": ["@esbuild/sunos-x64@0.28.1", "", { "os": "sunos", "cpu": "x64" }, "sha512-BEjgtECkL3vY+SaSQ6nzVfiALUeFxpawyp8Jmf5PtYhf1Ug40N1h/hxlhts+f1FvSvarEigdxS3BlSMI2PJLcQ=="],
"@esbuild/win32-arm64": ["@esbuild/win32-arm64@0.28.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-lCv9eK/H6ZJWbE7bh2nw54CZ9M2nupBxJcTsdk/QQnWkdSjKGuxmmH8/GWrlT1eMmZfn4dGcCjRte397WqfQXA=="],
"@esbuild/win32-ia32": ["@esbuild/win32-ia32@0.28.1", "", { "os": "win32", "cpu": "ia32" }, "sha512-zvb/mB2bSCoJOpoCBgYKKpX6YM6mJBlBUVUtVj41DlZJVEB6/0CKlRYxP5wWl1C1ILiCoAU5wZZ4q1P3qeS6Eg=="],
"@esbuild/win32-x64": ["@esbuild/win32-x64@0.28.1", "", { "os": "win32", "cpu": "x64" }, "sha512-bm4Mowrv+GXMlpWX++EcXw/iLyd1o3+bJkC2DkWXYVvgZCqD/bSj9ctZeAMC3cIxgjRVR2Dufaiu4YPxr5gW1A=="],
"@floating-ui/core": ["@floating-ui/core@1.7.5", "", { "dependencies": { "@floating-ui/utils": "^0.2.11" } }, "sha512-1Ih4WTWyw0+lKyFMcBHGbb5U5FtuHJuujoyyr5zTaWS5EYMeT6Jb2AuDeftsCsEuchO+mM2ij5+q9crhydzLhQ=="],
"@floating-ui/dom": ["@floating-ui/dom@1.7.6", "", { "dependencies": { "@floating-ui/core": "^1.7.5", "@floating-ui/utils": "^0.2.11" } }, "sha512-9gZSAI5XM36880PPMm//9dfiEngYoC6Am2izES1FF406YFsjvyBMmeJ2g4SAju3xWwtuynNRFL2s9hgxpLI5SQ=="],
"@floating-ui/react-dom": ["@floating-ui/react-dom@2.1.8", "", { "dependencies": { "@floating-ui/dom": "^1.7.6" }, "peerDependencies": { "react": ">=16.8.0", "react-dom": ">=16.8.0" } }, "sha512-cC52bHwM/n/CxS87FH0yWdngEZrjdtLW/qVruo68qg+prK7ZQ4YGdut2GyDVpoGeAYe/h899rVeOVm6Oi40k2A=="],
"@floating-ui/utils": ["@floating-ui/utils@0.2.11", "", {}, "sha512-RiB/yIh78pcIxl6lLMG0CgBXAZ2Y0eVHqMPYugu+9U0AeT6YBeiJpf7lbdJNIugFP5SIjwNRgo4DhR1Qxi26Gg=="],
"@fuma-translate/react": ["@fuma-translate/react@1.0.2", "", { "peerDependencies": { "@types/react": "*", "react": "^19.2.0", "react-dom": "^19.2.0" } }, "sha512-uOiOtBx3nRXR8Nu1GzBf1tApgF1FErDBTHxRIAQeyQdyOoZbrNRN6H4kDCWObY4qyGeGbHydG0DHzgeUgFDMIw=="],
"@fumadocs/tailwind": ["@fumadocs/tailwind@0.0.5", "", { "peerDependencies": { "@tailwindcss/oxide": "^4.0.0", "tailwindcss": "^4.0.0" } }, "sha512-ENKPWUDRmriccsrUDE4bDBq3FNr/ms3BP2rWlsAEMV1yP23pcCaan+ceGfeBUsAQjw7sj9Q3R4Kl3g/TCStPzQ=="],
"@img/colour": ["@img/colour@1.1.0", "", {}, ""],
"@img/sharp-darwin-arm64": ["@img/sharp-darwin-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.2.4" }, "os": "darwin", "cpu": "arm64" }, "sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w=="],
"@img/sharp-darwin-x64": ["@img/sharp-darwin-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-x64": "1.2.4" }, "os": "darwin", "cpu": "x64" }, "sha512-YNEFAF/4KQ/PeW0N+r+aVVsoIY0/qxxikF2SWdp+NRkmMB7y9LBZAVqQ4yhGCm/H3H270OSykqmQMKLBhBJDEw=="],
"@img/sharp-libvips-darwin-arm64": ["@img/sharp-libvips-darwin-arm64@1.2.4", "", { "os": "darwin", "cpu": "arm64" }, "sha512-zqjjo7RatFfFoP0MkQ51jfuFZBnVE2pRiaydKJ1G/rHZvnsrHAOcQALIi9sA5co5xenQdTugCvtb1cuf78Vf4g=="],
"@img/sharp-libvips-darwin-x64": ["@img/sharp-libvips-darwin-x64@1.2.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-1IOd5xfVhlGwX+zXv2N93k0yMONvUlANylbJw1eTah8K/Jtpi15KC+WSiaX/nBmbm2HxRM1gZ0nSdjSsrZbGKg=="],
"@img/sharp-libvips-linux-arm": ["@img/sharp-libvips-linux-arm@1.2.4", "", { "os": "linux", "cpu": "arm" }, "sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A=="],
"@img/sharp-libvips-linux-arm64": ["@img/sharp-libvips-linux-arm64@1.2.4", "", { "os": "linux", "cpu": "arm64" }, ""],
"@img/sharp-libvips-linux-ppc64": ["@img/sharp-libvips-linux-ppc64@1.2.4", "", { "os": "linux", "cpu": "ppc64" }, "sha512-FMuvGijLDYG6lW+b/UvyilUWu5Ayu+3r2d1S8notiGCIyYU/76eig1UfMmkZ7vwgOrzKzlQbFSuQfgm7GYUPpA=="],
"@img/sharp-libvips-linux-riscv64": ["@img/sharp-libvips-linux-riscv64@1.2.4", "", { "os": "linux", "cpu": "none" }, "sha512-oVDbcR4zUC0ce82teubSm+x6ETixtKZBh/qbREIOcI3cULzDyb18Sr/Wcyx7NRQeQzOiHTNbZFF1UwPS2scyGA=="],
"@img/sharp-libvips-linux-s390x": ["@img/sharp-libvips-linux-s390x@1.2.4", "", { "os": "linux", "cpu": "s390x" }, "sha512-qmp9VrzgPgMoGZyPvrQHqk02uyjA0/QrTO26Tqk6l4ZV0MPWIW6LTkqOIov+J1yEu7MbFQaDpwdwJKhbJvuRxQ=="],
"@img/sharp-libvips-linux-x64": ["@img/sharp-libvips-linux-x64@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw=="],
"@img/sharp-libvips-linuxmusl-arm64": ["@img/sharp-libvips-linuxmusl-arm64@1.2.4", "", { "os": "linux", "cpu": "arm64" }, ""],
"@img/sharp-libvips-linuxmusl-x64": ["@img/sharp-libvips-linuxmusl-x64@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg=="],
"@img/sharp-linux-arm": ["@img/sharp-linux-arm@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm": "1.2.4" }, "os": "linux", "cpu": "arm" }, "sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw=="],
"@img/sharp-linux-arm64": ["@img/sharp-linux-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, ""],
"@img/sharp-linux-ppc64": ["@img/sharp-linux-ppc64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-ppc64": "1.2.4" }, "os": "linux", "cpu": "ppc64" }, "sha512-7zznwNaqW6YtsfrGGDA6BRkISKAAE1Jo0QdpNYXNMHu2+0dTrPflTLNkpc8l7MUP5M16ZJcUvysVWWrMefZquA=="],
"@img/sharp-linux-riscv64": ["@img/sharp-linux-riscv64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-riscv64": "1.2.4" }, "os": "linux", "cpu": "none" }, "sha512-51gJuLPTKa7piYPaVs8GmByo7/U7/7TZOq+cnXJIHZKavIRHAP77e3N2HEl3dgiqdD/w0yUfiJnII77PuDDFdw=="],
"@img/sharp-linux-s390x": ["@img/sharp-linux-s390x@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-s390x": "1.2.4" }, "os": "linux", "cpu": "s390x" }, "sha512-nQtCk0PdKfho3eC5MrbQoigJ2gd1CgddUMkabUj+rBevs8tZ2cULOx46E7oyX+04WGfABgIwmMC0VqieTiR4jg=="],
"@img/sharp-linux-x64": ["@img/sharp-linux-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ=="],
"@img/sharp-linuxmusl-arm64": ["@img/sharp-linuxmusl-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, ""],
"@img/sharp-linuxmusl-x64": ["@img/sharp-linuxmusl-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q=="],
"@img/sharp-wasm32": ["@img/sharp-wasm32@0.34.5", "", { "dependencies": { "@emnapi/runtime": "^1.7.0" }, "cpu": "none" }, "sha512-OdWTEiVkY2PHwqkbBI8frFxQQFekHaSSkUIJkwzclWZe64O1X4UlUjqqqLaPbUpMOQk6FBu/HtlGXNblIs0huw=="],
"@img/sharp-win32-arm64": ["@img/sharp-win32-arm64@0.34.5", "", { "os": "win32", "cpu": "arm64" }, "sha512-WQ3AgWCWYSb2yt+IG8mnC6Jdk9Whs7O0gxphblsLvdhSpSTtmu69ZG1Gkb6NuvxsNACwiPV6cNSZNzt0KPsw7g=="],
"@img/sharp-win32-ia32": ["@img/sharp-win32-ia32@0.34.5", "", { "os": "win32", "cpu": "ia32" }, "sha512-FV9m/7NmeCmSHDD5j4+4pNI8Cp3aW+JvLoXcTUo0IqyjSfAZJ8dIUmijx1qaJsIiU+Hosw6xM5KijAWRJCSgNg=="],
"@img/sharp-win32-x64": ["@img/sharp-win32-x64@0.34.5", "", { "os": "win32", "cpu": "x64" }, "sha512-+29YMsqY2/9eFEiW93eqWnuLcWcufowXewwSNIT6UwZdUUCrM3oFjMWH/Z6/TMmb4hlFenmfAVbpWeup2jryCw=="],
"@jridgewell/gen-mapping": ["@jridgewell/gen-mapping@0.3.13", "", { "dependencies": { "@jridgewell/sourcemap-codec": "^1.5.0", "@jridgewell/trace-mapping": "^0.3.24" } }, ""],
"@jridgewell/remapping": ["@jridgewell/remapping@2.3.5", "", { "dependencies": { "@jridgewell/gen-mapping": "^0.3.5", "@jridgewell/trace-mapping": "^0.3.24" } }, ""],
"@jridgewell/resolve-uri": ["@jridgewell/resolve-uri@3.1.2", "", {}, ""],
"@jridgewell/sourcemap-codec": ["@jridgewell/sourcemap-codec@1.5.5", "", {}, ""],
"@jridgewell/trace-mapping": ["@jridgewell/trace-mapping@0.3.31", "", { "dependencies": { "@jridgewell/resolve-uri": "^3.1.0", "@jridgewell/sourcemap-codec": "^1.4.14" } }, ""],
"@mdx-js/mdx": ["@mdx-js/mdx@3.1.1", "", { "dependencies": { "@types/estree": "^1.0.0", "@types/estree-jsx": "^1.0.0", "@types/hast": "^3.0.0", "@types/mdx": "^2.0.0", "acorn": "^8.0.0", "collapse-white-space": "^2.0.0", "devlop": "^1.0.0", "estree-util-is-identifier-name": "^3.0.0", "estree-util-scope": "^1.0.0", "estree-walker": "^3.0.0", "hast-util-to-jsx-runtime": "^2.0.0", "markdown-extensions": "^2.0.0", "recma-build-jsx": "^1.0.0", "recma-jsx": "^1.0.0", "recma-stringify": "^1.0.0", "rehype-recma": "^1.0.0", "remark-mdx": "^3.0.0", "remark-parse": "^11.0.0", "remark-rehype": "^11.0.0", "source-map": "^0.7.0", "unified": "^11.0.0", "unist-util-position-from-estree": "^2.0.0", "unist-util-stringify-position": "^4.0.0", "unist-util-visit": "^5.0.0", "vfile": "^6.0.0" } }, ""],
"@next/env": ["@next/env@16.2.6", "", {}, "sha512-gd8HoHN4ufj73WmR3JmVolrpJR47ILK6LouP5xElPglaVxir6e1a7VzvTvDWkOoPXT9rkkTzyCxBu4yeZfZwcw=="],
"@next/swc-darwin-arm64": ["@next/swc-darwin-arm64@16.2.6", "", { "os": "darwin", "cpu": "arm64" }, "sha512-ZJGkkcNfYgrrMkqOdZ7zoLa1TOy0qpcMfk/z4Mh/FKUz40gVO+HNQWqmLxf67Z5WB64DRp0dhEbyHfel+6sJUg=="],
"@next/swc-darwin-x64": ["@next/swc-darwin-x64@16.2.6", "", { "os": "darwin", "cpu": "x64" }, "sha512-v/YLBHIY132Ced3puBJ7YJKw1lqsCrgcNo2aRJlCEyQrrCeRJlvGlnmxhPxNQI3KE3N1DN5r9TPNPvka3nq5RQ=="],
"@next/swc-linux-arm64-gnu": ["@next/swc-linux-arm64-gnu@16.2.6", "", { "os": "linux", "cpu": "arm64" }, "sha512-RPOvqlYBbcQjkz9VQQDZ2T2bARIjXZV1KFlt+V2Mr6SW/e4I9fcKsaA0hdyf2FHoTlsV2xnBd5Y912rP/1Ce6w=="],
"@next/swc-linux-arm64-musl": ["@next/swc-linux-arm64-musl@16.2.6", "", { "os": "linux", "cpu": "arm64" }, "sha512-URUTu1+dMkxJsPFgm+OeEvq9wf5sujw0EvgYy80TDGHTSLTnIHeqb0Eu8A3sC95IRgjejQL+kC4mw+4yPxiAXA=="],
"@next/swc-linux-x64-gnu": ["@next/swc-linux-x64-gnu@16.2.6", "", { "os": "linux", "cpu": "x64" }, "sha512-DOj182mPV8G3UkrayLoREM5YEYI+Dk5wv7Ox9xl1fFibAELEsFD0lDPfHIeILlutMMfdyhlzYPELG3peuKaurw=="],
"@next/swc-linux-x64-musl": ["@next/swc-linux-x64-musl@16.2.6", "", { "os": "linux", "cpu": "x64" }, "sha512-HKQ5SP/V/ub73UvF7n/zeJlxk2kLmtL7Wzrg4WfmkjmNos5onJ2tKu7yZOPdL18A6Svfn3max29ym+ry7NkK4g=="],
"@next/swc-win32-arm64-msvc": ["@next/swc-win32-arm64-msvc@16.2.6", "", { "os": "win32", "cpu": "arm64" }, "sha512-LZXpTlPyS5v7HhSmnvsLGP3iIYgYOBnc8r8ArlT55sGHV89bR2HlDdBjWQ+PY6SJMmk8TuVGFuxalnP3k/0Dwg=="],
"@next/swc-win32-x64-msvc": ["@next/swc-win32-x64-msvc@16.2.6", "", { "os": "win32", "cpu": "x64" }, "sha512-F0+4i0h9J6C4eE3EAPWsoCk7UW/dbzOjyzxY0qnDUOYFu6FFmdZ6l97/XdV3/Nz3VYyO7UWjyEJUXkGqcoXfMA=="],
"@opentelemetry/api": ["@opentelemetry/api@1.9.0", "", {}, ""],
"@orama/orama": ["@orama/orama@3.1.18", "", {}, ""],
"@radix-ui/number": ["@radix-ui/number@1.1.2", "", {}, "sha512-ceTwaxc4I5IOi97DgCotl3pqiyRGvffcc0oOsE2dQYaJOFIDsDt4VWG6xEbg1QePv9QWausCEIppud/tJ1wNig=="],
"@radix-ui/primitive": ["@radix-ui/primitive@1.1.4", "", {}, "sha512-7AdCK9PQyiljKoBDbN8OuctCbd/esdwZPQ8RtOE3SsyQtUpiPb+ND75q0jEhC1m1ecBI0MFNeLJvwIh9iKHRcQ=="],
"@radix-ui/react-accordion": ["@radix-ui/react-accordion@1.2.14", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-collapsible": "1.1.14", "@radix-ui/react-collection": "1.1.10", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-direction": "1.1.2", "@radix-ui/react-id": "1.1.2", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-controllable-state": "1.2.3" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-iE8YB9nmTBH8zd73ofBISZ8JCzgMoMkATJr7qDwa6u5F1+7mTM81V6fa71jgZ65rpjVpecDf1vSnwIFP9Ly1zw=="],
"@radix-ui/react-arrow": ["@radix-ui/react-arrow@1.1.10", "", { "dependencies": { "@radix-ui/react-primitive": "2.1.6" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-j2VTDz1vgCsmuG0k5lBfOcM8n5JPFqZBcMryasFjHYMhwxYL5SRUV5lMSUpRdNtw3D/Sv8pzJtrlAgkssYSsQQ=="],
"@radix-ui/react-collapsible": ["@radix-ui/react-collapsible@1.1.14", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-id": "1.1.2", "@radix-ui/react-presence": "1.1.6", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-controllable-state": "1.2.3", "@radix-ui/react-use-layout-effect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-9bT+FvifX1FK2Mj6UEsTdyu0cN3JaA3KdfhaBao+ONrYFy/pyOy3TU1TNw7iOk1o+0hOEq67RojlUUmoFGwxyA=="],
"@radix-ui/react-collection": ["@radix-ui/react-collection@1.1.10", "", { "dependencies": { "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-slot": "1.3.0" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-IVVz4EvBcKjrzKgof714qDnz/SzQAkLA2Emh5edlHbgcE6fNd3Un6CJLlaYcnm8N4JmAtzQgse4dOKxcD2yc9g=="],
"@radix-ui/react-compose-refs": ["@radix-ui/react-compose-refs@1.1.3", "", { "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-rYOP8OMnuuPMQF1uhPVlGNcCDlkokKqGFE3JcxFViIkAXP7EvFWUliJAstrapypaBLJNHbZL6jGhbVDGTwmVhA=="],
"@radix-ui/react-context": ["@radix-ui/react-context@1.1.4", "", { "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-QwH4PO5urrbO+FaGd5Aglg+YJgWTyyuZ3g/6mKvsqraLkglDdckw9JafgL5McL5VEJ6EPNduPaT3ZE9BttDAqg=="],
"@radix-ui/react-dialog": ["@radix-ui/react-dialog@1.1.17", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-dismissable-layer": "1.1.13", "@radix-ui/react-focus-guards": "1.1.4", "@radix-ui/react-focus-scope": "1.1.10", "@radix-ui/react-id": "1.1.2", "@radix-ui/react-portal": "1.1.12", "@radix-ui/react-presence": "1.1.6", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-slot": "1.3.0", "@radix-ui/react-use-controllable-state": "1.2.3", "aria-hidden": "^1.2.4", "react-remove-scroll": "^2.7.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-TDTYmpdq8dI2+Xgvgj9AJ8Ghqq+Eph/TRVEdaFQPDItIY+6QSkU7MJMeevw1568Yw/2Ijz8BTphPSP2XejKphw=="],
"@radix-ui/react-direction": ["@radix-ui/react-direction@1.1.2", "", { "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-C3vFhbyi4SW3PmbAi6Awpu4OzJtd0MxGurvSsYtr7p7nM8RNB3VAF3CUmnp2j50knpkrRcB7+ycVXzgLgF6yNA=="],
"@radix-ui/react-dismissable-layer": ["@radix-ui/react-dismissable-layer@1.1.13", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-callback-ref": "1.1.2", "@radix-ui/react-use-escape-keydown": "1.1.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-2v+zNAWWe0ySxgC0D0yeXMPQ23xZVgXZTerTz+JKlmdRj6gfTqmCcR29jb6d290DezXPGgruHWDX/vYUebtErg=="],
"@radix-ui/react-focus-guards": ["@radix-ui/react-focus-guards@1.1.4", "", { "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-cot/aB/mOm0IYVYTTmQcEEK1M48lZWi8FlYe5nDPQQ8NYZUlXEFgncJ9p2Kzer3RKSrY7cTTpEMLZKNo9QoP5Q=="],
"@radix-ui/react-focus-scope": ["@radix-ui/react-focus-scope@1.1.10", "", { "dependencies": { "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-callback-ref": "1.1.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-Fas/lXQqhVvqwAb64s5RFeHiHYElZ6SUQbZaNd6EkfhP/Al7wTIQ9WIR4QVX475tlu5yFCEdDcJH6/UwsZjMWw=="],
"@radix-ui/react-id": ["@radix-ui/react-id@1.1.2", "", { "dependencies": { "@radix-ui/react-use-layout-effect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-orBC88futVpqCmhX1p4cvquNHsELQ+w+vBJnuj3ftETI5bJb0bZn3Tqu3SWN2IOcPycTnMGnhwoermvISt72sA=="],
"@radix-ui/react-navigation-menu": ["@radix-ui/react-navigation-menu@1.2.16", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-collection": "1.1.10", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-direction": "1.1.2", "@radix-ui/react-dismissable-layer": "1.1.13", "@radix-ui/react-id": "1.1.2", "@radix-ui/react-presence": "1.1.6", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-callback-ref": "1.1.2", "@radix-ui/react-use-controllable-state": "1.2.3", "@radix-ui/react-use-layout-effect": "1.1.2", "@radix-ui/react-use-previous": "1.1.2", "@radix-ui/react-visually-hidden": "1.2.6" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-nJ0SkrSQgudyYhMiYeHA1ayLVuduEJCFLan1RZZN7c9kqzzCFLaU9kuy81uNtqzweM9YaQPgWzxi9MwQ9jZ04g=="],
"@radix-ui/react-popover": ["@radix-ui/react-popover@1.1.17", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-dismissable-layer": "1.1.13", "@radix-ui/react-focus-guards": "1.1.4", "@radix-ui/react-focus-scope": "1.1.10", "@radix-ui/react-id": "1.1.2", "@radix-ui/react-popper": "1.3.1", "@radix-ui/react-portal": "1.1.12", "@radix-ui/react-presence": "1.1.6", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-slot": "1.3.0", "@radix-ui/react-use-controllable-state": "1.2.3", "aria-hidden": "^1.2.4", "react-remove-scroll": "^2.7.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-/YSAOdJ7YJvdn7bn5sdSx2egW+SKY+u7O5RyAVs94Ymrg2fg5QTSFPMRkzvhGyFuE4/qsmPBdrwYoZMZh/4f+g=="],
"@radix-ui/react-popper": ["@radix-ui/react-popper@1.3.1", "", { "dependencies": { "@floating-ui/react-dom": "^2.0.0", "@radix-ui/react-arrow": "1.1.10", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-callback-ref": "1.1.2", "@radix-ui/react-use-layout-effect": "1.1.2", "@radix-ui/react-use-rect": "1.1.2", "@radix-ui/react-use-size": "1.1.2", "@radix-ui/rect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-bhnq/0DEPTi2lsOD3J5rTL65qUKHbKbhqHsmN9TMiclSXpipi651ooUKPPp6G5lF/WiHBdn1s0Wuqsn+myVAvw=="],
"@radix-ui/react-portal": ["@radix-ui/react-portal@1.1.12", "", { "dependencies": { "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-layout-effect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-m309havGzsjLHHaIX50G5PlvRs3xkgPCsGk/5PTvYm8D5q33yG0J7w/712PTOhid7NTaFETtnSXjngHQavvhVw=="],
"@radix-ui/react-presence": ["@radix-ui/react-presence@1.1.6", "", { "dependencies": { "@radix-ui/react-use-layout-effect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-zdTk4PlUO0E18HnZ3wYbW0KkJJxWCdiNYp6g6X1PtONFhxVkg01vliTJAmwIszU6mHiyBOoW9P0rAugl5/hULQ=="],
"@radix-ui/react-primitive": ["@radix-ui/react-primitive@2.1.6", "", { "dependencies": { "@radix-ui/react-slot": "1.3.0" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-wetd0QI77DbvrPpTAvH1SqOxsYF2wZe5TNxqwOd5Ty4XDpV3dpV0s8K/1MGMJBeY5o7lg8ub5VIt1Ub+yVen6g=="],
"@radix-ui/react-roving-focus": ["@radix-ui/react-roving-focus@1.1.13", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-collection": "1.1.10", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-direction": "1.1.2", "@radix-ui/react-id": "1.1.2", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-callback-ref": "1.1.2", "@radix-ui/react-use-controllable-state": "1.2.3" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-9gkwneI0guf8JDmrFxPjJF6Ozzgioyw+/lonYNCwefS9ZHA05er0BVHiXr+LbWGHxUfczvMY6G1oiZZi1VzjRw=="],
"@radix-ui/react-scroll-area": ["@radix-ui/react-scroll-area@1.2.12", "", { "dependencies": { "@radix-ui/number": "1.1.2", "@radix-ui/primitive": "1.1.4", "@radix-ui/react-compose-refs": "1.1.3", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-direction": "1.1.2", "@radix-ui/react-presence": "1.1.6", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-use-callback-ref": "1.1.2", "@radix-ui/react-use-layout-effect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-xuafVzQiTCLsyEjakowTdG3OgTXsmO7IdCiO77otIa+z44xoLNs9Do5eg7POFumIOCjtG6djfm6RKUKpUa/csA=="],
"@radix-ui/react-slot": ["@radix-ui/react-slot@1.3.0", "", { "dependencies": { "@radix-ui/react-compose-refs": "1.1.3" }, "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-MojKku4U/miO8Av4Dkb+ctMAQx7JmY96LmtDQlAarCRtd7rN52QCSzBF+XAvr5S6coSVj9HEPBgHAHKEJVk/WA=="],
"@radix-ui/react-tabs": ["@radix-ui/react-tabs@1.1.15", "", { "dependencies": { "@radix-ui/primitive": "1.1.4", "@radix-ui/react-context": "1.1.4", "@radix-ui/react-direction": "1.1.2", "@radix-ui/react-id": "1.1.2", "@radix-ui/react-presence": "1.1.6", "@radix-ui/react-primitive": "2.1.6", "@radix-ui/react-roving-focus": "1.1.13", "@radix-ui/react-use-controllable-state": "1.2.3" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-kxc9gI6/HfcU4nfMMVS3AmQK414kbU1IE6UCJmMmxjhO3cRPXOyYnmvyKD+ODt7q56nRq9l7Wovi6uaGwKgMlg=="],
"@radix-ui/react-use-callback-ref": ["@radix-ui/react-use-callback-ref@1.1.2", "", { "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-xCso9j1/u8sEgP1RNHjFrXJLApL8LiqOkI1R4ywuN00rxWdYg4oQXuwKLS3i0j5NWLromUD27/4nlxj2UFVvIw=="],
"@radix-ui/react-use-controllable-state": ["@radix-ui/react-use-controllable-state@1.2.3", "", { "dependencies": { "@radix-ui/react-use-effect-event": "0.0.3", "@radix-ui/react-use-layout-effect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-PLzC90MS+ReootmjC597dvopoelpZ8Q61HJkDXZSExitIq7PL55vHNnesAHwguHK0aPfBnpdNzQtv1uliaqQrA=="],
"@radix-ui/react-use-effect-event": ["@radix-ui/react-use-effect-event@0.0.3", "", { "dependencies": { "@radix-ui/react-use-layout-effect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-6c8ZqvPTWILEKnyVkP53EGRCcpnJiKTC21sS/6R1GF5xKyHJJWQEPfkqlcgUkdRQivd6tb23abUwe4ngWmY0JA=="],
"@radix-ui/react-use-escape-keydown": ["@radix-ui/react-use-escape-keydown@1.1.2", "", { "dependencies": { "@radix-ui/react-use-callback-ref": "1.1.2" }, "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-2uVLvLjgO7NZCWw01/FdqRwmA42J0BcjPMUCA+koFEOAb+zjqIP7SiFz/7zWPrKnVmSqr76Omq2ALyCuX4dhLw=="],
"@radix-ui/react-use-layout-effect": ["@radix-ui/react-use-layout-effect@1.1.2", "", { "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-jrBWOxZITuGcnjRCM2t2U5ZPkCLxD+Ym6DjfssS5haTj2iiak/DOb64JeN6OdLfLgptb6/e2kKR+ZuTrGoZTPA=="],
"@radix-ui/react-use-previous": ["@radix-ui/react-use-previous@1.1.2", "", { "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-IGBQPtRFdhN6MQ8dbegVmBq1LVZluya3F1jWY+puIcQC3MHctRwTDSBWCkL/3ZcnMJLTMJ++Z+ktmvg0F89iCw=="],
"@radix-ui/react-use-rect": ["@radix-ui/react-use-rect@1.1.2", "", { "dependencies": { "@radix-ui/rect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-d8a+bBY/FxikNPlgJJoaBHZX+zKVbWHYJGTLnLvveQgFSTntkGdEKv3JDtHrMS0DNYpllz2nRsTLGLKYttbpmw=="],
"@radix-ui/react-use-size": ["@radix-ui/react-use-size@1.1.2", "", { "dependencies": { "@radix-ui/react-use-layout-effect": "1.1.2" }, "peerDependencies": { "@types/react": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-giWQp+4mxjBPt4KZ0MmyuykFNWfbDxKt4x+fPkRYmgRFJSbCZFzUglvMb/Kjn38tm10YP4ufiQZDx3zna4LU6w=="],
"@radix-ui/react-visually-hidden": ["@radix-ui/react-visually-hidden@1.2.6", "", { "dependencies": { "@radix-ui/react-primitive": "2.1.6" }, "peerDependencies": { "@types/react": "*", "@types/react-dom": "*", "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" } }, "sha512-jCE0WljWifTI4niIMCll06kGpsJTAPiZVU9H4WR1N6qW7At9ystHbN7dDB+we2xH535roFHj7qKS+RGj0FMDWQ=="],
"@radix-ui/rect": ["@radix-ui/rect@1.1.2", "", {}, "sha512-xnXE7wG13PI+cxieVssYXlQJuYVRhH9NBoxt3KNwzghDIA69GMm7d4wXRouHIYjE+KvS6U/MsMO73NdS2MH9ZA=="],
"@reduxjs/toolkit": ["@reduxjs/toolkit@2.11.2", "", { "dependencies": { "@standard-schema/spec": "^1.0.0", "@standard-schema/utils": "^0.3.0", "immer": "^11.0.0", "redux": "^5.0.1", "redux-thunk": "^3.1.0", "reselect": "^5.1.0" }, "peerDependencies": { "react": "^16.9.0 || ^17.0.0 || ^18 || ^19", "react-redux": "^7.2.1 || ^8.1.3 || ^9.0.0" } }, ""],
"@shikijs/core": ["@shikijs/core@4.2.0", "", { "dependencies": { "@shikijs/primitive": "4.2.0", "@shikijs/types": "4.2.0", "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4", "hast-util-to-html": "^9.0.5" } }, "sha512-Hc87Ab1Ld/vEbZRCbwx344I5v+4RU8CVToUTRkqXL1+TjbuOp9U5Xa0M23V4GEWHxVn+yO5otb+HkQVm3ptWQQ=="],
"@shikijs/engine-javascript": ["@shikijs/engine-javascript@4.2.0", "", { "dependencies": { "@shikijs/types": "4.2.0", "@shikijs/vscode-textmate": "^10.0.2", "oniguruma-to-es": "^4.3.6" } }, "sha512-fjETeq1k5ffyXqRgS6+3hpvqseLalp1kjNfRbXpUgWR8FpZ1CmQfiNHovc5lncYjt/Vg5JK/WJEmLahjwMa0og=="],
"@shikijs/engine-oniguruma": ["@shikijs/engine-oniguruma@4.2.0", "", { "dependencies": { "@shikijs/types": "4.2.0", "@shikijs/vscode-textmate": "^10.0.2" } }, "sha512-hTorK1dffPkpbMUk6Z+828PgRo7d07HbnizoP0hNPFjhxMHctj0Px/qoHeGMYafc6ju+u9iMldN4JbVzNQM++g=="],
"@shikijs/langs": ["@shikijs/langs@4.2.0", "", { "dependencies": { "@shikijs/types": "4.2.0" } }, "sha512-bwrVRlJ0wUhZxAbVdvBbv2TTC9yLsh4C/IO5Ofz0T8MQntgDvyVnkbjw9vi50r1kx7RCIJdnJnjZAwmAsXFLZQ=="],
"@shikijs/primitive": ["@shikijs/primitive@4.2.0", "", { "dependencies": { "@shikijs/types": "4.2.0", "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, "sha512-NOq+DtUkVBJtZMVXL5A0vI0Xk8nvDYaXetFHSJFlOqjDZIVhIPRYFdGkSoElDqNuegikcc3A76SNUa8dTqtAYA=="],
"@shikijs/themes": ["@shikijs/themes@4.2.0", "", { "dependencies": { "@shikijs/types": "4.2.0" } }, "sha512-RX8IHYeLv8Cu2W6ruc3RxUqWn0IYCqSrMBzi/uRGAmfyDNOnNO5BF/Px7o97n4XTpmFTo5GbRaazuOWj+2ak2w=="],
"@shikijs/twoslash": ["@shikijs/twoslash@4.0.2", "", { "dependencies": { "@shikijs/core": "4.0.2", "@shikijs/types": "4.0.2", "twoslash": "^0.3.6" }, "peerDependencies": { "typescript": ">=5.5.0" } }, ""],
"@shikijs/types": ["@shikijs/types@4.2.0", "", { "dependencies": { "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, "sha512-VT/MKtlpOhEPZloSH3Pb9WCZEBDoQVMa9jedp5UAwmJOar1DVc9DRODAxmYPW9M93IK4ryuqRejFfmlvlVDemw=="],
"@shikijs/vscode-textmate": ["@shikijs/vscode-textmate@10.0.2", "", {}, ""],
"@stablelib/base64": ["@stablelib/base64@1.0.1", "", {}, "sha512-1bnPQqSxSuc3Ii6MhBysoWCg58j97aUjuCSZrGSmDxNqtytIi0k8utUenAwTZN4V5mXXYGsVUI9zeBqy+jBOSQ=="],
"@standard-schema/spec": ["@standard-schema/spec@1.1.0", "", {}, ""],
"@standard-schema/utils": ["@standard-schema/utils@0.3.0", "", {}, ""],
"@swc/helpers": ["@swc/helpers@0.5.15", "", { "dependencies": { "tslib": "^2.8.0" } }, ""],
"@tailwindcss/node": ["@tailwindcss/node@4.2.2", "", { "dependencies": { "@jridgewell/remapping": "^2.3.5", "enhanced-resolve": "^5.19.0", "jiti": "^2.6.1", "lightningcss": "1.32.0", "magic-string": "^0.30.21", "source-map-js": "^1.2.1", "tailwindcss": "4.2.2" } }, ""],
"@tailwindcss/oxide": ["@tailwindcss/oxide@4.2.2", "", { "optionalDependencies": { "@tailwindcss/oxide-android-arm64": "4.2.2", "@tailwindcss/oxide-darwin-arm64": "4.2.2", "@tailwindcss/oxide-darwin-x64": "4.2.2", "@tailwindcss/oxide-freebsd-x64": "4.2.2", "@tailwindcss/oxide-linux-arm-gnueabihf": "4.2.2", "@tailwindcss/oxide-linux-arm64-gnu": "4.2.2", "@tailwindcss/oxide-linux-arm64-musl": "4.2.2", "@tailwindcss/oxide-linux-x64-gnu": "4.2.2", "@tailwindcss/oxide-linux-x64-musl": "4.2.2", "@tailwindcss/oxide-wasm32-wasi": "4.2.2", "@tailwindcss/oxide-win32-arm64-msvc": "4.2.2", "@tailwindcss/oxide-win32-x64-msvc": "4.2.2" } }, ""],
"@tailwindcss/oxide-android-arm64": ["@tailwindcss/oxide-android-arm64@4.2.2", "", { "os": "android", "cpu": "arm64" }, "sha512-dXGR1n+P3B6748jZO/SvHZq7qBOqqzQ+yFrXpoOWWALWndF9MoSKAT3Q0fYgAzYzGhxNYOoysRvYlpixRBBoDg=="],
"@tailwindcss/oxide-darwin-arm64": ["@tailwindcss/oxide-darwin-arm64@4.2.2", "", { "os": "darwin", "cpu": "arm64" }, "sha512-iq9Qjr6knfMpZHj55/37ouZeykwbDqF21gPFtfnhCCKGDcPI/21FKC9XdMO/XyBM7qKORx6UIhGgg6jLl7BZlg=="],
"@tailwindcss/oxide-darwin-x64": ["@tailwindcss/oxide-darwin-x64@4.2.2", "", { "os": "darwin", "cpu": "x64" }, "sha512-BlR+2c3nzc8f2G639LpL89YY4bdcIdUmiOOkv2GQv4/4M0vJlpXEa0JXNHhCHU7VWOKWT/CjqHdTP8aUuDJkuw=="],
"@tailwindcss/oxide-freebsd-x64": ["@tailwindcss/oxide-freebsd-x64@4.2.2", "", { "os": "freebsd", "cpu": "x64" }, "sha512-YUqUgrGMSu2CDO82hzlQ5qSb5xmx3RUrke/QgnoEx7KvmRJHQuZHZmZTLSuuHwFf0DJPybFMXMYf+WJdxHy/nQ=="],
"@tailwindcss/oxide-linux-arm-gnueabihf": ["@tailwindcss/oxide-linux-arm-gnueabihf@4.2.2", "", { "os": "linux", "cpu": "arm" }, "sha512-FPdhvsW6g06T9BWT0qTwiVZYE2WIFo2dY5aCSpjG/S/u1tby+wXoslXS0kl3/KXnULlLr1E3NPRRw0g7t2kgaQ=="],
"@tailwindcss/oxide-linux-arm64-gnu": ["@tailwindcss/oxide-linux-arm64-gnu@4.2.2", "", { "os": "linux", "cpu": "arm64" }, ""],
"@tailwindcss/oxide-linux-arm64-musl": ["@tailwindcss/oxide-linux-arm64-musl@4.2.2", "", { "os": "linux", "cpu": "arm64" }, ""],
"@tailwindcss/oxide-linux-x64-gnu": ["@tailwindcss/oxide-linux-x64-gnu@4.2.2", "", { "os": "linux", "cpu": "x64" }, "sha512-rTAGAkDgqbXHNp/xW0iugLVmX62wOp2PoE39BTCGKjv3Iocf6AFbRP/wZT/kuCxC9QBh9Pu8XPkv/zCZB2mcMg=="],
"@tailwindcss/oxide-linux-x64-musl": ["@tailwindcss/oxide-linux-x64-musl@4.2.2", "", { "os": "linux", "cpu": "x64" }, "sha512-XW3t3qwbIwiSyRCggeO2zxe3KWaEbM0/kW9e8+0XpBgyKU4ATYzcVSMKteZJ1iukJ3HgHBjbg9P5YPRCVUxlnQ=="],
"@tailwindcss/oxide-wasm32-wasi": ["@tailwindcss/oxide-wasm32-wasi@4.2.2", "", { "cpu": "none" }, "sha512-eKSztKsmEsn1O5lJ4ZAfyn41NfG7vzCg496YiGtMDV86jz1q/irhms5O0VrY6ZwTUkFy/EKG3RfWgxSI3VbZ8Q=="],
"@tailwindcss/oxide-win32-arm64-msvc": ["@tailwindcss/oxide-win32-arm64-msvc@4.2.2", "", { "os": "win32", "cpu": "arm64" }, "sha512-qPmaQM4iKu5mxpsrWZMOZRgZv1tOZpUm+zdhhQP0VhJfyGGO3aUKdbh3gDZc/dPLQwW4eSqWGrrcWNBZWUWaXQ=="],
"@tailwindcss/oxide-win32-x64-msvc": ["@tailwindcss/oxide-win32-x64-msvc@4.2.2", "", { "os": "win32", "cpu": "x64" }, "sha512-1T/37VvI7WyH66b+vqHj/cLwnCxt7Qt3WFu5Q8hk65aOvlwAhs7rAp1VkulBJw/N4tMirXjVnylTR72uI0HGcA=="],
"@tailwindcss/postcss": ["@tailwindcss/postcss@4.2.2", "", { "dependencies": { "@alloc/quick-lru": "^5.2.0", "@tailwindcss/node": "4.2.2", "@tailwindcss/oxide": "4.2.2", "postcss": "^8.5.6", "tailwindcss": "4.2.2" } }, ""],
"@ts-morph/common": ["@ts-morph/common@0.28.1", "", { "dependencies": { "minimatch": "^10.0.1", "path-browserify": "^1.0.1", "tinyglobby": "^0.2.14" } }, ""],
"@turf/boolean-point-in-polygon": ["@turf/boolean-point-in-polygon@7.3.4", "", { "dependencies": { "@turf/helpers": "7.3.4", "@turf/invariant": "7.3.4", "@types/geojson": "^7946.0.10", "point-in-polygon-hao": "^1.1.0", "tslib": "^2.8.1" } }, ""],
"@turf/helpers": ["@turf/helpers@7.3.4", "", { "dependencies": { "@types/geojson": "^7946.0.10", "tslib": "^2.8.1" } }, ""],
"@turf/invariant": ["@turf/invariant@7.3.4", "", { "dependencies": { "@turf/helpers": "7.3.4", "@types/geojson": "^7946.0.10", "tslib": "^2.8.1" } }, ""],
"@types/d3-array": ["@types/d3-array@3.2.2", "", {}, ""],
"@types/d3-color": ["@types/d3-color@3.1.3", "", {}, ""],
"@types/d3-ease": ["@types/d3-ease@3.0.2", "", {}, ""],
"@types/d3-interpolate": ["@types/d3-interpolate@3.0.4", "", { "dependencies": { "@types/d3-color": "*" } }, ""],
"@types/d3-path": ["@types/d3-path@3.1.1", "", {}, ""],
"@types/d3-scale": ["@types/d3-scale@4.0.9", "", { "dependencies": { "@types/d3-time": "*" } }, ""],
"@types/d3-shape": ["@types/d3-shape@3.1.8", "", { "dependencies": { "@types/d3-path": "*" } }, ""],
"@types/d3-time": ["@types/d3-time@3.0.4", "", {}, ""],
"@types/d3-timer": ["@types/d3-timer@3.0.2", "", {}, ""],
"@types/debug": ["@types/debug@4.1.13", "", { "dependencies": { "@types/ms": "*" } }, ""],
"@types/estree": ["@types/estree@1.0.8", "", {}, ""],
"@types/estree-jsx": ["@types/estree-jsx@1.0.5", "", { "dependencies": { "@types/estree": "*" } }, ""],
"@types/geojson": ["@types/geojson@7946.0.16", "", {}, ""],
"@types/hast": ["@types/hast@3.0.4", "", { "dependencies": { "@types/unist": "*" } }, ""],
"@types/mdast": ["@types/mdast@4.0.4", "", { "dependencies": { "@types/unist": "*" } }, ""],
"@types/mdx": ["@types/mdx@2.0.13", "", {}, ""],
"@types/ms": ["@types/ms@2.1.0", "", {}, ""],
"@types/node": ["@types/node@25.5.2", "", { "dependencies": { "undici-types": "~7.18.0" } }, ""],
"@types/react": ["@types/react@19.2.14", "", { "dependencies": { "csstype": "^3.2.2" } }, ""],
"@types/react-dom": ["@types/react-dom@19.2.3", "", { "peerDependencies": { "@types/react": "^19.2.0" } }, ""],
"@types/unist": ["@types/unist@3.0.3", "", {}, ""],
"@types/use-sync-external-store": ["@types/use-sync-external-store@0.0.6", "", {}, ""],
"@typescript/vfs": ["@typescript/vfs@1.6.4", "", { "dependencies": { "debug": "^4.4.3" }, "peerDependencies": { "typescript": "*" } }, ""],
"@ungap/structured-clone": ["@ungap/structured-clone@1.3.0", "", {}, ""],
"@vercel/oidc": ["@vercel/oidc@3.1.0", "", {}, ""],
"acorn": ["acorn@8.16.0", "", { "bin": "bin/acorn" }, ""],
"acorn-jsx": ["acorn-jsx@5.3.2", "", { "peerDependencies": { "acorn": "^6.0.0 || ^7.0.0 || ^8.0.0" } }, ""],
"ai": ["ai@6.0.149", "", { "dependencies": { "@ai-sdk/gateway": "3.0.91", "@ai-sdk/provider": "3.0.8", "@ai-sdk/provider-utils": "4.0.23", "@opentelemetry/api": "1.9.0" }, "peerDependencies": { "zod": "^3.25.76 || ^4.1.8" } }, ""],
"argparse": ["argparse@2.0.1", "", {}, ""],
"aria-hidden": ["aria-hidden@1.2.6", "", { "dependencies": { "tslib": "^2.0.0" } }, ""],
"astring": ["astring@1.9.0", "", { "bin": "bin/astring" }, ""],
"bail": ["bail@2.0.2", "", {}, ""],
"balanced-match": ["balanced-match@4.0.4", "", {}, ""],
"baseline-browser-mapping": ["baseline-browser-mapping@2.10.16", "", { "bin": "dist/cli.cjs" }, ""],
"brace-expansion": ["brace-expansion@5.0.6", "", { "dependencies": { "balanced-match": "^4.0.2" } }, "sha512-kLpxurY4Z4r9sgMsyG0Z9uzsBlgiU/EFKhj/h91/8yHu0edo7XuixOIH3VcJ8kkxs6/jPzoI6U9Vj3WqbMQ94g=="],
"caniuse-lite": ["caniuse-lite@1.0.30001786", "", {}, ""],
"ccount": ["ccount@2.0.1", "", {}, ""],
"character-entities": ["character-entities@2.0.2", "", {}, ""],
"character-entities-html4": ["character-entities-html4@2.1.0", "", {}, ""],
"character-entities-legacy": ["character-entities-legacy@3.0.0", "", {}, ""],
"character-reference-invalid": ["character-reference-invalid@2.0.1", "", {}, ""],
"chokidar": ["chokidar@5.0.0", "", { "dependencies": { "readdirp": "^5.0.0" } }, ""],
"class-variance-authority": ["class-variance-authority@0.7.1", "", { "dependencies": { "clsx": "^2.1.1" } }, "sha512-Ka+9Trutv7G8M6WT6SeiRWz792K5qEqIGEGzXKhAE6xOWAY6pPH8U+9IY3oCMv6kqTmLsv7Xh/2w2RigkePMsg=="],
"client-only": ["client-only@0.0.1", "", {}, ""],
"clsx": ["clsx@2.1.1", "", {}, "sha512-eYm0QWBtUrBWZWG0d386OGAw16Z995PiOVo2B7bjWSbHedGl5e0ZWaq65kOGgUSNesEIDkB9ISbTg/JK9dhCZA=="],
"code-block-writer": ["code-block-writer@13.0.3", "", {}, ""],
"collapse-white-space": ["collapse-white-space@2.1.0", "", {}, ""],
"comma-separated-tokens": ["comma-separated-tokens@2.0.3", "", {}, ""],
"compute-scroll-into-view": ["compute-scroll-into-view@3.1.1", "", {}, ""],
"csstype": ["csstype@3.2.3", "", {}, ""],
"d3-array": ["d3-array@3.2.4", "", { "dependencies": { "internmap": "1 - 2" } }, ""],
"d3-color": ["d3-color@3.1.0", "", {}, ""],
"d3-ease": ["d3-ease@3.0.1", "", {}, ""],
"d3-format": ["d3-format@3.1.2", "", {}, ""],
"d3-interpolate": ["d3-interpolate@3.0.1", "", { "dependencies": { "d3-color": "1 - 3" } }, ""],
"d3-path": ["d3-path@3.1.0", "", {}, ""],
"d3-scale": ["d3-scale@4.0.2", "", { "dependencies": { "d3-array": "2.10.0 - 3", "d3-format": "1 - 3", "d3-interpolate": "1.2.0 - 3", "d3-time": "2.1.1 - 3", "d3-time-format": "2 - 4" } }, ""],
"d3-shape": ["d3-shape@3.2.0", "", { "dependencies": { "d3-path": "^3.1.0" } }, ""],
"d3-time": ["d3-time@3.1.0", "", { "dependencies": { "d3-array": "2 - 3" } }, ""],
"d3-time-format": ["d3-time-format@4.1.0", "", { "dependencies": { "d3-time": "1 - 3" } }, ""],
"d3-timer": ["d3-timer@3.0.1", "", {}, ""],
"debug": ["debug@4.4.3", "", { "dependencies": { "ms": "^2.1.3" } }, ""],
"decimal.js-light": ["decimal.js-light@2.5.1", "", {}, ""],
"decode-named-character-reference": ["decode-named-character-reference@1.3.0", "", { "dependencies": { "character-entities": "^2.0.0" } }, ""],
"dequal": ["dequal@2.0.3", "", {}, ""],
"detect-libc": ["detect-libc@2.1.2", "", {}, ""],
"detect-node-es": ["detect-node-es@1.1.0", "", {}, ""],
"devlop": ["devlop@1.1.0", "", { "dependencies": { "dequal": "^2.0.0" } }, ""],
"dotted-map": ["dotted-map@3.1.0", "", { "dependencies": { "@turf/boolean-point-in-polygon": "^7.3.4", "proj4": "^2.20.2" } }, ""],
"enhanced-resolve": ["enhanced-resolve@5.20.1", "", { "dependencies": { "graceful-fs": "^4.2.4", "tapable": "^2.3.0" } }, ""],
"entities": ["entities@6.0.1", "", {}, ""],
"es-toolkit": ["es-toolkit@1.45.1", "", {}, ""],
"esast-util-from-estree": ["esast-util-from-estree@2.0.0", "", { "dependencies": { "@types/estree-jsx": "^1.0.0", "devlop": "^1.0.0", "estree-util-visit": "^2.0.0", "unist-util-position-from-estree": "^2.0.0" } }, ""],
"esast-util-from-js": ["esast-util-from-js@2.0.1", "", { "dependencies": { "@types/estree-jsx": "^1.0.0", "acorn": "^8.0.0", "esast-util-from-estree": "^2.0.0", "vfile-message": "^4.0.0" } }, ""],
"esbuild": ["esbuild@0.28.1", "", { "optionalDependencies": { "@esbuild/aix-ppc64": "0.28.1", "@esbuild/android-arm": "0.28.1", "@esbuild/android-arm64": "0.28.1", "@esbuild/android-x64": "0.28.1", "@esbuild/darwin-arm64": "0.28.1", "@esbuild/darwin-x64": "0.28.1", "@esbuild/freebsd-arm64": "0.28.1", "@esbuild/freebsd-x64": "0.28.1", "@esbuild/linux-arm": "0.28.1", "@esbuild/linux-arm64": "0.28.1", "@esbuild/linux-ia32": "0.28.1", "@esbuild/linux-loong64": "0.28.1", "@esbuild/linux-mips64el": "0.28.1", "@esbuild/linux-ppc64": "0.28.1", "@esbuild/linux-riscv64": "0.28.1", "@esbuild/linux-s390x": "0.28.1", "@esbuild/linux-x64": "0.28.1", "@esbuild/netbsd-arm64": "0.28.1", "@esbuild/netbsd-x64": "0.28.1", "@esbuild/openbsd-arm64": "0.28.1", "@esbuild/openbsd-x64": "0.28.1", "@esbuild/openharmony-arm64": "0.28.1", "@esbuild/sunos-x64": "0.28.1", "@esbuild/win32-arm64": "0.28.1", "@esbuild/win32-ia32": "0.28.1", "@esbuild/win32-x64": "0.28.1" }, "bin": "bin/esbuild" }, "sha512-HrJrvZv5ayxBzPfwphOoNzkzOIIlifzk0KJrGK2c8R4+LKpMtpYLQeUdjnwjWv/LZlkH2laZk+4w78pi99D4Vw=="],
"escape-string-regexp": ["escape-string-regexp@5.0.0", "", {}, ""],
"estree-util-attach-comments": ["estree-util-attach-comments@3.0.0", "", { "dependencies": { "@types/estree": "^1.0.0" } }, ""],
"estree-util-build-jsx": ["estree-util-build-jsx@3.0.1", "", { "dependencies": { "@types/estree-jsx": "^1.0.0", "devlop": "^1.0.0", "estree-util-is-identifier-name": "^3.0.0", "estree-walker": "^3.0.0" } }, ""],
"estree-util-is-identifier-name": ["estree-util-is-identifier-name@3.0.0", "", {}, ""],
"estree-util-scope": ["estree-util-scope@1.0.0", "", { "dependencies": { "@types/estree": "^1.0.0", "devlop": "^1.0.0" } }, ""],
"estree-util-to-js": ["estree-util-to-js@2.0.0", "", { "dependencies": { "@types/estree-jsx": "^1.0.0", "astring": "^1.8.0", "source-map": "^0.7.0" } }, ""],
"estree-util-value-to-estree": ["estree-util-value-to-estree@3.5.0", "", { "dependencies": { "@types/estree": "^1.0.0" } }, ""],
"estree-util-visit": ["estree-util-visit@2.0.0", "", { "dependencies": { "@types/estree-jsx": "^1.0.0", "@types/unist": "^3.0.0" } }, ""],
"estree-walker": ["estree-walker@3.0.3", "", { "dependencies": { "@types/estree": "^1.0.0" } }, ""],
"eventemitter3": ["eventemitter3@5.0.4", "", {}, ""],
"eventsource-parser": ["eventsource-parser@3.0.6", "", {}, ""],
"extend": ["extend@3.0.2", "", {}, ""],
"fast-sha256": ["fast-sha256@1.3.0", "", {}, "sha512-n11RGP/lrWEFI/bWdygLxhI+pVeo1ZYIVwvvPkW7azl/rOy+F3HYRZ2K5zeE9mmkhQppyv9sQFx0JM9UabnpPQ=="],
"fdir": ["fdir@6.5.0", "", { "peerDependencies": { "picomatch": "^3 || ^4" } }, ""],
"framer-motion": ["framer-motion@12.40.0", "", { "dependencies": { "motion-dom": "^12.40.0", "motion-utils": "^12.39.0", "tslib": "^2.4.0" }, "peerDependencies": { "@emotion/is-prop-valid": "*", "react": "^18.0.0 || ^19.0.0", "react-dom": "^18.0.0 || ^19.0.0" }, "optionalPeers": ["@emotion/is-prop-valid"] }, "sha512-uaBd3qC1v3KQqBEjwTUd183K6PbS+j0yR9w9VmEOLWA/tnUcSn8Xa3uck7t4dgpDoUss8xQTcj8W2L07lrnLFg=="],
"fumadocs-core": ["fumadocs-core@16.10.3", "", { "dependencies": { "@orama/orama": "^3.1.18", "estree-util-value-to-estree": "^3.5.0", "github-slugger": "^2.0.0", "hast-util-to-estree": "^3.1.3", "hast-util-to-jsx-runtime": "^2.3.6", "js-yaml": "^4.2.0", "mdast-util-mdx": "^3.0.0", "mdast-util-to-markdown": "^2.1.2", "remark": "^15.0.1", "remark-gfm": "^4.0.1", "remark-rehype": "^11.1.2", "scroll-into-view-if-needed": "^3.1.0", "shiki": "^4.2.0", "tinyglobby": "^0.2.17", "unified": "^11.0.5", "unist-util-visit": "^5.1.0", "vfile": "^6.0.3" }, "peerDependencies": { "@mdx-js/mdx": "*", "@mixedbread/sdk": "0.x.x", "@orama/core": "1.x.x", "@oramacloud/client": "2.x.x", "@tanstack/react-router": "1.x.x", "@types/estree-jsx": "*", "@types/hast": "*", "@types/mdast": "*", "@types/react": "*", "algoliasearch": "5.x.x", "flexsearch": "*", "lucide-react": "*", "next": "16.x.x", "react": "^19.2.0", "react-dom": "^19.2.0", "react-router": "7.x.x", "waku": "*", "zod": "4.x.x" }, "optionalPeers": ["@mixedbread/sdk", "@orama/core", "@oramacloud/client", "@tanstack/react-router", "algoliasearch", "flexsearch", "react-router", "waku"] }, "sha512-xXhqz/fqbN7pLlshJb/B5L+vzMJOmWxoPj7+KMRTa/4A669hKeeCBPpRAiooMqjblWqIRSxLiO02/ds8ltvUPQ=="],
"fumadocs-mdx": ["fumadocs-mdx@15.0.12", "", { "dependencies": { "@mdx-js/mdx": "^3.1.1", "@standard-schema/spec": "^1.1.0", "chokidar": "^5.0.0", "esbuild": "^0.28.0", "estree-util-value-to-estree": "^3.5.0", "js-yaml": "^4.2.0", "mdast-util-mdx": "^3.0.0", "picocolors": "^1.1.1", "picomatch": "^4.0.4", "tinyexec": "^1.2.4", "tinyglobby": "^0.2.17", "unified": "^11.0.5", "unist-util-remove-position": "^5.0.0", "unist-util-visit": "^5.1.0", "vfile": "^6.0.3", "zod": "^4.4.3" }, "peerDependencies": { "@types/mdast": "*", "@types/mdx": "*", "@types/react": "*", "fumadocs-core": "^16.7.0", "mdast-util-directive": "*", "next": "^15.3.0 || ^16.0.0", "react": "^19.2.0", "rolldown": "*", "vite": "7.x.x || 8.x.x" }, "optionalPeers": ["mdast-util-directive", "rolldown", "vite"], "bin": "bin.js" }, "sha512-R4WenrNQxSKi+QU46Q1cscVWi+S90dj3As4jdN+vgChO2o0TVOj+FFIe3onWM7mglhPj53NxZp/upP+t/ryekQ=="],
"fumadocs-twoslash": ["fumadocs-twoslash@3.1.15", "", { "dependencies": { "@radix-ui/react-popover": "^1.1.15", "@shikijs/twoslash": "^4.0.2", "mdast-util-from-markdown": "^2.0.3", "mdast-util-gfm": "^3.1.0", "mdast-util-to-hast": "^13.2.1", "shiki": "^4.0.2", "tailwind-merge": "^3.5.0", "twoslash": "^0.3.6" }, "peerDependencies": { "@types/react": "*", "fumadocs-ui": "^15.0.0 || ^16.0.0", "react": "18.x.x || 19.x.x" } }, ""],
"fumadocs-typescript": ["fumadocs-typescript@4.0.14", "", { "dependencies": { "estree-util-value-to-estree": "^3.5.0", "hast-util-to-estree": "^3.1.3", "hast-util-to-jsx-runtime": "^2.3.6", "remark": "^15.0.1", "remark-rehype": "^11.1.2", "tinyglobby": "^0.2.15", "ts-morph": "^27.0.2", "unist-util-visit": "^5.0.0" }, "peerDependencies": { "@types/react": "*", "fumadocs-core": "^15.7.0 || ^16.0.0", "fumadocs-ui": "^15.7.0 || ^16.0.0", "typescript": "*" } }, ""],
"fumadocs-ui": ["fumadocs-ui@16.10.3", "", { "dependencies": { "@fuma-translate/react": "^1.0.2", "@fumadocs/tailwind": "0.0.5", "@radix-ui/react-accordion": "^1.2.13", "@radix-ui/react-collapsible": "^1.1.13", "@radix-ui/react-dialog": "^1.1.16", "@radix-ui/react-direction": "^1.1.2", "@radix-ui/react-navigation-menu": "^1.2.15", "@radix-ui/react-popover": "^1.1.16", "@radix-ui/react-presence": "^1.1.6", "@radix-ui/react-scroll-area": "^1.2.11", "@radix-ui/react-slot": "^1.2.5", "@radix-ui/react-tabs": "^1.1.14", "class-variance-authority": "^0.7.1", "lucide-react": "^1.17.0", "motion": "^12.40.0", "next-themes": "^0.4.6", "react-remove-scroll": "^2.7.2", "rehype-raw": "^7.0.0", "scroll-into-view-if-needed": "^3.1.0", "shiki": "^4.2.0", "tailwind-merge": "^3.6.0", "unist-util-visit": "^5.1.0" }, "peerDependencies": { "@takumi-rs/image-response": "*", "@types/mdx": "*", "@types/react": "*", "fumadocs-core": "16.10.3", "next": "16.x.x", "react": "^19.2.0", "react-dom": "^19.2.0" }, "optionalPeers": ["@takumi-rs/image-response"] }, "sha512-0aSLdQ73EWoCmYcQYr2uNHlSB/s2fD+NMugtdZF3vC4lqs0MfyOtwnZPyYAskUnXNs6HECly/Hu6oY5JqmlkHg=="],
"get-nonce": ["get-nonce@1.0.1", "", {}, ""],
"github-slugger": ["github-slugger@2.0.0", "", {}, ""],
"graceful-fs": ["graceful-fs@4.2.11", "", {}, ""],
"hast-util-from-parse5": ["hast-util-from-parse5@8.0.3", "", { "dependencies": { "@types/hast": "^3.0.0", "@types/unist": "^3.0.0", "devlop": "^1.0.0", "hastscript": "^9.0.0", "property-information": "^7.0.0", "vfile": "^6.0.0", "vfile-location": "^5.0.0", "web-namespaces": "^2.0.0" } }, ""],
"hast-util-parse-selector": ["hast-util-parse-selector@4.0.0", "", { "dependencies": { "@types/hast": "^3.0.0" } }, ""],
"hast-util-raw": ["hast-util-raw@9.1.0", "", { "dependencies": { "@types/hast": "^3.0.0", "@types/unist": "^3.0.0", "@ungap/structured-clone": "^1.0.0", "hast-util-from-parse5": "^8.0.0", "hast-util-to-parse5": "^8.0.0", "html-void-elements": "^3.0.0", "mdast-util-to-hast": "^13.0.0", "parse5": "^7.0.0", "unist-util-position": "^5.0.0", "unist-util-visit": "^5.0.0", "vfile": "^6.0.0", "web-namespaces": "^2.0.0", "zwitch": "^2.0.0" } }, ""],
"hast-util-to-estree": ["hast-util-to-estree@3.1.3", "", { "dependencies": { "@types/estree": "^1.0.0", "@types/estree-jsx": "^1.0.0", "@types/hast": "^3.0.0", "comma-separated-tokens": "^2.0.0", "devlop": "^1.0.0", "estree-util-attach-comments": "^3.0.0", "estree-util-is-identifier-name": "^3.0.0", "hast-util-whitespace": "^3.0.0", "mdast-util-mdx-expression": "^2.0.0", "mdast-util-mdx-jsx": "^3.0.0", "mdast-util-mdxjs-esm": "^2.0.0", "property-information": "^7.0.0", "space-separated-tokens": "^2.0.0", "style-to-js": "^1.0.0", "unist-util-position": "^5.0.0", "zwitch": "^2.0.0" } }, ""],
"hast-util-to-html": ["hast-util-to-html@9.0.5", "", { "dependencies": { "@types/hast": "^3.0.0", "@types/unist": "^3.0.0", "ccount": "^2.0.0", "comma-separated-tokens": "^2.0.0", "hast-util-whitespace": "^3.0.0", "html-void-elements": "^3.0.0", "mdast-util-to-hast": "^13.0.0", "property-information": "^7.0.0", "space-separated-tokens": "^2.0.0", "stringify-entities": "^4.0.0", "zwitch": "^2.0.4" } }, ""],
"hast-util-to-jsx-runtime": ["hast-util-to-jsx-runtime@2.3.6", "", { "dependencies": { "@types/estree": "^1.0.0", "@types/hast": "^3.0.0", "@types/unist": "^3.0.0", "comma-separated-tokens": "^2.0.0", "devlop": "^1.0.0", "estree-util-is-identifier-name": "^3.0.0", "hast-util-whitespace": "^3.0.0", "mdast-util-mdx-expression": "^2.0.0", "mdast-util-mdx-jsx": "^3.0.0", "mdast-util-mdxjs-esm": "^2.0.0", "property-information": "^7.0.0", "space-separated-tokens": "^2.0.0", "style-to-js": "^1.0.0", "unist-util-position": "^5.0.0", "vfile-message": "^4.0.0" } }, ""],
"hast-util-to-parse5": ["hast-util-to-parse5@8.0.1", "", { "dependencies": { "@types/hast": "^3.0.0", "comma-separated-tokens": "^2.0.0", "devlop": "^1.0.0", "property-information": "^7.0.0", "space-separated-tokens": "^2.0.0", "web-namespaces": "^2.0.0", "zwitch": "^2.0.0" } }, ""],
"hast-util-whitespace": ["hast-util-whitespace@3.0.0", "", { "dependencies": { "@types/hast": "^3.0.0" } }, ""],
"hastscript": ["hastscript@9.0.1", "", { "dependencies": { "@types/hast": "^3.0.0", "comma-separated-tokens": "^2.0.0", "hast-util-parse-selector": "^4.0.0", "property-information": "^7.0.0", "space-separated-tokens": "^2.0.0" } }, ""],
"headroom-ai": ["headroom-ai@file:../sdk/typescript", { "devDependencies": { "@ai-sdk/openai": "^3.0.48", "@ai-sdk/provider": "^1.0.0", "@anthropic-ai/sdk": "^0.104.1", "ai": "^6.0.0", "openai": "^4.80.0", "typescript": "^5.5.0" }, "peerDependencies": { "@ai-sdk/provider": ">=1.0.0", "@anthropic-ai/sdk": ">=0.30.0", "ai": ">=6.0.0", "openai": ">=4.0.0" } }],
"html-void-elements": ["html-void-elements@3.0.0", "", {}, ""],
"immer": ["immer@10.2.0", "", {}, ""],
"inline-style-parser": ["inline-style-parser@0.2.7", "", {}, ""],
"internmap": ["internmap@2.0.3", "", {}, ""],
"is-alphabetical": ["is-alphabetical@2.0.1", "", {}, ""],
"is-alphanumerical": ["is-alphanumerical@2.0.1", "", { "dependencies": { "is-alphabetical": "^2.0.0", "is-decimal": "^2.0.0" } }, ""],
"is-decimal": ["is-decimal@2.0.1", "", {}, ""],
"is-hexadecimal": ["is-hexadecimal@2.0.1", "", {}, ""],
"is-plain-obj": ["is-plain-obj@4.1.0", "", {}, ""],
"jiti": ["jiti@2.6.1", "", { "bin": "lib/jiti-cli.mjs" }, ""],
"js-yaml": ["js-yaml@4.2.0", "", { "dependencies": { "argparse": "^2.0.1" }, "bin": "bin/js-yaml.js" }, "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw=="],
"json-schema": ["json-schema@0.4.0", "", {}, ""],
"json-schema-to-ts": ["json-schema-to-ts@3.1.1", "", { "dependencies": { "@babel/runtime": "^7.18.3", "ts-algebra": "^2.0.0" } }, ""],
"lightningcss": ["lightningcss@1.32.0", "", { "dependencies": { "detect-libc": "^2.0.3" }, "optionalDependencies": { "lightningcss-android-arm64": "1.32.0", "lightningcss-darwin-arm64": "1.32.0", "lightningcss-darwin-x64": "1.32.0", "lightningcss-freebsd-x64": "1.32.0", "lightningcss-linux-arm-gnueabihf": "1.32.0", "lightningcss-linux-arm64-gnu": "1.32.0", "lightningcss-linux-arm64-musl": "1.32.0", "lightningcss-linux-x64-gnu": "1.32.0", "lightningcss-linux-x64-musl": "1.32.0", "lightningcss-win32-arm64-msvc": "1.32.0", "lightningcss-win32-x64-msvc": "1.32.0" } }, ""],
"lightningcss-android-arm64": ["lightningcss-android-arm64@1.32.0", "", { "os": "android", "cpu": "arm64" }, "sha512-YK7/ClTt4kAK0vo6w3X+Pnm0D2cf2vPHbhOXdoNti1Ga0al1P4TBZhwjATvjNwLEBCnKvjJc2jQgHXH0NEwlAg=="],
"lightningcss-darwin-arm64": ["lightningcss-darwin-arm64@1.32.0", "", { "os": "darwin", "cpu": "arm64" }, "sha512-RzeG9Ju5bag2Bv1/lwlVJvBE3q6TtXskdZLLCyfg5pt+HLz9BqlICO7LZM7VHNTTn/5PRhHFBSjk5lc4cmscPQ=="],
"lightningcss-darwin-x64": ["lightningcss-darwin-x64@1.32.0", "", { "os": "darwin", "cpu": "x64" }, "sha512-U+QsBp2m/s2wqpUYT/6wnlagdZbtZdndSmut/NJqlCcMLTWp5muCrID+K5UJ6jqD2BFshejCYXniPDbNh73V8w=="],
"lightningcss-freebsd-x64": ["lightningcss-freebsd-x64@1.32.0", "", { "os": "freebsd", "cpu": "x64" }, "sha512-JCTigedEksZk3tHTTthnMdVfGf61Fky8Ji2E4YjUTEQX14xiy/lTzXnu1vwiZe3bYe0q+SpsSH/CTeDXK6WHig=="],
"lightningcss-linux-arm-gnueabihf": ["lightningcss-linux-arm-gnueabihf@1.32.0", "", { "os": "linux", "cpu": "arm" }, "sha512-x6rnnpRa2GL0zQOkt6rts3YDPzduLpWvwAF6EMhXFVZXD4tPrBkEFqzGowzCsIWsPjqSK+tyNEODUBXeeVHSkw=="],
"lightningcss-linux-arm64-gnu": ["lightningcss-linux-arm64-gnu@1.32.0", "", { "os": "linux", "cpu": "arm64" }, ""],
"lightningcss-linux-arm64-musl": ["lightningcss-linux-arm64-musl@1.32.0", "", { "os": "linux", "cpu": "arm64" }, ""],
"lightningcss-linux-x64-gnu": ["lightningcss-linux-x64-gnu@1.32.0", "", { "os": "linux", "cpu": "x64" }, "sha512-V7Qr52IhZmdKPVr+Vtw8o+WLsQJYCTd8loIfpDaMRWGUZfBOYEJeyJIkqGIDMZPwPx24pUMfwSxxI8phr/MbOA=="],
"lightningcss-linux-x64-musl": ["lightningcss-linux-x64-musl@1.32.0", "", { "os": "linux", "cpu": "x64" }, "sha512-bYcLp+Vb0awsiXg/80uCRezCYHNg1/l3mt0gzHnWV9XP1W5sKa5/TCdGWaR/zBM2PeF/HbsQv/j2URNOiVuxWg=="],
"lightningcss-win32-arm64-msvc": ["lightningcss-win32-arm64-msvc@1.32.0", "", { "os": "win32", "cpu": "arm64" }, "sha512-8SbC8BR40pS6baCM8sbtYDSwEVQd4JlFTOlaD3gWGHfThTcABnNDBda6eTZeqbofalIJhFx0qKzgHJmcPTnGdw=="],
"lightningcss-win32-x64-msvc": ["lightningcss-win32-x64-msvc@1.32.0", "", { "os": "win32", "cpu": "x64" }, "sha512-Amq9B/SoZYdDi1kFrojnoqPLxYhQ4Wo5XiL8EVJrVsB8ARoC1PWW6VGtT0WKCemjy8aC+louJnjS7U18x3b06Q=="],
"longest-streak": ["longest-streak@3.1.0", "", {}, ""],
"lucide-react": ["lucide-react@1.20.0", "", { "peerDependencies": { "react": "^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0" } }, "sha512-jhXLeC/7m0/tjL1nzMdKk6x256zWA6AtbhTVreHOiKPoeX2d6MK4FbyIQPpVq0E6iPWBisyy1TW+pEge/uMEuQ=="],
"magic-string": ["magic-string@0.30.21", "", { "dependencies": { "@jridgewell/sourcemap-codec": "^1.5.5" } }, ""],
"markdown-extensions": ["markdown-extensions@2.0.0", "", {}, ""],
"markdown-table": ["markdown-table@3.0.4", "", {}, ""],
"mdast-util-find-and-replace": ["mdast-util-find-and-replace@3.0.2", "", { "dependencies": { "@types/mdast": "^4.0.0", "escape-string-regexp": "^5.0.0", "unist-util-is": "^6.0.0", "unist-util-visit-parents": "^6.0.0" } }, ""],
"mdast-util-from-markdown": ["mdast-util-from-markdown@2.0.3", "", { "dependencies": { "@types/mdast": "^4.0.0", "@types/unist": "^3.0.0", "decode-named-character-reference": "^1.0.0", "devlop": "^1.0.0", "mdast-util-to-string": "^4.0.0", "micromark": "^4.0.0", "micromark-util-decode-numeric-character-reference": "^2.0.0", "micromark-util-decode-string": "^2.0.0", "micromark-util-normalize-identifier": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0", "unist-util-stringify-position": "^4.0.0" } }, ""],
"mdast-util-gfm": ["mdast-util-gfm@3.1.0", "", { "dependencies": { "mdast-util-from-markdown": "^2.0.0", "mdast-util-gfm-autolink-literal": "^2.0.0", "mdast-util-gfm-footnote": "^2.0.0", "mdast-util-gfm-strikethrough": "^2.0.0", "mdast-util-gfm-table": "^2.0.0", "mdast-util-gfm-task-list-item": "^2.0.0", "mdast-util-to-markdown": "^2.0.0" } }, ""],
"mdast-util-gfm-autolink-literal": ["mdast-util-gfm-autolink-literal@2.0.1", "", { "dependencies": { "@types/mdast": "^4.0.0", "ccount": "^2.0.0", "devlop": "^1.0.0", "mdast-util-find-and-replace": "^3.0.0", "micromark-util-character": "^2.0.0" } }, ""],
"mdast-util-gfm-footnote": ["mdast-util-gfm-footnote@2.1.0", "", { "dependencies": { "@types/mdast": "^4.0.0", "devlop": "^1.1.0", "mdast-util-from-markdown": "^2.0.0", "mdast-util-to-markdown": "^2.0.0", "micromark-util-normalize-identifier": "^2.0.0" } }, ""],
"mdast-util-gfm-strikethrough": ["mdast-util-gfm-strikethrough@2.0.0", "", { "dependencies": { "@types/mdast": "^4.0.0", "mdast-util-from-markdown": "^2.0.0", "mdast-util-to-markdown": "^2.0.0" } }, ""],
"mdast-util-gfm-table": ["mdast-util-gfm-table@2.0.0", "", { "dependencies": { "@types/mdast": "^4.0.0", "devlop": "^1.0.0", "markdown-table": "^3.0.0", "mdast-util-from-markdown": "^2.0.0", "mdast-util-to-markdown": "^2.0.0" } }, ""],
"mdast-util-gfm-task-list-item": ["mdast-util-gfm-task-list-item@2.0.0", "", { "dependencies": { "@types/mdast": "^4.0.0", "devlop": "^1.0.0", "mdast-util-from-markdown": "^2.0.0", "mdast-util-to-markdown": "^2.0.0" } }, ""],
"mdast-util-mdx": ["mdast-util-mdx@3.0.0", "", { "dependencies": { "mdast-util-from-markdown": "^2.0.0", "mdast-util-mdx-expression": "^2.0.0", "mdast-util-mdx-jsx": "^3.0.0", "mdast-util-mdxjs-esm": "^2.0.0", "mdast-util-to-markdown": "^2.0.0" } }, ""],
"mdast-util-mdx-expression": ["mdast-util-mdx-expression@2.0.1", "", { "dependencies": { "@types/estree-jsx": "^1.0.0", "@types/hast": "^3.0.0", "@types/mdast": "^4.0.0", "devlop": "^1.0.0", "mdast-util-from-markdown": "^2.0.0", "mdast-util-to-markdown": "^2.0.0" } }, ""],
"mdast-util-mdx-jsx": ["mdast-util-mdx-jsx@3.2.0", "", { "dependencies": { "@types/estree-jsx": "^1.0.0", "@types/hast": "^3.0.0", "@types/mdast": "^4.0.0", "@types/unist": "^3.0.0", "ccount": "^2.0.0", "devlop": "^1.1.0", "mdast-util-from-markdown": "^2.0.0", "mdast-util-to-markdown": "^2.0.0", "parse-entities": "^4.0.0", "stringify-entities": "^4.0.0", "unist-util-stringify-position": "^4.0.0", "vfile-message": "^4.0.0" } }, ""],
"mdast-util-mdxjs-esm": ["mdast-util-mdxjs-esm@2.0.1", "", { "dependencies": { "@types/estree-jsx": "^1.0.0", "@types/hast": "^3.0.0", "@types/mdast": "^4.0.0", "devlop": "^1.0.0", "mdast-util-from-markdown": "^2.0.0", "mdast-util-to-markdown": "^2.0.0" } }, ""],
"mdast-util-phrasing": ["mdast-util-phrasing@4.1.0", "", { "dependencies": { "@types/mdast": "^4.0.0", "unist-util-is": "^6.0.0" } }, ""],
"mdast-util-to-hast": ["mdast-util-to-hast@13.2.1", "", { "dependencies": { "@types/hast": "^3.0.0", "@types/mdast": "^4.0.0", "@ungap/structured-clone": "^1.0.0", "devlop": "^1.0.0", "micromark-util-sanitize-uri": "^2.0.0", "trim-lines": "^3.0.0", "unist-util-position": "^5.0.0", "unist-util-visit": "^5.0.0", "vfile": "^6.0.0" } }, ""],
"mdast-util-to-markdown": ["mdast-util-to-markdown@2.1.2", "", { "dependencies": { "@types/mdast": "^4.0.0", "@types/unist": "^3.0.0", "longest-streak": "^3.0.0", "mdast-util-phrasing": "^4.0.0", "mdast-util-to-string": "^4.0.0", "micromark-util-classify-character": "^2.0.0", "micromark-util-decode-string": "^2.0.0", "unist-util-visit": "^5.0.0", "zwitch": "^2.0.0" } }, ""],
"mdast-util-to-string": ["mdast-util-to-string@4.0.0", "", { "dependencies": { "@types/mdast": "^4.0.0" } }, ""],
"mgrs": ["mgrs@1.0.0", "", {}, ""],
"micromark": ["micromark@4.0.2", "", { "dependencies": { "@types/debug": "^4.0.0", "debug": "^4.0.0", "decode-named-character-reference": "^1.0.0", "devlop": "^1.0.0", "micromark-core-commonmark": "^2.0.0", "micromark-factory-space": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-chunked": "^2.0.0", "micromark-util-combine-extensions": "^2.0.0", "micromark-util-decode-numeric-character-reference": "^2.0.0", "micromark-util-encode": "^2.0.0", "micromark-util-normalize-identifier": "^2.0.0", "micromark-util-resolve-all": "^2.0.0", "micromark-util-sanitize-uri": "^2.0.0", "micromark-util-subtokenize": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-core-commonmark": ["micromark-core-commonmark@2.0.3", "", { "dependencies": { "decode-named-character-reference": "^1.0.0", "devlop": "^1.0.0", "micromark-factory-destination": "^2.0.0", "micromark-factory-label": "^2.0.0", "micromark-factory-space": "^2.0.0", "micromark-factory-title": "^2.0.0", "micromark-factory-whitespace": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-chunked": "^2.0.0", "micromark-util-classify-character": "^2.0.0", "micromark-util-html-tag-name": "^2.0.0", "micromark-util-normalize-identifier": "^2.0.0", "micromark-util-resolve-all": "^2.0.0", "micromark-util-subtokenize": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-gfm": ["micromark-extension-gfm@3.0.0", "", { "dependencies": { "micromark-extension-gfm-autolink-literal": "^2.0.0", "micromark-extension-gfm-footnote": "^2.0.0", "micromark-extension-gfm-strikethrough": "^2.0.0", "micromark-extension-gfm-table": "^2.0.0", "micromark-extension-gfm-tagfilter": "^2.0.0", "micromark-extension-gfm-task-list-item": "^2.0.0", "micromark-util-combine-extensions": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-gfm-autolink-literal": ["micromark-extension-gfm-autolink-literal@2.1.0", "", { "dependencies": { "micromark-util-character": "^2.0.0", "micromark-util-sanitize-uri": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-gfm-footnote": ["micromark-extension-gfm-footnote@2.1.0", "", { "dependencies": { "devlop": "^1.0.0", "micromark-core-commonmark": "^2.0.0", "micromark-factory-space": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-normalize-identifier": "^2.0.0", "micromark-util-sanitize-uri": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-gfm-strikethrough": ["micromark-extension-gfm-strikethrough@2.1.0", "", { "dependencies": { "devlop": "^1.0.0", "micromark-util-chunked": "^2.0.0", "micromark-util-classify-character": "^2.0.0", "micromark-util-resolve-all": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-gfm-table": ["micromark-extension-gfm-table@2.1.1", "", { "dependencies": { "devlop": "^1.0.0", "micromark-factory-space": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-gfm-tagfilter": ["micromark-extension-gfm-tagfilter@2.0.0", "", { "dependencies": { "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-gfm-task-list-item": ["micromark-extension-gfm-task-list-item@2.1.0", "", { "dependencies": { "devlop": "^1.0.0", "micromark-factory-space": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-mdx-expression": ["micromark-extension-mdx-expression@3.0.1", "", { "dependencies": { "@types/estree": "^1.0.0", "devlop": "^1.0.0", "micromark-factory-mdx-expression": "^2.0.0", "micromark-factory-space": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-events-to-acorn": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-mdx-jsx": ["micromark-extension-mdx-jsx@3.0.2", "", { "dependencies": { "@types/estree": "^1.0.0", "devlop": "^1.0.0", "estree-util-is-identifier-name": "^3.0.0", "micromark-factory-mdx-expression": "^2.0.0", "micromark-factory-space": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-events-to-acorn": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0", "vfile-message": "^4.0.0" } }, ""],
"micromark-extension-mdx-md": ["micromark-extension-mdx-md@2.0.0", "", { "dependencies": { "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-mdxjs": ["micromark-extension-mdxjs@3.0.0", "", { "dependencies": { "acorn": "^8.0.0", "acorn-jsx": "^5.0.0", "micromark-extension-mdx-expression": "^3.0.0", "micromark-extension-mdx-jsx": "^3.0.0", "micromark-extension-mdx-md": "^2.0.0", "micromark-extension-mdxjs-esm": "^3.0.0", "micromark-util-combine-extensions": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-extension-mdxjs-esm": ["micromark-extension-mdxjs-esm@3.0.0", "", { "dependencies": { "@types/estree": "^1.0.0", "devlop": "^1.0.0", "micromark-core-commonmark": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-events-to-acorn": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0", "unist-util-position-from-estree": "^2.0.0", "vfile-message": "^4.0.0" } }, ""],
"micromark-factory-destination": ["micromark-factory-destination@2.0.1", "", { "dependencies": { "micromark-util-character": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-factory-label": ["micromark-factory-label@2.0.1", "", { "dependencies": { "devlop": "^1.0.0", "micromark-util-character": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-factory-mdx-expression": ["micromark-factory-mdx-expression@2.0.3", "", { "dependencies": { "@types/estree": "^1.0.0", "devlop": "^1.0.0", "micromark-factory-space": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-events-to-acorn": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0", "unist-util-position-from-estree": "^2.0.0", "vfile-message": "^4.0.0" } }, ""],
"micromark-factory-space": ["micromark-factory-space@2.0.1", "", { "dependencies": { "micromark-util-character": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-factory-title": ["micromark-factory-title@2.0.1", "", { "dependencies": { "micromark-factory-space": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-factory-whitespace": ["micromark-factory-whitespace@2.0.1", "", { "dependencies": { "micromark-factory-space": "^2.0.0", "micromark-util-character": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-util-character": ["micromark-util-character@2.1.1", "", { "dependencies": { "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-util-chunked": ["micromark-util-chunked@2.0.1", "", { "dependencies": { "micromark-util-symbol": "^2.0.0" } }, ""],
"micromark-util-classify-character": ["micromark-util-classify-character@2.0.1", "", { "dependencies": { "micromark-util-character": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-util-combine-extensions": ["micromark-util-combine-extensions@2.0.1", "", { "dependencies": { "micromark-util-chunked": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-util-decode-numeric-character-reference": ["micromark-util-decode-numeric-character-reference@2.0.2", "", { "dependencies": { "micromark-util-symbol": "^2.0.0" } }, ""],
"micromark-util-decode-string": ["micromark-util-decode-string@2.0.1", "", { "dependencies": { "decode-named-character-reference": "^1.0.0", "micromark-util-character": "^2.0.0", "micromark-util-decode-numeric-character-reference": "^2.0.0", "micromark-util-symbol": "^2.0.0" } }, ""],
"micromark-util-encode": ["micromark-util-encode@2.0.1", "", {}, ""],
"micromark-util-events-to-acorn": ["micromark-util-events-to-acorn@2.0.3", "", { "dependencies": { "@types/estree": "^1.0.0", "@types/unist": "^3.0.0", "devlop": "^1.0.0", "estree-util-visit": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0", "vfile-message": "^4.0.0" } }, ""],
"micromark-util-html-tag-name": ["micromark-util-html-tag-name@2.0.1", "", {}, ""],
"micromark-util-normalize-identifier": ["micromark-util-normalize-identifier@2.0.1", "", { "dependencies": { "micromark-util-symbol": "^2.0.0" } }, ""],
"micromark-util-resolve-all": ["micromark-util-resolve-all@2.0.1", "", { "dependencies": { "micromark-util-types": "^2.0.0" } }, ""],
"micromark-util-sanitize-uri": ["micromark-util-sanitize-uri@2.0.1", "", { "dependencies": { "micromark-util-character": "^2.0.0", "micromark-util-encode": "^2.0.0", "micromark-util-symbol": "^2.0.0" } }, ""],
"micromark-util-subtokenize": ["micromark-util-subtokenize@2.1.0", "", { "dependencies": { "devlop": "^1.0.0", "micromark-util-chunked": "^2.0.0", "micromark-util-symbol": "^2.0.0", "micromark-util-types": "^2.0.0" } }, ""],
"micromark-util-symbol": ["micromark-util-symbol@2.0.1", "", {}, ""],
"micromark-util-types": ["micromark-util-types@2.0.2", "", {}, ""],
"minimatch": ["minimatch@10.2.5", "", { "dependencies": { "brace-expansion": "^5.0.5" } }, ""],
"motion": ["motion@12.40.0", "", { "dependencies": { "framer-motion": "^12.40.0", "tslib": "^2.4.0" }, "peerDependencies": { "@emotion/is-prop-valid": "*", "react": "^18.0.0 || ^19.0.0", "react-dom": "^18.0.0 || ^19.0.0" }, "optionalPeers": ["@emotion/is-prop-valid"] }, "sha512-yjrHUrBFW6kQvjJwRsoiPSAhC5tRwRqNGJWmiJ4CrGnbKp0V88AdzkhBmDoqIsIPfarOe0Uddd37Xq43/gIocA=="],
"motion-dom": ["motion-dom@12.40.0", "", { "dependencies": { "motion-utils": "^12.39.0" } }, "sha512-HxU3ZaBwNPVQUBQf1xxgq+7JrPNZvjLVxgbpEZL7RrWJnsxOf0/OM+yrHG9ogLQ31Do/r57Oz2gQWPK+6q62mg=="],
"motion-utils": ["motion-utils@12.39.0", "", {}, "sha512-8nadJAJjTtqRkmRF36FoJTrywK9nnFmnPwnSMyxaOCU7GDjN9RTMJIxx9De8ErM+vpPhMccr/6fo5WciyQLnMQ=="],
"ms": ["ms@2.1.3", "", {}, ""],
"nanoid": ["nanoid@3.3.15", "", { "bin": "bin/nanoid.cjs" }, "sha512-y7Wygv/7mEOvxTuEQDB8StXdMRBWf1kR/tlhAzBRUFkB2jfcLOAxO/SHmOO2zgz1pVgK29/kyupn059/bCHdjA=="],
"next": ["next@16.2.6", "", { "dependencies": { "@next/env": "16.2.6", "@swc/helpers": "0.5.15", "baseline-browser-mapping": "^2.9.19", "caniuse-lite": "^1.0.30001579", "postcss": "8.4.31", "styled-jsx": "5.1.6" }, "optionalDependencies": { "@next/swc-darwin-arm64": "16.2.6", "@next/swc-darwin-x64": "16.2.6", "@next/swc-linux-arm64-gnu": "16.2.6", "@next/swc-linux-arm64-musl": "16.2.6", "@next/swc-linux-x64-gnu": "16.2.6", "@next/swc-linux-x64-musl": "16.2.6", "@next/swc-win32-arm64-msvc": "16.2.6", "@next/swc-win32-x64-msvc": "16.2.6", "sharp": "^0.34.5" }, "peerDependencies": { "@opentelemetry/api": "^1.1.0", "@playwright/test": "^1.51.1", "babel-plugin-react-compiler": "*", "react": "^18.2.0 || 19.0.0-rc-de68d2f4-20241204 || ^19.0.0", "react-dom": "^18.2.0 || 19.0.0-rc-de68d2f4-20241204 || ^19.0.0", "sass": "^1.3.0" }, "optionalPeers": ["@playwright/test", "babel-plugin-react-compiler", "sass"], "bin": "dist/bin/next" }, "sha512-qOVgKJg1+At15NpeUP+eJgCHvTCgXsogweq87Ri/Ix7PkqQHg4sdaXmSFqKlgaIXE4kW0g25LE68W87UANlHtw=="],
"next-themes": ["next-themes@0.4.6", "", { "peerDependencies": { "react": "^16.8 || ^17 || ^18 || ^19 || ^19.0.0-rc", "react-dom": "^16.8 || ^17 || ^18 || ^19 || ^19.0.0-rc" } }, ""],
"oniguruma-parser": ["oniguruma-parser@0.12.2", "", {}, "sha512-6HVa5oIrgMC6aA6WF6XyyqbhRPJrKR02L20+2+zpDtO5QAzGHAUGw5TKQvwi5vctNnRHkJYmjAhRVQF2EKdTQw=="],
"oniguruma-to-es": ["oniguruma-to-es@4.3.6", "", { "dependencies": { "oniguruma-parser": "^0.12.2", "regex": "^6.1.0", "regex-recursion": "^6.0.2" } }, "sha512-csuQ9x3Yr0cEIs/Zgx/OEt9iBw9vqIunAPQkx19R/fiMq2oGVTgcMqO/V3Ybqefr1TBvosI6jU539ksaBULJyA=="],
"openai": ["openai@6.33.0", "", { "peerDependencies": { "ws": "^8.18.0", "zod": "^3.25 || ^4.0" }, "optionalPeers": ["ws"], "bin": "bin/cli" }, ""],
"parse-entities": ["parse-entities@4.0.2", "", { "dependencies": { "@types/unist": "^2.0.0", "character-entities-legacy": "^3.0.0", "character-reference-invalid": "^2.0.0", "decode-named-character-reference": "^1.0.0", "is-alphanumerical": "^2.0.0", "is-decimal": "^2.0.0", "is-hexadecimal": "^2.0.0" } }, ""],
"parse5": ["parse5@7.3.0", "", { "dependencies": { "entities": "^6.0.0" } }, ""],
"path-browserify": ["path-browserify@1.0.1", "", {}, ""],
"picocolors": ["picocolors@1.1.1", "", {}, ""],
"picomatch": ["picomatch@4.0.4", "", {}, ""],
"point-in-polygon-hao": ["point-in-polygon-hao@1.2.4", "", { "dependencies": { "robust-predicates": "^3.0.2" } }, ""],
"postcss": ["postcss@8.5.15", "", { "dependencies": { "nanoid": "^3.3.12", "picocolors": "^1.1.1", "source-map-js": "^1.2.1" } }, "sha512-FfR8sjd4em2T6fb3I2MwAJU7HWVMr9zba+enmQeeWFfCbm+UOC/0X4DS8XtpUTMwWMGbjKYP7xjfNekzyGmB3A=="],
"proj4": ["proj4@2.20.8", "", { "dependencies": { "mgrs": "1.0.0", "wkt-parser": "^1.5.5" } }, ""],
"property-information": ["property-information@7.1.0", "", {}, ""],
"react": ["react@19.2.4", "", {}, ""],
"react-dom": ["react-dom@19.2.4", "", { "dependencies": { "scheduler": "^0.27.0" }, "peerDependencies": { "react": "^19.2.4" } }, ""],
"react-is": ["react-is@19.2.5", "", {}, ""],
"react-redux": ["react-redux@9.2.0", "", { "dependencies": { "@types/use-sync-external-store": "^0.0.6", "use-sync-external-store": "^1.4.0" }, "peerDependencies": { "@types/react": "^18.2.25 || ^19", "react": "^18.0 || ^19", "redux": "^5.0.0" } }, ""],
"react-remove-scroll": ["react-remove-scroll@2.7.2", "", { "dependencies": { "react-remove-scroll-bar": "^2.3.7", "react-style-singleton": "^2.2.3", "tslib": "^2.1.0", "use-callback-ref": "^1.3.3", "use-sidecar": "^1.1.3" }, "peerDependencies": { "@types/react": "*", "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" } }, ""],
"react-remove-scroll-bar": ["react-remove-scroll-bar@2.3.8", "", { "dependencies": { "react-style-singleton": "^2.2.2", "tslib": "^2.0.0" }, "peerDependencies": { "@types/react": "*", "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0" } }, ""],
"react-style-singleton": ["react-style-singleton@2.2.3", "", { "dependencies": { "get-nonce": "^1.0.0", "tslib": "^2.0.0" }, "peerDependencies": { "@types/react": "*", "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" } }, ""],
"readdirp": ["readdirp@5.0.0", "", {}, ""],
"recharts": ["recharts@3.8.1", "", { "dependencies": { "@reduxjs/toolkit": "^1.9.0 || 2.x.x", "clsx": "^2.1.1", "decimal.js-light": "^2.5.1", "es-toolkit": "^1.39.3", "eventemitter3": "^5.0.1", "immer": "^10.1.1", "react-redux": "8.x.x || 9.x.x", "reselect": "5.1.1", "tiny-invariant": "^1.3.3", "use-sync-external-store": "^1.2.2", "victory-vendor": "^37.0.2" }, "peerDependencies": { "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0", "react-dom": "^16.0.0 || ^17.0.0 || ^18.0.0 || ^19.0.0", "react-is": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0" } }, ""],
"recma-build-jsx": ["recma-build-jsx@1.0.0", "", { "dependencies": { "@types/estree": "^1.0.0", "estree-util-build-jsx": "^3.0.0", "vfile": "^6.0.0" } }, ""],
"recma-jsx": ["recma-jsx@1.0.1", "", { "dependencies": { "acorn-jsx": "^5.0.0", "estree-util-to-js": "^2.0.0", "recma-parse": "^1.0.0", "recma-stringify": "^1.0.0", "unified": "^11.0.0" }, "peerDependencies": { "acorn": "^6.0.0 || ^7.0.0 || ^8.0.0" } }, ""],
"recma-parse": ["recma-parse@1.0.0", "", { "dependencies": { "@types/estree": "^1.0.0", "esast-util-from-js": "^2.0.0", "unified": "^11.0.0", "vfile": "^6.0.0" } }, ""],
"recma-stringify": ["recma-stringify@1.0.0", "", { "dependencies": { "@types/estree": "^1.0.0", "estree-util-to-js": "^2.0.0", "unified": "^11.0.0", "vfile": "^6.0.0" } }, ""],
"redux": ["redux@5.0.1", "", {}, ""],
"redux-thunk": ["redux-thunk@3.1.0", "", { "peerDependencies": { "redux": "^5.0.0" } }, ""],
"regex": ["regex@6.1.0", "", { "dependencies": { "regex-utilities": "^2.3.0" } }, "sha512-6VwtthbV4o/7+OaAF9I5L5V3llLEsoPyq9P1JVXkedTP33c7MfCG0/5NOPcSJn0TzXcG9YUrR0gQSWioew3LDg=="],
"regex-recursion": ["regex-recursion@6.0.2", "", { "dependencies": { "regex-utilities": "^2.3.0" } }, "sha512-0YCaSCq2VRIebiaUviZNs0cBz1kg5kVS2UKUfNIx8YVs1cN3AV7NTctO5FOKBA+UT2BPJIWZauYHPqJODG50cg=="],
"regex-utilities": ["regex-utilities@2.3.0", "", {}, "sha512-8VhliFJAWRaUiVvREIiW2NXXTmHs4vMNnSzuJVhscgmGav3g9VDxLrQndI3dZZVVdp0ZO/5v0xmX516/7M9cng=="],
"rehype-raw": ["rehype-raw@7.0.0", "", { "dependencies": { "@types/hast": "^3.0.0", "hast-util-raw": "^9.0.0", "vfile": "^6.0.0" } }, ""],
"rehype-recma": ["rehype-recma@1.0.0", "", { "dependencies": { "@types/estree": "^1.0.0", "@types/hast": "^3.0.0", "hast-util-to-estree": "^3.0.0" } }, ""],
"remark": ["remark@15.0.1", "", { "dependencies": { "@types/mdast": "^4.0.0", "remark-parse": "^11.0.0", "remark-stringify": "^11.0.0", "unified": "^11.0.0" } }, ""],
"remark-gfm": ["remark-gfm@4.0.1", "", { "dependencies": { "@types/mdast": "^4.0.0", "mdast-util-gfm": "^3.0.0", "micromark-extension-gfm": "^3.0.0", "remark-parse": "^11.0.0", "remark-stringify": "^11.0.0", "unified": "^11.0.0" } }, ""],
"remark-mdx": ["remark-mdx@3.1.1", "", { "dependencies": { "mdast-util-mdx": "^3.0.0", "micromark-extension-mdxjs": "^3.0.0" } }, ""],
"remark-parse": ["remark-parse@11.0.0", "", { "dependencies": { "@types/mdast": "^4.0.0", "mdast-util-from-markdown": "^2.0.0", "micromark-util-types": "^2.0.0", "unified": "^11.0.0" } }, ""],
"remark-rehype": ["remark-rehype@11.1.2", "", { "dependencies": { "@types/hast": "^3.0.0", "@types/mdast": "^4.0.0", "mdast-util-to-hast": "^13.0.0", "unified": "^11.0.0", "vfile": "^6.0.0" } }, ""],
"remark-stringify": ["remark-stringify@11.0.0", "", { "dependencies": { "@types/mdast": "^4.0.0", "mdast-util-to-markdown": "^2.0.0", "unified": "^11.0.0" } }, ""],
"reselect": ["reselect@5.1.1", "", {}, ""],
"robust-predicates": ["robust-predicates@3.0.3", "", {}, ""],
"scheduler": ["scheduler@0.27.0", "", {}, ""],
"scroll-into-view-if-needed": ["scroll-into-view-if-needed@3.1.0", "", { "dependencies": { "compute-scroll-into-view": "^3.0.2" } }, ""],
"semver": ["semver@7.7.4", "", { "bin": "bin/semver.js" }, ""],
"sharp": ["sharp@0.34.5", "", { "dependencies": { "@img/colour": "^1.0.0", "detect-libc": "^2.1.2", "semver": "^7.7.3" }, "optionalDependencies": { "@img/sharp-darwin-arm64": "0.34.5", "@img/sharp-darwin-x64": "0.34.5", "@img/sharp-libvips-darwin-arm64": "1.2.4", "@img/sharp-libvips-darwin-x64": "1.2.4", "@img/sharp-libvips-linux-arm": "1.2.4", "@img/sharp-libvips-linux-arm64": "1.2.4", "@img/sharp-libvips-linux-ppc64": "1.2.4", "@img/sharp-libvips-linux-riscv64": "1.2.4", "@img/sharp-libvips-linux-s390x": "1.2.4", "@img/sharp-libvips-linux-x64": "1.2.4", "@img/sharp-libvips-linuxmusl-arm64": "1.2.4", "@img/sharp-libvips-linuxmusl-x64": "1.2.4", "@img/sharp-linux-arm": "0.34.5", "@img/sharp-linux-arm64": "0.34.5", "@img/sharp-linux-ppc64": "0.34.5", "@img/sharp-linux-riscv64": "0.34.5", "@img/sharp-linux-s390x": "0.34.5", "@img/sharp-linux-x64": "0.34.5", "@img/sharp-linuxmusl-arm64": "0.34.5", "@img/sharp-linuxmusl-x64": "0.34.5", "@img/sharp-wasm32": "0.34.5", "@img/sharp-win32-arm64": "0.34.5", "@img/sharp-win32-ia32": "0.34.5", "@img/sharp-win32-x64": "0.34.5" } }, ""],
"shiki": ["shiki@4.2.0", "", { "dependencies": { "@shikijs/core": "4.2.0", "@shikijs/engine-javascript": "4.2.0", "@shikijs/engine-oniguruma": "4.2.0", "@shikijs/langs": "4.2.0", "@shikijs/themes": "4.2.0", "@shikijs/types": "4.2.0", "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, "sha512-hjNax6o/ylDy9lefQEaSDtzaT3iVNtZ3WmpQnbuQNoG4xvnSKf2kSKbihZVO4JRG1TTMejs7CmNRYlWgAL66pQ=="],
"source-map": ["source-map@0.7.6", "", {}, ""],
"source-map-js": ["source-map-js@1.2.1", "", {}, ""],
"space-separated-tokens": ["space-separated-tokens@2.0.2", "", {}, ""],
"standardwebhooks": ["standardwebhooks@1.0.0", "", { "dependencies": { "@stablelib/base64": "^1.0.0", "fast-sha256": "^1.3.0" } }, "sha512-BbHGOQK9olHPMvQNHWul6MYlrRTAOKn03rOe4A8O3CLWhNf4YHBqq2HJKKC+sfqpxiBY52pNeesD6jIiLDz8jg=="],
"stringify-entities": ["stringify-entities@4.0.4", "", { "dependencies": { "character-entities-html4": "^2.0.0", "character-entities-legacy": "^3.0.0" } }, ""],
"style-to-js": ["style-to-js@1.1.21", "", { "dependencies": { "style-to-object": "1.0.14" } }, ""],
"style-to-object": ["style-to-object@1.0.14", "", { "dependencies": { "inline-style-parser": "0.2.7" } }, ""],
"styled-jsx": ["styled-jsx@5.1.6", "", { "dependencies": { "client-only": "0.0.1" }, "peerDependencies": { "react": ">= 16.8.0 || 17.x.x || ^18.0.0-0 || ^19.0.0-0" } }, ""],
"tailwind-merge": ["tailwind-merge@3.6.0", "", {}, "sha512-uxL7qAVQriqRQPAyK3pj66VqskWqoZ37PW94jwOTwNfq/z9oyu1V+eqrZqtR2+fCiXdYOZe/Modt8GtvqNzu+w=="],
"tailwindcss": ["tailwindcss@4.2.2", "", {}, ""],
"tapable": ["tapable@2.3.2", "", {}, ""],
"tiny-invariant": ["tiny-invariant@1.3.3", "", {}, ""],
"tinyexec": ["tinyexec@1.2.4", "", {}, "sha512-SHf/r48b7vOrjve9PxJo3MN5v5yuyjHvdUcrQffT3WXMUfnGmHDVbC4k3sHJaJTgZCwpUplIaAo5ANtMyp3YHg=="],
"tinyglobby": ["tinyglobby@0.2.17", "", { "dependencies": { "fdir": "^6.5.0", "picomatch": "^4.0.4" } }, "sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g=="],
"trim-lines": ["trim-lines@3.0.1", "", {}, ""],
"trough": ["trough@2.2.0", "", {}, ""],
"ts-algebra": ["ts-algebra@2.0.0", "", {}, ""],
"ts-morph": ["ts-morph@27.0.2", "", { "dependencies": { "@ts-morph/common": "~0.28.1", "code-block-writer": "^13.0.3" } }, ""],
"tslib": ["tslib@2.8.1", "", {}, ""],
"twoslash": ["twoslash@0.3.6", "", { "dependencies": { "@typescript/vfs": "^1.6.2", "twoslash-protocol": "0.3.6" }, "peerDependencies": { "typescript": "^5.5.0" } }, ""],
"twoslash-protocol": ["twoslash-protocol@0.3.6", "", {}, ""],
"typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, ""],
"undici-types": ["undici-types@7.18.2", "", {}, ""],
"unified": ["unified@11.0.5", "", { "dependencies": { "@types/unist": "^3.0.0", "bail": "^2.0.0", "devlop": "^1.0.0", "extend": "^3.0.0", "is-plain-obj": "^4.0.0", "trough": "^2.0.0", "vfile": "^6.0.0" } }, ""],
"unist-util-is": ["unist-util-is@6.0.1", "", { "dependencies": { "@types/unist": "^3.0.0" } }, ""],
"unist-util-position": ["unist-util-position@5.0.0", "", { "dependencies": { "@types/unist": "^3.0.0" } }, ""],
"unist-util-position-from-estree": ["unist-util-position-from-estree@2.0.0", "", { "dependencies": { "@types/unist": "^3.0.0" } }, ""],
"unist-util-remove-position": ["unist-util-remove-position@5.0.0", "", { "dependencies": { "@types/unist": "^3.0.0", "unist-util-visit": "^5.0.0" } }, ""],
"unist-util-stringify-position": ["unist-util-stringify-position@4.0.0", "", { "dependencies": { "@types/unist": "^3.0.0" } }, ""],
"unist-util-visit": ["unist-util-visit@5.1.0", "", { "dependencies": { "@types/unist": "^3.0.0", "unist-util-is": "^6.0.0", "unist-util-visit-parents": "^6.0.0" } }, ""],
"unist-util-visit-parents": ["unist-util-visit-parents@6.0.2", "", { "dependencies": { "@types/unist": "^3.0.0", "unist-util-is": "^6.0.0" } }, ""],
"use-callback-ref": ["use-callback-ref@1.3.3", "", { "dependencies": { "tslib": "^2.0.0" }, "peerDependencies": { "@types/react": "*", "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" } }, ""],
"use-sidecar": ["use-sidecar@1.1.3", "", { "dependencies": { "detect-node-es": "^1.1.0", "tslib": "^2.0.0" }, "peerDependencies": { "@types/react": "*", "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" } }, ""],
"use-sync-external-store": ["use-sync-external-store@1.6.0", "", { "peerDependencies": { "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0" } }, ""],
"vfile": ["vfile@6.0.3", "", { "dependencies": { "@types/unist": "^3.0.0", "vfile-message": "^4.0.0" } }, ""],
"vfile-location": ["vfile-location@5.0.3", "", { "dependencies": { "@types/unist": "^3.0.0", "vfile": "^6.0.0" } }, ""],
"vfile-message": ["vfile-message@4.0.3", "", { "dependencies": { "@types/unist": "^3.0.0", "unist-util-stringify-position": "^4.0.0" } }, ""],
"victory-vendor": ["victory-vendor@37.3.6", "", { "dependencies": { "@types/d3-array": "^3.0.3", "@types/d3-ease": "^3.0.0", "@types/d3-interpolate": "^3.0.1", "@types/d3-scale": "^4.0.2", "@types/d3-shape": "^3.1.0", "@types/d3-time": "^3.0.0", "@types/d3-timer": "^3.0.0", "d3-array": "^3.1.6", "d3-ease": "^3.0.1", "d3-interpolate": "^3.0.1", "d3-scale": "^4.0.2", "d3-shape": "^3.1.0", "d3-time": "^3.0.0", "d3-timer": "^3.0.1" } }, ""],
"web-namespaces": ["web-namespaces@2.0.1", "", {}, ""],
"wkt-parser": ["wkt-parser@1.5.5", "", {}, ""],
"zod": ["zod@4.4.3", "", {}, "sha512-ytENFjIJFl2UwYglde2jchW2Hwm4GJFLDiSXWdTrJQBIN9Fcyp7n4DhxJEiWNAJMV1/BqWfW/kkg71UDcHJyTQ=="],
"zwitch": ["zwitch@2.0.4", "", {}, ""],
"@reduxjs/toolkit/immer": ["immer@11.1.4", "", {}, ""],
"@shikijs/engine-javascript/@shikijs/types": ["@shikijs/types@4.2.0", "", { "dependencies": { "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, "sha512-VT/MKtlpOhEPZloSH3Pb9WCZEBDoQVMa9jedp5UAwmJOar1DVc9DRODAxmYPW9M93IK4ryuqRejFfmlvlVDemw=="],
"@shikijs/engine-oniguruma/@shikijs/types": ["@shikijs/types@4.2.0", "", { "dependencies": { "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, "sha512-VT/MKtlpOhEPZloSH3Pb9WCZEBDoQVMa9jedp5UAwmJOar1DVc9DRODAxmYPW9M93IK4ryuqRejFfmlvlVDemw=="],
"@shikijs/langs/@shikijs/types": ["@shikijs/types@4.2.0", "", { "dependencies": { "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, "sha512-VT/MKtlpOhEPZloSH3Pb9WCZEBDoQVMa9jedp5UAwmJOar1DVc9DRODAxmYPW9M93IK4ryuqRejFfmlvlVDemw=="],
"@shikijs/themes/@shikijs/types": ["@shikijs/types@4.2.0", "", { "dependencies": { "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, "sha512-VT/MKtlpOhEPZloSH3Pb9WCZEBDoQVMa9jedp5UAwmJOar1DVc9DRODAxmYPW9M93IK4ryuqRejFfmlvlVDemw=="],
"@shikijs/twoslash/@shikijs/core": ["@shikijs/core@4.0.2", "", { "dependencies": { "@shikijs/primitive": "4.0.2", "@shikijs/types": "4.0.2", "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4", "hast-util-to-html": "^9.0.5" } }, ""],
"@shikijs/twoslash/@shikijs/types": ["@shikijs/types@4.0.2", "", { "dependencies": { "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, ""],
"@shikijs/twoslash/twoslash": ["twoslash@0.3.6", "", { "dependencies": { "@typescript/vfs": "^1.6.2", "twoslash-protocol": "0.3.6" }, "peerDependencies": { "typescript": "^5.5.0" } }, ""],
"parse-entities/@types/unist": ["@types/unist@2.0.11", "", {}, ""],
"@shikijs/twoslash/@shikijs/core/@shikijs/primitive": ["@shikijs/primitive@4.0.2", "", { "dependencies": { "@shikijs/types": "4.0.2", "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, ""],
}
}
+147
View File
@@ -0,0 +1,147 @@
# Claude Code + AWS Bedrock, with Headroom compression
*Validated end-to-end on 2026-06-26 (Claude Code 2.1, Headroom 0.27.0, ap-southeast-2).*
This is the **working, tested** way to run **Claude Code** against **Claude models on
AWS Bedrock** with **Headroom compressing the context** in the middle.
## TL;DR
Run Claude Code in **normal Anthropic mode** (NOT Bedrock mode) pointed at a local
Headroom proxy, and let **Headroom** be the thing that talks to Bedrock:
```
Claude Code ──ANTHROPIC_BASE_URL──▶ Headroom proxy ──LiteLLM (bedrock)──▶ AWS Bedrock
(normal mode) (plain http) (compresses) (your AWS creds) (Claude)
```
One non-obvious requirement makes the difference between "works" and "silently bypasses
the proxy":
1. **`CLAUDE_CODE_USE_BEDROCK=0`** — Without this, Claude Code sees the
`CLAUDE_CODE_USE_BEDROCK=1` flag and calls Bedrock directly via the AWS SDK,
completely bypassing `ANTHROPIC_BASE_URL` and the proxy.
## Why not "just set CLAUDE_CODE_USE_BEDROCK=1"?
That approach **does not work** with Headroom. When `CLAUDE_CODE_USE_BEDROCK=1` is set,
Claude Code calls Bedrock directly using the AWS SDK — `ANTHROPIC_BASE_URL` is ignored
entirely and the proxy never receives a byte. Use the Anthropic-mode path below.
## Prerequisites
- **AWS credentials** configured for your environment (env vars, `~/.aws/credentials`,
instance profile, or SSO via `aws sso login`). Confirm direct access works before
involving Headroom:
```bash
aws bedrock-runtime invoke-model \
--region us-east-1 \
--model-id anthropic.claude-3-haiku-20240307-v1:0 \
--body '{"anthropic_version":"bedrock-2023-05-31","max_tokens":20,"messages":[{"role":"user","content":"hi"}]}' \
/tmp/out.json
```
- **boto3** in the proxy's Python environment (for dynamic inference profile discovery):
```bash
pip install boto3
```
- **IAM permissions** for the models you intend to use — at minimum
`bedrock:InvokeModel` and `bedrock:InvokeModelWithResponseStream`. For application
inference profiles, scope to the specific profile ARN:
```json
{
"Effect": "Allow",
"Action": ["bedrock:InvokeModel", "bedrock:InvokeModelWithResponseStream"],
"Resource": ["arn:aws:bedrock:<region>:<account>:application-inference-profile/<id>"]
}
```
## Terminal 1 — start the Headroom proxy (Bedrock backend)
```bash
headroom proxy --port 8787 \
--backend bedrock \
--region us-east-1
```
With a named AWS SSO profile:
```bash
headroom proxy --port 8787 \
--backend bedrock \
--region us-east-1 \
--bedrock-profile my-sso-profile
```
On startup the proxy calls `list_inference_profiles` to build a model map. Confirm it
is routing correctly by checking the LiteLLM log lines — you should see:
```
LiteLLM completion() model= converse/arn:aws:... provider = bedrock
```
## Terminal 2 — run Claude Code (normal Anthropic mode) against the proxy
```bash
export CLAUDE_CODE_USE_BEDROCK=0 # REQUIRED — prevents Claude Code bypassing the proxy
export ANTHROPIC_BASE_URL=http://127.0.0.1:8787
export ANTHROPIC_API_KEY=headroom # Claude Code needs *a* key to start; value is ignored
export ANTHROPIC_MODEL=claude-opus-4-6
export ANTHROPIC_DEFAULT_SONNET_MODEL=claude-sonnet-4-6
export ANTHROPIC_DEFAULT_OPUS_MODEL=claude-opus-4-6
export ANTHROPIC_DEFAULT_HAIKU_MODEL=claude-haiku-4-5-20251001
claude
```
Or via `~/.claude/settings.json`:
```json
{
"env": {
"CLAUDE_CODE_USE_BEDROCK": "0",
"ANTHROPIC_BASE_URL": "http://127.0.0.1:8787",
"ANTHROPIC_API_KEY": "headroom",
"ANTHROPIC_MODEL": "claude-opus-4-6",
"ANTHROPIC_DEFAULT_SONNET_MODEL": "claude-sonnet-4-6",
"ANTHROPIC_DEFAULT_OPUS_MODEL": "claude-opus-4-6",
"ANTHROPIC_DEFAULT_HAIKU_MODEL": "claude-haiku-4-5-20251001"
}
}
```
Claude Code now talks plain Anthropic `/v1/messages` to Headroom; Headroom compresses
and forwards to Bedrock via LiteLLM, then translates the answer back.
## Application inference profiles (account-specific ARNs)
If your IAM policy only permits **application inference profiles** (account-specific
ARNs) rather than system-defined cross-region profiles, pass the ARN directly as the
model value in `ANTHROPIC_DEFAULT_*_MODEL`. The proxy detects `arn:aws:` prefixed model
IDs and routes them via `bedrock/converse/<arn>` automatically — no extra configuration
required.
## Region prefix notes
| AWS region | Cross-region inference prefix |
|---|---|
| `us-*` | `us.` |
| `eu-*` | `eu.` |
| `ap-*` (except `ap-southeast-2`) | `apac.` |
| `ap-southeast-2` (Sydney) | `au.` |
The proxy uses the correct prefix automatically when constructing fallback model IDs.
## Verify compression is happening
- Dashboard: <http://localhost:8787/dashboard> — "tokens saved" climbs as you work.
- `curl -s localhost:8787/stats` → `tokens.saved` and `request_logs[].transforms_applied`.
## Troubleshooting
| Symptom | Cause | Fix |
|---|---|---|
| Proxy receives no requests | Claude Code is in Bedrock mode, bypassing proxy | Set `CLAUDE_CODE_USE_BEDROCK=0` |
| `400 The provided model identifier is invalid` | Bedrock rejected the model name format | Use standard cross-region profile names (`claude-sonnet-4-6`) or a valid application inference profile ARN |
| `403 AccessDeniedException` on system-defined profiles | IAM policy only permits application profiles | Use `--bedrock-profile` with an authorized profile and pass application inference profile ARNs as model values |
| `400 … Try calling via converse route` | Old proxy version routing ARNs to invoke path | Upgrade to headroom ≥ 0.27.1 |
| Model map empty at startup | boto3 not installed or wrong AWS profile | `pip install boto3`; check `--bedrock-profile` / `AWS_PROFILE` |
+59
View File
@@ -0,0 +1,59 @@
import * as React from "react";
import { Slot } from "@radix-ui/react-slot";
import { cva, type VariantProps } from "class-variance-authority";
import { cn } from "@/lib/cn";
const buttonVariants = cva(
"cursor-pointer active:scale-99 duration-200 font-medium inline-flex items-center justify-center gap-2 whitespace-nowrap rounded-full text-sm font-medium focus-visible:outline-none focus-visible:ring-1 focus-visible:ring-ring disabled:pointer-events-none disabled:opacity-50 [&_svg]:pointer-events-none [&_svg]:size-4 [&_svg]:shrink-0",
{
variants: {
variant: {
default: "bg-foreground text-background hover:brightness-95",
neutral: "bg-foreground text-background hover:brightness-95",
destructive:
"bg-destructive text-destructive-foreground shadow-md hover:bg-destructive/90",
outline:
"shadow-sm text-foreground shadow-black/6.5 border border-transparent bg-card ring-1 ring-foreground/15 duration-200 hover:bg-muted/50",
secondary:
"bg-secondary text-secondary-foreground hover:bg-secondary/80",
ghost: "hover:bg-foreground/5 text-foreground/75 hover:text-foreground",
link: "text-primary underline-offset-4 hover:underline",
},
size: {
default: "h-8 px-3 py-2",
sm: "h-7 px-2.5 text-sm",
lg: "h-11 px-6 font-medium text-base",
icon: "size-9",
"icon-sm": "size-7",
"icon-xs": "size-5",
},
},
defaultVariants: {
variant: "default",
size: "default",
},
},
);
export interface ButtonProps
extends
React.ButtonHTMLAttributes<HTMLButtonElement>,
VariantProps<typeof buttonVariants> {
asChild?: boolean;
}
const Button = React.forwardRef<HTMLButtonElement, ButtonProps>(
({ className, variant, size, asChild = false, ...props }, ref) => {
const Comp = asChild ? Slot : "button";
return (
<Comp
className={cn(buttonVariants({ variant, size, className }))}
ref={ref}
{...props}
/>
);
},
);
Button.displayName = "Button";
export { Button, buttonVariants };
+12
View File
@@ -0,0 +1,12 @@
interface CodeBlockProps {
code: string;
lang?: string;
}
export function CodeBlock({ code }: CodeBlockProps) {
return (
<pre className="mt-3 p-3 text-xs rounded-lg bg-fd-muted text-fd-foreground overflow-x-auto font-mono whitespace-pre-wrap break-words">
{code}
</pre>
);
}
+346
View File
@@ -0,0 +1,346 @@
'use client'
import { useState } from 'react'
import {
AreaChart, Area, BarChart, Bar,
XAxis, YAxis, Tooltip, ResponsiveContainer, CartesianGrid,
} from 'recharts'
// --- Embedded telemetry data ---
const DATA = {
total_tokens_saved: 41750654085,
total_cost_saved: 176635.62,
total_requests: 1194154,
unique_instances: 889,
active_days: 14,
daily_stats: [
{ date: '2026-03-30', requests: 1050, instances: 17, cost_saved: 42.29, tokens_saved: 7560502 },
{ date: '2026-03-31', requests: 26526, instances: 149, cost_saved: 1904.15, tokens_saved: 1004245860 },
{ date: '2026-04-01', requests: 53480, instances: 117, cost_saved: 3353.50, tokens_saved: 860455498 },
{ date: '2026-04-02', requests: 64906, instances: 123, cost_saved: 18007.10, tokens_saved: 3431477414 },
{ date: '2026-04-03', requests: 99872, instances: 162, cost_saved: 29238.20, tokens_saved: 6159073542 },
{ date: '2026-04-04', requests: 90992, instances: 147, cost_saved: 12863.30, tokens_saved: 1864667449 },
{ date: '2026-04-05', requests: 162541, instances: 142, cost_saved: 49802.60, tokens_saved: 11990261722 },
{ date: '2026-04-06', requests: 127494, instances: 174, cost_saved: 11533.70, tokens_saved: 3850552409 },
{ date: '2026-04-07', requests: 158840, instances: 201, cost_saved: 14260.90, tokens_saved: 4113326538 },
{ date: '2026-04-08', requests: 84459, instances: 186, cost_saved: 8383.12, tokens_saved: 2317153362 },
{ date: '2026-04-09', requests: 137856, instances: 208, cost_saved: 10663.80, tokens_saved: 2058570508 },
{ date: '2026-04-10', requests: 111043, instances: 192, cost_saved: 9893.92, tokens_saved: 1979992305 },
{ date: '2026-04-11', requests: 65851, instances: 147, cost_saved: 5853.92, tokens_saved: 1521911387 },
{ date: '2026-04-12', requests: 9244, instances: 27, cost_saved: 835.12, tokens_saved: 591405589 },
],
hourly_stats: [
{ hour: '2026-04-10 06:00', requests: 8548, instances: 9, cost_saved: 577.12, tokens_saved: 198669219 },
{ hour: '2026-04-10 07:00', requests: 8289, instances: 18, cost_saved: 456.34, tokens_saved: 172834374 },
{ hour: '2026-04-10 08:00', requests: 8066, instances: 23, cost_saved: 559.56, tokens_saved: 196181344 },
{ hour: '2026-04-10 09:00', requests: 4890, instances: 16, cost_saved: 206.95, tokens_saved: 124914685 },
{ hour: '2026-04-10 10:00', requests: 9531, instances: 21, cost_saved: 421.51, tokens_saved: 236470420 },
{ hour: '2026-04-10 11:00', requests: 5170, instances: 13, cost_saved: 174.45, tokens_saved: 114113494 },
{ hour: '2026-04-10 12:00', requests: 10982, instances: 20, cost_saved: 388.05, tokens_saved: 166308102 },
{ hour: '2026-04-10 13:00', requests: 8001, instances: 16, cost_saved: 228.53, tokens_saved: 134921875 },
{ hour: '2026-04-10 14:00', requests: 4950, instances: 15, cost_saved: 208.56, tokens_saved: 119763470 },
{ hour: '2026-04-10 15:00', requests: 9000, instances: 17, cost_saved: 1739.51, tokens_saved: 166911747 },
{ hour: '2026-04-10 16:00', requests: 12569, instances: 17, cost_saved: 1130.82, tokens_saved: 312958735 },
{ hour: '2026-04-10 17:00', requests: 12057, instances: 17, cost_saved: 443.07, tokens_saved: 203763560 },
{ hour: '2026-04-10 18:00', requests: 14011, instances: 18, cost_saved: 1018.06, tokens_saved: 294928335 },
{ hour: '2026-04-10 19:00', requests: 12599, instances: 17, cost_saved: 915.21, tokens_saved: 327535343 },
{ hour: '2026-04-10 20:00', requests: 10522, instances: 16, cost_saved: 2484.87, tokens_saved: 206735552 },
{ hour: '2026-04-10 21:00', requests: 7125, instances: 16, cost_saved: 928.27, tokens_saved: 289265979 },
{ hour: '2026-04-10 22:00', requests: 4481, instances: 9, cost_saved: 190.60, tokens_saved: 137301814 },
{ hour: '2026-04-10 23:00', requests: 4033, instances: 5, cost_saved: 154.82, tokens_saved: 126989500 },
{ hour: '2026-04-11 00:00', requests: 8998, instances: 12, cost_saved: 770.90, tokens_saved: 254824094 },
{ hour: '2026-04-11 01:00', requests: 13729, instances: 12, cost_saved: 668.12, tokens_saved: 330717273 },
{ hour: '2026-04-11 02:00', requests: 3738, instances: 2, cost_saved: 152.18, tokens_saved: 126547280 },
{ hour: '2026-04-11 03:00', requests: 4449, instances: 4, cost_saved: 185.03, tokens_saved: 134597314 },
{ hour: '2026-04-11 04:00', requests: 5961, instances: 9, cost_saved: 303.65, tokens_saved: 157734893 },
{ hour: '2026-04-11 05:00', requests: 6550, instances: 8, cost_saved: 248.84, tokens_saved: 158834743 },
{ hour: '2026-04-11 06:00', requests: 5429, instances: 8, cost_saved: 268.11, tokens_saved: 155129772 },
{ hour: '2026-04-11 07:00', requests: 5671, instances: 11, cost_saved: 225.25, tokens_saved: 143244185 },
{ hour: '2026-04-11 08:00', requests: 7258, instances: 12, cost_saved: 264.80, tokens_saved: 158618368 },
{ hour: '2026-04-11 09:00', requests: 8732, instances: 7, cost_saved: 301.46, tokens_saved: 157720064 },
{ hour: '2026-04-11 10:00', requests: 6058, instances: 6, cost_saved: 230.22, tokens_saved: 155936384 },
{ hour: '2026-04-11 11:00', requests: 7615, instances: 13, cost_saved: 444.21, tokens_saved: 288328690 },
{ hour: '2026-04-11 12:00', requests: 6749, instances: 10, cost_saved: 716.35, tokens_saved: 542054609 },
{ hour: '2026-04-11 13:00', requests: 6276, instances: 10, cost_saved: 2259.74, tokens_saved: 572384133 },
{ hour: '2026-04-11 14:00', requests: 8858, instances: 16, cost_saved: 846.04, tokens_saved: 602153567 },
{ hour: '2026-04-11 15:00', requests: 9284, instances: 12, cost_saved: 811.16, tokens_saved: 581585147 },
{ hour: '2026-04-11 16:00', requests: 7390, instances: 16, cost_saved: 749.69, tokens_saved: 566925112 },
{ hour: '2026-04-11 17:00', requests: 11558, instances: 13, cost_saved: 1023.99, tokens_saved: 623709492 },
{ hour: '2026-04-11 18:00', requests: 6154, instances: 10, cost_saved: 709.22, tokens_saved: 546162461 },
{ hour: '2026-04-11 19:00', requests: 5711, instances: 11, cost_saved: 660.33, tokens_saved: 548094492 },
{ hour: '2026-04-11 20:00', requests: 9287, instances: 13, cost_saved: 800.20, tokens_saved: 575078608 },
{ hour: '2026-04-11 21:00', requests: 5858, instances: 9, cost_saved: 675.77, tokens_saved: 547657183 },
{ hour: '2026-04-11 22:00', requests: 5523, instances: 13, cost_saved: 1133.82, tokens_saved: 638333840 },
{ hour: '2026-04-11 23:00', requests: 7668, instances: 8, cost_saved: 903.46, tokens_saved: 592286174 },
{ hour: '2026-04-12 00:00', requests: 5723, instances: 9, cost_saved: 655.44, tokens_saved: 543699655 },
{ hour: '2026-04-12 01:00', requests: 5291, instances: 8, cost_saved: 673.29, tokens_saved: 552129148 },
{ hour: '2026-04-12 02:00', requests: 6689, instances: 9, cost_saved: 698.83, tokens_saved: 551696076 },
{ hour: '2026-04-12 03:00', requests: 6532, instances: 8, cost_saved: 749.06, tokens_saved: 569436595 },
{ hour: '2026-04-12 04:00', requests: 4805, instances: 3, cost_saved: 644.29, tokens_saved: 540846848 },
{ hour: '2026-04-12 05:00', requests: 5181, instances: 6, cost_saved: 642.90, tokens_saved: 539173642 },
{ hour: '2026-04-12 06:00', requests: 4739, instances: 1, cost_saved: 637.17, tokens_saved: 537999942 },
],
top_instances: [
{ os: 'Windows', version: '0.5.18', cost_saved: 36325.40, instance_id: '1d5d8ed0', tokens_saved: 8852060567 },
{ os: 'Linux', version: '0.5.19', cost_saved: 2188.80, instance_id: '96d9632f', tokens_saved: 2395311403 },
{ os: 'Windows', version: '0.5.17', cost_saved: 11878.30, instance_id: '1e850cb3', tokens_saved: 2375786993 },
{ os: 'Windows', version: '0.5.17', cost_saved: 9080.36, instance_id: '0cdfb8e9', tokens_saved: 1816351278 },
{ os: 'Linux', version: '0.5.18', cost_saved: 7580.30, instance_id: '7456b0f9', tokens_saved: 1635744469 },
{ os: 'Darwin', version: '0.5.16', cost_saved: 1772.99, instance_id: '1661b732', tokens_saved: 1488844495 },
{ os: 'Darwin', version: '0.5.17', cost_saved: 4667.23, instance_id: '08bf5ae1', tokens_saved: 933450700 },
{ os: 'Darwin', version: '0.5.18', cost_saved: 0, instance_id: 'e20f01b6', tokens_saved: 565038755 },
{ os: 'Linux', version: '0.5.18', cost_saved: 538.37, instance_id: '5b0795b2', tokens_saved: 557489086 },
{ os: 'Windows', version: '0.5.18', cost_saved: 1773.69, instance_id: 'eff9e644', tokens_saved: 503362483 },
{ os: 'Windows', version: '0.5.18', cost_saved: 1812.66, instance_id: '388102c7', tokens_saved: 485305324 },
{ os: 'Linux', version: '0.5.17', cost_saved: 2297.45, instance_id: '3ee70444', tokens_saved: 468622655 },
{ os: 'Darwin', version: '0.5.16', cost_saved: 319.65, instance_id: '6e758d1e', tokens_saved: 437502391 },
{ os: 'Linux', version: '0.5.18', cost_saved: 1461.02, instance_id: 'b2e71a28', tokens_saved: 415224504 },
{ os: 'Darwin', version: '0.5.21', cost_saved: 509.44, instance_id: '8b3795aa', tokens_saved: 353583838 },
{ os: 'Linux', version: '0.5.17', cost_saved: 1663.48, instance_id: 'b6a1a735', tokens_saved: 334438796 },
{ os: 'Linux', version: '0.5.21', cost_saved: 1603.77, instance_id: 'd0c16fa0', tokens_saved: 322890921 },
{ os: 'Linux', version: '0.5.18', cost_saved: 1580.12, instance_id: '4140eb00', tokens_saved: 316804284 },
{ os: 'Darwin', version: '0.5.21', cost_saved: 1052.77, instance_id: '2b11b55c', tokens_saved: 308601543 },
{ os: 'Darwin', version: '0.5.17', cost_saved: 1473.78, instance_id: '1ede777b', tokens_saved: 296950447 },
],
}
// --- Formatters ---
function fmt(n: number): string {
if (n >= 1e9) return `${(n / 1e9).toFixed(1)}B`
if (n >= 1e6) return `${(n / 1e6).toFixed(1)}M`
if (n >= 1e3) return `${(n / 1e3).toFixed(1)}K`
return n.toFixed(0)
}
function fmtUsd(n: number): string {
if (n >= 1e3) return `$${(n / 1e3).toFixed(1)}K`
return `$${n.toFixed(0)}`
}
const MONTHS = ['Jan','Feb','Mar','Apr','May','Jun','Jul','Aug','Sep','Oct','Nov','Dec']
function fmtDateDaily(d: string): string {
const [, m, day] = d.split('-')
return `${MONTHS[parseInt(m, 10) - 1]} ${parseInt(day, 10)}`
}
function fmtDateHourly(d: string): string {
// "2026-04-10 06:00" → "Apr 10 6am"
const [date, time] = d.split(' ')
const [, m, day] = date.split('-')
const hour = parseInt(time.split(':')[0], 10)
const ampm = hour >= 12 ? 'pm' : 'am'
const h12 = hour === 0 ? 12 : hour > 12 ? hour - 12 : hour
return `${MONTHS[parseInt(m, 10) - 1]} ${parseInt(day, 10)} ${h12}${ampm}`
}
// --- Components ---
const PURPLE = 'hsl(262, 52%, 56%)'
const PURPLE_LIGHT = 'hsl(262, 60%, 65%)'
type Metric = 'cost_saved' | 'requests' | 'tokens_saved'
type TimeRange = 'daily' | 'hourly'
function ToggleButton({ active, onClick, children }: { active: boolean; onClick: () => void; children: React.ReactNode }) {
return (
<button
onClick={onClick}
className={`px-3 py-1 text-xs font-medium rounded-md transition ${
active
? 'bg-fd-primary text-fd-primary-foreground'
: 'bg-fd-muted text-fd-muted-foreground hover:text-fd-foreground'
}`}
>
{children}
</button>
)
}
function CustomTooltip({ active, payload, label }: any) {
if (!active || !payload?.length) return null
return (
<div className="rounded-lg border border-fd-border bg-fd-card px-3 py-2 text-xs shadow-lg">
<p className="font-medium text-fd-foreground mb-1">{label}</p>
{payload.map((p: any) => (
<p key={p.dataKey} className="text-fd-muted-foreground">
{p.dataKey === 'cost_saved' ? fmtUsd(p.value) : fmt(p.value)}
</p>
))}
</div>
)
}
function StatsCards() {
return (
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 not-prose">
{[
{ label: 'Cost Saved', value: fmtUsd(DATA.total_cost_saved) },
{ label: 'Requests Optimized', value: fmt(DATA.total_requests) },
{ label: 'Active Instances', value: fmt(DATA.unique_instances) },
{ label: 'Active Days', value: String(DATA.active_days) },
].map(s => (
<div key={s.label} className="flex flex-col items-center p-5 rounded-xl border border-fd-border bg-fd-card">
<span className="text-2xl font-bold text-fd-foreground">{s.value}</span>
<span className="mt-1 text-sm text-fd-muted-foreground">{s.label}</span>
</div>
))}
</div>
)
}
function AreaChartSection() {
const [metric, setMetric] = useState<Metric>('tokens_saved')
const [timeRange, setTimeRange] = useState<TimeRange>('hourly')
const isHourly = timeRange === 'hourly'
const chartData: { label: string; requests: number; instances: number; cost_saved: number; tokens_saved: number }[] = !isHourly
? DATA.daily_stats.map(d => ({ label: fmtDateDaily(d.date), requests: d.requests, instances: d.instances, cost_saved: d.cost_saved, tokens_saved: d.tokens_saved }))
: DATA.hourly_stats.map(d => ({ label: fmtDateHourly(d.hour), requests: d.requests, instances: d.instances, cost_saved: d.cost_saved, tokens_saved: d.tokens_saved }))
const metricLabels: Record<Metric, string> = {
cost_saved: 'Cost Saved',
requests: 'Requests',
tokens_saved: 'Tokens Saved',
}
const tickFormatter = (v: number) =>
metric === 'cost_saved' ? fmtUsd(v) : fmt(v)
return (
<div className="rounded-xl border border-fd-border bg-fd-card p-5 not-prose">
<div className="flex flex-wrap items-center justify-between gap-3 mb-5">
<div className="flex gap-1">
{(['cost_saved', 'requests', 'tokens_saved'] as Metric[]).map(m => (
<ToggleButton key={m} active={metric === m} onClick={() => setMetric(m)}>
{metricLabels[m]}
</ToggleButton>
))}
</div>
<div className="flex gap-1">
{(['hourly', 'daily'] as TimeRange[]).map(t => (
<ToggleButton key={t} active={timeRange === t} onClick={() => setTimeRange(t)}>
{t === 'hourly' ? 'Last 48h' : 'Daily'}
</ToggleButton>
))}
</div>
</div>
<div className="h-72">
<ResponsiveContainer width="100%" height="100%" minWidth={0} minHeight={0}>
<AreaChart data={chartData} margin={{ left: 0, right: 10, top: 5, bottom: isHourly ? 40 : 5 }}>
<defs>
<linearGradient id="purpleGrad" x1="0" y1="0" x2="0" y2="1">
<stop offset="0%" stopColor={PURPLE} stopOpacity={0.3} />
<stop offset="100%" stopColor={PURPLE} stopOpacity={0} />
</linearGradient>
</defs>
<CartesianGrid strokeDasharray="3 3" stroke="var(--color-fd-border)" />
<XAxis
dataKey="label"
tick={{ fontSize: 9, fill: 'var(--color-fd-muted-foreground)' }}
interval={isHourly ? 3 : 0}
angle={isHourly ? -45 : 0}
textAnchor={isHourly ? 'end' : 'middle'}
height={isHourly ? 60 : 30}
/>
<YAxis tickFormatter={tickFormatter} tick={{ fontSize: 10, fill: 'var(--color-fd-muted-foreground)' }} width={45} />
<Tooltip content={<CustomTooltip />} />
<Area
type="monotone"
dataKey={metric}
stroke={PURPLE}
strokeWidth={2}
fill="url(#purpleGrad)"
/>
</AreaChart>
</ResponsiveContainer>
</div>
</div>
)
}
function BarChartSection() {
const barData = DATA.top_instances.slice(0, 10).map(i => ({
name: `${i.instance_id} (${i.os})`,
tokens_saved: i.tokens_saved,
cost_saved: i.cost_saved,
}))
return (
<div className="rounded-xl border border-fd-border bg-fd-card p-5 not-prose">
<div className="h-80">
<ResponsiveContainer width="100%" height="100%" minWidth={0} minHeight={0}>
<BarChart data={barData} layout="vertical" margin={{ left: 0, right: 10, top: 5, bottom: 5 }}>
<CartesianGrid strokeDasharray="3 3" stroke="var(--color-fd-border)" horizontal={false} />
<XAxis type="number" tickFormatter={fmt} tick={{ fontSize: 10, fill: 'var(--color-fd-muted-foreground)' }} />
<YAxis type="category" dataKey="name" tick={{ fontSize: 9, fill: 'var(--color-fd-muted-foreground)' }} width={85} />
<Tooltip content={({ active, payload }: any) => {
if (!active || !payload?.length) return null
const d = payload[0].payload
return (
<div className="rounded-lg border border-fd-border bg-fd-card px-3 py-2 text-xs shadow-lg">
<p className="font-medium text-fd-foreground">{d.name}</p>
<p className="text-fd-muted-foreground">{fmt(d.tokens_saved)} tokens</p>
<p className="text-fd-muted-foreground">{fmtUsd(d.cost_saved)} saved</p>
</div>
)
}} />
<Bar dataKey="tokens_saved" fill={PURPLE} radius={[0, 4, 4, 0]} />
</BarChart>
</ResponsiveContainer>
</div>
</div>
)
}
function DataTable() {
return (
<div className="rounded-xl border border-fd-border bg-fd-card overflow-hidden not-prose">
<div className="overflow-x-auto">
<table className="w-full text-sm">
<thead>
<tr className="border-b border-fd-border text-left text-fd-muted-foreground">
<th className="px-5 py-2 font-medium">Instance</th>
<th className="px-5 py-2 font-medium">OS</th>
<th className="px-5 py-2 font-medium">Version</th>
<th className="px-5 py-2 font-medium text-right">Tokens Saved</th>
<th className="px-5 py-2 font-medium text-right">Cost Saved</th>
</tr>
</thead>
<tbody>
{DATA.top_instances.map((inst, i) => (
<tr
key={inst.instance_id}
className={`border-b border-fd-border last:border-0 ${i % 2 === 0 ? 'bg-fd-muted/30' : ''}`}
>
<td className="px-5 py-2.5 font-mono text-xs text-fd-foreground">{inst.instance_id}</td>
<td className="px-5 py-2.5 text-fd-muted-foreground">{inst.os}</td>
<td className="px-5 py-2.5 text-fd-muted-foreground">{inst.version}</td>
<td className="px-5 py-2.5 text-right text-fd-foreground font-medium">{fmt(inst.tokens_saved)}</td>
<td className="px-5 py-2.5 text-right text-fd-foreground font-medium">{fmtUsd(inst.cost_saved)}</td>
</tr>
))}
</tbody>
</table>
</div>
</div>
)
}
export function CommunityCharts({ section }: { section?: 'stats' | 'area' | 'bar' | 'table' }) {
if (section === 'stats') return <StatsCards />
if (section === 'area') return <AreaChartSection />
if (section === 'bar') return <BarChartSection />
if (section === 'table') return <DataTable />
// Render all if no section specified
return (
<div className="space-y-10">
<StatsCards />
<AreaChartSection />
<BarChartSection />
<DataTable />
</div>
)
}
@@ -0,0 +1,34 @@
import { fetchCommunityStats, fmtNum, fmtUsd } from '@/lib/telemetry';
/**
* Server component that renders the top-level stats for the community savings page.
* Fetches live data from Supabase at render time.
*/
export async function CommunityStatsHeader() {
const data = await fetchCommunityStats();
const stats = [
{ value: fmtNum(data.total_tokens_saved), label: 'Tokens Saved' },
{ value: fmtUsd(data.total_cost_saved), label: 'Cost Saved' },
{ value: fmtNum(data.total_requests), label: 'Requests Optimized' },
{ value: fmtNum(data.unique_instances), label: 'Active Instances' },
];
return (
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 not-prose">
{stats.map((s) => (
<div
key={s.label}
className="flex flex-col items-center p-5 rounded-xl border border-fd-border bg-fd-card"
>
<span className="text-2xl font-bold text-fd-foreground">
{s.value}
</span>
<span className="mt-1 text-sm text-fd-muted-foreground">
{s.label}
</span>
</div>
))}
</div>
);
}
+43
View File
@@ -0,0 +1,43 @@
import Link from 'next/link';
import { fetchCommunityStats, fmtNum, fmtUsd } from '@/lib/telemetry';
/**
* Server component that fetches live stats from Supabase at render time.
* Falls back to hardcoded data if the API is unreachable.
*/
export async function LiveStats() {
const data = await fetchCommunityStats();
const stats = [
{ value: fmtNum(data.total_tokens_saved), label: 'Tokens Saved' },
{ value: fmtUsd(data.total_cost_saved), label: 'Cost Saved' },
{ value: fmtNum(data.total_requests), label: 'Requests Optimized' },
{ value: fmtNum(data.unique_instances), label: 'Active Instances' },
];
return (
<div className="not-prose">
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 my-8">
{stats.map((s) => (
<div
key={s.label}
className="flex flex-col items-center p-5 rounded-xl border border-fd-border bg-fd-card"
>
<span className="text-2xl font-bold text-fd-foreground">
{s.value}
</span>
<span className="mt-1 text-sm text-fd-muted-foreground">
{s.label}
</span>
</div>
))}
</div>
<Link
href="/docs/community-savings"
className="text-sm font-medium hover:underline"
>
View detailed charts and breakdowns &rarr;
</Link>
</div>
);
}
+95
View File
@@ -0,0 +1,95 @@
'use client'
import DottedMap from 'dotted-map'
import { useEffect, useState } from 'react'
const pins = [
{ lat: 40.73061, lng: -73.935242 },
{ lat: 48.8534, lng: 2.3488 },
{ lat: 51.5074, lng: -0.1278 },
{ lat: 35.6895, lng: 139.6917 },
{ lat: 34.0522, lng: -118.2437 },
{ lat: 55.7558, lng: 37.6173 },
{ lat: 39.9042, lng: 116.4074 },
{ lat: 19.4326, lng: -99.1332 },
{ lat: 37.7749, lng: -122.4194 },
{ lat: -33.8688, lng: 151.2093 },
{ lat: 28.6139, lng: 77.209 },
{ lat: 52.52, lng: 13.405 },
{ lat: 41.9028, lng: 12.4964 },
{ lat: 43.65107, lng: -79.347015 },
{ lat: -23.55052, lng: -46.633308 },
{ lat: 31.2304, lng: 121.4737 },
{ lat: 55.9533, lng: -3.1883 },
{ lat: 35.6762, lng: 139.6503 },
{ lat: 1.3521, lng: 103.8198 },
{ lat: 37.5665, lng: 126.978 },
{ lat: 53.3498, lng: -6.2603 },
{ lat: 30.0444, lng: 31.2357 },
{ lat: 50.4501, lng: 30.5234 },
{ lat: -34.6037, lng: -58.3816 },
{ lat: 59.9343, lng: 30.3351 },
{ lat: 25.276987, lng: 55.296249 },
{ lat: 45.4642, lng: 9.19 },
{ lat: -22.9068, lng: -43.1729 },
{ lat: 40.4168, lng: -3.7038 },
{ lat: 41.3851, lng: 2.1734 },
{ lat: 13.7563, lng: 100.5018 },
{ lat: 52.3676, lng: 4.9041 },
{ lat: -37.8136, lng: 144.9631 },
{ lat: 60.1695, lng: 24.9354 },
{ lat: 47.4979, lng: 19.0402 },
{ lat: 59.3293, lng: 18.0686 },
{ lat: 35.9078, lng: 127.7669 },
{ lat: 46.2044, lng: 6.1432 },
{ lat: 29.7604, lng: -95.3698 },
{ lat: 39.7392, lng: -104.9903 },
{ lat: -11.6647, lng: 27.4794 },
{ lat: -10.7026, lng: 25.5122 },
{ lat: -4.4419, lng: 15.2663 },
]
function buildSvg(isDark: boolean) {
const map = new DottedMap({ height: 55, grid: 'diagonal' })
pins.forEach((pin) => {
map.addPin({
...pin,
svgOptions: {
color: isDark
? '#a78bfa' // bright purple on #050505
: '#7c3aed', // vivid purple on #FAFAFA
radius: 0.4,
},
})
})
return map.getSVG({
radius: 0.22,
color: isDark
? '#555555' // medium gray on #050505
: '#a0a0a0', // medium gray on #FAFAFA
shape: 'circle',
backgroundColor: 'transparent',
})
}
export const Map = () => {
const [isDark, setIsDark] = useState(false)
useEffect(() => {
const check = () => setIsDark(document.documentElement.classList.contains('dark'))
check()
const observer = new MutationObserver(check)
observer.observe(document.documentElement, { attributes: true, attributeFilter: ['class'] })
return () => observer.disconnect()
}, [])
const svgMap = buildSvg(isDark)
return (
<img
src={`data:image/svg+xml;utf8,${encodeURIComponent(svgMap)}`}
alt="map illustration"
/>
)
}
+224
View File
@@ -0,0 +1,224 @@
import Link from 'next/link';
import { Button } from './button';
import { CodeBlock } from './code-block';
// --- Live Stats Grid ---
const liveStats = [
{ value: '$176.6K', label: 'Cost Saved' },
{ value: '1.19M', label: 'Requests Optimized' },
{ value: '889', label: 'Active Instances' },
{ value: '14', label: 'Active Days' },
];
export function LiveStats() {
return (
<div className="not-prose">
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 my-8">
{liveStats.map((s) => (
<div
key={s.label}
className="flex flex-col items-center p-5 rounded-xl border border-fd-border bg-fd-card"
>
<span className="text-2xl font-bold text-fd-foreground">
{s.value}
</span>
<span className="mt-1 text-sm text-fd-muted-foreground">
{s.label}
</span>
</div>
))}
</div>
<Link
href="/docs/community-savings"
className="text-sm font-medium hover:underline"
>
View detailed charts and breakdowns &rarr;
</Link>
</div>
);
}
// --- Key Features Grid ---
const features: {
title: string;
description: string;
href: string;
code?: string;
lang?: string;
}[] = [
{
title: 'Lossless Compression (CCR)',
description:
'Compresses aggressively, stores originals, gives the LLM a tool to retrieve full details. Nothing is thrown away.',
href: '/docs/ccr',
},
{
title: 'Smart Content Detection',
description:
'Auto-detects JSON, code, logs, text, diffs, HTML. Routes each to the best compressor. Zero configuration needed.',
href: '/docs/how-compression-works',
},
{
title: 'Cache Optimization',
description:
"Stabilizes prefixes so provider KV caches hit. Tracks frozen messages to preserve the 90% read discount.",
href: '/docs/cache-optimization',
},
{
title: 'Image Compression',
description:
'40-90% token reduction via trained ML router. Automatically selects resize/quality tradeoff per image.',
href: '/docs/image-compression',
},
{
title: 'Persistent Memory',
description:
'Hierarchical memory (user/session/agent/turn) with SQLite + HNSW backends. Survives across conversations.',
href: '/docs/memory',
},
{
title: 'Failure Learning',
description:
'Reads past sessions, finds failed tool calls, correlates with what succeeded, writes learnings to CLAUDE.md.',
href: '/docs/failure-learning',
},
{
title: 'Multi-Agent Context',
description: 'Compress what moves between agents. Any framework.',
href: '/docs/shared-context',
code: 'ctx = SharedContext()\nctx.put("research", big_output)\nsummary = ctx.get("research")',
lang: 'python',
},
{
title: 'Metrics & Observability',
description:
'Prometheus endpoint, per-request logging, cost tracking, budget limits, pipeline timing breakdowns.',
href: '/docs/metrics',
},
];
export async function KeyFeatures() {
return (
<div className="grid grid-cols-1 md:grid-cols-2 gap-4 my-8 not-prose">
{await Promise.all(
features.map(async (f) => (
<div
key={f.title}
className="flex flex-col p-5 rounded-xl border border-fd-border bg-fd-card"
>
<h3 className="text-base font-semibold text-fd-foreground">
{f.title}
</h3>
<p className="mt-2 text-sm text-fd-muted-foreground flex-1">
{f.description}
</p>
{f.code && <CodeBlock code={f.code} lang={f.lang} />}
<Link
href={f.href}
className="mt-3 text-sm font-medium hover:underline"
>
Learn more &rarr;
</Link>
</div>
)),
)}
</div>
);
}
// --- Framework Integrations Bento ---
const integrations: {
title: string;
description: string;
code: string;
lang: string;
href: string;
}[] = [
{
title: 'LangChain',
description:
'Wrap any chat model. Supports memory, retrievers, tools, streaming, async.',
code: 'from headroom.integrations.langchain import HeadroomChatModel\nllm = HeadroomChatModel(ChatOpenAI())',
lang: 'python',
href: '/docs/langchain',
},
{
title: 'Agno',
description:
'Full agent framework integration with observability hooks.',
code: 'from headroom.integrations.agno import HeadroomAgnoModel\nmodel = HeadroomAgnoModel(Claude())\nagent = Agent(model=model)',
lang: 'python',
href: '/docs/agno',
},
{
title: 'Strands',
description:
'Model wrapping + tool output hook provider for Strands Agents.',
code: 'from headroom.integrations.strands import HeadroomStrandsModel\nmodel = HeadroomStrandsModel(...)\nagent = Agent(model=model)',
lang: 'python',
href: '/docs/strands',
},
{
title: 'MCP Tools',
description:
'Three tools for Claude Code, Cursor, or any MCP client: headroom_compress, headroom_retrieve, headroom_stats.',
code: 'headroom mcp install && claude',
lang: 'bash',
href: '/docs/mcp',
},
{
title: 'TypeScript SDK',
description:
'compress(), Vercel AI SDK middleware, OpenAI and Anthropic client wrappers.',
code: 'npm install headroom-ai',
lang: 'bash',
href: '/docs/vercel-ai-sdk',
},
{
title: 'Vercel AI SDK',
description:
'One-liner withHeadroom() or headroomMiddleware() for any Vercel AI SDK model.',
code: "import { withHeadroom } from 'headroom-ai/vercel-ai'\nconst model = withHeadroom(openai('gpt-4o'))",
lang: 'typescript',
href: '/docs/vercel-ai-sdk',
},
];
export async function FrameworkIntegrations() {
return (
<div className="not-prose">
<div className="grid grid-cols-1 md:grid-cols-2 gap-4 my-8">
{await Promise.all(
integrations.map(async (i) => (
<div
key={i.title}
className="flex flex-col p-5 rounded-xl border border-fd-border bg-fd-card"
>
<h3 className="text-base font-semibold text-fd-foreground">
{i.title}
</h3>
<p className="mt-2 text-sm text-fd-muted-foreground flex-1">
{i.description}
</p>
<CodeBlock code={i.code} lang={i.lang} />
<Link
href={i.href}
className="mt-3 text-sm font-medium hover:underline"
>
{i.title} Guide &rarr;
</Link>
</div>
)),
)}
</div>
<Button variant="link" size="sm" asChild>
<Link href="/docs/quickstart">
All integration patterns &rarr;
</Link>
</Button>
</div>
);
}
+43
View File
@@ -0,0 +1,43 @@
import defaultMdxComponents from 'fumadocs-ui/mdx';
import type { MDXComponents } from 'mdx/types';
import * as Twoslash from 'fumadocs-twoslash/ui';
import { AutoTypeTable, type AutoTypeTableProps } from 'fumadocs-typescript/ui';
import { createGenerator } from 'fumadocs-typescript';
import { TypeTable } from 'fumadocs-ui/components/type-table';
import { Tab, Tabs } from 'fumadocs-ui/components/tabs';
import {
KeyFeatures,
FrameworkIntegrations,
} from './marketing';
import { LiveStats } from './live-stats';
import { CommunityStatsHeader } from './community-stats-header';
import { StatsSection } from './stats';
import { CommunityCharts } from './community-charts';
const generator = createGenerator();
export function getMDXComponents(components?: MDXComponents) {
return {
...defaultMdxComponents,
...Twoslash,
AutoTypeTable: (props: Partial<AutoTypeTableProps>) => (
<AutoTypeTable {...props} generator={generator} />
),
TypeTable,
Tab,
Tabs,
StatsSection,
CommunityCharts,
CommunityStatsHeader,
LiveStats,
KeyFeatures,
FrameworkIntegrations,
...components,
} satisfies MDXComponents;
}
export const useMDXComponents = getMDXComponents;
declare global {
type MDXProvidedComponents = ReturnType<typeof getMDXComponents>;
}
+71
View File
@@ -0,0 +1,71 @@
import { Map } from '@/components/map'
export function StatsSection() {
return (
<section className="@container relative py-12 md:py-20 not-prose overflow-hidden">
<div className="mask-radial-to-75% absolute inset-0 max-md:hidden flex items-center justify-center">
<div className="w-[140%] min-w-[900px]">
<Map />
</div>
</div>
<div className="mx-auto max-w-5xl px-6">
<div className="md:max-w-3/5 lg:max-w-1/2 bg-fd-card ring-fd-border shadow-black/6.5 relative rounded-xl p-6 shadow-xl ring-1 sm:p-10">
<div className="mb-8 space-y-4">
<h2 className="text-fd-muted-foreground text-balance text-3xl font-semibold">
The Context Optimization Layer for{' '}
<strong className="text-fd-foreground font-semibold">
LLM Applications
</strong>
</h2>
<p className="text-fd-muted-foreground">
Compress everything your AI agent reads.{' '}
<strong className="text-fd-foreground font-semibold">
Same answers, fraction of the tokens.
</strong>
</p>
</div>
<div className="**:text-center *:bg-fd-muted/50 grid grid-cols-2 gap-1 *:rounded-md *:p-4">
<div className="space-y-2 *:block">
<span className="text-3xl font-semibold">
87 <span className="text-fd-muted-foreground text-lg">%</span>
</span>
<p className="text-fd-muted-foreground text-xs">
<strong className="text-fd-foreground font-medium">
Token Reduction
</strong>
</p>
</div>
<div className="space-y-2 *:block">
<span className="text-3xl font-semibold">
100 <span className="text-fd-muted-foreground text-lg">%</span>
</span>
<p className="text-fd-muted-foreground text-xs">
<strong className="text-fd-foreground font-medium">
Accuracy
</strong>
</p>
</div>
<div className="space-y-2 *:block">
<span className="text-3xl font-semibold">6</span>
<p className="text-fd-muted-foreground text-xs">
<strong className="text-fd-foreground font-medium">
Algorithms
</strong>
</p>
</div>
<div className="space-y-2 *:block">
<span className="text-3xl font-semibold">
100 <span className="text-fd-muted-foreground text-lg">+</span>
</span>
<p className="text-fd-muted-foreground text-xs">
<strong className="text-fd-foreground font-medium">
Providers
</strong>
</p>
</div>
</div>
</div>
</div>
</section>
)
}
+125
View File
@@ -0,0 +1,125 @@
---
title: Agent Orchestration
description: Keep repeated agent wakes cache-friendly while using CCR for lossless memory digest retrieval.
---
Repeated agent wakes usually rebuild the same expensive prompt shape. The useful split is simple: keep the byte-stable prefix first, keep the volatile wake-specific digest later, and keep run-specific context at the end so the repeated prefix stays cacheable.
## Repeated wake anatomy
Each wake tends to carry four different layers:
- stable instructions, project rules, and tool contracts
- the current task and other instruction-bearing fields
- the volatile per-wake memory digest
- recent loop state, tool output, and child-agent context
The stable layers should stay byte-identical across wakes. The volatile layers should move to the live zone, where they can change without invalidating the cacheable prefix.
## CacheAligner is detector-only
CacheAligner does not rewrite messages. It inspects the prefix, emits warnings for volatile content, and records observability data so callers can fix their own assembly logic.
The detector reports:
| Field | What it tells you |
|---|---|
| `warnings` | Which parts of the prefix look unstable |
| `cache_metrics.stable_prefix_bytes` | Size of the stable prefix in bytes |
| `cache_metrics.stable_prefix_tokens_est` | Estimated token size of the stable prefix |
| `cache_metrics.stable_prefix_hash` | Hash of the stable prefix for repeated-wake comparison |
| `cache_metrics.prefix_changed` | Whether the stable prefix drifted since the last wake |
| `cache_metrics.previous_hash` | Hash from the previous wake, when available |
| `markers` | The emitted `stable_prefix_hash` marker for downstream observability |
If CacheAligner warns about drift, keep the prefix stable in the caller. The transform is a detector, not a repair pass.
## Stable prefix layout
Put the byte-stable prefix first, then the live wake digest, then any run-specific context.
That layout keeps the provider cache path predictable:
1. Stable instructions stay identical.
2. The wake digest changes without disturbing the cacheable prefix.
3. Recent tool output and child-agent context stay outside the repeatable prefix.
## Real wake measurements
Use the issue comment guidance when measuring real wakes:
| Field | Record |
|---|---|
| Prompt section id | Name the section that changed or repeated |
| Byte/token estimate | Capture the section size before and after curation |
| Digest version | Track which digest schema was used |
| Cache-hit expectation | Note whether the section should stay cacheable |
| Compressed size | Record the compacted payload size |
| Section type | Mark the section as instruction-bearing or safe to compress |
That checklist helps separate the repeated prefix from the volatile digest before you decide where CCR belongs.
## CCR digest curation
CCR keeps compression reversible. The compressed digest can stay compact while the original backing detail remains recoverable from the local store.
The pieces that matter here are:
- `headroom_retrieve` for on-demand recovery of stored originals
- `HEADROOM_CCR_TTL_SECONDS` for sizing the local store lifetime
- `compression_strategy` as the authoritative discriminator on stored CCR entries
For routing decisions, the same rule in plain terms is: headroom_retrieve recovers originals, HEADROOM_CCR_TTL_SECONDS sizes the local lifetime, compression_strategy identifies the producing path, and shape inference is not the routing authority.
When a stored original expires, regenerate the digest or re-read the source content. Do not infer routing from payload shape. Use the stored `compression_strategy` metadata to understand how the original was produced.
## Digest routing
Route fields by how much exact wording they need at wake time.
| Field | Suggested handling | Why |
|---|---|---|
| Current task | Verbatim | It is instruction-bearing and changes the next action |
| Hard constraints | Verbatim | These are safety and acceptance boundaries |
| Definitions of done | Verbatim | The wording needs to survive every wake intact |
| Irreversible decisions | Verbatim summary plus retrievable backing detail | The summary stays short, the backing detail stays exact |
| Open threads | Compact summary plus CCR-backed backing detail | The live thread can shrink while the source stays recoverable |
| Learnings | Compact summary | They inform the next wake without needing exact prose |
| File/search/tool outputs | CCR-backed compression | These are large backing details that should stay retrievable |
| Prose notes | Compact summary, CCR-backed when bulky | Keep the digest readable without losing source detail |
| Bulk JSON-ish arrays | SmartCrusher or ContentRouter, with CCR-backed backing detail when needed | Structured blobs are usually the first thing to explode in size |
The important boundary is simple: instruction-bearing fields stay verbatim, or they get a verbatim compact summary plus retrievable backing detail. CCR-backed content is the backing detail, not the instruction itself.
Hard constraints stay verbatim; current task text stays verbatim; file/search/tool outputs use CCR-backed backing detail when they are large enough to compress.
## Integration modes
Choose the integration mode by where the orchestrator controls message assembly.
| Mode | Use when | Notes |
|---|---|---|
| Proxy | You want spawned agents and normal client traffic to pass through a local provider endpoint | Good for `headroom proxy --mode cache` and provider base URL routing |
| Library | The orchestrator owns message assembly and wants to shape the digest before launch | Best when the caller can decide what becomes stable prefix versus live digest |
| MCP | Agents need on-demand compression and retrieval tools | Best when `headroom_retrieve` should be available as a tool |
| Proxy plus MCP | You need traffic shaping and tool-level retrieval together | Useful when both the provider edge and the agent toolset matter |
## Local-first deployment
Keep the deployment local to the user or session:
- run one local proxy or MCP process per user session
- do not assume a central proxy
- treat the local CCR store as process-local unless the deployment explicitly shares it
- size the TTL for the longest realistic autonomous run
- assume a store can expire before the run finishes, then regenerate or re-read the source content
That model keeps the data boundary obvious. The orchestrator can still recover backing detail, but the cacheable prefix stays small and stable.
## Source-backed caveats
- Prompt-cache hits require a byte-identical stable prefix.
- CacheAligner identifies drift, it does not repair prompt assembly.
- CCR retrieval depends on store lifetime and stored hashes.
- Provider-neutral guidance still applies, so keep the guide away from Orcha-specific runtime claims.
- Use `compression_strategy` to read stored CCR intent, not payload shape.
+154
View File
@@ -0,0 +1,154 @@
---
title: Agno
description: Automatic context compression for Agno AI agents with model wrapping and observability hooks.
---
Headroom integrates with [Agno](https://github.com/agno-agi/agno) (formerly Phidata) to compress context for AI agents. Wrap any Agno model for automatic optimization, and use hooks for observability.
## Installation
```bash
pip install "headroom-ai[agno]" agno
```
## Quick start
```python
from agno.agent import Agent
from agno.models.openai import OpenAIChat
from headroom.integrations.agno import HeadroomAgnoModel
model = HeadroomAgnoModel(OpenAIChat(id="gpt-4o"))
agent = Agent(model=model)
response = agent.run("What's the capital of France?")
print(f"Tokens saved: {model.total_tokens_saved}")
print(model.get_savings_summary())
# {'total_requests': 1, 'total_tokens_saved': 245, 'average_savings_percent': 12.3}
```
Works with any Agno provider:
```python
from agno.models.anthropic import Claude
from agno.models.google import Gemini
claude_model = HeadroomAgnoModel(Claude(id="claude-sonnet-4-20250514"))
gemini_model = HeadroomAgnoModel(Gemini(id="gemini-2.0-flash"))
```
## Observability hooks
Use hooks for detailed tracking without modifying your model:
```python
from headroom.integrations.agno import (
HeadroomAgnoModel,
HeadroomPreHook,
HeadroomPostHook,
)
model = HeadroomAgnoModel(OpenAIChat(id="gpt-4o"))
pre_hook = HeadroomPreHook()
post_hook = HeadroomPostHook(token_alert_threshold=10000)
agent = Agent(
model=model,
pre_hooks=[pre_hook],
post_hooks=[post_hook],
)
response = agent.run("Analyze this large dataset...")
# Check for alerts
if post_hook.alerts:
print(f"{len(post_hook.alerts)} requests exceeded threshold")
```
Or use the convenience factory:
```python
from headroom.integrations.agno import create_headroom_hooks
pre_hook, post_hook = create_headroom_hooks(
token_alert_threshold=5000,
log_level="DEBUG",
)
```
## Tool-heavy agents
Tool outputs (JSON, logs, search results) see the biggest compression gains at 70-90% reduction:
```python
from agno.tools.duckduckgo import DuckDuckGoTools
model = HeadroomAgnoModel(OpenAIChat(id="gpt-4o"))
agent = Agent(
model=model,
tools=[DuckDuckGoTools()],
show_tool_calls=True,
)
response = agent.run("Research the latest AI developments")
print(f"Tokens saved: {model.total_tokens_saved}")
```
## Async support
```python
import asyncio
async def process():
model = HeadroomAgnoModel(OpenAIChat(id="gpt-4o"))
response = await model.aresponse(messages)
async for chunk in model.aresponse_stream(messages):
print(chunk, end="", flush=True)
asyncio.run(process())
```
## Standalone message optimization
Optimize messages without wrapping a model:
```python
from headroom.integrations.agno import optimize_messages
optimized, metrics = optimize_messages(messages, model="gpt-4o")
print(f"Tokens saved: {metrics['tokens_saved']}")
```
## Session management
Reset metrics between sessions:
```python
model = HeadroomAgnoModel(OpenAIChat(id="gpt-4o"))
# Session 1
agent.run("First conversation...")
print(model.get_savings_summary())
# Reset for new session
model.reset()
# Session 2 starts fresh
agent.run("Second conversation...")
```
## Supported providers
| Provider | Agno Model | Auto-Detected |
|----------|-----------|---------------|
| OpenAI | `OpenAIChat`, `OpenAILike` | Yes |
| Anthropic | `Claude`, `AwsBedrock` | Yes |
| Google | `Gemini`, `VertexAI` | Yes |
| Groq | `Groq` | Yes |
| Mistral | `Mistral` | Yes |
| Ollama | `Ollama` | Yes |
+127
View File
@@ -0,0 +1,127 @@
---
title: Anthropic SDK
description: Auto-compress messages in the Anthropic TypeScript SDK with a single withHeadroom() wrapper.
---
Headroom wraps the Anthropic TypeScript SDK to automatically compress messages before every `messages.create()` call. All other methods pass through unchanged.
## Installation
```bash
npm install headroom-ai @anthropic-ai/sdk
```
<Callout type="info" title="Proxy required">
The TypeScript SDK sends messages to a local Headroom proxy for compression. Start the proxy before using the SDK:
```bash
pip install "headroom-ai[proxy]"
headroom proxy
```
</Callout>
## Quick start
```ts twoslash
import { withHeadroom } from 'headroom-ai/anthropic';
import Anthropic from '@anthropic-ai/sdk';
const client = withHeadroom(new Anthropic());
const response = await client.messages.create({
model: 'claude-sonnet-4-5-20250929',
messages: longConversation,
max_tokens: 1024,
});
```
Every call to `client.messages.create()` compresses messages first. The response format is identical to the unwrapped client.
## How it works
`withHeadroom()` returns a proxy around your Anthropic client that intercepts `messages.create()`:
1. Converts Anthropic-format messages to OpenAI format (the compression engine's native format)
2. Sends them to the Headroom proxy's `/v1/compress` endpoint
3. Converts the compressed messages back to Anthropic format
4. Forwards the request to Anthropic as normal
### Message format conversion
The adapter handles the full Anthropic message format including content blocks:
| Anthropic format | OpenAI format |
|-----------------|---------------|
| `{ type: "text", text: "..." }` | `{ role: "user", content: "..." }` |
| `{ type: "tool_use", id, name, input }` | `{ tool_calls: [{ id, function: { name, arguments } }] }` |
| `{ type: "tool_result", tool_use_id, content }` | `{ role: "tool", tool_call_id, content }` |
This conversion is lossless. Your request and response behave identically to an unwrapped client.
## Options
Pass compression options as the second argument:
```ts twoslash
import { withHeadroom } from 'headroom-ai/anthropic';
import Anthropic from '@anthropic-ai/sdk';
const client = withHeadroom(new Anthropic(), {
model: 'claude-sonnet-4-5-20250929',
baseUrl: 'http://localhost:8787',
});
```
## Streaming
Streaming works normally. Compression happens before the request:
```ts twoslash
import { withHeadroom } from 'headroom-ai/anthropic';
import Anthropic from '@anthropic-ai/sdk';
const client = withHeadroom(new Anthropic());
const stream = await client.messages.create({
model: 'claude-sonnet-4-5-20250929',
messages: longConversation,
max_tokens: 1024,
stream: true,
});
```
## Tool use
Tool results are where compression has the biggest impact. Large JSON payloads from tool calls are compressed automatically:
```ts twoslash
import { withHeadroom } from 'headroom-ai/anthropic';
import Anthropic from '@anthropic-ai/sdk';
const client = withHeadroom(new Anthropic());
const response = await client.messages.create({
model: 'claude-sonnet-4-5-20250929',
max_tokens: 1024,
messages: [
{ role: 'user', content: 'What went wrong?' },
{
role: 'assistant',
content: [
{ type: 'tool_use', id: 'toolu_1', name: 'get_logs', input: { service: 'api' } },
],
},
{
role: 'user',
content: [
{
type: 'tool_result',
tool_use_id: 'toolu_1',
content: hugeLogOutput, // Compressed automatically
},
],
},
],
tools: [{ name: 'get_logs', description: 'Get logs', input_schema: { type: 'object', properties: {} } }],
});
```
+637
View File
@@ -0,0 +1,637 @@
---
title: API Reference
description: Complete API reference for the Headroom Python and TypeScript SDKs. Core client, configuration types, result types, errors, and utilities.
---
Complete API reference for the Headroom Python and TypeScript SDKs.
## Core
### HeadroomClient
The main entry point for the Headroom SDK.
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
<TypeTable type={{
baseUrl: { type: 'string', description: 'Base URL for the Headroom proxy' },
apiKey: { type: 'string', description: 'API key for authentication' },
timeout: { type: 'number', description: 'Request timeout in milliseconds' },
fallback: { type: 'boolean', description: 'Return original messages on failure instead of throwing' },
retries: { type: 'number', description: 'Number of retry attempts on failure' },
}} />
```ts twoslash
import { HeadroomClient } from 'headroom-ai';
const client = new HeadroomClient({
baseUrl: 'http://localhost:8787',
apiKey: 'your-api-key',
timeout: 30_000,
fallback: true,
retries: 2,
});
```
</Tab>
<Tab value="Python">
**Constructor Parameters**
<TypeTable type={{
original_client: { type: 'OpenAI | Anthropic', description: 'The underlying LLM client', default: 'Required' },
provider: { type: 'Provider', description: 'Token counting provider', default: 'Auto-detected' },
default_mode: { type: '"audit" | "optimize"', description: 'Default compression mode', default: '"audit"' },
store_url: { type: 'str | None', description: 'Storage URL for metrics database', default: 'None' },
smart_crusher_config: { type: 'SmartCrusherConfig', description: 'Compression settings', default: 'Default config' },
cache_aligner_config: { type: 'CacheAlignerConfig', description: 'Cache alignment settings', default: 'Default config' },
enable_cache_optimizer: { type: 'bool', description: 'Enable provider-specific cache optimization', default: 'True' },
enable_semantic_cache: { type: 'bool', description: 'Enable query-level semantic caching', default: 'False' },
model_context_limits: { type: 'dict[str, int]', description: 'Override context limits per model', default: '{}' },
}} />
```python
from headroom import HeadroomClient, OpenAIProvider
from openai import OpenAI
client = HeadroomClient(
original_client=OpenAI(),
provider=OpenAIProvider(),
default_mode="optimize",
)
```
</Tab>
</Tabs>
### chat.completions.create()
Create a chat completion with optional optimization.
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
The TypeScript SDK uses `compress()` to optimize messages before sending them to your LLM client:
```ts twoslash
import { compress } from 'headroom-ai';
const result = await compress(messages, {
model: 'gpt-4o',
tokenBudget: 100_000,
});
// Then pass result.messages to your LLM client
```
</Tab>
<Tab value="Python">
Accepts all standard OpenAI/Anthropic parameters plus Headroom-specific overrides:
<TypeTable type={{
headroom_mode: { type: '"audit" | "optimize" | "simulate"', description: 'Override mode for this request', default: 'Client default' },
headroom_query: { type: 'str', description: 'Query for relevance scoring', default: 'None' },
headroom_output_buffer_tokens: { type: 'int', description: 'Reserve tokens for output', default: '4000' },
headroom_keep_turns: { type: 'int', description: 'Keep last N turns uncompressed', default: '2' },
headroom_tool_profiles: { type: 'dict', description: 'Per-tool compression overrides', default: '{}' },
}} />
```python
response = client.chat.completions.create(
model="gpt-4o",
messages=[...],
headroom_mode="optimize",
headroom_keep_turns=5,
headroom_tool_profiles={
"important_tool": {"skip_compression": True},
},
)
```
</Tab>
</Tabs>
### chat.completions.simulate()
Preview optimization without making an API call.
```python
plan = client.chat.completions.simulate(
model="gpt-4o",
messages=[...],
)
print(f"Tokens: {plan.tokens_before} -> {plan.tokens_after}")
print(f"Savings: {plan.savings_percent:.1f}%")
print(f"Transforms: {plan.transforms_applied}")
```
**Returns:** `SimulationResult`
### compress() (TypeScript)
Top-level function to compress messages via the Headroom proxy.
<TypeTable type={{
model: { type: 'string', description: 'Model name for token counting and context limits' },
baseUrl: { type: 'string', description: 'Base URL for the Headroom proxy' },
apiKey: { type: 'string', description: 'API key for authentication' },
timeout: { type: 'number', description: 'Request timeout in milliseconds' },
fallback: { type: 'boolean', description: 'Return original messages on failure instead of throwing' },
retries: { type: 'number', description: 'Number of retry attempts on failure' },
client: { type: 'HeadroomClientInterface', description: 'Pre-configured client instance to use' },
tokenBudget: { type: 'number', description: 'Token budget — compress to fit within this limit' },
hooks: { type: 'CompressionHooks', description: 'Compression hooks for pre/post processing' },
}} />
```ts twoslash
import { compress } from 'headroom-ai';
const result = await compress(messages, {
model: 'gpt-4o',
baseUrl: 'http://localhost:8787',
timeout: 15_000,
fallback: true,
retries: 2,
tokenBudget: 100_000,
});
```
### get_stats()
Quick stats for the current session (no database query).
```python
stats = client.get_stats()
# Returns dict with "session", "config", and "transforms" keys
```
### get_metrics()
Query stored metrics from the database.
```python
from datetime import datetime, timedelta
metrics = client.get_metrics(
start_time=datetime.utcnow() - timedelta(hours=1),
limit=100,
)
```
### get_summary()
Aggregate statistics across all stored metrics.
```python
summary = client.get_summary()
# Returns dict with total_requests, total_tokens_saved,
# avg_compression_ratio, total_cost_saved_usd
```
### validate_setup()
Validate that the client is configured correctly.
```python
result = client.validate_setup()
if not result["valid"]:
for issue in result["issues"]:
print(f" - {issue}")
```
---
## Configuration
### SmartCrusherConfig
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
<TypeTable type={{
enabled: { type: 'boolean', description: 'Enable/disable the smart crusher' },
minItemsToAnalyze: { type: 'number', description: 'Minimum items before analyzing for compression' },
minTokensToCrush: { type: 'number', description: 'Minimum tokens before applying compression' },
varianceThreshold: { type: 'number', description: 'Variance threshold for analysis' },
uniquenessThreshold: { type: 'number', description: 'Uniqueness threshold for deduplication' },
similarityThreshold: { type: 'number', description: 'Similarity threshold for grouping' },
maxItemsAfterCrush: { type: 'number', description: 'Maximum items to keep after compression' },
preserveChangePoints: { type: 'boolean', description: 'Preserve change points in data' },
useFeedbackHints: { type: 'boolean', description: 'Use feedback hints for scoring' },
toinConfidenceThreshold: { type: 'number', description: 'TOIN confidence threshold' },
relevance: { type: 'RelevanceScorerConfig', description: 'Relevance scoring configuration' },
anchor: { type: 'AnchorConfig', description: 'Anchor selection configuration' },
dedupIdenticalItems: { type: 'boolean', description: 'Deduplicate identical items' },
firstFraction: { type: 'number', description: 'Fraction of items to keep from the start' },
lastFraction: { type: 'number', description: 'Fraction of items to keep from the end' },
}} />
</Tab>
<Tab value="Python">
<TypeTable type={{
min_tokens_to_crush: { type: 'int', description: 'Minimum tokens before applying compression', default: '200' },
min_items_to_analyze: { type: 'int', description: 'Minimum items before analyzing for compression', default: '5' },
max_items_after_crush: { type: 'int', description: 'Maximum items to keep after compression', default: '15' },
variance_threshold: { type: 'float', description: 'Variance threshold for analysis', default: '2.0' },
uniqueness_threshold: { type: 'float', description: 'Uniqueness threshold for deduplication', default: '0.1' },
similarity_threshold: { type: 'float', description: 'Similarity threshold for grouping', default: '0.8' },
preserve_change_points: { type: 'bool', description: 'Preserve significant change points in data', default: 'True' },
use_feedback_hints: { type: 'bool', description: 'Use TOIN feedback hints for scoring', default: 'True' },
dedup_identical_items: { type: 'bool', description: 'Deduplicate identical items', default: 'True' },
first_fraction: { type: 'float', description: 'Fraction of items to keep from the start', default: '0.3' },
last_fraction: { type: 'float', description: 'Fraction of items to keep from the end', default: '0.15' },
}} />
```python
from headroom import SmartCrusherConfig
config = SmartCrusherConfig(
min_tokens_to_crush=200,
max_items_after_crush=15,
variance_threshold=2.0,
preserve_change_points=True,
)
```
</Tab>
</Tabs>
### CacheAlignerConfig
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
<TypeTable type={{
enabled: { type: 'boolean', description: 'Enable/disable cache alignment' },
useDynamicDetector: { type: 'boolean', description: 'Use dynamic content detector' },
detectionTiers: { type: '("regex" | "ner" | "semantic")[]', description: 'Detection tiers to apply' },
extraDynamicLabels: { type: 'string[]', description: 'Additional labels for dynamic content detection' },
entropyThreshold: { type: 'number', description: 'Entropy threshold for dynamic detection' },
datePatterns: { type: 'string[]', description: 'Regex patterns for date extraction' },
normalizeWhitespace: { type: 'boolean', description: 'Normalize whitespace for stable prefix' },
collapseBlankLines: { type: 'boolean', description: 'Collapse consecutive blank lines' },
dynamicTailSeparator: { type: 'string', description: 'Separator between static and dynamic content' },
}} />
</Tab>
<Tab value="Python">
<TypeTable type={{
enabled: { type: 'bool', description: 'Enable/disable cache alignment (off by default)', default: 'False' },
extract_dates: { type: 'bool', description: 'Extract date patterns from system prompt', default: 'True' },
normalize_whitespace: { type: 'bool', description: 'Normalize whitespace for stable prefix', default: 'True' },
stable_prefix_min_tokens: { type: 'int', description: 'Minimum prefix tokens for caching', default: '100' },
dynamic_patterns: { type: 'list[str]', description: 'Regex patterns to extract as dynamic content', default: '[]' },
}} />
```python
from headroom import CacheAlignerConfig
config = CacheAlignerConfig(
enabled=True,
extract_dates=True,
normalize_whitespace=True,
stable_prefix_min_tokens=100,
)
```
</Tab>
</Tabs>
### Context management
Context management is now handled automatically inside the pipeline (live-zone-only compression). Headroom never drops messages from the conversation history; it compresses only the newest content blocks (latest user message, latest tool result) and keeps the cache hot zone — system prompt, tools, and older turns — untouched. Use the `headroom_keep_turns` / `headroom_output_buffer_tokens` per-request overrides to tune behavior. The `RollingWindowConfig`, `IntelligentContextConfig`, and `ScoringWeights` classes are no longer part of Headroom.
### HeadroomConfig
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
<TypeTable type={{
storeUrl: { type: 'string', description: 'Storage URL for metrics database' },
defaultMode: { type: 'HeadroomMode', description: 'Default compression mode' },
modelContextLimits: { type: 'Record<string, number>', description: 'Override context limits per model' },
smartCrusher: { type: 'SmartCrusherConfig', description: 'Smart crusher configuration' },
cacheAligner: { type: 'CacheAlignerConfig', description: 'Cache aligner configuration' },
cacheOptimizer: { type: 'CacheOptimizerConfig', description: 'Cache optimizer configuration' },
ccr: { type: 'CCRConfig', description: 'CCR (Compress-Cache-Retrieve) configuration' },
prefixFreeze: { type: 'PrefixFreezeConfig', description: 'Prefix freeze configuration' },
contentRouterEnabled: { type: 'boolean', description: 'Enable content-type routing' },
generateDiffArtifact: { type: 'boolean', description: 'Generate diff artifacts for debugging' },
}} />
</Tab>
<Tab value="Python">
The top-level config object that contains all sub-configurations:
```python
from headroom import HeadroomConfig
config = HeadroomConfig()
config.smart_crusher.min_tokens_to_crush = 100
config.cache_aligner.enabled = True
# Note: rolling_window has been removed — use headroom_keep_turns per-request instead
```
</Tab>
</Tabs>
### RelevanceScorerConfig
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
<TypeTable type={{
tier: { type: 'RelevanceTier', description: 'Scoring method: "bm25" | "embedding" | "hybrid"' },
bm25K1: { type: 'number', description: 'BM25 k1 parameter' },
bm25B: { type: 'number', description: 'BM25 b parameter' },
embeddingModel: { type: 'string', description: 'Model name for embedding scorer' },
hybridAlpha: { type: 'number', description: 'Weight for hybrid scoring (0=embedding, 1=bm25)' },
adaptiveAlpha: { type: 'boolean', description: 'Automatically adapt alpha based on query' },
relevanceThreshold: { type: 'number', description: 'Minimum relevance score to keep' },
}} />
</Tab>
<Tab value="Python">
<TypeTable type={{
scorer_type: { type: '"bm25" | "embedding" | "hybrid"', description: 'Scoring method', default: '"bm25"' },
embedding_model: { type: 'str | None', description: 'Model name for embedding scorer', default: 'None' },
hybrid_alpha: { type: 'float', description: 'Weight for hybrid scoring (0=embedding, 1=bm25)', default: '0.5' },
}} />
</Tab>
</Tabs>
---
## Results
### CompressResult (TypeScript)
<TypeTable type={{
messages: { type: 'any[]', description: 'Compressed messages in the same format as input' },
tokensBefore: { type: 'number', description: 'Token count before compression' },
tokensAfter: { type: 'number', description: 'Token count after compression' },
tokensSaved: { type: 'number', description: 'Tokens removed by compression' },
compressionRatio: { type: 'number', description: 'Ratio of tokens after to tokens before' },
transformsApplied: { type: 'string[]', description: 'Names of transforms that were applied' },
ccrHashes: { type: 'string[]', description: 'CCR hashes for Compress-Cache-Retrieve' },
compressed: { type: 'boolean', description: 'Whether compression was actually applied' },
}} />
### SimulationResult (Python)
<TypeTable type={{
tokens_before: { type: 'int', description: 'Token count before compression' },
tokens_after: { type: 'int', description: 'Token count after compression' },
tokens_saved: { type: 'int', description: 'Tokens removed by compression' },
savings_percent: { type: 'float', description: 'Percentage of tokens saved' },
transforms_applied: { type: 'list[str]', description: 'Names of transforms that were applied' },
waste_signals: { type: 'WasteSignals', description: 'Detected waste in the request' },
}} />
### WasteSignals (Python)
<TypeTable type={{
json_bloat_tokens: { type: 'int', description: 'Tokens from JSON formatting waste' },
html_noise_tokens: { type: 'int', description: 'Tokens from HTML tags and noise' },
whitespace_tokens: { type: 'int', description: 'Tokens from excessive whitespace' },
dynamic_date_tokens: { type: 'int', description: 'Tokens from dynamic date strings' },
repetition_tokens: { type: 'int', description: 'Tokens from repeated content' },
}} />
### RequestMetrics (Python)
<TypeTable type={{
request_id: { type: 'str', description: 'Unique request identifier' },
timestamp: { type: 'datetime', description: 'When the request was processed' },
model: { type: 'str', description: 'Model name used' },
tokens_input_before: { type: 'int', description: 'Input tokens before compression' },
tokens_input_after: { type: 'int', description: 'Input tokens after compression' },
tokens_output: { type: 'int', description: 'Output tokens from the model' },
cost_before: { type: 'float', description: 'Cost before compression (USD)' },
cost_after: { type: 'float', description: 'Cost after compression (USD)' },
transforms_applied: { type: 'list[str]', description: 'Transforms that were applied' },
}} />
---
## Providers
### OpenAIProvider
```python
from headroom import OpenAIProvider
provider = OpenAIProvider(
enable_prefix_caching=True,
)
counter = provider.get_token_counter("gpt-4o")
tokens = counter.count_text("Hello, world!")
limit = provider.get_context_limit("gpt-4o") # 128000
cost = provider.estimate_cost(input_tokens=1000, output_tokens=500, model="gpt-4o")
```
### AnthropicProvider
```python
from headroom import AnthropicProvider
from anthropic import Anthropic
provider = AnthropicProvider(
client=Anthropic(),
enable_cache_control=True,
)
counter = provider.get_token_counter("claude-3-5-sonnet-latest")
tokens = counter.count_messages(messages) # Accurate count via API
```
### GoogleProvider
```python
from headroom.providers import GoogleProvider
provider = GoogleProvider(
enable_context_caching=True,
)
```
---
## Relevance Scoring
### create_scorer()
Factory function to create scorers:
```python
from headroom import create_scorer
# Auto-select best available scorer
scorer = create_scorer()
# Explicitly choose type
scorer = create_scorer(scorer_type="hybrid", alpha=0.7)
```
### BM25Scorer
Fast keyword-based scoring (zero dependencies):
```python
from headroom import BM25Scorer
scorer = BM25Scorer()
scores = scorer.score_items(items=["item 1", "item 2"], query="search query")
```
### EmbeddingScorer
Semantic similarity scoring (requires `headroom-ai[relevance]`):
```python
from headroom import EmbeddingScorer, embedding_available
if embedding_available():
scorer = EmbeddingScorer(model_name="BAAI/bge-small-en-v1.5")
scores = scorer.score_items(items, query)
```
### HybridScorer
Combines BM25 and embeddings:
```python
from headroom import HybridScorer
scorer = HybridScorer(alpha=0.5) # 50% BM25, 50% embedding
scores = scorer.score_items(items, query)
```
---
## Transforms (Direct Use)
### SmartCrusher
```python
from headroom import SmartCrusher
crusher = SmartCrusher()
result = crusher.crush(data={"results": [...]}, query="user query")
```
### CacheAligner
```python
from headroom import CacheAligner
aligner = CacheAligner()
result = aligner.align(messages)
```
### TransformPipeline
```python
from headroom import TransformPipeline
pipeline = TransformPipeline([
SmartCrusher(),
CacheAligner(),
])
result = pipeline.transform(messages)
```
---
## Errors
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
| Exception | Meaning |
|-----------|---------|
| `HeadroomError` | Base class for all errors |
| `HeadroomConnectionError` | Cannot reach proxy |
| `HeadroomAuthError` | 401 from proxy |
| `HeadroomCompressError` | Compression failed (includes `statusCode`, `errorType`) |
| `ConfigurationError` | Invalid configuration |
| `ProviderError` | Provider issues |
| `StorageError` | Storage failures |
| `TokenizationError` | Token counting failed |
| `CacheError` | Cache operations failed |
| `ValidationError` | Validation failures |
| `TransformError` | Transform execution failed |
Use `mapProxyError(status, type, message)` to convert proxy error responses to the correct class.
</Tab>
<Tab value="Python">
| Exception | Meaning |
|-----------|---------|
| `HeadroomError` | Base class for all Headroom errors |
| `ConfigurationError` | Invalid config values |
| `ProviderError` | Provider issue (unknown model, etc.) |
| `StorageError` | Database issue |
| `CompressionError` | Compression failed (rare) |
| `ValidationError` | Setup validation failed |
All exceptions include a `details` dict with additional context.
</Tab>
</Tabs>
---
## Utilities
### Tokenizer
```python
from headroom import Tokenizer, count_tokens_text, count_tokens_messages
# Quick counting
tokens = count_tokens_text("Hello, world!", model="gpt-4o")
# With tokenizer instance
tokenizer = Tokenizer(model="gpt-4o")
tokens = tokenizer.count_text("Hello")
tokens = tokenizer.count_messages(messages)
```
### generate_report()
Generate HTML/Markdown reports from stored metrics:
```python
from headroom import generate_report
report = generate_report(
store_url="sqlite:///headroom.db",
format="html",
period="day",
)
```
---
## TypeScript Message Types
<TypeTable type={{
role: { type: '"system" | "user" | "assistant" | "tool"', description: 'The role of the message sender' },
content: { type: 'string | ContentPart[] | null', description: 'Message content (string, content parts array, or null for tool-calling assistant messages)' },
tool_calls: { type: 'ToolCall[]', description: 'Tool calls made by the assistant (assistant messages only)' },
tool_call_id: { type: 'string', description: 'ID of the tool call this message responds to (tool messages only)' },
}} />
The TypeScript SDK uses the standard OpenAI message format with `SystemMessage`, `UserMessage`, `AssistantMessage`, and `ToolMessage` variants.
+132
View File
@@ -0,0 +1,132 @@
---
title: Architecture
description: How Headroom's three-stage compression pipeline works, from message parsing through transform execution to provider cache optimization.
---
Headroom sits between your application and the LLM provider. It intercepts messages, compresses them intelligently, and forwards the optimized request. The response comes back unchanged.
## High-Level Flow
```
+---------------------------------------------------------------+
| YOUR APPLICATION |
+---------------------------------------------------------------+
|
v
+---------------------------------------------------------------+
| HEADROOM CLIENT |
| +-----------+ +------------+ +---------+ |
| | ANALYZE | > | TRANSFORM | > | CALL | |
| | (Parser) | | (Pipeline)| | (API) | |
| +-----------+ +------------+ +---------+ |
| | | | |
| v v v |
| Count tokens Apply compressions Send to LLM provider |
| Detect waste Preserve meaning Log metrics |
+---------------------------------------------------------------+
|
v
+---------------------------------------------------------------+
| OPENAI / ANTHROPIC / GOOGLE |
+---------------------------------------------------------------+
```
## Entry Points
Headroom can be used in three ways, all feeding into the same pipeline:
| Entry Point | How It Works | Code Changes |
|-------------|-------------|--------------|
| **SDK Mode** | Wrap your LLM client with `HeadroomClient` | Minimal -- swap client constructor |
| **Proxy Mode** | Run `headroom proxy` and point your client at it | Zero -- just change the base URL |
| **Integrations** | LangChain, Vercel AI SDK, Agno adapters | Framework-specific setup |
## The Transform Pipeline
Messages flow through a sequence of transforms. Each transform is independent, safe to skip, and fails gracefully (returns original content unchanged).
### Stage 1: Cache Aligner
Detects dynamic content (dates, UUIDs, session tokens) in your system prompt and reports prefix metrics. Keep the stable prefix and live context separated in the caller so provider caches (Anthropic `cache_control`, OpenAI prefix caching) can hit on repeated calls.
```
Observed: "You are helpful. Current Date: 2024-12-15"
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
Changes daily = cache miss every day
Caller layout:
"You are helpful." [stable prefix]
"[Context: Current Date: 2024-12-15]" [live context]
```
Overhead: sub-millisecond.
### Stage 2: Smart Crusher
Analyzes tool output content and compresses it using statistical methods. This is where the bulk of token savings come from.
**What it does:**
1. Parses JSON arrays in tool outputs
2. Runs field-level statistical analysis (variance, uniqueness, change points)
3. Selects a representative subset using the Kneedle algorithm on bigram coverage
4. Preserves errors, anomalies, and distribution boundaries unconditionally
5. Factors out constant fields shared by all items
**Strategies by content type:**
| Content | Strategy | Typical Savings |
|---------|----------|-----------------|
| JSON arrays of dicts | Statistical sampling + anomaly preservation | 83--95% |
| JSON arrays of strings | Dedup + adaptive sampling | 60--90% |
| JSON arrays of numbers | Statistical summary + outlier preservation | 70--85% |
| Build/test logs | Pattern clustering | 85--94% |
| HTML | Article extraction (trafilatura-based) | ~95% |
**Item retention split:** 30% from array start (schema), 15% from end (recency), 55% by importance score. Error items are always kept regardless of budget.
Overhead: 1--50ms for typical payloads. Scales linearly with input size.
### Stage 3: Context Manager
Ensures the final message array fits within the model's context window.
**Rolling Window** (default): Drops oldest messages first, preserving system prompt and recent turns. Tool calls and their responses are dropped as atomic units.
**Intelligent Context** (advanced): Scores every message on six dimensions (recency, semantic similarity, TOIN importance, error indicators, forward references, token density) and drops the lowest-scored messages first. Dropped messages are stored in CCR for potential retrieval.
Overhead: sub-millisecond for Rolling Window; depends on scoring config for Intelligent Context.
## Provider Cache Optimization
After the pipeline, Headroom applies provider-specific cache hints:
| Provider | Mechanism | Savings |
|----------|-----------|---------|
| Anthropic | `cache_control` blocks on stable prefix | Up to 90% on cached tokens |
| OpenAI | Prefix alignment for automatic caching | Up to 50% on cached tokens |
| Google | `CachedContent` API | Up to 75% on cached tokens |
## CCR: Compress-Cache-Retrieve
When SmartCrusher compresses a tool output or Intelligent Context drops messages, the original content is stored in a local compression cache. If the LLM needs the full data, it can request retrieval via a `headroom_retrieve` tool call. This makes compression reversible.
```
Compress: 1000 items -> 15 items (stored original in CCR)
Cache: Hash-indexed local store (SQLite)
Retrieve: LLM calls headroom_retrieve("abc123") -> original 1000 items
```
## TOIN: Tool Output Intelligence Network
TOIN learns compression patterns across sessions and users. When a tool is used repeatedly, TOIN builds up statistics about which fields matter, which items get retrieved, and what compression strategies work best. These learned patterns feed back into SmartCrusher and Intelligent Context scoring.
Cold start: For new tool types, TOIN falls back to statistical heuristics. Patterns build up over time as tools are used.
## What Headroom Does NOT Touch
- **User messages**: Never compressed (the user's intent must be preserved exactly)
- **System prompts**: Content preserved; dynamic parts are reported so callers can keep them outside the stable prefix
- **Code**: Passes through unchanged unless tree-sitter AST compression is explicitly enabled
- **Model responses**: Returned unchanged from the provider
- **Short content**: Tool outputs under 200 tokens pass through (overhead exceeds savings)
+152
View File
@@ -0,0 +1,152 @@
---
title: Benchmarks
description: Compression performance, accuracy preservation, latency overhead, and real-world production telemetry from 250+ Headroom proxy instances.
---
Headroom's core promise: compress context without losing accuracy. This page covers compression benchmarks, accuracy evaluations, latency overhead, and production telemetry.
For local inference, the main benefit is often faster prompt processing rather than lower API spend. See [Local LLM prefill benchmarking](/docs/local-llm-prefill) for a reproducible passthrough-vs-optimized proxy workflow.
## Compression Performance
Tested on Apple M-series (CPU), Headroom v0.5.18. Each test runs `compress()` on realistic tool outputs.
| Content Type | Original | Compressed | Saved | Ratio | Latency |
|---|---|---|---|---|---|
| JSON array (100 items) | 3,163 | 297 | 2,866 | **90.6%** | 1ms |
| JSON array (500 items) | 9,526 | 1,614 | 7,912 | **83.1%** | 2ms |
| Shell output (200 lines) | 3,238 | 469 | 2,769 | **85.5%** | 1ms |
| Build log (200 lines) | 2,412 | 148 | 2,264 | **93.9%** | 1ms |
| grep results (150 hits) | 2,624 | 2,624 | 0 | 0.0% | &lt;1ms |
| Python source (~480 lines) | 2,958 | 2,958 | 0 | 0.0% | &lt;1ms |
| **Total** | **23,921** | **8,110** | **15,811** | **66.1%** | **5ms** |
<Callout type="info" title="Zero compression is intentional">
grep results and Python source show 0% compression. These are already compact structured formats. SmartCrusher only compresses JSON arrays; code passes through to preserve correctness.
</Callout>
## Accuracy Benchmarks
### HTML Extraction
**Dataset**: Scrapinghub Article Extraction Benchmark (181 HTML pages with ground truth)
| Metric | Value |
|---|---|
| **F1 Score** | 0.919 |
| **Precision** | 0.879 |
| **Recall** | 0.982 |
| **Compression** | 94.9% |
For LLM applications, recall is critical -- 98.2% means nearly all article content is preserved. The slight precision drop (some extra content) does not hurt LLM accuracy.
### JSON Compression (SmartCrusher)
**Test**: 100 production log entries with critical error at position 67. Task: find the error, error code, resolution, and affected count.
| Metric | Baseline | Headroom |
|---|---|---|
| Input tokens | 10,144 | 1,260 |
| Correct answers | 4/4 | **4/4** |
| Compression | -- | **87.6%** |
SmartCrusher preserves first N items (schema), last N items (recency), all anomalies (errors, warnings), and statistical distribution.
### QA Accuracy Preservation
| Metric | Original HTML | Extracted | Delta |
|---|---|---|---|
| F1 Score | 0.85 | 0.87 | +0.02 |
| Exact Match | 60% | 62% | +2% |
<Callout type="info" title="Extraction can improve accuracy">
Removing HTML noise sometimes helps LLMs focus on relevant content, leading to slightly higher scores on extraction benchmarks.
</Callout>
## Latency Overhead
### SDK Compression Latency
Measured per-scenario on Apple M-series (CPU):
| Scenario | Tokens In | Tokens Out | Saved | p50 (ms) | p95 (ms) |
|----------|-----------|------------|-------|----------|----------|
| JSON: Search Results (100 items) | 10.2K | 1.5K | 8.7K | 189 | 231 |
| JSON: Search Results (500 items) | 50.2K | 1.5K | 48.7K | 943 | 955 |
| JSON: Search Results (1K items) | 100.5K | 1.5K | 99.0K | 2,012 | 2,198 |
| JSON: API Responses (500 items) | 38.9K | 1.1K | 37.8K | 743 | 776 |
| JSON: Database Rows (1K rows) | 43.7K | 605 | 43.1K | 961 | 1,104 |
| JSON: String Array (100 strings) | 1.1K | 231 | 820 | 15 | 15 |
| JSON: String Array (500 strings) | 4.9K | 233 | 4.6K | 72 | 80 |
| JSON: Number Array (200 numbers) | 1.2K | 192 | 1.1K | 31 | 62 |
| JSON: Mixed Array (250 items) | 2.3K | 368 | 1.9K | 38 | 40 |
### Cost-Benefit Analysis
Net latency benefit = LLM time saved from fewer tokens minus compression overhead (at Claude Sonnet pricing, $3.0/MTok):
| Scenario | Compress (ms) | LLM Saved (ms) | Net Benefit | Savings per 1K Requests |
|----------|---------------|-----------------|-------------|------------------------|
| JSON: Search Results (100 items) | 189 | 261 | **+72ms** | $26 |
| JSON: Search Results (500 items) | 943 | 1,461 | **+518ms** | $146 |
| JSON: Search Results (1K items) | 2,012 | 2,969 | **+957ms** | $297 |
| JSON: API Responses (500 items) | 743 | 1,134 | **+391ms** | $113 |
| JSON: Database Rows (1K rows) | 961 | 1,292 | **+331ms** | $129 |
Compression pays for itself in latency for 11 of 12 tested scenarios against Claude Sonnet. Slower and more expensive models (Opus) benefit even more.
### Pipeline Step Timing
| Step | Median | P90 | Description |
|------|--------|-----|-------------|
| `pipeline_total` | 16.9ms | 289ms | Full compression pipeline |
| `content_router` | 11.7ms | 259ms | Content detection + routing |
| `smart_crusher` | 50.1ms | 50ms | JSON array compression |
| `text_compressor` | 32.0ms | 576ms | Text compression (Kompress ONNX) |
| `initial_token_count` | 2.9ms | 16ms | Token counting (tiktoken) |
ContentRouter accounts for 91--98% of pipeline cost on average. CacheAligner is sub-millisecond.
## Production Telemetry
Real-world data from **50,000+ proxy sessions** across 250+ unique instances (March--April 2026). Collected via anonymous telemetry (opt-in: `HEADROOM_TELEMETRY=on`; telemetry is off by default).
### Proxy Overhead
| Percentile | Latency |
|---|---|
| **Median (P50)** | **52ms** |
| P90 | 309ms |
| P99 | 4,172ms |
| Mean | 161ms |
The median 52ms overhead is negligible compared to LLM inference time (typically 2--10 seconds).
### Compression Rate
| Percentile | Compression |
|---|---|
| P25 | 4.8% |
| **Median** | **4.8%** |
| P75 | 6.9% |
| Mean | 11.3% |
Median compression is modest because many requests are short conversational turns. Heavy tool-use sessions (file reads, shell output) see 40--80% compression.
### Fleet Summary
| Metric | Value |
|---|---|
| Clean instances | 249 |
| Total tokens saved | 1.4 billion |
| Total savings | ~$4,000 |
| OS distribution | Linux 57%, macOS 38%, Windows 5% |
## Reproducing Results
```bash
git clone https://github.com/chopratejas/headroom.git
cd headroom
pip install -e ".[evals,html]"
pytest tests/test_evals/ -v -s
```
+69
View File
@@ -0,0 +1,69 @@
---
title: Cache Optimization
description: Stabilize message prefixes for provider KV cache hits and configure provider-specific caching strategies.
---
LLM providers cache prompt prefixes to avoid reprocessing identical input on repeated calls. Headroom's **CacheAligner** is detector-only, so it surfaces prefix drift, reports observability data, and leaves message assembly to the caller.
## What CacheAligner reports
System prompts often contain dynamic content, such as dates, session IDs, and timestamps, that changes between requests. Even a single character difference at the start of a prompt invalidates the entire provider cache.
CacheAligner does not extract, move, normalize, reorder, strip, compress, or rewrite content. It detects volatile content and reports the stable prefix hash plus cache metrics so you can fix the prefix at the source:
| Signal | Meaning |
|---|---|
| `warnings` | The prefix contains unstable content |
| `cache_metrics.stable_prefix_bytes` | Stable prefix size in bytes |
| `cache_metrics.stable_prefix_tokens_est` | Stable prefix size in estimated tokens |
| `cache_metrics.stable_prefix_hash` | Stable prefix hash for repeated-wake comparison |
| `cache_metrics.prefix_changed` | The prefix drifted since the previous wake |
| `markers` | The emitted `stable_prefix_hash` marker for observability |
The prefix must stay byte-identical across requests for provider KV caches to reuse previously computed attention states.
## Provider-specific strategies
Each LLM provider implements caching differently. Headroom applies the optimal strategy for each.
### Anthropic
Anthropic supports explicit `cache_control` blocks that mark content as cacheable. Cached input tokens cost **90% less** than regular input tokens.
Keep the stable prefix byte-identical, then place provider cache markers where your client or orchestrator already assembles the request. Headroom's job is to surface prefix instability, not repair it.
| Metric | Value |
|---|---|
| Cache read discount | 90% off input price |
| Cache write cost | 25% premium on first write |
| Cache TTL | 5 minutes (extended on hit) |
### OpenAI
OpenAI uses automatic **prefix caching**. If consecutive requests share the same message prefix, the provider reuses cached KV states. No explicit API markers are needed, but the prefix must be byte-identical.
CacheAligner tells you when the prefix changed, which is the only signal you need to keep OpenAI prefix caching effective.
| Metric | Value |
|---|---|
| Cache read discount | 50% off input price |
| Activation | Automatic (prefix match) |
| Min prefix length | 1024 tokens |
### Google
Google provides the **CachedContent API**, which lets you explicitly cache large context (system instructions, documents, tools) and reference it across requests. Cached tokens cost **75% less**.
Keep the prefix stable in your integration layer; Headroom reports when the cacheable zone drifts. The CachedContent lifecycle itself also stays in your integration layer.
| Metric | Value |
|---|---|
| Cache read discount | 75% off input price |
| Mechanism | Explicit CachedContent API objects |
| Min cache size | 32,768 tokens |
## What this means in practice
Keep the stable prefix first, keep volatile content out of it, and treat CacheAligner warnings as a signal that the caller needs to move assembly logic.
CacheAligner surfaces prefix instability, provider caches reward byte-identical prefixes, and the caller owns the actual message layout.
+179
View File
@@ -0,0 +1,179 @@
---
title: Reversible Compression (CCR)
description: Compress-Cache-Retrieve architecture that makes compression lossless — the LLM can always get the original data back.
---
Headroom's CCR (Compress-Cache-Retrieve) architecture makes compression **reversible**. When content is compressed, the original data is cached locally. If the LLM needs the full data, it retrieves it instantly.
<Callout type="info" title="Nothing is ever thrown away">
Unlike traditional lossy compression, CCR guarantees that every piece of original data remains accessible. You get 70-90% token savings with zero risk of permanent data loss.
</Callout>
## The problem with traditional compression
Traditional compression forces a difficult tradeoff:
- **Aggressive compression** risks losing data the LLM needs
- **Conservative compression** misses out on token savings
CCR eliminates this tradeoff entirely. Compress aggressively, retrieve on demand.
## Architecture
CCR flows through four phases:
```
TOOL OUTPUT (1000 items)
-> SmartCrusher compresses to 20 items
-> Original cached with hash=abc123
-> Retrieval tool injected into context
LLM PROCESSING
Option A: LLM solves task with 20 items -> Done (90% savings)
Option B: LLM calls headroom_retrieve(hash=abc123)
-> Response Handler returns full data automatically
```
## Phase 1: Compression Store
When SmartCrusher compresses tool output:
1. The original content is stored in an LRU cache
2. A hash key is generated for retrieval
3. A marker is added to the compressed output:
```
[1000 items compressed to 20. Retrieve more: hash=abc123]
```
## Phase 2: Tool Injection
Headroom injects a `headroom_retrieve` tool into the LLM's available tools:
```json
{
"name": "headroom_retrieve",
"description": "Retrieve original uncompressed data from Headroom cache",
"parameters": {
"hash": "The hash key from the compression marker"
}
}
```
The LLM sees this tool alongside your application's tools and can call it whenever the compressed data is insufficient.
## Phase 3: Response Handler
When the LLM calls `headroom_retrieve`:
1. The Response Handler intercepts the tool call
2. Data is retrieved from the local cache (around 1ms)
3. The result is added to the conversation
4. The API call continues automatically
The client never sees CCR tool calls -- they are handled transparently by Headroom.
## Phase 4: Context Tracker
Across multiple turns, the Context Tracker maintains awareness of all compressed content:
1. Remembers what was compressed in earlier turns
2. Analyzes new queries for relevance to compressed content
3. Proactively expands relevant data before the LLM asks
```
Turn 1: User searches for files
-> 500 files compressed to 15, cached (hash=abc123)
-> LLM answers with 15 files
Turn 5: User asks "What about the auth middleware?"
-> Context Tracker detects "auth" may match cached content
-> Proactively expands compressed data
-> LLM finds auth_middleware.py in the full list
```
## Retrieving originals
CCR works automatically through the proxy, but you can also retrieve cached data programmatically:
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from "headroom-ai";
import type { CCRConfig } from "headroom-ai";
// CCR is enabled by default when compressing through the proxy.
const result = await compress(messages, {
model: "gpt-4o",
});
// Access compressed messages — CCR markers are embedded automatically
console.log(result.messages);
// CCR configuration options
const ccrConfig: CCRConfig = {
enabled: true,
injectTool: true, // Inject headroom_retrieve tool
injectRetrievalMarker: true, // Add retrieval markers to compressed output
feedbackEnabled: true, // Learn from retrieval patterns
storeMaxEntries: 1000, // Max cached items
storeTtlSeconds: 3600, // Cache TTL
};
```
</Tab>
<Tab value="Python">
```python
from headroom import HeadroomClient, OpenAIProvider
from openai import OpenAI
client = HeadroomClient(
original_client=OpenAI(),
provider=OpenAIProvider(),
default_mode="optimize",
)
# CCR happens automatically during chat completions.
# The LLM calls headroom_retrieve when it needs more data.
response = client.chat.completions.create(
model="gpt-4o",
messages=messages,
)
# CCR is enabled by default in the current proxy path. Use --no-optimize for
# full passthrough behavior.
```
</Tab>
</Tabs>
## Retention
Proxy CCR originals are kept for 1800 seconds (30 minutes) by default. For longer autonomous
agent runs, set `HEADROOM_CCR_TTL_SECONDS` before starting the proxy:
```bash
HEADROOM_CCR_TTL_SECONDS=7200 headroom proxy
```
Check the effective setting at `/v1/retrieve/stats` under
`store.default_ttl_seconds`.
## Message-level CCR
> **Retired:** The "Message-level CCR via IntelligentContext" feature (where `IntelligentContext` would store dropped messages in CCR with a retrieval marker) was part of the `IntelligentContextConfig` API that has since been removed. Context management is now handled automatically by the pipeline without a separate configurable IntelligentContext stage. Tool-output CCR via SmartCrusher and ContentRouter remains fully supported.
## CCR-enabled components
| Component | What it compresses | CCR integration |
|---|---|---|
| **SmartCrusher** | JSON arrays (tool outputs) | Stores original array, marker includes hash |
| **ContentRouter** | Code, logs, search results, text | Stores original content by strategy |
## Why CCR matters
| Approach | Risk | Savings |
|---|---|---|
| No compression | None | 0% |
| Traditional compression | Data loss | 70-90% |
| CCR compression | None (reversible) | 70-90% |
CCR gives you the savings of aggressive compression with zero risk. The LLM can always retrieve the original data if needed.
+242
View File
@@ -0,0 +1,242 @@
---
title: CI/CD Flow Diagrams
description: Visual decision trees for pull requests, release publishing, Docker images, docs deploys, and manual validation.
---
## Purpose
This page is the quick visual map for Headroom automation. Use it when opening a PR, reviewing a PR, cutting a release, or deciding which workflow owns a failure.
The short version:
- Pull requests are gated by PR governance, path-filtered CI, targeted e2e workflows, and human review.
- Release publishing is not triggered by every merge to `main`. `release-please` maintains a release PR; merging that release PR creates the tag and GitHub Release that trigger publishing.
- Docker images are built as multi-architecture digests first, then merged into tagged manifests.
- Docs deploy only after docs changes land on `main`.
## PR Flow
```mermaid
flowchart TD
A[Open, edit, synchronize, or mark PR ready] --> B[PR Governance]
B --> B1{Template complete and ready?}
B1 -- no --> B2[Add needs author action label and governance comment]
B1 -- yes --> B3[Add ready for review label when no blocking status exists]
A --> C[Path filters decide workflow surface]
C --> D{Code paths changed?}
D -- yes --> E[CI changes job]
E --> F[lint: ruff, format, mypy]
E --> G[build-wheel: Rust extension wheel]
E --> H[prefetch-model: Hugging Face cache]
G --> I[test shards 1 to 4]
H --> I
G --> J[test-extras, test-agno, dashboard UI]
E --> K[build: release-profile wheel and sdist smoke]
C --> L{E2E paths changed?}
L -- yes --> M[Docker native e2e and platform wrapper checks]
C --> N{Init or wrap paths changed?}
N -- yes --> O[Init E2E, Wrap E2E, native init or wrap smoke tests]
C --> P{Rust paths changed?}
P -- yes --> Q[Rust fmt, clippy, tests, wheel build, audit]
C --> R{Devcontainer paths changed?}
R -- yes --> S[Devcontainer validation and linked worktree smoke]
C --> T{Release-critical paths changed?}
T -- yes --> U[Release dry-run: wheel matrix and smoke-import gates]
C --> V{Workflow files changed?}
V -- yes --> W[Workflow validation with actionlint and act dry-run]
F --> X{All required checks green?}
I --> X
J --> X
K --> X
M --> X
O --> X
Q --> X
S --> X
U --> X
W --> X
B3 --> X
X -- no --> Y[Fix, rebase, or request changes]
X -- yes --> Z[Human review and merge when approved]
```
## PR Decision Tree
```mermaid
flowchart TD
A[Review an open PR] --> B{Draft?}
B -- yes --> C[Do not approve. Comment with remaining readiness steps.]
B -- no --> D{Governance label says needs author action?}
D -- yes --> E[Fix or ask author to complete template and real behavior proof]
D -- no --> F{Merge state dirty or behind?}
F -- dirty --> G[Resolve conflicts before reviewing final code]
F -- behind --> H[Rebase or update branch, then rerun checks]
F -- clean or unknown --> I{Any failing check?}
I -- yes --> J[Read failing logs, classify as stale-main, infra, or code bug]
J --> K{Can maintainer safely fix without changing author intent?}
K -- yes --> L[Patch, test locally, push with lease]
K -- no --> M[Request changes with exact file and line references]
I -- no --> N{Code review complete?}
N -- no --> O[Review diff, tests, docs, and behavior proof]
N -- yes --> P[Approve]
```
## Fork Workflow Approval
GitHub may leave product workflows in `action_required` for first-time or fork contributors. Approve only after the diff is safe enough to execute in CI.
```mermaid
flowchart TD
A[Fork PR has action_required workflows] --> B{Diff is understandable and not suspicious?}
B -- no --> C[Do not approve. Ask for changes or close if unsafe.]
B -- yes --> D{Workflow runs use pull_request with read-scoped token?}
D -- no --> E[Inspect workflow permissions before approving]
D -- yes --> F[Approve queued CI runs]
F --> G[Wait for fresh check results on latest head SHA]
```
## Release Flow
```mermaid
flowchart TD
A[Merge ordinary PR to main] --> B[Release Please on push to main]
B --> C{Commit is releasable?}
C -- docs, ci, chore only --> D[No release PR change]
C -- fix, feat, breaking change --> E[Create or update Release PR]
E --> F[Release PR contains version bump and changelog]
F --> G{Ready to ship?}
G -- no --> H[Keep merging regular PRs; bot updates Release PR]
G -- yes --> I[Merge Release PR]
I --> J[release-please tags vX.Y.Z and publishes GitHub Release]
J --> K[release.yml starts on release: published]
K --> L[detect-version]
L --> M[build: sync versions, verify versions, changelog, npm packs]
M --> N[build-wheels matrix]
N --> O[collect-dist]
O --> P[smoke-import wheels]
P --> Q{Smoke import green?}
Q -- no --> R[Stop before publishing broken wheels]
Q -- yes --> S[publish PyPI]
Q -- yes --> T[publish npm packages]
Q -- yes --> U[publish GitHub Package Registry packages]
Q -- yes --> V[publish Docker images through docker.yml]
S --> W{PyPI published or PYPI_SKIP=true?}
W -- no --> X[Do not update release assets]
W -- yes --> Y[Create or update GitHub Release assets and notes]
T --> Y
U --> Y
V --> Y
```
## Release Decision Tree
```mermaid
flowchart TD
A[Need a release?] --> B{Release PR exists?}
B -- no --> C[Merge at least one releasable conventional commit to main]
B -- yes --> D{Release PR checks green and changelog correct?}
D -- no --> E[Fix source PRs or release config, then let release-please update]
D -- yes --> F{Registry skip variables needed?}
F -- yes --> G[Set PYPI_SKIP, NPM_SKIP, or GH_PACKAGES_SKIP deliberately]
F -- no --> H[Merge Release PR]
G --> H
H --> I[Watch release.yml]
I --> J{Failure before publish?}
J -- yes --> K[Fix and rerun before any package is public]
J -- no --> L{Failure after partial publish?}
L -- yes --> M[Use rerun or skip variables to reach consistent GitHub Release state]
L -- no --> N[Release complete]
```
## Docker Publish Flow
`docker.yml` can run directly on `push`, `workflow_dispatch`, or `release: published`, and it is also called by `release.yml`.
```mermaid
flowchart TD
A[Docker workflow starts] --> B[Matrix: variant x architecture]
B --> C[Build each platform image by digest only]
C --> D[Smoke-test image imports pydantic_core and headroom._core]
D --> E{Smoke test green?}
E -- no --> F[Stop before tag manifest]
E -- yes --> G[Upload digest marker]
G --> H[Per-variant manifest merge]
H --> I[Apply tags to multi-arch manifest]
I --> J{Release event?}
J -- yes --> K[Version tags and latest retag rules]
J -- no --> L[Branch, PR, dev, or manual tags as configured]
```
## Docs Deploy Flow
```mermaid
flowchart TD
A[Docs change in PR] --> B[PR review and normal checks]
B --> C[Merge to main]
C --> D{docs/**, mkdocs.yml, or docs workflow changed?}
D -- no --> E[No docs deploy]
D -- yes --> F[Deploy Documentation workflow]
F --> G[Build docs site]
G --> H[Deploy generated site]
```
## Manual Validation Flow
Use this when editing workflows or release automation.
```mermaid
flowchart TD
A[Edit workflow or release scripts] --> B[Run local workflow validation]
B --> C[actionlint]
B --> D[act dry-run fixtures]
C --> E{Local validation green?}
D --> E
E -- no --> F[Fix before opening or updating PR]
E -- yes --> G[Push PR]
G --> H[workflow-validation job reruns same validation in CI]
```
Recommended local command:
```bash
bash scripts/validate-workflows.sh
```
For release dry-runs:
```bash
act workflow_dispatch -W .github/workflows/release.yml -e .github/act/dry-run.json
```
## Gate Summary
| Flow | Trigger | Main gates | Success condition |
|------|---------|------------|-------------------|
| PR governance | `pull_request_target`, schedule, manual | Template, readiness labels, merge state, check labels | PR has no governance blockers |
| CI | PR, push to `main`, manual | Path filter, lint, mypy, wheel build, model prefetch, test shards, package smoke | Required jobs green or path-skipped |
| Rust | Rust paths, schedule | fmt, clippy, cargo test, wheel build, audit | Rust checks green; nightly parity is allowed to fail in Phase 0 |
| E2E | CLI, install, wrap, Docker, package paths | Docker init/wrap, native init/wrap/install, platform smoke | Relevant lifecycle checks green |
| Release dry-run | PRs touching release-critical paths | Version detection, wheel matrix, smoke imports | Publish path can build before merge |
| Release publish | GitHub Release published by release-please | Version sync, changelog, wheels, smoke import, PyPI gate, npm, GPR, Docker | Public packages and GitHub Release assets are consistent |
| Docs deploy | Push to `main` with docs paths | Docs build | Site deploy completes |
## Workflow Ownership
| Workflow | Owns |
|----------|------|
| `.github/workflows/pr-health.yml` | PR body governance, readiness labels, rebase/conflict/failing-check labels |
| `.github/workflows/ci.yml` | Python lint, type checks, wheel build, test shards, package smoke, workflow validation |
| `.github/workflows/rust.yml` | Rust workspace quality gates and native wheel smoke artifacts |
| `.github/workflows/init-e2e.yml` | Dockerized `headroom init` behavior |
| `.github/workflows/wrap-e2e.yml` | Dockerized `headroom wrap` behavior |
| `.github/workflows/init-native-e2e.yml` | Host-specific `headroom init -g` smoke tests |
| `.github/workflows/install-native-e2e.yml` | Host-specific install CLI smoke tests |
| `.github/workflows/wrap-native-e2e.yml` | Host-specific wrap prepare-only smoke tests |
| `.github/workflows/devcontainers.yml` | Devcontainer startup and linked worktree compatibility |
| `.github/workflows/release-please.yml` | Release PR aggregation from conventional commits |
| `.github/workflows/release.yml` | Release build, wheel smoke-import gates, registry publishing, GitHub Release assets |
| `.github/workflows/docker.yml` | GHCR multi-architecture image builds and manifests |
| `.github/workflows/docs.yml` | Documentation deploy after docs changes merge |
@@ -0,0 +1,102 @@
---
title: Claude Code on Azure AI Foundry
description: Run Claude Code against Claude models on Azure AI Foundry, with Headroom compressing your prompts — fewer input tokens, same answers, your own Azure credentials.
---
If your Claude models live on **Azure AI Foundry**, you can still get Headroom's
prompt compression. Headroom sits between Claude Code and Azure: it shrinks the
big stuff in each request (file reads, logs, tool output) and forwards the rest to
Azure using **your own Azure credentials**. You keep your Azure setup; Headroom
just makes each call cheaper.
## What you get
- **Fewer input tokens** on every Claude Code request to Azure AI Foundry (often
3060% on agent workloads), so you pay for less.
- **Same answers** — compression is reversible and content-aware.
- **No new secrets** — Headroom never holds your Azure credentials. Claude Code
keeps authenticating to Azure AI Foundry with its own `api-key` or Entra Bearer
token; Headroom passes it through.
## Before you start
You should already have Claude Code working against Azure AI Foundry **without**
Headroom. That means these are set in your shell (or your `~/.claude/settings.json`
`env` block):
```bash
export CLAUDE_CODE_USE_FOUNDRY=1
export ANTHROPIC_FOUNDRY_RESOURCE=<your-azure-resource-name>
# e.g. ANTHROPIC_FOUNDRY_RESOURCE=my-org-claude
```
No `ANTHROPIC_API_KEY` is needed — Foundry mode uses your Azure credentials.
## Run it (one command)
```bash
pip install headroom-ai
headroom wrap claude
```
That's it. Because `CLAUDE_CODE_USE_FOUNDRY=1` is set, `headroom wrap claude`
automatically:
1. derives your Azure AI Foundry endpoint from `ANTHROPIC_FOUNDRY_RESOURCE`,
2. starts the Headroom proxy with that endpoint as the upstream,
3. points Claude Code's Foundry endpoint at the proxy (`ANTHROPIC_FOUNDRY_BASE_URL`),
4. leaves your Azure resource, model, and credentials untouched.
You'll see a line like:
```
Foundry mode: ANTHROPIC_FOUNDRY_BASE_URL=http://127.0.0.1:8787/anthropic
→ upstream https://my-org-claude.services.ai.azure.com/anthropic
```
Use Claude Code exactly as you normally would.
## How it works
```
Claude Code ──(Foundry request)──▶ Headroom ──(compressed)──▶ Azure AI Foundry (Claude)
in Foundry mode compresses your resource
(your api-key / Entra) ──────── passed through ───────────▶ authenticates you
```
Claude Code sends its Foundry request (Anthropic API format) to Headroom. Headroom
compresses the messages, then forwards to your Azure AI Foundry resource endpoint —
`https://{ANTHROPIC_FOUNDRY_RESOURCE}.services.ai.azure.com/anthropic` — with your
auth headers passed through unchanged.
## If you have ANTHROPIC_FOUNDRY_BASE_URL set explicitly
If your environment already has `ANTHROPIC_FOUNDRY_BASE_URL` set to the full Azure
endpoint URL, Headroom uses it directly and `ANTHROPIC_FOUNDRY_RESOURCE` is not
needed. Either configuration works.
## Check that compression is working
1. Open the dashboard: [http://localhost:8787/dashboard](http://localhost:8787/dashboard).
"Tokens saved" should climb as you use Claude Code.
2. Or check response headers: `x-headroom-tokens-before`, `x-headroom-tokens-after`,
`x-headroom-tokens-saved`.
## Troubleshooting
**Headroom says "ANTHROPIC_BASE_URL" instead of "Foundry mode"**
`CLAUDE_CODE_USE_FOUNDRY` is not set in the shell where you ran `headroom wrap
claude`. Make sure to export it before running, or set it in your shell profile.
**Claude Code fails with a 401 / auth error**
Your Azure credentials are not being forwarded correctly. Verify that Claude Code
works against Azure AI Foundry directly (without Headroom) before wrapping. If it
works direct but not through Headroom, open an issue with the proxy log.
**"tokens saved" is always 0**
Check the [dashboard](http://localhost:8787/dashboard) — if requests are flowing
but savings are 0, content may be below the compression threshold or the Rust
extension may not be installed (`pip install "headroom-ai[proxy]"` includes it).
+117
View File
@@ -0,0 +1,117 @@
---
title: Claude Code on Vertex AI
description: Run Claude Code against Claude models on Google Vertex AI, with Headroom compressing your prompts — fewer input tokens, same answers, your own GCP login.
---
If your Claude models live on **Google Vertex AI**, you can still get Headroom's
prompt compression. Headroom sits between Claude Code and Vertex: it shrinks the
big stuff in each request (file reads, logs, tool output) and forwards the rest to
Vertex using **your own Google credentials**. You keep your GCP setup; Headroom
just makes each call cheaper.
## What you get
- **Fewer input tokens** on every Claude Code request to Vertex (often 3060% on
agent workloads), so you pay Vertex for less.
- **Same answers** — compression is reversible and content-aware.
- **No new secrets** — Headroom never holds your Google credentials. Claude Code
keeps authenticating to Vertex with its own ADC token; Headroom passes it through.
## Before you start
You should already have Claude Code working against Vertex **without** Headroom.
That means these are set in your shell:
```bash
export CLAUDE_CODE_USE_VERTEX=1
export ANTHROPIC_VERTEX_PROJECT_ID=<your-gcp-project>
export CLOUD_ML_REGION=us-east5 # your Vertex region (or "global")
gcloud auth application-default login # or set GOOGLE_APPLICATION_CREDENTIALS
```
No `ANTHROPIC_API_KEY` is needed — Vertex mode uses your Google login.
## Run it (one command)
```bash
pip install headroom-ai
headroom wrap claude
```
That's it. Because `CLAUDE_CODE_USE_VERTEX=1` is set, `headroom wrap claude`
automatically:
1. starts the Headroom proxy,
2. points Claude Code's Vertex endpoint at it (`ANTHROPIC_VERTEX_BASE_URL`),
3. leaves your project, region, and Google login untouched.
You'll see a line like:
```
Vertex mode: ANTHROPIC_VERTEX_BASE_URL=http://127.0.0.1:8787
→ compress, then forward to Vertex with your GCP ADC token
```
Use Claude Code exactly as you normally would.
## How it works
```
Claude Code ──(Vertex request)──▶ Headroom ──(compressed)──▶ Vertex AI (Claude)
in Vertex mode compresses your project + region
(your ADC token) ───────────── passed through ───────────▶ authenticates you
```
Claude Code sends its normal Vertex `…:rawPredict` / `:streamRawPredict` request to
Headroom. Headroom compresses the messages (keeping the Vertex request shape
intact), then forwards to the correct regional Vertex host — derived from the
request itself, so multi-region and `global` both work — using the Google token
Claude Code already attached.
## Check that compression is working
1. Open the dashboard: [http://localhost:8787/dashboard](http://localhost:8787/dashboard).
"Tokens saved" should climb as you use Claude Code.
2. Or look at the response headers on a request: `x-headroom-tokens-before`,
`x-headroom-tokens-after`, `x-headroom-tokens-saved`.
If "tokens saved" stays at 0 on large prompts, see Troubleshooting below.
## Troubleshooting
- **It still goes straight to Google (no savings).** Make sure `CLAUDE_CODE_USE_VERTEX=1`
is exported *in the same shell* before `headroom wrap claude`. The wrapper only
switches to Vertex mode when it sees that variable.
- **Wrong region / 404 from Vertex.** Confirm `CLOUD_ML_REGION` matches a region
where your Claude model is enabled. `global` is supported and maps to the
non-regional host.
- **Auth errors.** Headroom forwards your token as-is — if `gcloud auth
application-default login` (or `GOOGLE_APPLICATION_CREDENTIALS`) works for Claude
Code without Headroom, it works with it.
## Alternative: let Headroom talk to Vertex for you
If you'd rather **not** run Claude Code in Vertex mode, you can have Headroom be the
translator instead: Claude Code speaks plain Anthropic to Headroom, and Headroom
calls Vertex on your behalf.
```bash
export HEADROOM_BACKEND=litellm-vertex_ai # note the _ai suffix
export HEADROOM_REGION=us-east5
export VERTEXAI_PROJECT=<your-gcp-project>
export GOOGLE_APPLICATION_CREDENTIALS=/path/sa.json # or gcloud ADC
export ANTHROPIC_API_KEY=placeholder # Claude Code needs *a* key to start
headroom wrap claude --backend litellm-vertex_ai --region us-east5
```
The native Vertex-mode flow above is recommended — it keeps your existing GCP auth
and has the smallest moving parts. Use this alternative only if you can't set
`CLAUDE_CODE_USE_VERTEX`.
## Notes
- Pick a Claude model that is enabled in your Vertex project/region
(e.g. `claude-sonnet-4-6`, `claude-haiku-4-5`).
- Streaming, tool use, and prompt caching all work through Headroom.
- Want to point at a private Vertex gateway instead of Google's host? Start the
proxy with `--vertex-api-url https://your-gateway` and Headroom will forward there.
+173
View File
@@ -0,0 +1,173 @@
---
title: Code Compression
description: AST-aware compression that preserves imports, signatures, and types while compressing function bodies. Powered by tree-sitter.
---
Headroom's CodeAwareCompressor uses tree-sitter to parse source code into an AST, then selectively compresses function bodies while preserving the structural elements that LLMs need -- imports, signatures, type annotations, and error handlers.
## Why AST-Aware Compression?
Naive truncation breaks code. Cutting a function in half leaves invalid syntax that confuses the LLM. CodeAwareCompressor guarantees:
- **Syntax validity** -- output always parses correctly
- **Structural preservation** -- imports, signatures, types, decorators are kept intact
- **Lightweight** -- ~50MB (tree-sitter) vs ~1GB for LLMLingua
## Supported Languages
| Tier | Languages | Support Level |
|---|---|---|
| Tier 1 | Python, JavaScript, TypeScript | Full AST analysis |
| Tier 2 | Go, Rust, Java, C, C++ | Function body compression |
## What Gets Preserved vs Compressed
**Always preserved:**
- Import statements
- Function and method signatures
- Class definitions
- Type annotations
- Decorators
- Error handlers (`try`/`except`, `try`/`catch`)
**Compressed:**
- Function bodies (implementations)
- Comments (unless configured to preserve)
- Verbose docstrings (configurable: full, first line, or removed)
## Example
```python
from headroom.transforms import CodeAwareCompressor
compressor = CodeAwareCompressor()
code = '''
import os
from typing import List
def process_items(items: List[str]) -> List[str]:
"""Process a list of items."""
results = []
for item in items:
if not item:
continue
processed = item.strip().lower()
results.append(processed)
return results
'''
result = compressor.compress(code, language="python")
print(result.compressed)
# import os
# from typing import List
#
# def process_items(items: List[str]) -> List[str]:
# """Process a list of items."""
# results = []
# for item in items:
# # ... (5 lines compressed)
# pass
print(f"Compression: {result.compression_ratio:.0%}") # ~55%
print(f"Syntax valid: {result.syntax_valid}") # True
```
## Configuration
```python
from headroom.transforms import CodeAwareCompressor, CodeCompressorConfig, DocstringMode
config = CodeCompressorConfig(
preserve_imports=True, # Always keep imports
preserve_signatures=True, # Always keep function signatures
preserve_type_annotations=True, # Keep type hints
preserve_error_handlers=True, # Keep try/except blocks
preserve_decorators=True, # Keep decorators
docstring_mode=DocstringMode.FIRST_LINE, # FULL, FIRST_LINE, REMOVE
target_compression_rate=0.2, # Keep 20% of tokens
max_body_lines=5, # Lines to keep per function body
min_tokens_for_compression=100, # Skip small content
language_hint=None, # Auto-detect if None
fallback_to_kompress=True, # Use Kompress for unknown langs
)
compressor = CodeAwareCompressor(config)
result = compressor.compress(code)
```
### Configuration Options
| Option | Default | Description |
|---|---|---|
| `preserve_imports` | `True` | Keep all import statements |
| `preserve_signatures` | `True` | Keep function/method signatures |
| `preserve_type_annotations` | `True` | Keep type hints |
| `preserve_error_handlers` | `True` | Keep try/except blocks |
| `preserve_decorators` | `True` | Keep decorators |
| `docstring_mode` | `FIRST_LINE` | How to handle docstrings: `FULL`, `FIRST_LINE`, `REMOVE` |
| `target_compression_rate` | `0.2` | Fraction of tokens to keep (0.2 = keep 20%) |
| `max_body_lines` | `5` | Max lines to keep per function body |
| `min_tokens_for_compression` | `100` | Skip files smaller than this |
| `language_hint` | `None` | Override language detection |
| `fallback_to_kompress` | `True` | Use Kompress for unsupported languages |
## Before and After
```python
# Before (full source file)
def process_data(items: List[str]) -> Dict[str, int]:
"""Process items and count occurrences."""
result = {}
for item in items:
item = item.strip().lower()
if item in result:
result[item] += 1
else:
result[item] = 1
return result
# After (signature preserved, body compressed)
def process_data(items: List[str]) -> Dict[str, int]:
"""Process items and count occurrences."""
result = {}
for item in items:
# ... (5 lines compressed)
pass
```
The LLM sees the function's purpose, its input/output types, and the general approach -- enough to reason about the code without needing every implementation line.
## Installation
```bash
# Install tree-sitter language pack
pip install "headroom-ai[code]"
```
## Memory Management
Tree-sitter parsers are lazy-loaded and cached. You can free memory when done:
```python
from headroom.transforms import is_tree_sitter_available, unload_tree_sitter
# Check if tree-sitter is installed
print(is_tree_sitter_available()) # True
# Free memory when done
unload_tree_sitter()
```
## Performance
| Metric | Value |
|---|---|
| Compression | 40-70% token reduction |
| Speed | ~10-50ms per file |
| Memory | ~50MB (tree-sitter parsers) |
| Syntax validity | Guaranteed |
<Callout type="info" title="Automatic routing">
When you use the Headroom proxy or call `compress()`, source code is automatically detected and routed to CodeAwareCompressor. Direct usage gives you control over compression settings per language.
</Callout>
+22
View File
@@ -0,0 +1,22 @@
---
title: Community Savings
description: Aggregate savings from Headroom instances across the community. Anonymous telemetry data — no prompts, no content, no PII.
---
Real-time aggregate metrics from Headroom proxy instances worldwide. All data is anonymous — only token counts, compression ratios, and cost estimates are collected. Telemetry is off by default; [opt in](https://github.com/chopratejas/headroom/blob/main/headroom/telemetry/beacon.py) with `HEADROOM_TELEMETRY=on`.
## Overview
<CommunityStatsHeader />
## Savings Over Time
<CommunityCharts section="area" />
## Top Savings by Instance
<CommunityCharts section="bar" />
## Instance Details
<CommunityCharts section="table" />
+468
View File
@@ -0,0 +1,468 @@
---
title: Configuration
description: All configuration options for the Headroom Python and TypeScript SDKs, proxy server, and per-request overrides.
---
Headroom can be configured via the SDK constructor, proxy command line, environment variables, or per-request overrides.
## CLI Context Tool
`headroom wrap ...` uses RTK for local shell-output filtering by default.
Set `HEADROOM_CONTEXT_TOOL=lean-ctx` to have wrap commands install or reuse
`lean-ctx` and run `lean-ctx init --agent <tool>` instead of RTK setup.
```bash
export HEADROOM_CONTEXT_TOOL=lean-ctx
headroom wrap claude
headroom wrap codex --prepare-only
```
Supported values are `rtk` and `lean-ctx`; unset defaults to `rtk`.
The proxy reads RTK lifetime savings with global scope by default so a shared
daemon reports savings across the operator's projects. Set
`HEADROOM_RTK_GAIN_SCOPE=project` to query `rtk gain --project` from the
proxy process working directory.
## SDK Modes (`default_mode` / `headroom_mode`)
These modes apply to SDK usage via `HeadroomClient(default_mode=...)` or per-request `headroom_mode=...`. They are **not** the same as the proxy `--mode` flag.
| Mode | Behavior | Use Case |
|------|----------|----------|
| `audit` | Observes and logs, no modifications | Production monitoring, baseline measurement |
| `optimize` | Applies safe, deterministic transforms | Production optimization |
| `simulate` | Returns plan without API call | Testing, cost estimation |
> **Proxy `--mode` is a separate axis**: `headroom proxy --mode token` (maximize compression) or `--mode cache` (freeze prior turns for prefix-cache stability). The proxy does not accept `audit`, `optimize`, or `simulate`.
## SDK Configuration
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { HeadroomClient } from 'headroom-ai';
// Reads from HEADROOM_BASE_URL and HEADROOM_API_KEY automatically
const client = new HeadroomClient();
// Or configure explicitly
const explicit = new HeadroomClient({
baseUrl: 'http://localhost:8787',
apiKey: 'your-api-key',
timeout: 30_000,
fallback: true,
retries: 2,
});
```
</Tab>
<Tab value="Python">
```python
from headroom import HeadroomClient, OpenAIProvider
from openai import OpenAI
client = HeadroomClient(
original_client=OpenAI(),
provider=OpenAIProvider(),
# Mode: "audit" (observe only) or "optimize" (apply transforms)
default_mode="optimize",
# Enable provider-specific cache optimization
enable_cache_optimizer=True,
# Enable query-level semantic caching
enable_semantic_cache=False,
# Override default context limits per model
model_context_limits={
"gpt-4o": 128000,
"gpt-4o-mini": 128000,
},
# Database location (defaults to temp directory)
# store_url="sqlite:////absolute/path/to/headroom.db",
)
```
</Tab>
</Tabs>
## Per-Request Overrides
Override configuration for individual requests:
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from 'headroom-ai';
const result = await compress(messages, {
model: 'gpt-4o',
tokenBudget: 100_000,
timeout: 15_000,
});
```
</Tab>
<Tab value="Python">
```python
response = client.chat.completions.create(
model="gpt-4o",
messages=[...],
# Override mode for this request
headroom_mode="audit",
# Reserve more tokens for output
headroom_output_buffer_tokens=8000,
# Keep last N turns (don't compress)
headroom_keep_turns=5,
# Skip compression for specific tools
headroom_tool_profiles={
"important_tool": {"skip_compression": True}
},
)
```
</Tab>
</Tabs>
### Proxy upstream override (`x-headroom-base-url`)
When using the proxy, send the `x-headroom-base-url` request header to route a
single request to a different upstream instead of the configured provider URL.
This lets a client that speaks a provider's wire format authenticate against a
compatible gateway (for example an OpenAI-compatible endpoint, or an
Anthropic-Messages gateway such as OpenCode Zen) without changing the proxy
configuration.
The header is honored by the OpenAI-compatible routes, the Anthropic Messages
route (`POST /v1/messages`), and the generic passthrough route. The proxy
forwards the request to `<x-headroom-base-url>` + the original request path
(e.g. `/v1/messages`). An empty or whitespace-only value is ignored and the
configured upstream is used.
```bash
curl http://127.0.0.1:8787/v1/messages \
-H "content-type: application/json" \
-H "anthropic-version: 2023-06-01" \
-H "x-headroom-base-url: https://opencode.ai/zen/go" \
-H "x-api-key: <gateway-api-key>" \
-d '{"model":"glm-5.2","max_tokens":16,"messages":[{"role":"user","content":"hi"}]}'
```
When `HEADROOM_STRIP_INTERNAL_HEADERS` is `enabled` (the default), the proxy
reads this header for routing and then strips it before forwarding upstream.
## SmartCrusher Configuration
Fine-tune JSON compression behavior:
```python
from headroom.transforms import SmartCrusherConfig
config = SmartCrusherConfig(
# Maximum items to keep after compression
max_items_after_crush=15,
# Minimum tokens before applying compression
min_tokens_to_crush=200,
# Fraction of items always kept from the start/end
first_fraction=0.3,
last_fraction=0.15,
# Variance threshold for statistical analysis
variance_threshold=2.0,
)
```
## CacheAligner Configuration
Control prefix stabilization for provider cache hit rates:
```python
from headroom.transforms import CacheAlignerConfig
config = CacheAlignerConfig(
# Enable/disable cache alignment
enabled=True,
# Patterns to extract from system prompt
dynamic_patterns=[
r"Today is \w+ \d+, \d{4}",
r"Current time: .*",
],
)
```
## Context Window Management
Context management is now automatic. Use per-request overrides to control behavior:
```python
response = client.chat.completions.create(
model="gpt-4o",
messages=messages,
# Reserve tokens for model output
headroom_output_buffer_tokens=4000,
# Keep last N turns uncompressed
headroom_keep_turns=3,
)
```
The `RollingWindowConfig`, `IntelligentContextConfig`, and `ScoringWeights` classes are no longer part of Headroom. Context management now happens automatically inside the pipeline (live-zone-only compression).
## Pipeline Extensions
Use a `headroom.pipeline_extension` entry point when you need to normalize or annotate requests before they leave Headroom. The `PRE_SEND` stage is the right place for provider-specific request cleanup, such as turning `content: null` into `content: ""` for upstreams that reject OpenAI-spec tool-call messages.
```python
from headroom.pipeline import PipelineEvent, PipelineStage
class NormalizeNullContent:
def on_pipeline_event(self, event: PipelineEvent) -> PipelineEvent:
if event.stage is not PipelineStage.PRE_SEND or not event.messages:
return event
for message in event.messages:
if (
message.get("role") == "assistant"
and message.get("content") is None
and message.get("tool_calls")
):
message["content"] = ""
return event
```
Register it in `pyproject.toml`:
```toml
[project.entry-points."headroom.pipeline_extension"]
normalize_null_content = "my_pkg.normalize:NormalizeNullContent"
```
If the upstream base URL itself must vary per request, use the `x-headroom-base-url` override header in addition to the normalization hook.
## Proxy Configuration
### Command Line Options
```bash
headroom proxy \
--port 8787 \ # Port to listen on
--host 0.0.0.0 \ # Host to bind to
--mode token \ # token compression mode; use cache for prefix-cache stability
--budget 10.00 \ # Daily budget limit in USD
--log-file headroom.jsonl # Log file path
```
### Feature Flags
```bash
# Disable optimization (passthrough mode)
headroom proxy --no-optimize
# Disable semantic caching
headroom proxy --no-cache
# Preserve provider prefix-cache stability instead of maximizing token removal
headroom proxy --mode cache
# Enable memory and live learning
headroom proxy --memory
headroom proxy --learn --min-evidence 3
```
## Environment Variables
| Variable | Description | Default |
|----------|-------------|---------|
| `HEADROOM_HOST` | Proxy bind host | `127.0.0.1` |
| `HEADROOM_PORT` | Proxy bind port | `8787` |
| `HEADROOM_MODE` | Proxy optimization mode: `token` or `cache` | `token` |
| `HEADROOM_WORKERS` | Uvicorn worker count | `1` |
| `HEADROOM_LIMIT_CONCURRENCY` | Maximum concurrent connections before 503 | `1000` |
| `HEADROOM_MAX_CONNECTIONS` | Maximum upstream HTTP connections | `500` |
| `HEADROOM_MAX_KEEPALIVE` | Maximum upstream keep-alive connections | `100` |
| `HEADROOM_KEEPALIVE_EXPIRY` | Seconds an idle upstream keep-alive connection is kept open | `90` |
| `HEADROOM_HTTP_PROXY` | HTTP proxy URL for upstream provider requests only; HTTPS provider APIs use CONNECT | -- |
| `HEADROOM_BUDGET` | Daily budget limit in USD | -- |
| `HEADROOM_TELEMETRY` | Set to `on` to opt in to anonymous telemetry | `off` (opt-in) |
| `HEADROOM_STATELESS` | Set to `true` to disable filesystem writes | `false` |
| `HEADROOM_MODEL_LIMITS` | Custom model config (JSON string or file path) | -- |
| `HEADROOM_BASE_URL` | Base URL of the Headroom proxy (TypeScript SDK) | `http://localhost:8787` |
| `HEADROOM_API_KEY` | Optional API key for authenticated Headroom endpoints (TypeScript SDK) | -- |
| `HEADROOM_CONFIG_DIR` | Canonical config (read-mostly) root. Derives `models.json` and per-plugin config paths when set. | `~/.headroom/config` |
| `HEADROOM_WORKSPACE_DIR` | Canonical workspace (read-write state) root. Derives savings, memory DB, logs, TOIN, subscription state, and more when set. | `~/.headroom` |
| `HEADROOM_SAVINGS_PATH` | Override persistent savings file location. Always wins when set. | derived from `${HEADROOM_WORKSPACE_DIR}` |
| `HEADROOM_TOIN_PATH` | Override TOIN telemetry file location. Always wins when set. | derived from `${HEADROOM_WORKSPACE_DIR}` |
| `HEADROOM_SUBSCRIPTION_STATE_PATH` | Override subscription tracker state file. Always wins when set. | derived from `${HEADROOM_WORKSPACE_DIR}` |
| `HEADROOM_TELEMETRY` | Set to `on` to opt in to anonymous telemetry | `off` |
| `HEADROOM_PERIODIC_TOIN_STATS` | Controls periodic TOIN stats logging in long-lived proxy workers. Set to `0`, `false`, `off`, or `no` to disable the 5-minute stats loop without disabling TOIN learning or request-time feedback. | `true` |
| `HEADROOM_MEMORY_INJECTION_MODE` | Memory-context routing mode: `live_zone_tail` (default) or `disabled`. The legacy `system_prompt` mode was retired by PR-A2; supplying it raises. | `live_zone_tail` |
| `HEADROOM_PROXY_PYTHON_FORWARDER_MODE` | Python forwarder serialization mode. `byte_faithful` (default) forwards original request bytes verbatim when no transform mutated the body and re-serializes canonically only when needed — keeps Anthropic prompt-cache hit-rate intact. `legacy_json_kwarg` is an explicit operator opt-in for emergency rollback to the historical `httpx ... json=body` behavior. NOT a fallback — only flip on explicit operator decision. | `byte_faithful` |
| `HEADROOM_STRIP_INTERNAL_HEADERS` | Python proxy: whether to strip internal `x-headroom-*` request headers (e.g. `x-headroom-bypass`, `x-headroom-mode`, `x-headroom-user-id`, `x-headroom-stack`, `x-headroom-base-url`) before every upstream forwarder call (PR-A5, fixes P5-49). `enabled` (default) stops fingerprinting / leakage. `disabled` is an explicit operator opt-in for diagnostic shadow tracing — NOT a fallback. Inbound reads of these headers (bypass gating, memory user-id resolution) are unaffected because they read `request.headers` directly. | `enabled` |
| `HEADROOM_PROXY_STRIP_INTERNAL_HEADERS` | Rust proxy: same policy as `HEADROOM_STRIP_INTERNAL_HEADERS` but for the Rust transparent proxy. Stripping happens inside `build_forward_request_headers` so both HTTP and WebSocket upstream calls are gated by one flag. `enabled` default; `disabled` operator opt-in for diagnostic shadow tracing. Response-side `X-Headroom-*` injection (e.g. `x-headroom-tokens-saved`) is unrelated and stays. | `enabled` |
| `HEADROOM_EMBEDDER_RUNTIME` | Set to `pytorch_mps` to run the memory embedder via the torch sentence-transformers backend on the Apple GPU (MPS). Only engages when Apple MPS is actually available; otherwise it logs a warning and uses the existing default embedder selection path. `pytorch_mps` is the only accepted value. Requires the `[pytorch-mps]` extra. See [Memory](/docs/memory#embedding-runtime--gpu-offload-apple-silicon). | default embedder selection |
| `ORT_DYLIB_PATH` | Windows: path to the `onnxruntime.dll` loaded by the Rust core (magika detection, fastembed embeddings). Auto-pinned at `import headroom` to the DLL inside the `onnxruntime` pip package; set it yourself to override. Without a pin the bare Windows DLL search resolves to the Windows ML System32 build (1.17.x on Win11 24H2+), which deadlocks ONNX session init — see [Troubleshooting](/docs/troubleshooting#windows-ml-content-detection-hangs-or-silently-falls-back). | auto-pinned on Windows |
| `HEADROOM_MAGIKA_INIT_TIMEOUT_SECS` | Upper bound (integer seconds, > 0) on magika's one-time ONNX session init in the Rust detection chain. On timeout the init error is cached and detection uses the non-ML fallback tiers for the rest of the process; a warning is logged. Safety net for environments where the dylib pin above does not apply. | `5` |
| `HEADROOM_REQUEST_TIMEOUT` | Request timeout in seconds | `300` |
| `HEADROOM_BETA_HEADER_STICKY` | Controls per-session `anthropic-beta` / `OpenAI-Beta` re-echo. `enabled` (default): the proxy unions beta tokens across turns within a session — if the client sends a token in turn N and omits it in turn N+1, the proxy re-injects it to preserve prefix-cache stability. `disabled`: the client's value is forwarded verbatim with no accumulation. Any other value raises at request time. See [Session Beta Header Tracking](/docs/configuration#session-beta-header-tracking). | `enabled` |
| `HEADROOM_BETA_TRACKER_MAX_SESSIONS` | LRU capacity of the in-memory session beta tracker. Once full, the oldest session entry is evicted. | `1000` |
For provider-only proxying, prefer `HEADROOM_HTTP_PROXY` over process-wide variables such as `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, or `NO_PROXY`. HTTPX reads those global variables, but Headroom also passes them through to tool executions.
### Session Beta Header Tracking
When running as a proxy, Headroom maintains a per-session union of `anthropic-beta` (and `OpenAI-Beta`) tokens via `SessionBetaTracker`. The session key is derived from the `x-headroom-session-id` header if present, otherwise from `md5(model + system_prompt[:500])[:16]` — stable across turns of the same conversation.
**Why:** clients such as Claude Code and Codex CLI may drop a beta token between consecutive turns. Because `anthropic-beta` is part of the request bytes that determine the upstream prefix-cache key, a dropped token would bust the cache mid-conversation. The tracker re-injects any token seen earlier in the session so the cache key stays stable.
**Trade-off:** once the proxy has seen a beta token in a session it will continue re-sending it for the rest of that session, even if the client stops including it. Stopping the token on the client side alone is not sufficient — the proxy re-injects it. Set `HEADROOM_BETA_HEADER_STICKY=disabled` to pass the client's `anthropic-beta` value verbatim and bypass this accumulation.
```bash
# Disable sticky beta re-echo
export HEADROOM_BETA_HEADER_STICKY=disabled
headroom proxy ...
```
Note: disabling sticky mode may reduce prefix-cache hit rates for clients that legitimately drop-and-re-add beta tokens across turns.
### Filesystem Contract
Headroom resolves every on-disk resource through a two-root model
(`HEADROOM_CONFIG_DIR` + `HEADROOM_WORKSPACE_DIR`) with additive
precedence rules: explicit argument > per-resource env var > derived
from canonical root > default. Every legacy env var continues to work
unchanged.
See the **[Filesystem Contract](https://github.com/chopratejas/headroom/blob/main/wiki/filesystem-contract.md)**
page for the full bucket table, plugin-author guidance, and the Docker
naming overlap note (`HEADROOM_WORKSPACE` is *not* the same as
`HEADROOM_WORKSPACE_DIR`).
## Custom Model Configuration
Configure context limits and pricing for new or custom models:
```json
{
"anthropic": {
"context_limits": {
"claude-4-opus-20250301": 200000,
"claude-custom-finetune": 128000
},
"pricing": {
"claude-4-opus-20250301": {
"input": 15.00,
"output": 75.00,
"cached_input": 1.50
}
}
},
"openai": {
"context_limits": {
"gpt-5": 256000,
"ft:gpt-4o:my-org": 128000
}
}
}
```
Save as `${HEADROOM_CONFIG_DIR}/models.json` (defaults to
`~/.headroom/config/models.json`), or set `HEADROOM_MODEL_LIMITS` to a
JSON string or file path. Installs that still have
`~/.headroom/models.json` (the legacy location) continue to work.
Settings are resolved in this order (later overrides earlier):
1. Built-in defaults
2. `${HEADROOM_CONFIG_DIR}/models.json` (new canonical location); falls
back to `~/.headroom/models.json` (legacy) when the canonical file
is absent
3. `HEADROOM_MODEL_LIMITS` environment variable
4. SDK constructor arguments
### Pattern-Based Inference
Unknown models are automatically inferred from naming patterns:
| Pattern | Inferred Settings |
|---------|-------------------|
| `*opus*` | 200K context, Opus-tier pricing |
| `*sonnet*` | 200K context, Sonnet-tier pricing |
| `*haiku*` | 200K context, Haiku-tier pricing |
| `gpt-4o*` | 128K context, GPT-4o pricing |
| `o1*`, `o3*` | 200K context, reasoning model pricing |
## Provider-Specific Settings
<Tabs groupId="lang" items={['OpenAI', 'Anthropic', 'Google']}>
<Tab value="OpenAI">
```python
from headroom import OpenAIProvider
provider = OpenAIProvider(
enable_prefix_caching=True,
)
```
</Tab>
<Tab value="Anthropic">
```python
from headroom import AnthropicProvider
provider = AnthropicProvider(
enable_cache_control=True,
)
```
</Tab>
<Tab value="Google">
```python
from headroom.providers import GoogleProvider
provider = GoogleProvider(
enable_context_caching=True,
)
```
</Tab>
</Tabs>
## Tool Profiles
Skip or customize compression for specific tools:
```python
response = client.chat.completions.create(
model="gpt-4o",
messages=messages,
headroom_tool_profiles={
"important_tool": {"skip_compression": True},
"search_tool": {"max_items_after_crush": 25},
},
)
```
## Configuration Precedence
Settings are applied in this order (later overrides earlier):
1. Default values
2. Environment variables
3. SDK constructor arguments
4. Per-request overrides
## Validation
Validate your configuration at startup:
```python
result = client.validate_setup()
if not result["valid"]:
print("Configuration issues:")
for issue in result["issues"]:
print(f" - {issue}")
```
+83
View File
@@ -0,0 +1,83 @@
---
title: Context Management
description: Automatic live-zone-only context management that compresses the newest content blocks while preserving the provider cache hot zone.
---
Context management is now handled automatically inside the pipeline (live-zone-only compression). Headroom **never drops messages** from the conversation history and does not do position-based or score-based context management.
## How It Works
Headroom compresses only the **newest content blocks** — the latest user message and the latest tool result / tool output. Compression is type-aware and reversible via [CCR](/docs/ccr), so the LLM can retrieve the original content on demand.
The **cache hot zone** — the system prompt, tool definitions, and older turns — is never mutated. Leaving the prefix untouched preserves provider prompt caching, so cache hit rates stay stable across turns.
```
Conversation with a large latest tool result
-> Identify the live zone (newest user message + latest tool output)
-> Compress the live zone type-aware, cache original in CCR (hash=def456)
-> Insert marker: "compressed, retrieve: def456"
-> Older turns, tools, and system prompt are forwarded byte-for-byte
```
## Protection rules
Headroom enforces several protections to ensure model output quality:
### Output buffer reservation
A configurable number of tokens is reserved for the model's response. The context budget is calculated as:
```
context_budget = model_context_limit - output_buffer_tokens
```
This prevents the input from consuming the entire context window and leaving no room for the model to respond.
### System message protection
System messages are never dropped. They contain critical instructions, persona definitions, and tool descriptions that the model needs throughout the conversation.
### Turn protection
The last N user/assistant turns are always preserved, ensuring the model has immediate conversational context. By default, the last 2 turns are protected.
## Configuration
Context management is now automatic. Use per-request overrides to control behavior:
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from "headroom-ai";
const result = await compress(messages, {
model: "gpt-4o",
tokenBudget: 32000,
});
console.log(`Compressed: ${result.tokensBefore} -> ${result.tokensAfter}`);
```
</Tab>
<Tab value="Python">
```python
from headroom import HeadroomClient, OpenAIProvider
from openai import OpenAI
client = HeadroomClient(
original_client=OpenAI(),
provider=OpenAIProvider(),
default_mode="optimize",
)
# Per-request overrides
response = client.chat.completions.create(
model="gpt-4o",
messages=messages,
headroom_output_buffer_tokens=8000, # More room for long responses
headroom_keep_turns=5, # Protect last 5 turns
)
```
</Tab>
</Tabs>
> **Note:** The `IntelligentContextConfig`, `ScoringWeights`, and `RollingWindowConfig` classes are no longer part of Headroom. Context management is now handled automatically inside the pipeline (live-zone-only compression).
+204
View File
@@ -0,0 +1,204 @@
---
title: Docker-Native Install
description: Run Headroom without installing Python or Node.js on the host. A native `headroom` wrapper keeps the proxy and CLI in Docker while your other tools run on the host OS.
---
Run Headroom without installing Python or Node.js on the host. The install scripts add a native `headroom` wrapper that keeps **Headroom itself** in Docker while orchestrating the rest of your workflow on the host OS.
## One-line install
### Linux
```bash
curl -fsSL https://raw.githubusercontent.com/chopratejas/headroom/main/scripts/install.sh | bash
```
### macOS (bash 4.3+)
```bash
curl -fsSL https://raw.githubusercontent.com/chopratejas/headroom/main/scripts/install.sh | "$(brew --prefix bash)/bin/bash"
```
Stock `/bin/bash` on macOS is 3.2, so install a newer bash first (for example via Homebrew) and run the installer with that shell. The installed wrapper pins that same bash interpreter so later invocations stay on the supported runtime.
### Windows PowerShell
```powershell
irm https://raw.githubusercontent.com/chopratejas/headroom/main/scripts/install.ps1 | iex
```
## What the installer does
1. Verifies Docker is installed and available.
2. Pulls `ghcr.io/chopratejas/headroom:latest` by default, or reuses / pulls `HEADROOM_DOCKER_IMAGE` when you set a custom image override.
3. Installs a `headroom` wrapper into `~/.local/bin` or `~/bin`.
4. Updates shell startup files so the wrapper directory is on `PATH`.
The wrapper keeps Headroom inside Docker and mounts host state back into the container so native behavior stays consistent:
- project workspace → `/workspace`
- `~/.headroom`
- `~/.claude`
- `~/.codex`
- `~/.gemini`
Port `8787` stays the default, so `http://localhost:8787` works the same way as a native install.
Published releases also push versioned GHCR tags such as `ghcr.io/chopratejas/headroom:0.5.26`, and those images are built with the same synced package version used for the matching PyPI and npm release.
## How the wrapper behaves
### Native Headroom commands
These run directly inside the container:
```bash
headroom proxy
headroom learn
headroom mcp install
headroom memory list
```
For `proxy`, the wrapper publishes the selected port back to the host:
```bash
docker run --rm -it \
-p 8787:8787 \
-v "$PWD:/workspace" \
-w /workspace \
ghcr.io/chopratejas/headroom:latest \
headroom proxy --host 0.0.0.0 --port 8787
```
### `wrap` commands
`wrap` is host-oriented in Docker-native mode:
- the wrapper starts the Headroom proxy in Docker
- container-side prep writes Headroom config, memory, and selected CLI context-tool setup into mounted host files
- the target CLI itself is launched on the host by the wrapper
Supported host wrap flows:
- `headroom wrap claude`
- `headroom wrap codex`
- `headroom wrap aider`
- `headroom wrap cursor`
- `headroom wrap openclaw`
- `headroom unwrap openclaw`
OpenClaw remains host-native in Docker-native mode:
- the host must already have the `openclaw` CLI installed
- `headroom wrap openclaw` installs/configures the Headroom plugin through the host `openclaw` CLI
- plugin auto-start still launches the installed host `headroom` wrapper from `PATH`, which then runs Headroom in Docker
- local plugin source mode (`--plugin-path`) is also supported, but it may require host `npm` when build steps are needed
## Persistent Docker lifecycle from the native wrapper
The Docker-native `headroom` wrapper exposes the persistent Docker lifecycle directly:
```bash
headroom install apply --profile default --preset persistent-docker
headroom install status
headroom install restart
headroom install remove
```
In Docker-native mode this surface is intentionally scoped to **persistent-docker**:
- supported: `apply`, `status`, `start`, `stop`, `restart`, `remove`
- supported flags: `--profile`, `--port`, `--backend`, `--anyllm-provider`, `--region`, `--mode`, `--memory`, `--no-telemetry`, `--image`
- not supported: `persistent-service`, `persistent-task`, or provider/user/system mutation flags such as `--scope`, `--providers`, and `--target`
Those broader lifecycle and config-mutation flows still belong to the Python-native `headroom install ...` command.
Persistent Docker deployments launched by the wrapper also tag the proxy process with deployment metadata, so `/health` reports the active `profile`, `preset`, `runtime`, `supervisor`, and `scope` the same way the Python install subsystem does.
## Docker Compose support
Use `docker/docker-compose.native.yml` when you want an explicit compose-managed proxy or CLI shell, or when you prefer compose over the native wrapper's `headroom install ...` surface.
### Persistent Docker runtime
The `proxy` service uses `restart: unless-stopped`, so compose can act as the always-on Docker runtime for Headroom:
```bash
export HEADROOM_HOST_HOME="$HOME"
export HEADROOM_WORKSPACE="$PWD"
docker compose -f docker/docker-compose.native.yml up -d proxy
```
```powershell
$env:HEADROOM_HOST_HOME = $HOME
$env:HEADROOM_WORKSPACE = (Get-Location).Path
docker compose -f docker/docker-compose.native.yml up -d proxy
```
This is a supported persistent-Docker path when you want the proxy managed explicitly through Compose instead of the installed wrapper.
<Callout type="info" title="HEADROOM_WORKSPACE vs HEADROOM_WORKSPACE_DIR">
These are two different variables — both are set by the compose file, and both are retained for backward compatibility:
- **`HEADROOM_WORKSPACE`** (host-side) is the directory the compose file bind-mounts into the container as `/workspace`. It behaves like CWD in a native (non-Docker) run.
- **`HEADROOM_WORKSPACE_DIR`** (inside the container) is the canonical Headroom state root from the [filesystem contract](/docs/filesystem-contract). The compose file sets it to `/tmp/headroom-home/.headroom` so the proxy resolves savings, logs, TOIN, and memory under the bind-mounted `${HOME}/.headroom`.
You do not need to set `HEADROOM_WORKSPACE_DIR` manually when using the shipped compose file — it is already in the `environment:` block.
</Callout>
### macOS / Linux
```bash
export HEADROOM_HOST_HOME="$HOME"
export HEADROOM_WORKSPACE="$PWD"
docker compose -f docker/docker-compose.native.yml up proxy
```
### Windows PowerShell
```powershell
$env:HEADROOM_HOST_HOME = $HOME
$env:HEADROOM_WORKSPACE = (Get-Location).Path
docker compose -f docker/docker-compose.native.yml up proxy
```
You can also run one-off CLI commands through compose:
```bash
docker compose -f docker/docker-compose.native.yml run --rm cli learn
docker compose -f docker/docker-compose.native.yml run --rm cli mcp install
```
## Environment passthrough
The wrapper forwards Headroom and provider environment variables into the container, including common prefixes such as:
- `HEADROOM_`
- `ANTHROPIC_`
- `OPENAI_`
- `GEMINI_`
- `AWS_`
- `GOOGLE_` / `GOOGLE_CLOUD_`
- `AZURE_`
- `OTEL_`
That keeps provider auth and runtime config working without maintaining a separate env file for the container.
## Notes
- Docker is the only required Headroom runtime dependency on the host.
- Wrapped tools like Claude Code, Codex CLI, Aider, and Cursor still run on the host when you use `headroom wrap ...`.
- The install scripts are idempotent: rerunning them refreshes the wrapper and image without duplicating shell profile blocks.
- For persistent service and task installs, use the Python-native `headroom install ...` workflow — see [Persistent Installs](/docs/persistent-installs).
- For Docker-native `headroom install ...`, the wrapper persists its profile manifest under `~/.headroom/deploy/<profile>/`.
- The `rtk` binary is not bundled in the Docker image. Dashboard CLI-filtering
savings figures show as "not installed" (not `0`) until `rtk` is installed
inside the container.
## Next steps
<Cards>
<Card title="Quickstart" href="/docs/quickstart" />
<Card title="Proxy Server" href="/docs/proxy" />
<Card title="Installation (pip / npm / plain Docker)" href="/docs/installation" />
</Cards>
+276
View File
@@ -0,0 +1,276 @@
---
title: Error Handling
description: How to catch and handle Headroom errors in Python and TypeScript. Error hierarchy, proxy error mapping, and safety guarantees.
---
Headroom provides explicit exceptions for debugging, with a core safety guarantee: **compression failures never break your LLM calls**. If compression fails, the original content passes through unchanged.
## Error Hierarchy
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```
HeadroomError (base class)
+-- HeadroomConnectionError # Cannot reach proxy
+-- HeadroomAuthError # 401 from proxy
+-- HeadroomCompressError # Compression failed (with statusCode)
+-- ConfigurationError # Invalid configuration
+-- ProviderError # Provider issues
+-- StorageError # Storage failures
+-- TokenizationError # Token counting failed
+-- CacheError # Cache operations failed
+-- ValidationError # Validation failures
+-- TransformError # Transform execution failed
```
```ts twoslash
import {
HeadroomError,
HeadroomConnectionError,
HeadroomAuthError,
HeadroomCompressError,
ConfigurationError,
ProviderError,
mapProxyError,
} from 'headroom-ai';
```
</Tab>
<Tab value="Python">
```
HeadroomError (base class)
+-- ConfigurationError # Invalid configuration
+-- ProviderError # Provider issues (unknown model, etc.)
+-- StorageError # Database/storage failures
+-- CompressionError # Compression failures (rare)
+-- ValidationError # Setup validation failures
```
```python
from headroom import (
HeadroomError,
ConfigurationError,
ProviderError,
StorageError,
CompressionError,
ValidationError,
)
```
</Tab>
</Tabs>
## Catching Errors
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress, HeadroomConnectionError, HeadroomAuthError, HeadroomCompressError, HeadroomError } from 'headroom-ai';
try {
const result = await compress(messages, { model: 'gpt-4o' });
} catch (e) {
if (e instanceof HeadroomConnectionError) {
console.error('Cannot reach proxy:', e.message);
} else if (e instanceof HeadroomAuthError) {
console.error('Auth failed:', e.message);
} else if (e instanceof HeadroomCompressError) {
console.error(`Compress failed (${e.statusCode}):`, e.message);
} else if (e instanceof HeadroomError) {
console.error('Headroom error:', e.message, e.details);
}
}
```
</Tab>
<Tab value="Python">
```python
from headroom import (
HeadroomClient,
HeadroomError,
ConfigurationError,
StorageError,
)
try:
client = HeadroomClient(...)
response = client.chat.completions.create(...)
except ConfigurationError as e:
print(f"Config issue: {e}")
print(f"Details: {e.details}")
except StorageError as e:
print(f"Storage issue: {e}")
# Headroom continues to work, just without metrics persistence
except HeadroomError as e:
print(f"Headroom error: {e}")
```
</Tab>
</Tabs>
## Error Types in Detail
### ConfigurationError
Raised when configuration is invalid.
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { ConfigurationError } from 'headroom-ai';
// ConfigurationError is thrown when the proxy returns
// a configuration_error type in its error response
```
</Tab>
<Tab value="Python">
```python
try:
client = HeadroomClient(
original_client=OpenAI(),
provider=OpenAIProvider(),
default_mode="invalid_mode", # Will raise ConfigurationError
)
except ConfigurationError as e:
print(f"Config error: {e}")
print(f"Field: {e.details.get('field')}")
```
</Tab>
</Tabs>
### ProviderError
Raised for provider-specific issues (unknown model, API error, token counting failure).
```python
try:
response = client.chat.completions.create(
model="unknown-model-xyz",
messages=[...],
)
except ProviderError as e:
print(f"Provider error: {e}")
print(f"Provider: {e.details.get('provider')}")
```
### StorageError
Raised when database operations fail. Storage errors do not affect core functionality -- the application can continue without historical metrics.
```python
try:
metrics = client.get_metrics()
except StorageError as e:
metrics = [] # Continue without historical metrics
```
### CompressionError
Raised when compression fails (rare). In practice, compression errors are caught internally and the original content passes through unchanged. This exception is only raised in strict mode.
### HeadroomConnectionError (TypeScript)
Raised when the TypeScript SDK cannot connect to the Headroom proxy.
```ts twoslash
import { compress, HeadroomConnectionError } from 'headroom-ai';
try {
await compress(messages, { model: 'gpt-4o' });
} catch (e) {
if (e instanceof HeadroomConnectionError) {
console.error('Is the proxy running? Start with: headroom proxy');
}
}
```
## Proxy Error Mapping
The TypeScript SDK automatically maps proxy error responses to the correct error class:
| HTTP Status | Proxy Error Type | TypeScript Class |
|-------------|-----------------|-----------------|
| 401 | -- | `HeadroomAuthError` |
| 4xx/5xx | `configuration_error` | `ConfigurationError` |
| 4xx/5xx | `provider_error` | `ProviderError` |
| 4xx/5xx | `storage_error` | `StorageError` |
| 4xx/5xx | `tokenization_error` | `TokenizationError` |
| 4xx/5xx | `validation_error` | `ValidationError` |
| 4xx/5xx | `transform_error` | `TransformError` |
| 4xx/5xx | (other) | `HeadroomCompressError` |
The `mapProxyError()` function handles this mapping:
```ts twoslash
import { mapProxyError } from 'headroom-ai';
const error = mapProxyError(400, 'configuration_error', 'Invalid mode');
// Returns a ConfigurationError instance
```
## Error Details
All Headroom exceptions include a `details` dict/object with additional context:
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { HeadroomError } from 'headroom-ai';
// HeadroomError.details is Record<string, any> | undefined
// HeadroomCompressError also has .statusCode and .errorType
```
</Tab>
<Tab value="Python">
```python
try:
client = HeadroomClient(...)
except HeadroomError as e:
print(f"Error: {e}")
print(f"Type: {type(e).__name__}")
print(f"Details: {e.details}")
# Details might include:
# - field: which config field caused the error
# - provider: which provider was involved
# - model: which model was requested
# - original_error: underlying exception
```
</Tab>
</Tabs>
## Safety Guarantee
If compression fails, the original content passes through unchanged. Your LLM calls never fail due to Headroom:
```python
messages = [
{"role": "tool", "content": "malformed json {{{"}
]
# This will NOT raise an exception
# The malformed content passes through unchanged
response = client.chat.completions.create(
model="gpt-4o",
messages=messages,
)
```
## Best Practices
1. **Catch specific exceptions** rather than broad `Exception` to avoid hiding real bugs
2. **Let StorageError pass** -- storage errors do not affect core compression functionality
3. **Validate on startup** with `client.validate_setup()` to catch configuration issues early
4. **Enable logging** at WARNING level to see when compression is skipped
```python
import logging
logging.basicConfig(level=logging.WARNING)
# WARNING:headroom.transforms.smart_crusher:Skipping compression: invalid JSON
```
+157
View File
@@ -0,0 +1,157 @@
---
title: Failure Learning
description: Offline failure analysis for coding agents. Analyzes past sessions, finds what went wrong, correlates with what fixed it, and writes project-level learnings.
---
`headroom learn` analyzes past coding agent sessions, finds what went wrong, correlates each failure with what eventually worked, and writes specific project-level learnings that prevent the same mistakes next session.
## Quick Start
```bash
# See recommendations for current project (dry-run, no changes)
headroom learn
# Write recommendations to CLAUDE.local.md and MEMORY.md
headroom learn --apply
# Analyze a specific project
headroom learn --project ~/my-project --apply
# Analyze all projects
headroom learn --all --apply
# Write to the team-shared CLAUDE.md instead of CLAUDE.local.md
headroom learn --apply --target CLAUDE.md
```
## Success Correlation
The core innovation. Instead of cataloging failures ("Read failed 5 times"), Headroom finds what the model did to **fix** each failure:
- **Failed**: `Read axion-formats/src/main/java/.../FirstClassEntity.java`
- **Then succeeded**: `Read axion-scala-common/src/main/scala/.../FirstClassEntity.scala`
- **Learning**: "`FirstClassEntity` is at `axion-scala-common/`, not `axion-formats/`"
This produces specific, actionable corrections -- not generic advice.
## What It Learns
### Environment Facts
Which runtime commands work vs fail.
```markdown
### Environment
- **Python**: use `uv run python` (not `python3` -- modules not available outside venv)
```
### File Path Corrections
Wrong paths the model keeps guessing, with the correct locations.
```markdown
### File Path Corrections
- `axion-common/src/.../AxionSparkConstants.scala`
-> actually at `axion-spark-common/src/.../AxionSparkConstants.scala`
```
### Search Scope
Which directories to search in (narrow paths fail, broader ones work).
```markdown
### Search Scope
- Don't search `axion-model/` -> use `axion/` (the repo root)
```
### Command Patterns
How commands should (and should not) be run.
```markdown
### Command Patterns
- **user_prefers_manual**: User rejected gradle 18 times -- show the command, don't execute
- **python_runtime**: Use `uv run python` not `python3` (ModuleNotFoundError)
```
### Known Large Files
Files that need `offset`/`limit` with Read.
```markdown
### Known Large Files
- `proxy/server.py` (~8000 lines) -- always use offset/limit
```
## Where Learnings Go
| Pattern | Destination | Why |
|---------|-------------|-----|
| Environment, paths, search scope, commands, large files | **CLAUDE.local.md** | Personal facts (machine-specific paths), gitignored by default |
| Missing paths, retry patterns, permissions | **MEMORY.md** | May change, agent-specific |
`CLAUDE.local.md` lives in your project directory. It is the **personal** local
memory file in Claude Code's [memory convention](https://docs.claude.com/en/docs/claude-code/memory)
— meant to be gitignored — so machine-specific learnings (absolute paths,
tool-discovery byproducts) don't pollute the team-shared `CLAUDE.md`. Make sure
`CLAUDE.local.md` is listed in your `.gitignore`. Pass `--target CLAUDE.md` to
opt into the shared file instead, or `--target <path>` for any custom location.
MEMORY.md lives in `~/.claude/projects/*/memory/`.
If an older Headroom version already wrote a learned-patterns block into your
team-shared `CLAUDE.md`, the next `headroom learn --apply` moves it into
`CLAUDE.local.md` and prints a warning so you can review the diff before
committing. If `CLAUDE.md` contained nothing but the Headroom block, it is
removed entirely.
## Marker-Based Updates
Headroom manages a clearly-delimited section in each file:
```markdown
<!-- headroom:learn:start -->
## Headroom Learned Patterns
*Auto-generated by `headroom learn` -- do not edit manually*
...
<!-- headroom:learn:end -->
```
On re-run, only the content between markers is replaced. Your existing file content is preserved.
## Architecture
The system is built with an adapter pattern so it can support multiple agent systems:
- **Scanners** read tool-specific log formats (e.g., `~/.claude/projects/*.jsonl`) and produce normalized `ToolCall` sequences
- **Analyzers** work on `ToolCall` data -- same analysis logic for any agent system
- **Writers** output to tool-specific context injection mechanisms (e.g., CLAUDE.md)
To add support for a new agent (e.g., Cursor), you write a Scanner that reads its log format and a Writer that outputs to `.cursorrules`. The analyzers stay the same.
## CLI Reference
```bash
headroom learn [OPTIONS]
Options:
--project PATH Project directory to analyze (default: current directory)
--all Analyze all discovered projects
--apply Write recommendations (default: dry-run)
--target TEXT Context file to write to, Claude Code only (default:
CLAUDE.local.md). Relative to project root, or absolute.
--agent TEXT Agent to analyze (e.g. claude, codex, gemini, opencode)
--model TEXT LLM model to use for analysis
--workers INT Number of parallel workers
```
## Real-World Results
Tested on 67,583 tool calls across 23 projects:
| Metric | Value |
|--------|-------|
| Failure rate | 7.5% (5,066 failures) |
| Corrections extracted | 164 per project (avg) |
| Path corrections | 22 (axion project) |
| Search scope corrections | 24 (axion project) |
| Command patterns learned | 5 (axion project) |
+147
View File
@@ -0,0 +1,147 @@
---
title: Filesystem Contract
description: Where Headroom writes config, runtime state, logs, and caches — the canonical two-root model, precedence rules, and Docker behavior.
---
Headroom writes configuration, runtime state, logs, and caches to a small set of well-known paths under the user's home directory. This page is the source of truth for where those paths live, how to override them, and how they behave inside Docker containers.
## Two-root model
| Variable | Default | Purpose | Typical access |
|---|---|---|---|
| `HEADROOM_CONFIG_DIR` | `~/.headroom/config` | User/admin-authored configuration (model catalogs, plugin settings, etc.) | Read-mostly |
| `HEADROOM_WORKSPACE_DIR` | `~/.headroom` | Runtime state written by the proxy and CLI (savings, logs, memory DB, telemetry, caches) | Read-write |
Both variables are recognized by the Python proxy / CLI and the npm SDK. They are **additive** — every pre-existing per-resource env var (`HEADROOM_SAVINGS_PATH`, `HEADROOM_TOIN_PATH`, `HEADROOM_SUBSCRIPTION_STATE_PATH`, `HEADROOM_MODEL_LIMITS`, ...) continues to work with identical semantics.
## Precedence
For every per-resource helper, resolution follows this order:
```
explicit argument
│ falls through when None/""
per-resource env var (e.g. HEADROOM_SAVINGS_PATH)
│ falls through when unset/blank
derived from canonical root
│ e.g. ${HEADROOM_WORKSPACE_DIR}/proxy_savings.json
default (e.g. ~/.headroom/proxy_savings.json)
```
Examples:
- `HEADROOM_WORKSPACE_DIR=/mnt/state` → savings land at `/mnt/state/proxy_savings.json` unless `HEADROOM_SAVINGS_PATH` overrides.
- `HEADROOM_SAVINGS_PATH=/custom/savings.json` always wins, even when `HEADROOM_WORKSPACE_DIR` is set.
- Unset both and the default is `~/.headroom/proxy_savings.json`.
## Bucket assignments
### Workspace bucket (`HEADROOM_WORKSPACE_DIR`)
| Resource | Default path | Legacy env var |
|---|---|---|
| Proxy savings ledger | `${WORKSPACE_DIR}/proxy_savings.json` | `HEADROOM_SAVINGS_PATH` |
| TOIN telemetry JSON | `${WORKSPACE_DIR}/toin.json` | `HEADROOM_TOIN_PATH` |
| Subscription tracker state | `${WORKSPACE_DIR}/subscription_state.json` | `HEADROOM_SUBSCRIPTION_STATE_PATH` |
| Memory SQLite | `${WORKSPACE_DIR}/memory.db` | CLI `--memory-db-path` |
| Native memory directory | `${WORKSPACE_DIR}/memories/` | `MemoryConfig.native_memory_dir` |
| License cache | `${WORKSPACE_DIR}/license_cache.json` | — |
| Session stats JSONL | `${WORKSPACE_DIR}/session_stats.jsonl` | — |
| Memory sync state | `${WORKSPACE_DIR}/sync_state.json` | — |
| Memory bridge state | `${WORKSPACE_DIR}/bridge_state.json` | — |
| Proxy log directory | `${WORKSPACE_DIR}/logs/` | — |
| HTTP 400 debug dumps | `${WORKSPACE_DIR}/logs/debug_400/` | — |
| Vendored `rtk` binary | `${WORKSPACE_DIR}/bin/rtk[.exe]` | — |
| Vendored `lean-ctx` binary | `${WORKSPACE_DIR}/bin/lean-ctx[.exe]` | — |
| Deployment profiles | `${WORKSPACE_DIR}/deploy/` | — |
| Beacon lock file | `${WORKSPACE_DIR}/.beacon_lock_<port>` | — |
### Config bucket (`HEADROOM_CONFIG_DIR`)
| Resource | Default path | Legacy env var |
|---|---|---|
| Models catalog | `${CONFIG_DIR}/models.json` | `HEADROOM_MODEL_LIMITS` (content override) |
| Plugin settings | `${CONFIG_DIR}/plugins/<name>/...` | — |
### Backward compatibility — models.json
`models.json` historically lived at `~/.headroom/models.json` (i.e. in the workspace root, not in `config/`). For a seamless migration the Python providers check **both** locations in this order:
1. `${HEADROOM_CONFIG_DIR}/models.json` (new canonical location)
2. `${HEADROOM_WORKSPACE_DIR}/models.json` (legacy fallback)
Existing installs continue to work unchanged. New installs are encouraged to put `models.json` in the config bucket.
## Plugin authors
Two helpers give plugins isolated, per-plugin directories under both roots:
### Python
```python
from headroom import paths
cfg_dir = paths.plugin_config_dir("my-plugin")
# → ~/.headroom/config/plugins/my-plugin
state_dir = paths.plugin_workspace_dir("my-plugin")
# → ~/.headroom/plugins/my-plugin
cfg_dir.mkdir(parents=True, exist_ok=True)
(cfg_dir / "settings.json").write_text("{}")
```
### npm SDK
```typescript
import { pluginConfigDir, pluginWorkspaceDir } from "@headroom/sdk";
const cfgDir = pluginConfigDir("my-plugin");
const stateDir = pluginWorkspaceDir("my-plugin");
```
Plugin-author helpers reject names containing `/` or `\` to keep the namespace flat.
## Docker naming overlap: `HEADROOM_WORKSPACE` vs `HEADROOM_WORKSPACE_DIR`
These are **two different variables** with different semantics, both retained for backward compatibility:
| Variable | Scope | Meaning |
|---|---|---|
| `HEADROOM_WORKSPACE` | Host-side (Docker) | Directory to bind-mount into the container as `/workspace` (equivalent to CWD in native runs). Used by `docker-compose.native.yml`. |
| `HEADROOM_WORKSPACE_DIR` | Inside the container | Canonical Headroom state root. Resolves to `/tmp/headroom-home/.headroom` inside the official container image, which in turn bind-mounts to `${HOME}/.headroom` on the host. |
The official Docker bootstrap (compose file, `scripts/install.sh`, and the Python `install` command) sets `HEADROOM_WORKSPACE_DIR` and `HEADROOM_CONFIG_DIR` inside the container so the proxy resolves state to the bind-mounted path without any user action.
## Project-scoped `.headroom/` directories
A few code paths deliberately use **project-local** `.headroom/` paths resolved relative to the current working directory rather than the canonical workspace root:
- `headroom/proxy/server.py` — project-scoped memory DB default
- `headroom/memory/mcp_server.py` — project-scoped memory DB default
- `headroom/cli/wrap.py` — project-scoped memory and hook artifacts
These **do not obey** `HEADROOM_WORKSPACE_DIR`. This is intentional: it preserves the "project memory lives in the project directory" invariant. Users who want a single centrally located memory store can pass `--memory-db-path <path>` explicitly or set the path via the plugin API.
## Legacy per-resource env vars
Every legacy env var continues to work with its original semantics (raw string in, raw string out — no tilde expansion, no path-separator normalization), ensuring byte-for-byte backward compatibility.
Full list:
- `HEADROOM_SAVINGS_PATH`
- `HEADROOM_TOIN_PATH`
- `HEADROOM_SUBSCRIPTION_STATE_PATH`
- `HEADROOM_MODEL_LIMITS` (content override — JSON string or file path)
## See also
<Cards>
<Card title="Configuration" href="/docs/configuration" />
<Card title="Docker-Native Install" href="/docs/docker-install" />
<Card title="Persistent Installs" href="/docs/persistent-installs" />
<Card title="Memory" href="/docs/memory" />
</Cards>
+161
View File
@@ -0,0 +1,161 @@
---
title: How Compression Works
description: Understand Headroom's three-stage compression pipeline, automatic content routing, and how different content types are compressed.
---
Headroom automatically detects what kind of content you're sending and routes it to the right compressor. You don't need to configure anything -- just call `compress()` and the pipeline handles the rest.
## The Three-Stage Pipeline
Every request flows through three stages:
```
┌──────────────┐ ┌────────────────┐
│ CacheAligner │────>│ ContentRouter │
│ │ │ │
│ Report │ │ Detect type & │
│ prefix drift │ │ route to best │
│ for cache │ │ compressor │
└──────────────┘ └────────────────┘
```
1. **CacheAligner** detects dynamic content (dates, user context) in your system prompt and reports prefix drift so the caller can keep the static prefix cacheable across requests.
2. **ContentRouter** inspects each tool output and routes it to the optimal compressor -- SmartCrusher for JSON arrays, CodeAwareCompressor for source code, LogCompressor for build output, and so on.
## Content Type Detection
The router auto-detects content type by analyzing structure and patterns. No manual hints required.
| Content Type | Detection Signal | Compressor | Typical Savings |
|---|---|---|---|
| JSON arrays | Valid JSON with array elements | SmartCrusher | 70-90% |
| Source code | Syntax patterns, indentation, keywords | CodeAwareCompressor | 40-70% (opt-in; disabled by default) |
| Search results | `file:line:content` format | SearchCompressor | 80-95% |
| Build/test logs | Timestamps, log levels, pytest/npm markers | LogCompressor | 85-95% |
| Diffs | Unified diff format | DiffCompressor | 60-80% |
| HTML | Tag structure | HTMLCompressor | 50-70% |
| Plain text | Fallback | TextCompressor | 60-80% |
## Quick Start
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from "headroom-ai";
const messages = [
{ role: "system" as const, content: "You are a helpful assistant." },
{ role: "user" as const, content: "Summarize this data" },
{ role: "tool" as const, content: '{"results": [...]}', tool_call_id: "call_1" },
];
const result = await compress(messages);
console.log(`Tokens saved: ${result.tokensSaved}`);
console.log(`Compression ratio: ${result.compressionRatio}`);
```
</Tab>
<Tab value="Python">
```python
from headroom import compress
result = compress(content)
print(result.compressed)
print(f"Saved {result.savings_percentage:.0f}% tokens")
```
</Tab>
</Tabs>
## Configuring the Compressor
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from "headroom-ai";
const result = await compress(messages, {
model: "gpt-4o",
tokenBudget: 50000,
});
console.log(`Before: ${result.tokensBefore} tokens`);
console.log(`After: ${result.tokensAfter} tokens`);
console.log(`Transforms: ${result.transformsApplied.join(", ")}`);
```
</Tab>
<Tab value="Python">
```python
# Advanced / Internal API -- prefer `from headroom import compress` for typical use
from headroom.compression import UniversalCompressor, UniversalCompressorConfig
config = UniversalCompressorConfig(
compression_ratio_target=0.5, # Keep 50% of content
use_entropy_preservation=True, # Preserve UUIDs, hashes
use_magika=True, # ML-based content detection
ccr_enabled=True, # Store originals for retrieval
)
compressor = UniversalCompressor(config=config)
result = compressor.compress(content)
print(f"Type: {result.content_type}")
print(f"Handler: {result.handler_used}")
print(f"Saved: {result.savings_percentage:.0f}%")
```
</Tab>
</Tabs>
## Structure Preservation
Headroom doesn't blindly truncate. It identifies what matters in each content type and preserves it:
| Content Type | What's Preserved | What's Compressed |
|---|---|---|
| **JSON** | Keys, brackets, booleans, nulls, short values, UUIDs | Long string values, whitespace |
| **Code** | Imports, function signatures, class definitions, types | Function bodies, comments |
| **Logs** | Timestamps, log levels, error messages, stack traces | Repeated patterns, verbose details |
| **Text** | High-entropy tokens (IDs, hashes), headers | Low-information content |
## Real Compression Ratios
| Content Type | Compression | Speed | What's Preserved |
|---|---|---|---|
| JSON (large arrays) | 70-90% | ~1ms | All keys, structure |
| Source code (Python) | 50-70% | ~10ms | Signatures, imports |
| Search results | 80-95% | ~2ms | Relevant matches |
| Build logs | 85-95% | ~3ms | Errors, stack traces |
| Plain text | 60-80% | ~5ms | High-entropy tokens |
## Batch Compression
For multiple contents, batch compression is more efficient:
```python
# Advanced / Internal API -- prefer `from headroom import compress` for typical use
from headroom.compression import UniversalCompressor
compressor = UniversalCompressor()
contents = [
'{"users": [...]}',
'def hello(): pass',
'Plain text content',
]
results = compressor.compress_batch(contents)
for result in results:
print(f"{result.content_type}: {result.savings_percentage:.0f}% saved")
```
## What Happens Under the Hood
When you call `compress()`, here is the full sequence:
1. **Content detection** -- Magika (ML-based) or pattern matching identifies the content type
2. **Structure extraction** -- A handler extracts a structure mask marking what to preserve
3. **Compression** -- Non-structural content is compressed (SmartCrusher, LLMLingua, or text utilities)
4. **CCR storage** -- If enabled, the original is stored for retrieval when the LLM needs full context
<Callout type="info" title="Zero-config by default">
The pipeline works out of the box with no configuration. All detection, routing, and compression happens automatically. Configuration is available when you need fine-grained control.
</Callout>
+152
View File
@@ -0,0 +1,152 @@
---
title: Image Compression
description: ML-powered image compression that reduces vision model token usage by 40-90% while maintaining answer accuracy.
---
Vision models charge by the token, and images are expensive. A single 1024x1024 image costs ~765 tokens on OpenAI. Headroom's image compression uses a trained ML router to analyze your query and automatically select the optimal compression technique, saving 40-90% of image tokens.
## How It Works
```
User uploads image + asks question
|
[Query Analysis]
TrainedRouter (MiniLM from HuggingFace)
Classifies: "What animal is this?" -> full_low
|
[Image Analysis]
SigLIP analyzes image properties
(has text? complex? fine details?)
|
[Apply Compression]
OpenAI: detail="low"
Anthropic: Resize to 512px
Google: Resize to 768px
|
Compressed request to LLM
```
The router is a fine-tuned MiniLM classifier (`chopratejas/technique-router` on HuggingFace) with 93.7% accuracy across 1,157 training examples.
## Compression Techniques
| Technique | Savings | When Used | Example Query |
|---|---|---|---|
| `full_low` | ~87% | General understanding | "What is this?", "Describe the scene" |
| `preserve` | 0% | Fine details needed | "Count the whiskers", "Read the serial number" |
| `crop` | 50-90% | Region-specific queries | "What's in the corner?", "Focus on the background" |
| `transcode` | ~99% | Text extraction | "Read the sign", "Transcribe the document" |
## Quick Start
### With Headroom Proxy (Zero Code Changes)
```bash
# Start the proxy
headroom proxy --port 8787
# Connect your client -- images are compressed automatically
ANTHROPIC_BASE_URL=http://localhost:8787 claude
```
### With HeadroomClient
```python
from headroom import HeadroomClient
client = HeadroomClient(provider="openai")
response = client.chat.completions.create(
model="gpt-4o",
messages=[{
"role": "user",
"content": [
{"type": "text", "text": "What animal is this?"},
{"type": "image_url", "image_url": {"url": "data:image/jpeg;base64,..."}}
]
}]
)
# Image automatically compressed with detail="low" (87% savings)
```
### Direct API
```python
from headroom.image import ImageCompressor
compressor = ImageCompressor()
# Compress images in messages
compressed_messages = compressor.compress(messages, provider="openai")
# Check savings
print(f"Saved {compressor.last_savings:.0f}% tokens")
print(f"Technique: {compressor.last_result.technique.value}")
```
## Provider Support
The compressor adapts its strategy per provider:
| Provider | Compression Method | Details |
|---|---|---|
| **OpenAI** | Sets `detail="low"` | Native detail parameter |
| **Anthropic** | Resizes to 512px | PIL-based resize |
| **Google Gemini** | Resizes to 768px | Optimized for Gemini's 768x768 tile system |
### Token Savings by Provider
**OpenAI** (1024x1024 image):
| Technique | Before | After | Savings |
|---|---|---|---|
| `full_low` | 765 tokens | 85 tokens | 89% |
| `preserve` | 765 tokens | 765 tokens | 0% |
**Anthropic** (1024x1024 image):
| Before | After | Savings |
|---|---|---|
| ~1,398 tokens | ~349 tokens | 75% |
**Google Gemini** (1536x1536 image):
| Before | After | Savings |
|---|---|---|
| 1,032 tokens (4 tiles) | 258 tokens (1 tile) | 75% |
## Configuration
```python
from headroom.image import ImageCompressor
compressor = ImageCompressor(
model_id="chopratejas/technique-router", # HuggingFace model
use_siglip=True, # Enable image analysis
device="cuda", # Use GPU if available (auto, cuda, cpu, mps)
)
```
### Proxy Configuration
```bash
# Image optimization is part of the ContentRouter path when the optional image
# dependencies are installed. There are no public proxy CLI image toggles in
# the current release.
headroom proxy
```
## Performance
| Metric | Value |
|---|---|
| Router inference | ~10ms (CPU), ~2ms (GPU) |
| Image resize | ~5-20ms |
| First request | +2-3s (model download, cached after) |
| Router accuracy | 93.7% |
| Model size | ~128MB |
| GPU memory (SigLIP) | ~400MB |
<Callout type="info" title="Automatic with the proxy">
When using the Headroom proxy, image compression happens automatically on every request that contains images. No code changes needed.
</Callout>
+109
View File
@@ -0,0 +1,109 @@
---
title: Introduction
description: Headroom is the context optimization layer for LLM applications. Compress tool outputs, DB results, file reads, and RAG results before they reach the model. Same answers, fraction of the tokens.
---
<StatsSection />
Headroom compresses everything your AI agent reads -- tool outputs, database results, file reads, RAG retrievals, API responses -- before it reaches the LLM. The model sees less noise, responds faster, and costs less.
## Quick preview
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from 'headroom-ai';
const messages = [
{ role: 'user' as const, content: 'Analyze these results' },
];
const result = await compress(messages, { model: 'gpt-4o' });
console.log(`Saved ${result.tokensSaved} tokens (${(result.compressionRatio * 100).toFixed(0)}%)`);
```
</Tab>
<Tab value="Python">
```python
from headroom import compress
result = compress(messages, model="gpt-4o")
response = client.messages.create(
model="gpt-4o",
messages=result.messages,
)
print(f"Saved {result.tokens_saved} tokens ({result.compression_ratio:.0%})")
```
</Tab>
</Tabs>
## Community stats
<LiveStats />
## What gets compressed
| Content type | What happens | Typical savings |
|---|---|---|
| JSON arrays (tool outputs) | Statistical analysis keeps errors, anomalies, boundaries | 70--90% |
| Source code | AST-aware compression preserves signatures, collapses bodies | 40--70% (opt-in; disabled by default) |
| Build/test logs | Keeps failures and errors, drops passing noise | 80--95% |
| Search results | Ranks by relevance, keeps top matches | 60--80% |
| Plain text | ModernBERT token classification removes redundancy | 30--50% |
| Git diffs | Preserves change hunks, drops unchanged context | 40--60% |
| Images | ML router selects optimal resize/quality tradeoff | 40--90% |
## Where Headroom fits
```
Your Agent / App
|
| tool outputs, logs, DB reads, RAG results, file reads, API responses
v
Headroom <-- proxy, Python library, TS SDK, or framework integration
|
v
LLM Provider (OpenAI, Anthropic, Google, Bedrock, 100+ via LiteLLM)
```
Headroom works as a **transparent proxy** (zero code changes), a **Python function** (`compress()`), a **TypeScript function** (`compress()`), or a **framework integration** (LangChain, Agno, Strands, LiteLLM, Vercel AI SDK, MCP).
## Real-world results
**100 production log entries. One critical error buried at position 67.**
| Metric | Baseline | Headroom |
|---|---|---|
| Input tokens | 10,144 | 1,260 |
| Correct answers | **4/4** | **4/4** |
87.6% fewer tokens. Same answer. The FATAL error was automatically preserved -- not by keyword matching, but by statistical analysis of field variance.
| Scenario | Before | After | Savings |
|---|---|---|---|
| Code search (100 results) | 17,765 | 1,408 | **92%** |
| SRE incident debugging | 65,694 | 5,118 | **92%** |
| Codebase exploration | 78,502 | 41,254 | **47%** |
| GitHub issue triage | 54,174 | 14,761 | **73%** |
## Key Features
<KeyFeatures />
## Framework Integrations
<FrameworkIntegrations />
## Nothing is lost
Compressed content goes into the CCR store (Compress-Cache-Retrieve). The LLM gets a `headroom_retrieve` tool and can fetch full originals when it needs more detail. Compression is aggressive but reversible.
## Next steps
<Cards>
<Card title="Quickstart" href="/docs/quickstart" />
<Card title="Installation" href="/docs/installation" />
<Card title="Proxy Server" href="/docs/proxy" />
<Card title="Vercel AI SDK" href="/docs/vercel-ai-sdk" />
<Card title="LangChain" href="/docs/langchain" />
<Card title="How Compression Works" href="/docs/how-compression-works" />
</Cards>
+296
View File
@@ -0,0 +1,296 @@
---
title: Installation
description: Install Headroom via pip, npm, or Docker. Includes all Python extras, TypeScript setup, Docker image tags, and environment variables.
---
---
## Install options
**pip** - you're writing Python, or you need the CLI (`headroom proxy`, `wrap`, `mcp`, `learn`, `perf`), regardless of what language your app is in.
**npm** - you're writing TypeScript/Node and want inline `compress()`, SDK wrapping (`withHeadroom`), or Vercel AI SDK middleware.
## Python
Headroom requires **Python 3.10+** and is published as `headroom-ai` on PyPI.
Current release wheels are built for Python **3.10 through 3.13** on Linux
(manylinux_2_28 x86_64 / aarch64) and macOS (Apple Silicon). Other targets —
**Windows** and **Intel macOS** — fall back to building the Rust extension
from the sdist and need a working native toolchain. If your installer is
using a newer Python, force a supported interpreter.
### Core package
```bash
pip install headroom-ai
```
The core package includes the `compress()` function, SmartCrusher, CacheAligner, and live-zone ContentRouter compression. No heavy dependencies.
> **Note:** IntelligentContext / RollingWindow (score-based history dropping) were retired in PR-B1. Headroom compresses fresh tool output and new turns only — it does not drop conversation history.
### Extras
Install only what you need, or grab everything with `[all]`:
```bash
pip install "headroom-ai[all]"
```
| Extra | What it adds | Install command |
|---|---|---|
| `proxy` | Proxy server, MCP tools, HTTP API | `pip install "headroom-ai[proxy]"` |
| `ml` | Kompress (ModernBERT text compression, requires PyTorch) | `pip install "headroom-ai[ml]"` |
| `code` | CodeCompressor (tree-sitter AST parsing) | `pip install "headroom-ai[code]"` |
| `memory` | Persistent memory (sqlite-vec, sentence-transformers) — pure-Python default backend, no compiler | `pip install "headroom-ai[memory]"` |
| `vector` | Optional HNSW vector backend (hnswlib) — needs a C++ toolchain; **not in `[all]`** | `pip install "headroom-ai[vector]"` |
| `relevance` | fastembed-based relevance scoring (BAAI/bge-small-en-v1.5, ONNX) | `pip install "headroom-ai[relevance]"` |
| `image` | Image compression (Pillow, ONNX runtime, OCR) | `pip install "headroom-ai[image]"` |
| `reports` | HTML/Markdown report generation (Jinja2) | `pip install "headroom-ai[reports]"` |
| `otel` | OpenTelemetry exporter (OTLP) | `pip install "headroom-ai[otel]"` |
| `voice` | Voice/audio support | `pip install "headroom-ai[voice]"` |
| `mcp` | MCP server tools (`headroom_compress`, `headroom_retrieve`, `headroom_stats`) | `pip install "headroom-ai[mcp]"` |
| `langchain` | LangChain `HeadroomChatModel` wrapper | `pip install "headroom-ai[langchain]"` |
| `agno` | Agno `HeadroomAgnoModel` wrapper | `pip install "headroom-ai[agno]"` |
| `evals` | Evaluation framework (GSM8K, SQuAD, BFCL benchmarks) | `pip install "headroom-ai[evals]"` |
| `pytorch-mps` | Apple-GPU (MPS) memory-embedder offload — **macOS only**, not in `[all]` (torch + sentence-transformers); opt in with `HEADROOM_EMBEDDER_RUNTIME=pytorch_mps` | `pip install "headroom-ai[pytorch-mps]"` |
| `all` | Everything above | `pip install "headroom-ai[all]"` |
You can combine extras:
```bash
pip install "headroom-ai[proxy,langchain,ml]"
```
### Windows
There are no prebuilt Windows wheels yet, so `pip install headroom-ai`
(and `uv tool install`, `pipx install`, …) will fall back to building the
Rust extension from the sdist. The build needs the MSVC toolchain on
`PATH`. Without it you'll see:
```text
error: linker `link.exe` not found
note: please ensure that Visual Studio 2017 or later, or Build Tools for
Visual Studio were installed with the Visual C++ option
```
To install the prerequisites:
1. **MSVC toolchain** — install **[Build Tools for Visual Studio](https://visualstudio.microsoft.com/visual-cpp-build-tools/)**
and select the *"Desktop development with C++"* workload (this gives
you `link.exe`). VS Code on its own is not enough.
2. **Rust** — install via [rustup](https://rustup.rs/). Choose the
`stable-x86_64-pc-windows-msvc` toolchain so Cargo uses the MSVC
linker you just installed.
3. **Open a fresh PowerShell** so the installer's PATH updates take
effect, then run the install:
```powershell
uv tool install --python 3.13 "headroom-ai[all]"
# or
pip install "headroom-ai[all]"
```
If you'd rather avoid the native toolchain entirely, run Headroom
through Docker — see the [Docker](#docker) section below. Native
Windows wheels are tracked in [#636](https://github.com/chopratejas/headroom/issues/636).
### pipx
`pipx` creates one virtual environment per app. If that environment uses an
unsupported Python version, `pipx` may resolve an older compatible Headroom
release instead of the newest one.
Use Python 3.13 explicitly:
```bash
pipx install --python python3.13 "headroom-ai[all]"
```
For a pinned release:
```bash
pipx install --python python3.13 "headroom-ai[all]==0.21.4"
```
Check which Python an existing `pipx` environment uses:
```bash
pipx list
```
### Verify the install
```bash
python -c "import headroom; print(headroom.__version__)"
```
## TypeScript / Node.js
The TypeScript SDK is published as `headroom-ai` on npm. It requires **Node.js 18+**. It is a library you import — it does **not** install the `headroom` CLI (`headroom wrap`, `headroom proxy`, etc.), which ships only with the Python package above.
```bash
npm install headroom-ai
```
Or with other package managers:
```bash
pnpm add headroom-ai
yarn add headroom-ai
```
<Callout type="info" title="The TS SDK needs a running proxy">
The TypeScript SDK sends messages to the Headroom proxy over HTTP for compression. The proxy runs the full compression pipeline (Python). Start it before using the SDK:
```bash
pip install "headroom-ai[proxy]"
headroom proxy --port 8787
```
Then point the SDK at it:
```ts
import { compress } from 'headroom-ai';
const result = await compress(messages, {
baseUrl: 'http://localhost:8787',
});
```
</Callout>
### Verify the install
```bash
node -e "const h = require('headroom-ai'); console.log('headroom-ai loaded')"
```
## Docker
Pre-built images are published to GitHub Container Registry on every release.
```bash
docker pull ghcr.io/chopratejas/headroom:latest
docker run -p 8787:8787 ghcr.io/chopratejas/headroom:latest
```
<Callout type="info" title="Running Headroom without installing Python or Node?">
If you want a host `headroom` CLI that keeps Headroom itself inside a container — with mounted state, a one-line installer, and a persistent Docker lifecycle — see [Docker-Native Install](/docs/docker-install).
</Callout>
### Image tags
| Tag | Extras | Base image | Description |
|---|---|---|---|
| `latest` | `proxy` | Debian slim | Default image, runs the proxy |
| `<version>` | `proxy` | Debian slim | Pinned version |
| `nonroot` | `proxy` | Debian slim | Runs as non-root user |
| `code` | `proxy,code` | Debian slim | Includes tree-sitter for code compression |
| `code-nonroot` | `proxy,code` | Debian slim | Code compression, non-root |
| `slim` | `proxy` | Distroless | Minimal image, no shell |
| `slim-nonroot` | `proxy` | Distroless | Minimal, non-root |
| `code-slim` | `proxy,code` | Distroless | Code compression, minimal |
| `code-slim-nonroot` | `proxy,code` | Distroless | Code compression, minimal, non-root |
### Build from source
Use Docker Bake for multi-variant builds:
```bash
# List all targets
docker buildx bake --list targets
# Build the default runtime image
docker buildx bake runtime-default
# Build a specific variant with custom registry
docker buildx bake runtime-code-slim-nonroot \
--set '*.tags=my-registry/headroom:code-slim-nonroot'
```
## Environment variables
These variables configure Headroom at runtime. Set them in your shell, `.env` file, or container environment.
### LLM provider keys
| Variable | Description |
|---|---|
| `OPENAI_API_KEY` | OpenAI API key (used when proxying to OpenAI) |
| `ANTHROPIC_API_KEY` | Anthropic API key (used when proxying to Anthropic) |
| `AWS_ACCESS_KEY_ID` / `AWS_SECRET_ACCESS_KEY` | AWS credentials for Bedrock backend |
| `GOOGLE_APPLICATION_CREDENTIALS` | Google Cloud credentials for Vertex AI backend |
### Proxy configuration
| Variable | Default | Description |
|---|---|---|
| `HEADROOM_PORT` | `8787` | Port the proxy listens on |
| `HEADROOM_HOST` | `127.0.0.1` | Host the proxy binds to |
| `HEADROOM_MODE` | `token` | Default optimization mode: `token` or `cache` |
| `HEADROOM_TELEMETRY` | `off` (opt-in) | Set to `on` to opt in to anonymous telemetry |
| `HEADROOM_REQUEST_TIMEOUT` | `300` | Request timeout in seconds |
### TypeScript SDK
| Variable | Default | Description |
|---|---|---|
| `HEADROOM_BASE_URL` | `http://localhost:8787` | Proxy URL for the TypeScript SDK |
| `HEADROOM_API_KEY` | _(none)_ | API key if the proxy requires auth |
## Troubleshooting
These are common issues faced during initial setup and how to resolve them.
### Python version error
This project requires **Python 3.10+**.
Check your version:
```bash
python3 --version
```
If needed (Mac with Homebrew):
```bash
brew install python@3.11
```
### Editable install fails (`pip install -e`)
Upgrade pip to the latest version:
```bash
python3 -m pip install --upgrade pip
```
### Missing `cargo` (Rust error)
Some tests require Rust tooling.
The recommended way to install rust is using `rustup`. You can find the official installation instructions [here](https://rust-lang.org/tools/install/).
## Dashboard
Headroom serves a live savings dashboard while the proxy is running. Open it with:
```bash
headroom dashboard # opens http://localhost:8787/dashboard in your browser
headroom dashboard --no-open # just print the URL
```
Or browse to `http://localhost:8787/dashboard` directly (use `--port` / `HEADROOM_PORT` if you
run the proxy on a different port).
## Next steps
<Cards>
<Card title="Quickstart" href="/docs/quickstart" />
<Card title="Proxy Server" href="/docs/proxy" />
<Card title="Configuration" href="/docs/configuration" />
<Card title="Vercel AI SDK" href="/docs/vercel-ai-sdk" />
</Cards>
+188
View File
@@ -0,0 +1,188 @@
---
title: LangChain
description: Automatic context compression for LangChain chat models, memory, retrievers, and agents.
---
Headroom integrates with LangChain to compress context across all LangChain patterns: chat models, memory, retrievers, agents, and streaming.
## Installation
```bash
pip install "headroom-ai[langchain]"
```
## Quick start
Wrap any chat model in one line:
```python
from langchain_openai import ChatOpenAI
from headroom.integrations import HeadroomChatModel
llm = HeadroomChatModel(ChatOpenAI(model="gpt-4o"))
# Use exactly like before
response = llm.invoke("Hello!")
# Check savings
print(llm.get_metrics())
# {'tokens_saved': 12500, 'savings_percent': 45.2, 'requests': 50}
```
Works with any provider:
```python
from langchain_anthropic import ChatAnthropic
llm = HeadroomChatModel(ChatAnthropic(model="claude-sonnet-4-20250514"))
```
## Memory integration
`HeadroomChatMessageHistory` wraps any chat history with automatic compression. Long conversations stay under your token budget:
```python
from langchain.memory import ConversationBufferMemory
from langchain_community.chat_message_histories import ChatMessageHistory
from headroom.integrations import HeadroomChatMessageHistory
base_history = ChatMessageHistory()
compressed_history = HeadroomChatMessageHistory(
base_history,
compress_threshold_tokens=4000, # Compress when over 4K tokens
keep_recent_turns=5, # Always keep last 5 turns
)
memory = ConversationBufferMemory(chat_memory=compressed_history)
```
After usage:
```python
print(compressed_history.get_compression_stats())
# {'compression_count': 12, 'total_tokens_saved': 28000}
```
## Retriever integration
`HeadroomDocumentCompressor` filters retrieved documents by relevance. Retrieve many for recall, keep the best for precision:
```python
from langchain.retrievers import ContextualCompressionRetriever
from langchain_community.vectorstores import FAISS
from headroom.integrations import HeadroomDocumentCompressor
base_retriever = vectorstore.as_retriever(search_kwargs={"k": 50})
compressor = HeadroomDocumentCompressor(
max_documents=10,
min_relevance=0.3,
prefer_diverse=True, # MMR-style diversity
)
retriever = ContextualCompressionRetriever(
base_compressor=compressor,
base_retriever=base_retriever,
)
# Retrieves 50 docs, returns best 10
docs = retriever.invoke("What is Python?")
```
## Agent tool wrapping
`wrap_tools_with_headroom` compresses tool outputs before they re-enter the agent's context:
```python
from langchain_core.tools import tool
from headroom.integrations import wrap_tools_with_headroom
@tool
def search_database(query: str) -> str:
"""Search the database."""
return json.dumps({"results": [...], "total": 1000})
wrapped_tools = wrap_tools_with_headroom(
[search_database],
min_chars_to_compress=1000,
)
agent = create_openai_tools_agent(llm, wrapped_tools, prompt)
executor = AgentExecutor(agent=agent, tools=wrapped_tools)
```
Per-tool metrics:
```python
from headroom.integrations import get_tool_metrics
metrics = get_tool_metrics()
print(metrics.get_summary())
# {'total_invocations': 25, 'total_compressions': 18, 'total_chars_saved': 450000}
```
## LangGraph ReAct agent
```python
from langchain_openai import ChatOpenAI
from langgraph.prebuilt import create_react_agent
from headroom.integrations import HeadroomChatModel, wrap_tools_with_headroom
llm = HeadroomChatModel(ChatOpenAI(model="gpt-4o"))
tools = wrap_tools_with_headroom([search_web, query_database])
agent = create_react_agent(llm, tools)
result = agent.invoke({
"messages": [("user", "Find users who signed up last week")]
})
```
## LangGraph custom graph
Insert a compression node between tools and the agent in a custom `StateGraph`:
```python
from langgraph.graph import StateGraph, MessagesState, START, END
from headroom.integrations.langchain import create_compress_tool_messages_node
graph = StateGraph(MessagesState)
graph.add_node("agent", agent_node)
graph.add_node("tools", tools_node)
graph.add_node("compress", create_compress_tool_messages_node(
min_tokens_to_compress=100,
))
# Wire: tools -> compress -> agent
graph.add_edge(START, "agent")
graph.add_edge("tools", "compress")
graph.add_edge("compress", "agent")
```
## Streaming
Full async support:
```python
# Async invoke
response = await llm.ainvoke("Hello!")
# Async streaming
async for chunk in llm.astream("Tell me a story"):
print(chunk.content, end="", flush=True)
```
## Custom configuration
```python
from headroom import HeadroomConfig, HeadroomMode
config = HeadroomConfig(
default_mode=HeadroomMode.OPTIMIZE,
smart_crusher_target_ratio=0.3,
)
llm = HeadroomChatModel(
ChatOpenAI(model="gpt-4o"),
headroom_config=config,
)
```
+143
View File
@@ -0,0 +1,143 @@
---
title: Limitations
description: When Headroom helps, when it does not, and what to watch out for. Honest documentation of compression constraints and safety gates.
---
Headroom is designed to compress LLM context without losing accuracy. This page documents when it helps, when it does not, and the safety gates that prevent harmful compression.
## When Headroom Helps vs. Does Not
| Content Type | Compression | Latency Impact | Best For |
|---|---|---|---|
| **JSON: Arrays of dicts** (search results, API responses, DB rows) | 86--100% | Net latency win on Sonnet/Opus | Primary use case |
| **JSON: Arrays of strings** (file paths, log lines, tags) | 60--90% | Net latency win | String dedup + sampling |
| **JSON: Arrays of numbers** (metrics, time series) | 70--85% | Net latency win | Statistical summary |
| **JSON: Mixed-type arrays** | 50--70% | Net latency win | Group-by-type compression |
| **Structured logs** (as JSON) | 82--95% | Net latency win | Log entries in tool outputs |
| **Agentic conversations** (25--50 turns) | 56--81% | Break-even to net win | Multi-tool agent sessions |
| **Plain text** (documentation, articles) | 43--46% | Adds latency (cost savings only) | Cost optimization |
| **Code** | Passthrough | Minimal overhead | See below |
| **RAG document contexts** | Passthrough | Minimal overhead | Not compressed |
### Where Headroom Adds the Most Value
- Long agent sessions with accumulated tool outputs (40--80% compression)
- JSON-heavy workflows -- API responses, database queries (83--94% compression)
- Build and test output (85--94% compression)
- Multi-tool agents (60--76% compression across tool results)
### Where Headroom Adds Little Value
- Short conversational exchanges (median 4.8% compression)
- Code-only sessions (reading/writing files) -- code passes through
- Single-turn requests with no accumulated context
## What Headroom Does NOT Compress
- **Short messages** (< 300 tokens) -- overhead exceeds savings
- **Source code** -- passes through unchanged to preserve correctness
- **grep/search results** -- compact structured format, already minimal
- **Images** -- counted at fixed token cost (~1,600 tokens), not compressed
- **System prompts** -- preserved for prefix cache compatibility
## Code Compression
Headroom includes an AST-aware CodeCompressor (tree-sitter, 8 languages) but it is gated behind safety protections that prevent it from firing in most real-world scenarios. This is intentional.
**Why code mostly passes through:**
1. **Word count gate**: Content under 50 words is silently skipped
2. **Recent code protection** (`protect_recent_code=4`): Code in the last 4 messages is never compressed
3. **Analysis intent protection** (`protect_analysis_context=True`): If the most recent user message contains keywords like "analyze", "review", "explain", "fix", "debug" -- ALL code in the conversation is protected
**Why this is the right default**: Code is almost always fetched because the user wants to work with it. Compressing function bodies would remove exactly what they need.
**Where code savings come from**: Headroom compresses code in the live zone — the newest tool outputs and content blocks — with the AST-aware CodeCompressor, while keeping recent and analysis-context code fully intact. It never drops messages from the conversation history or strips function bodies.
**Override**: Set `protect_analysis_context=False` in `ContentRouterConfig` for aggressive code compression. Requires `headroom-ai[code]` for tree-sitter.
## JSON Compression Constraints
### What Gets Compressed
- Arrays of **dicts**: Full statistical analysis with adaptive K (Kneedle algorithm)
- Arrays of **strings**: Dedup + adaptive sampling + error preservation
- Arrays of **numbers**: Statistical summary + outlier/change-point preservation
- **Mixed-type** arrays: Grouped by type, each group compressed independently
- **Nested** objects: Recursed into, arrays within are compressed (up to depth 5)
### What Passes Through
- Arrays below 5 items (`min_items_to_analyze`)
- Content below 200 tokens (`min_tokens_to_crush`)
- Bool-only arrays
- JSON objects without array values
- Malformed JSON (silently passes through, no error)
### Edge Cases
- **NaN/Infinity** in numeric fields: Filtered out before statistics are computed
- **Nesting depth > 5**: Inner arrays not examined for compression
- **Mixed-type arrays with small groups**: Groups below `min_items_to_analyze` are kept as-is
## Safety Gates
All compressors follow the same principle: **fail gracefully, return original content unchanged**.
- Invalid JSON passes through (no error raised)
- AST parse failure falls back to original or LLMLingua
- Compression that makes output larger returns the original
- Missing optional dependencies (tree-sitter, LLMLingua) cause a passthrough with warning log
- Errors are logged at WARNING level and never propagated to callers
<Callout type="info" title="One exception">
LLMLingua out-of-memory during model loading raises a `RuntimeError`. All other failures are silently handled.
</Callout>
## Adaptive K: How Item Retention Works
SmartCrusher does not use fixed K values. It uses information-theoretic sizing:
1. **Kneedle algorithm** on bigram coverage curves finds the point where adding more items stops providing new information
2. **SimHash** fingerprinting detects near-duplicate items
3. **zlib validation** ensures the subset captures the full set's diversity
The resulting K is split: 30% from array start, 15% from end, 55% for importance-scored items.
**Safety guarantees (additive, never dropped):**
- Error items (containing "error", "exception", "failed", "critical") -- across ALL array types
- Numeric anomalies (> 2 standard deviations from mean)
- String length anomalies (> 2 standard deviations from mean length)
- Change points (sudden shifts in running values)
These are kept even if they exceed the K budget.
## Configuration Tuning
| Parameter | Default | Effect |
|---|---|---|
| `min_items_to_analyze` | 5 | Arrays below this pass through |
| `min_tokens_to_crush` | 200 | Content below this passes through |
| `max_items_after_crush` | 15 | Upper bound on retained items |
| `variance_threshold` | 2.0 | Std devs for anomaly detection (lower = more preserved) |
| `protect_analysis_context` | True | Protect code when user asks about it |
| `protect_recent_code` | 4 | Messages from end to protect code in |
| `skip_user_messages` | True | Never compress user messages |
| `toin_confidence_threshold` | 0.3 | Minimum TOIN confidence to apply hints |
## Provider Interactions
- CacheAligner maximizes Anthropic/OpenAI prefix cache hit rates
- Token counting uses model-specific tokenizers (tiktoken for OpenAI, calibrated estimation for Anthropic)
- Compression works with all providers -- no provider-specific limitations
- Compressed content is valid JSON -- downstream tools and parsers work unchanged
## TOIN Cold Start
The Tool Output Intelligence Network (TOIN) learns compression patterns from usage. For new tool types:
- No learned patterns exist -- falls back to statistical heuristics
- Confidence below `toin_confidence_threshold` (default 0.3) -- TOIN hints ignored
- Patterns build up over time as tools are used repeatedly
- Cross-session learning requires persistence (`TelemetryConfig.storage_path`)
+89
View File
@@ -0,0 +1,89 @@
---
title: LiteLLM
description: Add Headroom compression to LiteLLM with a single callback. Works with all 100+ supported providers.
---
Headroom integrates with [LiteLLM](https://github.com/BerriAI/litellm) as a callback that compresses messages before they reach any provider. One line to enable, works with all 100+ LiteLLM-supported providers.
## Installation
```bash
pip install headroom-ai litellm
```
## Quick start
```python
import litellm
from headroom.integrations.litellm_callback import HeadroomCallback
litellm.callbacks = [HeadroomCallback()]
# All calls now compressed automatically
response = litellm.completion(model="gpt-4o", messages=[...])
response = litellm.completion(model="bedrock/claude-sonnet", messages=[...])
response = litellm.completion(model="azure/gpt-4o", messages=[...])
```
The callback compresses messages in LiteLLM's `pre_call_hook` before they reach the provider.
## How it works
1. You call `litellm.completion()` with your messages
2. `HeadroomCallback.pre_call_hook` compresses the messages
3. LiteLLM sends the compressed messages to the provider
4. The response comes back unchanged
This works with every provider LiteLLM supports: OpenAI, Anthropic, Bedrock, Azure, Vertex AI, Cohere, Groq, Mistral, Together, Ollama, and more.
## With LiteLLM Proxy
If you run LiteLLM as a proxy server, use the ASGI middleware:
```python
from litellm.proxy.proxy_server import app
from headroom.integrations.asgi import CompressionMiddleware
app.add_middleware(CompressionMiddleware)
```
Or configure via YAML:
```yaml
# litellm_config.yaml
litellm_settings:
callbacks: ["headroom.integrations.litellm_callback.HeadroomCallback"]
```
## Direct compress() with LiteLLM
You can also use `compress()` directly instead of the callback:
```python
import litellm
from headroom import compress
messages = [{"role": "user", "content": large_content}]
compressed = compress(messages, model="bedrock/claude-sonnet")
response = litellm.completion(
model="bedrock/claude-sonnet",
messages=compressed.messages,
)
print(f"Saved {compressed.tokens_saved} tokens")
```
## ASGI middleware
Drop-in middleware for any ASGI application. Intercepts `/v1/messages`, `/v1/chat/completions`, `/v1/responses`, and `/chat/completions`:
```python
from fastapi import FastAPI
from headroom.integrations.asgi import CompressionMiddleware
app = FastAPI()
app.add_middleware(CompressionMiddleware)
```
Response headers include `x-headroom-compressed: true` and `x-headroom-tokens-saved: 1234`.
+119
View File
@@ -0,0 +1,119 @@
---
title: Local LLM Prefill Benchmark
description: Measure local LLM prompt-processing savings by running Headroom in passthrough and optimized proxy modes against an OpenAI-compatible local server.
---
Local models do not charge per token, but they still pay for every prompt token during prefill. On Apple Silicon and other local inference setups, long coding-agent sessions often bottleneck on prompt processing rather than generation speed. Headroom can help by sending fewer prompt tokens to the local server.
This workflow measures that effect with the proxy dashboard: run the same task once with optimization disabled, reset the agent state, run it again with optimization enabled, and compare token counts.
<Callout type="info" title="Community demo">
Joe Maddalone demonstrated this workflow in [Cut Local LLM Prompt Processing 30% on a Mac with Headroom](https://www.youtube.com/watch?v=j6U_kKiMXgo). His June 2026 demo used an OpenAI-compatible local server, a coding-agent refactor task, `--no-optimize` for the baseline, and the dashboard to compare sessions.
</Callout>
## Setup
Start your local OpenAI-compatible model server first. Examples include MLX/OMLX, vLLM, LM Studio, Ollama's OpenAI-compatible endpoint, or another server that accepts `/v1/chat/completions` or `/v1/responses`.
For this guide, assume the local server is listening on `http://127.0.0.1:8000`.
```bash
pip install "headroom-ai[proxy]"
```
## 1. Baseline passthrough run
Start Headroom as a transparent proxy with optimization disabled:
```bash
headroom proxy \
--port 8787 \
--openai-api-url http://127.0.0.1:8000 \
--no-optimize
```
Point your agent or app at Headroom, not directly at the local server:
```bash
export OPENAI_BASE_URL=http://127.0.0.1:8787/v1
export OPENAI_API_KEY=local
```
Run a realistic task. Coding-agent refactors are good benchmark candidates because they produce repeated file reads, tool results, lint/test output, and a growing conversation context.
Open the dashboard while the run is active:
```bash
headroom dashboard --port 8787 --no-open
# or open http://127.0.0.1:8787/dashboard
```
Record the baseline session totals. With `--no-optimize`, before and after token counts should match because Headroom is only forwarding traffic.
## 2. Reset the task
Before the optimized run, reset the benchmark state so the second run is comparable:
- revert the code or data changes made by the first run
- start a fresh agent session
- use the same model and local server
- use the same prompt
- avoid changing unrelated flags or server settings
For coding-agent tests, a clean git worktree is the simplest reset point.
## 3. Optimized run
Restart Headroom without `--no-optimize`:
```bash
headroom proxy \
--port 8787 \
--openai-api-url http://127.0.0.1:8000
```
Run the same task again with the same `OPENAI_BASE_URL` and prompt. Watch the dashboard's before/after token counts for the session.
The savings percentage is the prompt-token reduction sent upstream to the local model. That does not make the model's prefill kernel faster; it reduces how much prompt the kernel has to process.
## Optional: traffic learning
After you have a baseline, you can test learning-enabled runs:
```bash
headroom proxy \
--port 8787 \
--openai-api-url http://127.0.0.1:8000 \
--learn
```
`--learn` implies memory and lets Headroom learn recurring traffic patterns from proxy sessions. Treat this as a separate benchmark condition: compare passthrough, optimized, and optimized-with-learning runs independently.
## What to report
For a useful local prefill benchmark, include:
| Field | Example |
|---|---|
| Local server | MLX, vLLM, LM Studio, Ollama-compatible endpoint |
| Model | local model name and quantization, if relevant |
| Hardware | Mac model, RAM, or GPU/CPU target |
| Agent/client | coding agent or app name |
| Task | short description of the repeated task |
| Baseline tokens | dashboard before/after total with `--no-optimize` |
| Optimized tokens | dashboard before/after total without `--no-optimize` |
| Savings | dashboard percentage |
| Notes | whether `--learn`, `--memory`, or other flags were enabled |
## Interpreting results
Local inference changes the value proposition:
- Hosted APIs: fewer input tokens usually means lower cost and lower latency.
- Local models: fewer input tokens primarily means less prefill work and lower memory pressure.
Long-running agent sessions tend to show larger gains than short chat turns because repeated file reads, tool outputs, and logs create more compressible context. If a task is mostly short natural-language turns, expect smaller savings.
<Callout type="warning" title="Keep the comparison honest">
Do not compare a cold first run against a warmed second run and attribute all improvement to compression. Keep the server, model, prompt, and agent task stable, and use the dashboard token counts as the primary measurement.
</Callout>
+228
View File
@@ -0,0 +1,228 @@
---
title: MCP Tools
description: Compression, retrieval, and stats as MCP tools for Claude Code, Cursor, and any MCP-compatible host.
---
Headroom's MCP server exposes compression, retrieval, and observability as tools that any MCP-compatible AI coding tool can call -- Claude Code, Cursor, Codex, and more. No proxy required.
## Installation
```bash
# MCP tools only (lightweight)
pip install "headroom-ai[mcp]"
# Or with the proxy
pip install "headroom-ai[proxy]"
```
## Setup for Claude Code
```bash
# Register with Claude Code (one-time)
headroom mcp install
# Start Claude Code — it now has headroom tools
claude
```
Claude Code can now compress content on demand, retrieve originals, and check session stats.
For automatic compression of **all** traffic, also run the proxy:
```bash
# Terminal 1
headroom proxy
# Terminal 2
ANTHROPIC_BASE_URL=http://127.0.0.1:8787 claude
```
## Tools
### headroom_compress
Compress content on demand. The LLM calls this when it wants to shrink large content before reasoning over it.
**Parameters:**
- `content` (required) -- text to compress (files, JSON, logs, search results)
**Returns:**
- `compressed` -- compressed text
- `hash` -- key for retrieving the original later
- `original_tokens` / `compressed_tokens` / `savings_percent`
- `transforms` -- which compression algorithms were applied
Example flow:
```
Claude: Let me compress this large output to save context space.
-> headroom_compress(content="[5000 lines of grep results...]")
<- {
"compressed": "[key matches with context...]",
"hash": "a1b2c3d4e5f6...",
"original_tokens": 12000,
"compressed_tokens": 3200,
"savings_percent": 73.3,
"transforms": ["router:search:0.27"]
}
```
The original is stored locally for 1 hour. If the LLM needs the full content later, it calls `headroom_retrieve`.
### headroom_retrieve
Retrieve original uncompressed content by hash.
**Parameters:**
- `hash` (required) -- hash key from a previous compression
- `query` (optional) -- search within the original to return only matching items
**Returns:**
- `original_content` (full retrieval) or `results` (filtered search)
- `source` -- `"local"` or `"proxy"`
Retrieval checks the local store first, then falls back to the proxy's store. Hashes from either source work transparently.
### headroom_stats
Session compression statistics.
**Returns:**
- `compressions`, `retrievals`, `tokens_saved`, `savings_percent`
- `estimated_cost_saved_usd`
- `recent_events` -- last 10 compression/retrieval events
- `sub_agents` -- stats from sub-agent MCP instances
- `combined` -- main + sub-agent totals
- `proxy` -- request count, cache hits, cost saved (if proxy is running)
Sub-agent stats are aggregated via a shared stats file at `~/.headroom/session_stats.jsonl`.
## CLI commands
```bash
# Install (registers with Claude Code)
headroom mcp install
headroom mcp install --proxy-url http://host:9000 # Custom proxy URL
headroom mcp install --force # Overwrite existing
# Check status
headroom mcp status
# Uninstall
headroom mcp uninstall
# Debug mode
headroom mcp serve --debug
```
## MCP host configuration
For MCP hosts that let you configure a local stdio server, point them at `headroom mcp serve`. If you also run the proxy, pass the proxy URL explicitly so retrieval and stats come from the intended proxy instance.
```json
{
"mcpServers": {
"headroom": {
"type": "stdio",
"command": "headroom",
"args": ["mcp", "serve", "--proxy-url", "http://127.0.0.1:8787"]
}
}
}
```
For multiple proxy instances, register one stdio MCP server per proxy URL:
```json
{
"mcpServers": {
"headroom": {
"type": "stdio",
"command": "headroom",
"args": ["mcp", "serve", "--proxy-url", "http://127.0.0.1:8787"]
},
"headroom-azure": {
"type": "stdio",
"command": "headroom",
"args": ["mcp", "serve", "--proxy-url", "http://127.0.0.1:8788"]
}
}
}
```
Do not assume that a running proxy exposes an HTTP MCP endpoint at `/mcp`. If `http://127.0.0.1:<port>/mcp` returns `404`, use the stdio configuration above.
### `command: "headroom"` fails to start
The configurations above use `"command": "headroom"`, which only works if the `headroom`
executable is on the PATH your MCP host (Codex, etc.) sees at startup. If you installed Headroom
into a project virtualenv — for example with `uv add headroom-ai` — the CLI lives only inside
that venv, and the host fails at launch with:
```text
MCP client for `headroom` failed to start: MCP startup failed: No such file or directory (os error 2)
```
Install Headroom so it's globally on PATH — `uv tool install "headroom-ai[mcp]"` (or
`pipx install "headroom-ai[mcp]"`) — or replace `"headroom"` with the absolute path to the binary
(`command -v headroom`, or `where headroom` on Windows).
## Cross-tool compatibility
| Tool | MCP Support | Setup |
|------|-------------|-------|
| Claude Code | Native | `headroom mcp install` |
| Cursor | Supported | Add to Cursor MCP settings |
| Codex | If supported | Configure MCP server |
| Any MCP host | Yes | Point to `headroom mcp serve` |
## Architecture
### MCP only (no proxy)
The LLM calls `headroom_compress` on demand. Compression happens locally in the MCP process. Originals are stored in a local `CompressionStore` with 1-hour TTL.
### MCP + Proxy (full setup)
The proxy compresses all traffic at the HTTP level (before the LLM sees content). MCP tools operate after the LLM receives content. They handle different data and do not double-compress.
`headroom_retrieve` checks the local store first, then falls back to the proxy's store.
## Troubleshooting
**"MCP SDK not installed"** -- Run `pip install "headroom-ai[mcp]"`.
**"Proxy not running"** -- Start the proxy with `headroom proxy` in another terminal. Only needed for proxy-backed retrieval.
**"Entry not found or expired"** -- Local content expires after 1 hour, proxy content after 5 minutes.
**Claude doesn't see headroom tools** -- Run `headroom mcp status`, restart Claude Code, and verify with `/mcp` inside Claude Code.
### Claude Code `/usage` attributes a large share to `headroom` MCP
Claude Code counts MCP tool calls and MCP tool results as session context. If a
long-running workflow or a subagent-heavy command calls `headroom_compress`,
`headroom_retrieve`, or `headroom_stats` many times, `/usage` can show a visible
share under the `headroom` MCP server even when Headroom is saving tokens inside
individual tool results.
That number is not a direct "Headroom overhead" bill. It means Claude Code kept
Headroom MCP interactions in the conversation context. Deep research workflows
can amplify this because each subagent has its own requests and may keep its own
MCP results in context.
Use these checks when the MCP share looks high:
- Run `headroom_stats` and compare `tokens_saved` with the number of MCP calls.
- Use `/compact` after large MCP-backed investigation steps so old MCP tool
results stop occupying the active context window.
- Prefer the proxy path for automatic compression of normal Claude Code traffic:
`headroom proxy` plus `ANTHROPIC_BASE_URL=http://127.0.0.1:8787 claude`.
- Disable the MCP server for sessions where you only want proxy-level
compression and do not need on-demand `headroom_compress` or
`headroom_retrieve`.
- For deep research or custom subagent workflows, reduce unnecessary subagent
fan-out first; subagent traffic usually dominates the usage picture before
MCP overhead does.
+273
View File
@@ -0,0 +1,273 @@
---
title: Persistent Memory
description: Hierarchical, temporal memory for LLM applications. Enable your AI to remember across conversations with intelligent scoping and versioning.
---
LLMs have two fundamental limitations: context windows overflow with too much history, and every conversation starts from zero. Persistent Memory solves both by extracting key facts, persisting them, and injecting them when relevant.
This is **temporal compression** -- instead of carrying 10,000 tokens of conversation history, carry 100 tokens of extracted memories.
## Quick Start
```python
from openai import OpenAI
from headroom import with_memory
# One line -- that's it
client = with_memory(OpenAI(), user_id="alice")
# Use exactly like normal
response = client.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": "I prefer Python for backend work"}]
)
# Memory extracted INLINE -- zero extra latency
# Later, in a new conversation...
response = client.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": "What language should I use?"}]
)
# Response uses the Python preference from memory
```
## How It Works
The `with_memory()` wrapper intercepts every chat completion call:
1. **Inject** -- Semantic search finds relevant memories and prepends them to the user message
2. **Instruct** -- Adds a memory extraction instruction to the system prompt
3. **Call** -- Forwards the request to the LLM
4. **Parse** -- Extracts the `<memory>` block from the response
5. **Store** -- Saves with embeddings, vector index, and full-text search index
6. **Return** -- Cleans the response (strips the memory block before returning)
Memory extraction happens **inline** as part of the LLM response. No extra API calls, no extra latency.
## Hierarchical Scoping
Memories exist at four scope levels, from broadest to narrowest:
| Scope | Persists Across | Use Case |
|-------|-----------------|----------|
| **User** | All sessions, all time | Long-term preferences, identity |
| **Session** | Current session only | Current task context |
| **Agent** | Current agent in session | Agent-specific context |
| **Turn** | Single turn only | Ephemeral working memory |
```python
from openai import OpenAI
from headroom import with_memory
# Session 1: Morning
client1 = with_memory(
OpenAI(),
user_id="bob",
session_id="morning-session",
)
response = client1.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": "I prefer Go for performance-critical code"}]
)
# Memory stored at USER level (persists across sessions)
# Session 2: Afternoon (different session, same user)
client2 = with_memory(
OpenAI(),
user_id="bob",
session_id="afternoon-session",
)
response = client2.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": "What language for my new microservice?"}]
)
# Recalls Go preference from morning session
```
## Memory Categories
Memories are categorized for better organization and retrieval:
| Category | Description | Examples |
|----------|-------------|----------|
| `PREFERENCE` | Likes, dislikes, preferred approaches | "Prefers Python", "Likes dark mode" |
| `FACT` | Identity, role, constraints | "Works at fintech startup", "Senior engineer" |
| `CONTEXT` | Current goals, ongoing tasks | "Migrating to microservices", "Working on auth" |
| `ENTITY` | Information about entities | "Project Apollo uses React", "Team lead is Sarah" |
| `DECISION` | Decisions made | "Chose PostgreSQL over MySQL" |
| `INSIGHT` | Derived insights | "User tends to prefer typed languages" |
## Memory API
The `with_memory()` wrapper exposes a `.memory` attribute for direct access:
```python
client = with_memory(OpenAI(), user_id="alice")
# Search memories (semantic)
results = client.memory.search("python preferences", top_k=5)
for memory in results:
print(f"{memory.content}")
# Add a memory manually
client.memory.add(
"User is a senior engineer",
importance=0.9,
)
# Get all memories for this user
all_memories = client.memory.get_all()
# Clear all memories
client.memory.clear()
# Get stats
stats = client.memory.stats()
print(f"Total memories: {stats['total']}")
print(f"By category: {stats['categories']}")
```
## Temporal Versioning
When facts change, Headroom creates a **supersession chain** that preserves history:
```python
from headroom.memory import HierarchicalMemory, MemoryCategory
memory = await HierarchicalMemory.create()
# Original fact
orig = await memory.add(
content="User works at Google",
user_id="alice",
category=MemoryCategory.FACT,
)
# User changes jobs -- supersede the old memory
new = await memory.supersede(
old_memory_id=orig.id,
new_content="User now works at Anthropic",
)
# Query current state (excludes superseded by default)
current = await memory.query(MemoryFilter(
user_id="alice",
include_superseded=False,
))
# Returns only "User now works at Anthropic"
# Get the full chain
chain = await memory.get_history(new.id)
# [
# Memory(content="User works at Google", is_current=False),
# Memory(content="User now works at Anthropic", is_current=True),
# ]
```
This gives you an audit trail, the ability to debug why the LLM made certain decisions, and rollback if needed.
## Backends
### Embedder Backends
```python
from headroom.memory import MemoryConfig, EmbedderBackend
# ONNX embeddings (recommended -- fast, free, private; ~30 MB int8-quantized)
# Note: EmbedderBackend.LOCAL requires PyTorch (~2 GB); use ONNX instead.
config = MemoryConfig(
embedder_backend=EmbedderBackend.ONNX,
embedder_model="BAAI/bge-small-en-v1.5",
)
# OpenAI embeddings (higher quality, costs money)
config = MemoryConfig(
embedder_backend=EmbedderBackend.OPENAI,
openai_api_key="sk-...",
embedder_model="text-embedding-3-small",
)
# Ollama embeddings (local server, many models)
config = MemoryConfig(
embedder_backend=EmbedderBackend.OLLAMA,
ollama_base_url="http://localhost:11434",
embedder_model="nomic-embed-text",
)
```
### Embedding Runtime / GPU Offload (Apple Silicon)
By default the proxy's memory embedder runs on the **ONNX CPU** backend -- fast
and dependency-light, but CPU-only. Under sustained load the embedding step can
saturate the CPU and make the proxy less responsive.
On Apple Silicon you can opt in to running the embedder on the **Apple GPU
(MPS)** instead, which offloads that work off the CPU and keeps the proxy
responsive. Install the extra and set the env var:
```bash
pip install "headroom-ai[pytorch-mps]" # also works as [pytorch_mps]
export HEADROOM_EMBEDDER_RUNTIME=pytorch_mps
```
When set, the embedder runs via the torch sentence-transformers backend on the
Apple GPU instead of the default ONNX CPU embedder. Notes:
- **Strictly opt-in.** `pytorch_mps` is the only accepted value; anything else
(or unset) keeps the default ONNX CPU embedder. Default behavior is unchanged,
and there is no CLI flag -- it is env-var only.
- **Auto-fallback.** It only activates when Apple MPS is actually available
(Apple Silicon + torch). If MPS is unavailable or torch/sentence-transformers
is not installed, it logs a warning and uses the existing default embedder
selection path: ONNX when available, then the pre-existing local
sentence-transformers fallback.
- **MPS serialization.** torch-MPS is not thread-safe, so the embedder
serializes MPS encode calls internally via a single-worker executor. This is
automatic -- there is nothing to configure.
### Storage
Storage uses **SQLite** for CRUD and filtering, **HNSW** for vector similarity search, and **FTS5** for full-text keyword search. All embedded -- no external services required.
```python
config = MemoryConfig(
db_path="memory.db",
vector_dimension=384,
hnsw_ef_construction=200,
hnsw_m=16,
hnsw_ef_search=50,
cache_enabled=True,
cache_max_size=1000,
)
```
## Provider Compatibility
Memory works with any OpenAI-compatible client:
```python
from openai import OpenAI
from headroom import with_memory
# OpenAI
client = with_memory(OpenAI(), user_id="alice")
# Azure OpenAI
client = with_memory(
OpenAI(base_url="https://your-resource.openai.azure.com/..."),
user_id="alice",
)
# Groq
from groq import Groq
client = with_memory(Groq(), user_id="alice")
```
## Performance
| Operation | Latency | Notes |
|-----------|---------|-------|
| Memory injection | &lt;50ms | Local embeddings + HNSW search |
| Memory extraction | +50-100 tokens | Part of LLM response (inline) |
| Memory storage | &lt;10ms | SQLite + HNSW + FTS5 indexing |
| Cache hit | &lt;1ms | LRU cache lookup |
+61
View File
@@ -0,0 +1,61 @@
{
"pages": [
"---Getting Started---",
"index",
"quickstart",
"installation",
"docker-install",
"persistent-installs",
"community-savings",
"---Compression---",
"how-compression-works",
"smart-crusher",
"code-compression",
"image-compression",
"text-and-logs",
"---Reversible Compression---",
"ccr",
"---Cache & Context---",
"cache-optimization",
"agent-orchestration",
"context-management",
"---Memory---",
"memory",
"shared-context",
"failure-learning",
"---Proxy Server---",
"proxy",
"local-llm-prefill",
"---Integrations---",
"vercel-ai-sdk",
"openai-sdk",
"anthropic-sdk",
"langchain",
"agno",
"strands",
"litellm",
"claude-code-vertex",
"claude-code-azure-foundry",
"opencode",
"mcp",
"---Configuration---",
"configuration",
"pipeline-extensions",
"filesystem-contract",
"---Observability---",
"savings",
"metrics",
"simulation",
"---API Reference---",
"api-reference",
"---Architecture---",
"architecture",
"ci-cd-flows",
"releases",
"benchmarks",
"limitations",
"---Help---",
"errors",
"troubleshooting"
]
}
+276
View File
@@ -0,0 +1,276 @@
---
title: Metrics & Monitoring
description: Monitor compression performance, cost savings, and system health with Headroom's built-in metrics, Prometheus endpoint, and SDK APIs.
---
Headroom provides comprehensive metrics for monitoring compression performance, cost savings, and system health through both the proxy server and the SDK.
## Proxy Endpoints
### Stats Endpoint
```bash
curl http://localhost:8787/stats
```
```json
{
"persistent_savings": {
"lifetime": {
"tokens_saved": 12500,
"compression_savings_usd": 0.04
}
},
"requests": {
"total": 42,
"cached": 5,
"rate_limited": 0,
"failed": 0
},
"tokens": {
"input": 50000,
"output": 8000,
"saved": 12500,
"savings_percent": 25.0
},
"cost": {
"total_cost_usd": 0.15,
"total_savings_usd": 0.04
},
"cache": {
"entries": 10,
"total_hits": 5
}
}
```
Persistent savings are stored at `~/.headroom/proxy_savings.json` and survive proxy restarts. Override the path with `HEADROOM_SAVINGS_PATH`.
### Historical Savings
```bash
curl http://localhost:8787/stats-history
```
Returns durable compression history with hourly, daily, weekly, and monthly rollups. Supports CSV export:
```bash
curl "http://localhost:8787/stats-history?format=csv&series=daily"
curl "http://localhost:8787/stats-history?format=csv&series=monthly"
```
### Prometheus Metrics
```bash
curl http://localhost:8787/metrics
```
```
# HELP headroom_requests_total Total requests processed
headroom_requests_total{mode="optimize"} 1234
# HELP headroom_tokens_saved_total Total tokens saved
headroom_tokens_saved_total 5678900
# HELP headroom_persistent_savings_tokens_saved_total Durable lifetime input tokens saved by proxy compression
headroom_persistent_savings_tokens_saved_total 5678900
# HELP headroom_compression_ratio Compression ratio histogram
headroom_compression_ratio_bucket{le="0.5"} 890
headroom_compression_ratio_bucket{le="0.7"} 1100
headroom_compression_ratio_bucket{le="0.9"} 1200
# HELP headroom_latency_seconds Request latency histogram
headroom_latency_seconds_bucket{le="0.01"} 800
headroom_latency_seconds_bucket{le="0.1"} 1150
# HELP headroom_cache_hits_total Cache hit counter
headroom_cache_hits_total 456
```
### Health Check
```bash
curl http://localhost:8787/health
```
```json
{
"status": "healthy",
"version": "0.1.0",
"uptime_seconds": 3600
}
```
## SDK Metrics
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
### Proxy Stats
The TypeScript SDK queries the proxy for stats:
```ts twoslash
import { HeadroomClient } from 'headroom-ai';
const client = new HeadroomClient();
// Get proxy stats
const stats = await client.proxyStats();
console.log(`Tokens saved: ${stats.tokens.saved}`);
console.log(`Savings: ${stats.tokens.savings_percent}%`);
```
### Compression Result Metrics
Every `compress()` call returns metrics:
```ts twoslash
import { compress } from 'headroom-ai';
const result = await compress(messages, { model: 'gpt-4o' });
console.log(`Tokens: ${result.tokensBefore} -> ${result.tokensAfter}`);
console.log(`Saved: ${result.tokensSaved} (${(result.compressionRatio * 100).toFixed(1)}%)`);
console.log(`Transforms: ${result.transformsApplied.join(', ')}`);
```
</Tab>
<Tab value="Python">
### Session Stats
Quick stats for the current session (no database query):
```python
stats = client.get_stats()
print(f"Mode: {stats['config']['mode']}")
print(f"Tokens saved: {stats['session']['tokens_saved_total']}")
print(f"Avg compression: {stats['session']['compression_ratio_avg']:.1%}")
```
Returns:
```python
{
"session": {
"requests_total": 10,
"tokens_input_before": 50000,
"tokens_input_after": 35000,
"tokens_saved_total": 15000,
"tokens_output_total": 8000,
"cache_hits": 3,
"compression_ratio_avg": 0.70,
},
"config": {
"mode": "optimize",
"provider": "openai",
"cache_optimizer_enabled": True,
"semantic_cache_enabled": False,
},
"transforms": {
"smart_crusher_enabled": True,
"cache_aligner_enabled": True,
"rolling_window_enabled": True,
},
}
```
### Historical Metrics
Query stored metrics from the database:
```python
from datetime import datetime, timedelta
metrics = client.get_metrics(
start_time=datetime.utcnow() - timedelta(hours=1),
limit=100,
)
for m in metrics:
print(f"{m.timestamp}: {m.tokens_input_before} -> {m.tokens_input_after}")
```
### Summary Statistics
Aggregate statistics across all stored metrics:
```python
summary = client.get_summary()
print(f"Total requests: {summary['total_requests']}")
print(f"Total tokens saved: {summary['total_tokens_saved']}")
print(f"Average compression: {summary['avg_compression_ratio']:.1%}")
print(f"Total cost savings: ${summary['total_cost_saved_usd']:.2f}")
```
</Tab>
</Tabs>
## Logging
<Tabs groupId="lang" items={['Python', 'Proxy']}>
<Tab value="Python">
```python
import logging
# INFO level shows compression summaries
logging.basicConfig(level=logging.INFO)
# DEBUG level shows detailed transform decisions
logging.basicConfig(level=logging.DEBUG)
```
Example output:
```
INFO:headroom.transforms.pipeline:Pipeline complete: 45000 -> 4500 tokens (saved 40500, 90.0% reduction)
INFO:headroom.transforms.smart_crusher:SmartCrusher applied top_n strategy: kept 15 of 1000 items
DEBUG:headroom.transforms.smart_crusher:Kept items: [0,1,2,42,77,97,98,99] (errors at 42, warnings at 77)
```
</Tab>
<Tab value="Proxy">
```bash
# Log to file
headroom proxy --log-file headroom.jsonl
# Increase verbosity
headroom proxy --log-level debug
```
</Tab>
</Tabs>
## Cost Tracking
### Budget Alerts
Set a budget limit in the proxy:
```bash
headroom proxy --budget 10.00
```
When the budget is exceeded, requests return a budget exceeded error, the `/stats` endpoint shows budget status, and logs indicate the budget state.
## Key Metrics to Monitor
| Metric | What It Tells You | Target |
|--------|-------------------|--------|
| `headroom_tokens_saved_total` | Runtime tokens saved since this proxy process started | Higher is better |
| `headroom_persistent_savings_tokens_saved_total` | Durable lifetime tokens saved from `/stats.persistent_savings` | Higher is better |
| `compression_ratio_avg` | Efficiency | 0.7--0.9 typical |
| `cache_hit_rate` | Cache effectiveness | >20% is good |
| `latency_p99` | Performance impact | &lt;10ms |
| `failed_requests` | Reliability | 0 |
## Grafana Dashboard
Example Prometheus queries for a Grafana dashboard:
| Panel | PromQL |
|-------|--------|
| Runtime Tokens Saved | `headroom_tokens_saved_total` |
| Lifetime Tokens Saved | `headroom_persistent_savings_tokens_saved_total` |
| Compression Ratio (median) | `histogram_quantile(0.5, headroom_compression_ratio_bucket)` |
| Request Latency (p99) | `histogram_quantile(0.99, headroom_latency_seconds_bucket)` |
| Cache Hit Rate | `headroom_cache_hits_total / (headroom_cache_hits_total + headroom_cache_misses_total)` |
+126
View File
@@ -0,0 +1,126 @@
---
title: OpenAI SDK
description: Auto-compress messages in the OpenAI Node.js SDK with a single withHeadroom() wrapper.
---
Headroom wraps the OpenAI Node.js SDK to automatically compress messages before every `chat.completions.create()` call. All other methods (embeddings, images, audio) pass through unchanged.
## Installation
```bash
npm install headroom-ai openai
```
<Callout type="info" title="Proxy required">
The TypeScript SDK sends messages to a local Headroom proxy for compression. Start the proxy before using the SDK:
```bash
pip install "headroom-ai[proxy]"
headroom proxy
```
</Callout>
## Quick start
```ts twoslash
import { withHeadroom } from 'headroom-ai/openai';
import OpenAI from 'openai';
const client = withHeadroom(new OpenAI());
// Messages are compressed automatically before sending
const response = await client.chat.completions.create({
model: 'gpt-4o',
messages: longConversation,
});
```
That's it. Every call to `client.chat.completions.create()` compresses the messages first. The response format is identical to the unwrapped client.
## How it works
`withHeadroom()` returns a proxy around your OpenAI client that intercepts `chat.completions.create()`:
1. Extracts `messages` from the request params
2. Sends them to the Headroom proxy's `/v1/compress` endpoint
3. Replaces the original messages with the compressed result
4. Forwards the request to OpenAI as normal
All other client methods are untouched:
```ts twoslash
import { withHeadroom } from 'headroom-ai/openai';
import OpenAI from 'openai';
const client = withHeadroom(new OpenAI());
// These pass through unchanged
const embedding = await client.embeddings.create({
model: 'text-embedding-3-small',
input: 'Hello world',
});
```
## Options
Pass compression options as the second argument:
```ts twoslash
import { withHeadroom } from 'headroom-ai/openai';
import OpenAI from 'openai';
const client = withHeadroom(new OpenAI(), {
model: 'gpt-4o',
baseUrl: 'http://localhost:8787',
});
```
## Streaming
Streaming works normally. Compression happens before the request is sent:
```ts twoslash
import { withHeadroom } from 'headroom-ai/openai';
import OpenAI from 'openai';
const client = withHeadroom(new OpenAI());
const stream = await client.chat.completions.create({
model: 'gpt-4o',
messages: longConversation,
stream: true,
});
for await (const chunk of stream) {
process.stdout.write(chunk.choices[0]?.delta?.content ?? '');
}
```
## Tool calling
Tool call messages and tool results are compressed like any other message content. Large tool outputs (JSON arrays, logs) see the biggest savings:
```ts twoslash
import { withHeadroom } from 'headroom-ai/openai';
import OpenAI from 'openai';
const client = withHeadroom(new OpenAI());
const response = await client.chat.completions.create({
model: 'gpt-4o',
messages: [
{ role: 'user', content: 'Search for recent errors' },
{
role: 'assistant',
content: null,
tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'search', arguments: '{"q":"errors"}' } }],
},
{
role: 'tool',
tool_call_id: 'call_1',
content: hugeJsonResult, // Compressed automatically
},
],
tools: [{ type: 'function', function: { name: 'search', parameters: {} } }],
});
```
+161
View File
@@ -0,0 +1,161 @@
---
title: OpenCode Integration
description: Route OpenCode traffic through Headroom for token compression, MCP tools, and cached model access. One command to wrap, one to unwrap.
---
Use `headroom wrap opencode` to route OpenCode LLM traffic through the Headroom proxy with a single command. The wrapper starts or reuses the proxy, writes OpenCode config, injects Headroom MCP tools, adds RTK context filtering, and launches OpenCode with the generated config.
The `headroom-opencode` npm package also exports a native OpenCode plugin. The plugin can be used directly from OpenCode config when you want in-process transport interception plus the Headroom retrieve tool.
## Quick Start
```bash
headroom wrap opencode
```
When you are done:
```bash
headroom unwrap opencode
```
## What `wrap opencode` Does
| Step | What happens |
|---|---|
| Proxy | Starts the Headroom proxy unless `--no-proxy` is set |
| Provider injection | Writes a `headroom` provider using `@ai-sdk/openai-compatible` into `opencode.json`, pointing at `http://127.0.0.1:<port>/v1` |
| Runtime env | Sets `OPENCODE_CONFIG_CONTENT` with provider, plugin, and optional local MCP config so OpenCode picks up Headroom at launch |
| Provider compatibility | Leaves `OPENAI_BASE_URL` and `ANTHROPIC_BASE_URL` untouched so OpenCode `/connect` providers keep their own routing |
| Context tool | Injects RTK (or `lean-ctx`) instructions into `~/.config/opencode/AGENTS.md` and project `AGENTS.md` |
| MCP setup | Registers the Headroom MCP server (`headroom_compress`, `headroom_retrieve`, `headroom_stats`) |
| Serena MCP | Optionally registers Serena code graph tools (`--no-serena` to skip) |
| Backup | Snapshots `opencode.json` to `opencode.json.headroom-backup` before making any changes |
| Launch | Starts the `opencode` binary through the proxy |
## Options
```bash
headroom wrap opencode \
--port 8787 \
--no-rtk \
--no-mcp \
--no-serena \
--code-graph \
--no-proxy \
--learn \
--memory \
--backend anthropic \
--anyllm-provider ... \
--region ... \
-- <opencode args>
```
## Provider Model Mapping
The generated `headroom` provider exposes these models through the proxy:
| Provider model | Upstream model |
|---|---|
| `headroom/claude-sonnet-4-6` | Claude Sonnet 4.6, 200K context, 16K output |
| `headroom/claude-opus-4-6` | Claude Opus 4.6, 200K context, 16K output |
| `headroom/claude-haiku-4-5-20251001` | Claude Haiku 4.5, 200K context, 8K output |
| `headroom/gpt-4o` | GPT-4o, 128K context, 16K output |
| `headroom/gpt-4.1` | GPT-4.1, 1M context, 32K output |
The default model is `headroom/claude-sonnet-4-6`. Change it in `opencode.json` or in the generated `OPENCODE_CONFIG_CONTENT` payload.
## Environment Variables
| Variable | Description |
|---|---|
| `OPENCODE_CONFIG_CONTENT` | JSON payload with provider, plugin, and optional local MCP config injected by `wrap` |
| `HEADROOM_PROXY_URL` | Proxy URL passed to Headroom MCP when a non-default port is used, and to the native plugin when configured |
| `HEADROOM_CONTEXT_TOOL` | Set to `lean-ctx` to use lean-ctx instead of RTK |
## Failure Learning
`headroom learn` supports OpenCode as a scan target. It reads past sessions from `~/.local/share/opencode/opencode.db` and writes corrections to your project's `AGENTS.md`.
```bash
headroom learn --agent opencode --apply
```
See [Failure Learning](/docs/failure-learning) for details on the learn system.
## Persistent Installs
`headroom install` supports OpenCode as a target for persistent provider wiring. Use provider scope when you want Headroom to edit `opencode.json` directly:
```bash
headroom install apply --preset persistent-service --scope provider --providers manual --target opencode
```
This writes the Headroom provider into `~/.config/opencode/opencode.json` and keeps the proxy running on port 8787.
The default user scope only writes shell environment configuration. For OpenCode, direct provider config requires `--scope provider`.
## Native OpenCode Plugin
The `headroom-opencode` package exports `HeadroomPlugin` for direct OpenCode plugin registration. The plugin installs Headroom transport interception inside OpenCode, exposes the `headroom_retrieve` tool, and publishes Headroom metadata through the OpenCode plugin output env.
Example:
```ts
import { HeadroomPlugin } from "headroom-opencode";
export default async function plugin(input) {
return HeadroomPlugin(input, {
proxyUrl: process.env.HEADROOM_PROXY_URL ?? "http://127.0.0.1:8787",
});
}
```
Use this plugin when OpenCode should intercept provider traffic in-process. Use `headroom wrap opencode` when you want the CLI to manage the proxy, config injection, MCP registration, backups, and unwrap behavior.
## Programmatic Config Helpers
The package also exports helpers for custom integrations:
```ts
import {
buildOpencodeConfigContent,
createHeadroomProvider,
createHeadroomRetrieveTool,
} from "headroom-opencode";
const provider = createHeadroomProvider({ proxyPort: 8787 });
const config = buildOpencodeConfigContent({
proxyPort: 8787,
defaultModel: "claude-sonnet-4-6",
});
const retrieve = createHeadroomRetrieveTool({
proxyBaseUrl: "http://127.0.0.1:8787",
});
```
## How It Works Under The Hood
1. **Config injection**. The wrapper writes a `provider.headroom` block into `opencode.json`. The provider uses `@ai-sdk/openai-compatible`, which OpenCode supports natively. Model mappings route requests through `http://127.0.0.1:<port>/v1`.
2. **Runtime config**. `OPENCODE_CONFIG_CONTENT` is set as an env var containing provider, plugin, and optional local MCP JSON. OpenCode reads it at startup and merges it with on-disk config.
3. **MCP tools**. Headroom registers `headroom_compress`, `headroom_retrieve`, and `headroom_stats` through `headroom mcp serve` unless `--no-mcp` is set.
4. **Native plugin path**. `HeadroomPlugin` installs Headroom transport interception and uses `HEADROOM_PROXY_URL` or `http://127.0.0.1:8787` to reach the proxy.
5. **Unwrap**. `headroom unwrap opencode` restores `opencode.json` from the pre-wrap backup when present, strips Headroom marker blocks when no backup exists, and unregisters Headroom MCP servers.
## Troubleshooting
**OpenCode does not use the headroom provider.**
Check that `OPENCODE_CONFIG_CONTENT` is set and contains the `provider.headroom` block. The wrap command prints the env vars it sets.
**The native plugin cannot reach Headroom.**
Set `HEADROOM_PROXY_URL` to the running proxy URL, for example `http://127.0.0.1:8787`.
**Provider not found after unwrap.**
If unwrap left the provider configured, run `headroom unwrap opencode` again, or manually restore from `~/.config/opencode/opencode.json.headroom-backup`.
**Proxy port conflict.**
Use `--port` to select a specific port, or let the proxy auto-select an available one.
+164
View File
@@ -0,0 +1,164 @@
---
title: Persistent Installs
description: Install Headroom as a durable local runtime — background service, scheduled watchdog, or restartable Docker container — instead of starting it ad hoc.
---
Headroom can be installed as a durable local runtime instead of only being started ad hoc with `headroom proxy` or `headroom wrap ...`.
Use the Python-native `headroom install` CLI when you want supported tools to keep talking to an always-on proxy at `http://127.0.0.1:8787` and have `wrap` reuse or recover that deployment instead of starting a second ephemeral proxy.
## Runtime matrix
| Mode | What stays running | Primary entrypoint |
|---|---|---|
| Persistent Service | Native background service | `headroom install apply --preset persistent-service` |
| Persistent Task | Scheduled watchdog + on-demand runner | `headroom install apply --preset persistent-task` |
| Persistent Docker | Restartable Docker container | `headroom install apply --preset persistent-docker` |
| On-Demand CLI (Python) | Nothing after command exits | `headroom proxy` |
| On-Demand CLI (Docker) | Nothing after container exits | Docker-native wrapper / compose CLI |
| Wrapped (Python) | Proxy lasts for wrapped session | `headroom wrap ...` |
| Wrapped (Docker) | Containerized proxy + host tool session | Docker-native wrapper |
## Quick examples
### Persistent service on the local machine
```bash
headroom install apply --preset persistent-service --providers auto
headroom install status
```
This installs a background service on the current machine, applies persistent tool wiring, and keeps the proxy healthy on port `8787`.
### Persistent watchdog task
```bash
headroom install apply --preset persistent-task --providers manual --target claude --target codex
```
This installs a scheduled recovery path instead of a traditional always-running service.
### Persistent Docker
```bash
headroom install apply --preset persistent-docker --scope user --providers auto
```
This uses Docker's restart policy instead of an OS supervisor.
If you are using the Docker-native host wrapper instead of a Python install, you can use `headroom install apply|status|start|stop|restart|remove` for the `persistent-docker` preset directly from the installed wrapper. Service/task installs and provider/user/system mutation flows still belong to the Python-native CLI.
## Command surface
```text
headroom install apply
headroom install status
headroom install start
headroom install stop
headroom install restart
headroom install remove
```
`apply` creates or updates a named deployment profile, stores its manifest under `~/.headroom/deploy/<profile>/manifest.json`, applies reversible configuration changes, and starts the selected runtime.
## Presets and runtime kinds
### Presets
- `persistent-service` → native service supervisor
- `persistent-task` → scheduled watchdog / recovery supervisor
- `persistent-docker` → Docker restart policy with no extra OS supervisor
### Runtime kinds
- `--runtime python` runs `headroom proxy` directly
- `--runtime docker` runs Headroom inside Docker while keeping the deployment managed locally
For `persistent-docker`, the runtime is always Docker.
## Configuration scopes
| Scope | What changes |
|---|---|
| `provider` | Tool-specific config surfaces where Headroom can make a precise reversible edit |
| `user` | User-level shell or environment surfaces |
| `system` | Machine-wide shell or environment surfaces |
### Provider scope today
Provider scope is intentionally conservative. The current direct adapters are:
- Claude Code → `~/.claude/settings.json` `env`
- Codex → managed block in `~/.codex/config.toml`
- OpenClaw → existing `wrap openclaw` / `unwrap openclaw` flow
- OpenCode → managed block in `~/.config/opencode/opencode.json`
For Copilot, Aider, Cursor, and broader env-driven setups, prefer `--scope user` or `--scope system`.
## Provider selection
| Option | Meaning |
|---|---|
| `--providers auto` | Detect supported tools on the host and configure the best available defaults |
| `--providers all` | Configure all known targets |
| `--providers manual --target ...` | Configure only the named tools |
Examples:
```bash
headroom install apply --providers auto
headroom install apply --providers all --scope user
headroom install apply --providers manual --target claude --target copilot
```
## Health and wrap behavior
Persistent deployments publish the same `readyz` and `health` endpoints as ad hoc proxy runs.
`/health` also exposes deployment metadata when the proxy was launched through the install subsystem:
```json
{
"deployment": {
"profile": "default",
"preset": "persistent-service",
"runtime": "python",
"supervisor": "service",
"scope": "user"
}
}
```
The Python-native `headroom wrap ...` flow checks for a matching persistent deployment on the requested port before it starts a new ephemeral proxy. If an installed deployment exists but is stopped or unhealthy, it attempts to recover it first.
The Docker-native host wrapper does **not** yet reuse or recover persistent profiles automatically; it still starts a fresh proxy container unless you opt into `--no-proxy`.
## Docker-native relationship
The Docker-native host wrapper and the Python install CLI solve different layers of the runtime story:
- [Docker-Native Install](/docs/docker-install) → containerized on-demand CLI, wrapped host-tool flows, and Docker-native `persistent-docker` lifecycle commands
- `headroom install ...` → full persistent service, task, and Docker lifecycle management, including provider/user/system mutation
For a no-Python persistent Docker workflow, use the compose-managed proxy path from `docker/docker-compose.native.yml`:
```bash
export HEADROOM_HOST_HOME="$HOME"
export HEADROOM_WORKSPACE="$PWD"
docker compose -f docker/docker-compose.native.yml up -d proxy
```
That keeps `localhost:8787` stable and restarts the proxy automatically.
<Callout type="info" title="HEADROOM_WORKSPACE vs HEADROOM_WORKSPACE_DIR">
`HEADROOM_WORKSPACE` (the host-side bind-mount source used by the compose file) is **not** the same variable as `HEADROOM_WORKSPACE_DIR` (the canonical Headroom state root inside the container). Both are retained; the compose file sets the latter automatically. See [Filesystem Contract](/docs/filesystem-contract) for the full bucket model.
</Callout>
## Related guides
<Cards>
<Card title="Docker-Native Install" href="/docs/docker-install" />
<Card title="Proxy Server" href="/docs/proxy" />
<Card title="Filesystem Contract" href="/docs/filesystem-contract" />
<Card title="Configuration" href="/docs/configuration" />
</Cards>
+79
View File
@@ -0,0 +1,79 @@
---
title: Pipeline Extensions
description: Write a request-normalization extension for a quirky upstream provider, and route requests to different upstream bases per request with x-headroom-base-url.
---
Headroom emits lifecycle events at every stage of the canonical request pipeline. Third-party packages can hook these events — without forking Headroom — by registering a **pipeline extension** under the `headroom.pipeline_extension` entry-point group. Extensions can mutate `messages`, `tools`, `headers`, or `metadata` in place before the request is forwarded upstream.
Both the SDK client and the proxy dispatch the same events, so one extension covers both deployments.
## Lifecycle stages
Extensions receive a `PipelineEvent` for each stage in `headroom.pipeline.PipelineStage`:
| Stage | When |
|-------|------|
| `SETUP`, `PRE_START`, `POST_START` | Process/pipeline startup |
| `INPUT_RECEIVED` | Raw request accepted |
| `INPUT_CACHED`, `INPUT_ROUTED`, `INPUT_COMPRESSED`, `INPUT_REMEMBERED` | Cache, routing, compression, memory stages |
| `PRE_SEND` | Last hook before the request is forwarded upstream |
| `POST_SEND`, `RESPONSE_RECEIVED` | After forwarding / on response |
`PRE_SEND` is the right stage for normalizing requests to fit a quirky upstream: compression and caching are done, and whatever you write into `event.messages` is exactly what the provider receives.
## Recipe: normalize requests for a quirky upstream provider
Some OpenAI-compatible gateways reject valid OpenAI-spec payloads. A real example: an upstream returns `400 "Message content is null"` for assistant messages that carry `content: null` alongside `tool_calls` — a combination the OpenAI spec explicitly produces when the model returns only tool calls. The provider-recommended workaround is to send `content: ""` instead.
An extension that rewrites those messages at `PRE_SEND`:
```python
# my_headroom_ext/normalize.py
from headroom.pipeline import PipelineEvent, PipelineStage
class NullContentNormalizer:
"""Rewrite `content: null` + tool_calls to `content: ""` before send."""
def on_pipeline_event(self, event: PipelineEvent) -> PipelineEvent | None:
if event.stage is not PipelineStage.PRE_SEND or not event.messages:
return None
for message in event.messages:
if (
message.get("role") == "assistant"
and message.get("content") is None
and message.get("tool_calls")
):
message["content"] = ""
return None # mutated in place; returning None keeps the event
```
Register it as an entry point in your extension package:
```toml
# pyproject.toml of your extension package
[project.entry-points."headroom.pipeline_extension"]
null-content-normalizer = "my_headroom_ext.normalize:NullContentNormalizer"
```
Install the package into the same environment as Headroom (`pip install my-headroom-ext`) and it is discovered automatically — entry points are loaded on startup, and a failing extension is isolated and logged rather than breaking the pipeline.
Notes on the contract:
- An extension is either an object with an `on_pipeline_event(event)` method or a class Headroom instantiates with no arguments.
- Return `None` (mutate in place) or return a replacement `PipelineEvent`.
- Exceptions raised by an extension are caught and logged (`fail-open`); the request proceeds unmodified.
- Discovery can be disabled with the SDK config flag `discover_pipeline_extensions=False`, and explicit instances can be passed via `pipeline_extensions=[...]` (SDK `HeadroomConfig` and proxy `ProxyConfig` both expose these fields).
## Per-request upstream routing with `x-headroom-base-url`
To route different models through one Headroom instance to different OpenAI-compatible upstream bases — instead of one global `OPENAI_API_URL` / `OPENAI_TARGET_API_URL` per proxy process — send the `x-headroom-base-url` request header. The dedicated OpenAI handlers (`/v1/chat/completions`, `/v1/responses`) and the generic passthrough route all honor it, falling back to the configured upstream when absent:
```bash
curl http://localhost:8787/v1/chat/completions \
-H "content-type: application/json" \
-H "x-headroom-base-url: https://api.example-gateway.ai/gemini-3-flash" \
-d '{"model": "gemini-3-flash", "messages": [{"role": "user", "content": "hi"}]}'
```
Internal `x-headroom-*` headers (including this one) are stripped before the request is forwarded upstream by default — see `HEADROOM_STRIP_INTERNAL_HEADERS` in [Configuration](/docs/configuration).
+338
View File
@@ -0,0 +1,338 @@
---
title: Proxy Server
description: Run the Headroom proxy to compress LLM traffic for any client — Claude Code, Cursor, OpenAI SDK, or custom apps.
---
The Headroom proxy is a standalone HTTP server that compresses all LLM traffic passing through it. Point any client at the proxy and get automatic context optimization.
Running a local OpenAI-compatible model? See [Local LLM prefill benchmarking](/docs/local-llm-prefill) for a baseline-vs-optimized workflow that measures prompt-processing savings with the dashboard.
## Starting the proxy
```bash
# Basic usage
headroom proxy
# Custom host and port
headroom proxy --host 0.0.0.0 --port 8080
# With logging and budget
headroom proxy \
--log-file /var/log/headroom.jsonl \
--budget 100.0
```
Telemetry is **off by default** (opt-in). Opt in with `HEADROOM_TELEMETRY=on` or `--telemetry`.
## CLI options
### Core
| Option | Default | Description |
|--------|---------|-------------|
| `--host` | `127.0.0.1` | Host to bind to |
| `--port` | `8787` | Port to bind to |
| `--workers` | `1` | Number of Uvicorn worker processes |
| `--limit-concurrency` | `1000` | Maximum concurrent connections before Uvicorn returns 503 |
| `--max-connections` | `500` | Maximum upstream HTTP connections |
| `--max-keepalive` | `100` | Maximum upstream keep-alive connections |
| `--http-proxy` | None | HTTP proxy URL for upstream provider requests only; HTTPS provider APIs use CONNECT |
| `--mode` | `token` | Optimization mode: `token` prioritizes compression, `cache` preserves provider prefix-cache stability |
| `--no-optimize` | `false` | Disable optimization (passthrough mode) |
| `--no-cache` | `false` | Disable semantic caching |
| `--no-rate-limit` | `false` | Disable rate limiting |
| `--log-file` | None | Path to JSONL log file |
| `--log-messages` | `false` | Store full request/response content for the live feed |
| `--budget` | None | Daily budget limit in USD |
| `--openai-api-url` | `https://api.openai.com` | Custom OpenAI API URL |
| `--anthropic-api-url` | Anthropic default | Custom Anthropic API URL |
| `--gemini-api-url` | Gemini default | Custom Gemini API URL |
| `--backend` | `anthropic` | Backend: `anthropic`, `bedrock`, `openrouter`, `anyllm`, or `litellm-<provider>` |
| `--bedrock-api-url` | None | Bedrock InvokeModel upstream for the `/model/{id}/invoke` passthrough routes (see [Bedrock via a local gateway](#bedrock-via-a-local-gateway)) |
| `--telemetry` | `false` | Opt in to anonymous telemetry (off by default) |
| `--no-telemetry` | `false` | Force anonymous telemetry off (already the default) |
| `--stateless` | `false` | Disable filesystem writes and keep runtime state in memory |
Use `--http-proxy` or `HEADROOM_HTTP_PROXY` when only provider API traffic should go through a proxy:
```bash
headroom proxy --http-proxy http://proxy.internal:8080
```
Avoid setting process-wide variables such as `HTTP_PROXY`, `HTTPS_PROXY`, `ALL_PROXY`, or `NO_PROXY` for this use case. HTTPX reads those variables too, but Headroom also inherits them into tool executions, so they can proxy unrelated tool traffic.
### Context management
| Option | Default | Description |
|--------|---------|-------------|
| `--mode token` | `token` | Prioritize token compression. This is the default. |
| `--mode cache` | `token` | Preserve prior turns to maximize provider prefix-cache hit rate. |
| `--intercept-tool-results` | `false` | Opt into tool-result interceptors such as ast-grep Read outlining. |
| `--no-read-lifecycle` | `false` | Disable stale/superseded Read-output compression. |
| `--code-aware` / `--no-code-aware` | disabled | Enable or disable AST-based code compression. Requires `headroom-ai[code]`. |
| `--code-graph` | `false` | Force a tokensave code-graph index of the current project (tokensave is the default coding-task compressor registered by `headroom wrap`). |
#### tokensave binary trust model
`headroom wrap` registers **tokensave** (a local code-graph MCP server) as the default coding-task compressor. tokensave ships as a single prebuilt Rust binary, so `wrap` downloads the release asset for your platform from GitHub and runs it locally. Because the binary is executed, every supported asset is **pinned to a SHA-256 digest in Headroom** (`headroom/graph/tokensave_installer.py`); the downloaded bytes are verified against that digest before extraction, and a mismatch aborts the install (Headroom falls back to the Serena backup) rather than running unverified code.
- Set `HEADROOM_BINARIES_OFFLINE=1` to never reach the network — `wrap` then uses an already-installed tokensave or falls back to Serena.
- `HEADROOM_TOKENSAVE_VERSION` overrides the pinned release tag. Since an overridden version has no pinned digest, the download is **refused** unless you also set `HEADROOM_TOKENSAVE_ALLOW_UNVERIFIED=1`.
- Pass `--no-tokensave` to skip the primary compressor entirely, or `--serena` to force the Serena backup on.
By default, the proxy uses the shared **ContentRouter** pipeline. It routes text, logs, JSON, code, images, and tool outputs through the currently enabled compressors and preserves reversible CCR markers where applicable.
```bash
# Maximize compression
headroom proxy --mode token
# Preserve provider prefix cache stability
headroom proxy --mode cache
```
### Optional features
| Option | Default | Description |
|--------|---------|-------------|
| `--memory` | `false` | Enable persistent user memory and provider-appropriate memory tools |
| `--memory-db-path` | `{cwd}/.headroom/memory.db` | Override the memory SQLite path |
| `--no-memory-tools` | `false` | Disable automatic memory tool injection |
| `--no-memory-context` | `false` | Disable automatic memory context injection |
| `--memory-top-k` | `10` | Number of memories to inject as context |
| `--learn` | `false` | Enable live traffic learning; implies `--memory` |
| `--no-learn` | `false` | Explicitly disable traffic learning |
| `--min-evidence` | `5` | Minimum observations before a learned pattern is persisted |
| `--codex-wire-debug` | `false` | Write local Codex wire snapshots and matching proxy log traces |
```bash
headroom proxy --memory
headroom proxy --learn --min-evidence 3
headroom proxy --codex-wire-debug
```
<Callout type="info" title="LLMLingua removed from the proxy CLI">
The old LLMLingua proxy toggles are no longer part of the CLI. Headroom's proxy compression path uses ContentRouter plus the current built-in compressors, including Kompress where applicable.
</Callout>
## API endpoints
### `GET /health`
```bash
curl http://localhost:8787/health
```
```json
{
"status": "healthy",
"optimize": true,
"stats": {
"total_requests": 42,
"tokens_saved": 15000,
"savings_percent": 45.2
}
}
```
### `GET /stats`
Live session statistics plus durable `persistent_savings` totals. Stored at `~/.headroom/proxy_savings.json` (override with `HEADROOM_SAVINGS_PATH`).
```bash
curl http://localhost:8787/stats
```
### `GET /stats-history`
Durable history with hourly, daily, weekly, and monthly rollups. Powers the `/dashboard` view.
```bash
curl http://localhost:8787/stats-history
curl "http://localhost:8787/stats-history?format=csv&series=weekly"
```
### `GET /metrics`
Prometheus-format metrics for monitoring.
```bash
curl http://localhost:8787/metrics
```
```
headroom_requests_total{mode="optimize"} 1234
headroom_tokens_saved_total 5678900
headroom_persistent_savings_tokens_saved_total 5678900
headroom_compression_ratio_bucket{le="0.5"} 890
headroom_latency_seconds_bucket{le="0.01"} 800
headroom_cache_hits_total 456
```
`headroom_tokens_saved_total` is the runtime counter for the current proxy process. Use `headroom_persistent_savings_tokens_saved_total` for durable lifetime savings that match `/stats.persistent_savings`.
### `POST /v1/messages`
Anthropic API format. The proxy compresses messages, forwards to Anthropic, and returns the response.
### `POST /v1/chat/completions`
OpenAI API format. The proxy compresses messages, forwards to OpenAI, and returns the response.
### `POST /v1/responses`
OpenAI Responses API format. The proxy compresses `input` payloads where applicable, forwards the request, and returns the response.
For Codex-compatible clients, the proxy also accepts these alias paths and routes them through the same handler:
- `POST /v1/codex/responses`
- `POST /backend-api/responses`
- `POST /backend-api/codex/responses`
Matching WebSocket and subpath aliases are also supported for Codex flows.
### `POST /v1internal:streamGenerateContent`
Google Cloud Code Assist / Antigravity compatibility endpoint used by Pi-style `google-gemini-cli` and `google-antigravity` providers.
The proxy also accepts:
- `POST /v1/v1internal:streamGenerateContent`
### `POST /v1/compress`
Compression-only endpoint. Compresses messages without calling any LLM. Used by the TypeScript SDK.
**Request:**
```json
{
"messages": [{ "role": "user", "content": "..." }],
"model": "gpt-4o"
}
```
**Response:**
```json
{
"messages": [{ "role": "user", "content": "..." }],
"tokens_before": 15000,
"tokens_after": 3500,
"tokens_saved": 11500,
"compression_ratio": 0.23,
"transforms_applied": ["router:smart_crusher:0.35"],
"ccr_hashes": ["a1b2c3"]
}
```
Set `x-headroom-bypass: true` to skip compression.
## Agent wrapping
Use `headroom wrap` to launch supported CLI agents through the local proxy:
```bash
# Claude Code
headroom wrap claude
# OpenAI Codex
headroom wrap codex
# Aider
headroom wrap aider
# Cursor (starts the proxy and prints settings to paste into Cursor)
headroom wrap cursor
```
Cursor reads model endpoints from its settings UI, so `headroom wrap cursor`
does not rewrite Cursor configuration or launch the app. After it starts the
proxy, copy the printed base URL into Cursor's model settings.
For environment-driven clients, you can also set the base URL manually:
```bash
# Claude Code
ANTHROPIC_BASE_URL=http://localhost:8787 claude
# Any OpenAI-compatible CLI client that reads OPENAI_BASE_URL
OPENAI_BASE_URL=http://localhost:8787/v1 your-client
```
## Cloud providers
```bash
# AWS Bedrock
headroom proxy --backend bedrock --region us-east-1
# Google Vertex AI
headroom proxy --backend vertex_ai --region us-central1
# Azure OpenAI
headroom proxy --backend azure
# OpenRouter (400+ models)
OPENROUTER_API_KEY=sk-or-... headroom proxy --backend openrouter
```
### Bedrock via a local gateway
`--backend bedrock` accepts **Anthropic** input (`/v1/messages`) and re-signs to AWS. Some setups are the other way around: the client already speaks **Bedrock** (e.g. Claude Code with `CLAUDE_CODE_USE_BEDROCK=1`, or any AWS SDK pointed at a custom endpoint), sending `POST /model/{id}/invoke` to a local gateway that re-signs and forwards to AWS (LiteLLM, LocalStack, a corporate Bedrock proxy).
`--bedrock-api-url` lets Headroom sit in that chain. It registers passthrough routes for `/model/{id}/invoke` and `/model/{id}/invoke-with-response-stream`, compresses the request body with the same pipeline as `/v1/messages`, and forwards to the gateway:
```bash
headroom proxy --bedrock-api-url http://127.0.0.1:4000
# then point the client's Bedrock endpoint at Headroom:
AWS_ENDPOINT_URL_BEDROCK_RUNTIME=http://127.0.0.1:8787 your-bedrock-client
```
The routes are registered **only** when `--bedrock-api-url` (or `BEDROCK_TARGET_API_URL`) is set — otherwise Bedrock requests fall through unchanged.
<Callout type="warn">
Rewriting the request body invalidates the caller's **SigV4** signature (it covers a hash of the body). Point `--bedrock-api-url` at a gateway that re-signs or does not verify the inbound signature — **never raw AWS**, which would reject the request with 403. For direct-to-AWS compression, use `--backend bedrock` (which re-signs). The two are complementary.
</Callout>
## Environment variables
```bash
export HEADROOM_HOST=0.0.0.0
export HEADROOM_PORT=8787
export HEADROOM_BUDGET=100.0
# Route OpenAI passthrough requests to a custom endpoint
export OPENAI_TARGET_API_URL=https://custom.openai.endpoint.com
# Route Anthropic passthrough requests to a custom endpoint
export ANTHROPIC_TARGET_API_URL=https://litellm.company.internal
# Compress Bedrock InvokeModel traffic, forwarding to a re-signing gateway
export BEDROCK_TARGET_API_URL=http://127.0.0.1:4000
headroom proxy
```
## Production deployment
### gunicorn
```bash
pip install gunicorn
gunicorn headroom.proxy.server:app \
--workers 4 \
--bind 0.0.0.0:8787 \
--worker-class uvicorn.workers.UvicornWorker
```
### Docker
```dockerfile
FROM python:3.11-slim
RUN apt-get update && apt-get install -y --no-install-recommends build-essential \
&& pip install "headroom-ai[proxy]" \
&& apt-get purge -y build-essential && apt-get autoremove -y \
&& rm -rf /var/lib/apt/lists/*
EXPOSE 8787
CMD ["headroom", "proxy", "--host", "0.0.0.0"]
```
<Callout type="info" title="Build dependencies">
`build-essential` is required at install time because `headroom-ai` includes `hnswlib`, a C++ extension compiled from source. It is removed after installation to keep the image slim.
</Callout>
+240
View File
@@ -0,0 +1,240 @@
---
title: Quickstart
description: Get Headroom running in 5 minutes. Install, compress, and send to your LLM with fewer tokens.
---
This guide gets you from zero to compressed LLM calls in under 5 minutes.
## 1. Install
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```bash
npm install headroom-ai
```
</Tab>
<Tab value="Python">
```bash
pip install "headroom-ai[all]"
```
</Tab>
</Tabs>
<Callout type="info" title="TypeScript SDK requires the proxy">
The TypeScript SDK sends messages to a local Headroom proxy for compression. Start the proxy before using the TS SDK:
```bash
pip install "headroom-ai[proxy]"
headroom proxy --port 8787
```
The proxy runs the compression pipeline (Python) and exposes an HTTP API that the TS SDK calls.
</Callout>
## 2. Compress messages
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from 'headroom-ai';
const messages = [
{ role: 'system' as const, content: 'You analyze search results.' },
{ role: 'user' as const, content: 'Search for Python tutorials.' },
{
role: 'assistant' as const,
content: null,
tool_calls: [{
id: 'call_1',
type: 'function' as const,
function: { name: 'search', arguments: '{"q": "python"}' },
}],
},
{
role: 'tool' as const,
tool_call_id: 'call_1',
content: JSON.stringify({
results: Array.from({ length: 500 }, (_, i) => ({
title: `Result ${i}`,
snippet: `Description ${i}`,
score: 100 - i,
})),
}),
},
{ role: 'user' as const, content: 'What are the top 3 results?' },
];
const result = await compress(messages, {
model: 'gpt-4o',
baseUrl: 'http://localhost:8787',
});
```
</Tab>
<Tab value="Python">
```python
from headroom import compress
import json
messages = [
{"role": "system", "content": "You analyze search results."},
{"role": "user", "content": "Search for Python tutorials."},
{
"role": "assistant",
"content": None,
"tool_calls": [{
"id": "call_1",
"type": "function",
"function": {"name": "search", "arguments": '{"q": "python"}'},
}],
},
{
"role": "tool",
"tool_call_id": "call_1",
"content": json.dumps({
"results": [
{"title": f"Result {i}", "snippet": f"Description {i}", "score": 100 - i}
for i in range(500)
]
}),
},
{"role": "user", "content": "What are the top 3 results?"},
]
result = compress(messages, model="gpt-4o")
```
</Tab>
</Tabs>
## 3. Send to your LLM
Use the compressed messages exactly like the originals:
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import OpenAI from 'openai';
const client = new OpenAI();
// result.messages from the previous step
const messages: any[] = [];
const response = await client.chat.completions.create({
model: 'gpt-4o',
messages,
});
console.log(response.choices[0].message.content);
```
</Tab>
<Tab value="Python">
```python
from openai import OpenAI
client = OpenAI()
response = client.chat.completions.create(
model="gpt-4o",
messages=result.messages,
)
print(response.choices[0].message.content)
```
</Tab>
</Tabs>
## 4. Check your savings
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
const result = {
tokensBefore: 45000,
tokensAfter: 4500,
tokensSaved: 40500,
compressionRatio: 0.9,
transformsApplied: ['smart_crusher', 'cache_aligner'],
messages: [],
ccrHashes: [],
compressed: true,
};
// ---cut---
console.log(`Tokens before: ${result.tokensBefore}`);
console.log(`Tokens after: ${result.tokensAfter}`);
console.log(`Tokens saved: ${result.tokensSaved}`);
console.log(`Compression: ${(result.compressionRatio * 100).toFixed(0)}%`);
console.log(`Transforms: ${result.transformsApplied.join(', ')}`);
```
Example output:
```
Tokens before: 45000
Tokens after: 4500
Tokens saved: 40500
Compression: 90%
Transforms: smart_crusher, cache_aligner
```
</Tab>
<Tab value="Python">
```python
print(f"Tokens before: {result.tokens_before}")
print(f"Tokens after: {result.tokens_after}")
print(f"Tokens saved: {result.tokens_saved}")
print(f"Compression: {result.compression_ratio:.0%}")
print(f"Transforms: {result.transforms_applied}")
```
Example output:
```
Tokens before: 45000
Tokens after: 4500
Tokens saved: 40500
Compression: 90%
Transforms: ['smart_crusher', 'cache_aligner']
```
</Tab>
</Tabs>
## Alternative: proxy mode (zero code changes)
If you do not want to change any code, run Headroom as a proxy and point your existing client at it:
```bash
# Start the proxy
headroom proxy --port 8787
# Point Claude Code at it
ANTHROPIC_BASE_URL=http://localhost:8787 claude
# Or any OpenAI-compatible client
OPENAI_BASE_URL=http://localhost:8787/v1 your-app
```
All requests flow through Headroom automatically. Check savings at any time:
```bash
curl http://localhost:8787/stats
# {"requests_total": 42, "tokens_saved_total": 125000, ...}
```
## What gets compressed
The biggest savings come from tool outputs -- search results, database rows, log files, API responses. Headroom auto-detects the content type and routes it to the best compressor. No configuration needed.
| Content type | Compressor | Typical savings |
|---|---|---|
| JSON arrays | SmartCrusher | 70--90% |
| Source code | CodeCompressor | 40--70% |
| Build/test logs | LogCompressor | 80--95% |
| Search results | SearchCompressor | 60--80% |
| Plain text | Kompress | 30--50% |
## Next steps
<Cards>
<Card title="Installation" href="/docs/installation" />
<Card title="Proxy Server" href="/docs/proxy" />
<Card title="How Compression Works" href="/docs/how-compression-works" />
<Card title="Configuration" href="/docs/configuration" />
</Cards>
+297
View File
@@ -0,0 +1,297 @@
---
title: Releases & CI/CD
description: Automated release pipeline with release-please, semantic versioning, multi-package publishing, and changelog generation.
---
## Overview
Headroom uses `release-please` to maintain a release PR from conventional commits on `main`. Merging that release PR creates the release tag and GitHub Release, which triggers `.github/workflows/release.yml` to publish all packages, build version-matched Docker images, and attach release assets.
The release workflow also calls `.github/workflows/docker.yml` as a reusable workflow so GHCR images are published in the same release run with the exact same synced version as PyPI, npm, and GitHub release assets.
For the end-to-end visual flow, see [CI/CD Flow Diagrams](/docs/ci-cd-flows).
## Packages & Registries
| Package | Type | Registry | Environment Variable |
|---------|------|----------|----------------------|
| `headroom-ai` | Python | PyPI | `PYPI_PACKAGE` |
| `headroom-ai` | TypeScript SDK | npmjs.org | `NPM_SDK_PACKAGE` |
| `headroom-openclaw` | TypeScript plugin | npmjs.org | `NPM_OPENCLAW_PACKAGE` |
| `@{owner}/headroom-ai` | TypeScript SDK | GitHub Package Registry | — |
| `@{owner}/headroom-openclaw` | TypeScript plugin | GitHub Package Registry | — |
| `headroom-ai-{version}.tar.gz` / `headroom_ai-{version}-py3-none-any.whl` | Python package distributions | GitHub Release (`{owner}/headroom`) | — |
| `headroom-ai-{version}.tgz` / `headroom-openclaw-{version}.tgz` | Node release assets | GitHub Release (`{owner}/headroom`) | — |
| `ghcr.io/{owner}/headroom` | Docker image | GitHub Container Registry | — |
## Version Strategy
Release Please calculates the release version from conventional commits and the release manifest. The release workflow still computes and verifies the version it is about to publish:
1. `.release-please-config.json` defines release-please behavior.
2. `.release-please-manifest.json` tracks current package versions.
3. The release PR updates versions and changelog content.
4. Merging the release PR publishes a GitHub Release tagged `vX.Y.Z`.
5. `release.yml` uses that tag as the manual version for the publish run.
### Version Files
- `.release-please-config.json` - release-please package configuration
- `.release-please-manifest.json` - release-please version manifest
- `pyproject.toml` - `[project].version`
- `headroom/_version.py` - `__version__`, synced at build time
- `plugins/openclaw/package.json` - `version`, synced at build time
- `sdk/typescript/package.json` - `version`, synced at build time
`release.yml` does not commit back to the repo. Version synchronization happens inside the release build workspace.
## Conventional Commits & Semantic Bumping
Release Please analyzes unreleased conventional commits and applies the highest required bump level:
| Commit | Bump |
|--------|------|
| `fix:` | patch |
| `feat:` | minor |
| Any conventional commit with `!` or any commit with `BREAKING CHANGE` in the body | major |
| `docs:`, `ci:`, `chore:`, `refactor:` | no release note by default unless configured |
Commits are linted in CI via `commitlint` using `@commitlint/config-conventional`.
The release PR is the place where version and changelog changes are reviewed before publishing.
## Release Workflow
The `release.yml` workflow runs when a GitHub Release is published, which normally happens when the release-please PR is merged. It also supports manual `workflow_dispatch` and PR dry-runs for release-critical workflow/package changes.
```
detect-version → build → build-wheels → collect-dist → smoke-import-wheels
↘ publish-pypi
↘ publish-npm
↘ publish-github-packages
↘ publish-docker
→ create-release
```
The workflow never commits back to the repo.
### detect-version
Resolves the release version from the trigger. On `release: published`, it uses the published tag (`vX.Y.Z`) as the manual version for the run. On `workflow_dispatch`, it uses the optional `version` input when provided. PR dry-runs compute a version without publishing.
### build
1. Syncs version across package files via `scripts/version-sync.py --version {npm_version}`
2. Verifies package versions with `scripts/verify-versions.py`
3. Generates the changelog artifact
4. Builds npm release packages for the TypeScript SDK and OpenClaw plugin
5. Uploads release asset artifacts for downstream publish jobs
### build-wheels
Builds the Python wheel matrix for Linux x86_64, Linux arm64, and Apple Silicon macOS, plus one source distribution. Linux wheels are audited for glibc symbol compatibility.
### collect-dist
Collects the wheel matrix, source distribution, and npm tarballs into the canonical artifacts used by publishing and GitHub Release asset upload.
### smoke-import-wheels
Installs the built wheels into representative customer environments and imports `headroom._core`. This blocks publishing if a wheel builds successfully but cannot import on its promised platform floor.
### publish-pypi
Downloads the Python dist artifact and publishes to PyPI via `pypa/gh-action-pypi-publish@release/v1` (trusted publisher).
### publish-npm
Publishes both TypeScript packages to npmjs.org:
- `sdk/typescript/` as `headroom-ai`
- `plugins/openclaw/` as `headroom-openclaw`
### publish-github-packages
Publishes both Node packages to GitHub Package Registry (`npm.pkg.github.com`) using the current repository owner as the npm scope:
- `sdk/typescript/` as `@{owner}/headroom-ai`
- `plugins/openclaw/` as `@{owner}/headroom-openclaw`
### GitHub release assets
Uploads the built Python distributions and both npm tarballs to the GitHub Release created in the current repository. GitHub Packages does not provide a PyPI-compatible package registry, so the workflow publishes Python wheels and sdists to GitHub as release assets while npm packages go to GitHub Package Registry and Docker images go to GHCR.
### publish-docker
Calls the reusable Docker workflow to publish GHCR images with the same semantic version and synced package metadata as the rest of the release.
### create-release
Creates or updates the GitHub Release in the current repo and uploads the built Python distributions and npm tarballs as release assets. PyPI publish is a hard gate unless `PYPI_SKIP=true`, so release notes and assets do not advertise a version that failed to publish to PyPI.
## Configuration
All package names, registry URLs, and environment names are defined as top-level `env` constants:
```yaml
env:
PYPI_PACKAGE: headroom-ai
PYPI_ENVIRONMENT: pypi
NPM_REGISTRY_URL: https://registry.npmjs.org
NPM_SDK_PACKAGE: headroom-ai
NPM_OPENCLAW_PACKAGE: headroom-openclaw
GITHUB_PACKAGES_REGISTRY_URL: https://npm.pkg.github.com
```
To rename a package, update the corresponding constant — all references throughout the workflow update automatically.
## Safety Gates
Each publish job requires **both** of the following to be false:
```yaml
if: github.event.inputs.dry_run != 'true' && vars.PYPI_SKIP != 'true'
```
To skip a publish target, set the corresponding GitHub Actions variable:
| Variable | Effect |
|----------|--------|
| `PYPI_SKIP=true` | Skip PyPI publish |
| `NPM_SKIP=true` | Skip both npm publishes |
| `GH_PACKAGES_SKIP=true` | Skip GitHub Package Registry publish |
Set these in: **GitHub repo → Settings → Variables → Actions Variables → New repository variable**.
PyPI publishing is a hard gate for GitHub Releases unless `PYPI_SKIP=true`.
If the PyPI upload fails, the workflow stops before creating or updating the
GitHub Release, so release notes cannot advertise a version that was not
published to PyPI.
Before release artifacts are built, the workflow runs:
```bash
python scripts/verify-versions.py
```
That gate fails on cross-package version drift. The sdist build is also checked
for a top-level `LICENSE` file before any publish job can consume it.
## Workflow Triggers
Release Please runs on pushes to `main` and maintains the release PR:
```yaml
on:
push:
branches: [main]
```
The publish workflow runs when a GitHub Release is published, on PR dry-runs for release-critical paths, and by manual dispatch:
```yaml
on:
release:
types: [published]
pull_request:
paths:
- ".github/workflows/release.yml"
- ".github/workflows/docker.yml"
- "crates/headroom-py/**"
- "pyproject.toml"
- "scripts/verify-versions.py"
- "scripts/version-sync.py"
- "Cargo.toml"
- "Cargo.lock"
workflow_dispatch:
inputs:
version:
description: "Manual version override"
required: false
dry_run:
description: "Skip publish"
type: boolean
default: false
```
- **Normal release:** merge the release-please PR; the bot publishes a GitHub Release, which triggers `release.yml`.
- **PR dry-run:** release-critical PRs build and smoke-import wheels before merge, but do not publish.
- **Manual dispatch:** use `version` to override the release version and `dry_run: true` to skip publish steps.
## Local Testing with `act`
### Prerequisites
```bash
# Install act
winget install act
# Optional: install actionlint for schema validation
winget install actionlint
```
### Dry-run Test
```bash
act workflow_dispatch -W .github/workflows/release.yml -e .github/act/dry-run.json
```
This runs the full workflow end-to-end with `dry_run=true`, skipping all publish steps.
### Simulate Release Please
```bash
act push -W .github/workflows/release-please.yml -e .github/act/push-feat.json
```
The `push-feat.json` event file simulates a `feat:` commit on `main` so the release-please workflow can be validated locally.
### Simulate a Published Release
```bash
act release -W .github/workflows/release.yml -e .github/act/release-published.json -n
```
The `release-published.json` event file simulates the event emitted when the release-please PR is merged.
### Validate the Release and Docker Workflows
```bash
bash scripts/validate-workflows.sh
```
This runs `actionlint` plus `act -n` against the release and Docker workflows using the checked-in `.github/act/*.json` event fixtures. CI runs the same script in the `workflow-validation` job so branch changes to release automation are validated before merge.
### Local Secrets
```bash
cp .env.act.example .env.act
# Edit .env.act and add your test tokens
```
`act` automatically reads `.env` and passes values as workflow secrets.
## Workflow Files Reference
| File | Purpose |
|------|---------|
| `.github/workflows/release.yml` | Main release pipeline |
| `.github/workflows/release-please.yml` | Release PR aggregation from conventional commits |
| `.github/workflows/ci.yml` | CI — lint, test, commitlint |
| `.github/workflows/publish.yml` | Manual-only PyPI fallback (superseded by `release.yml`) |
| `.commitlintrc.json` | Conventional commit rules |
| `scripts/version-sync.py` | Sync version across all packages |
| `scripts/changelog-gen.py` | Generate changelog from git log |
| `scripts/verify-versions.py` | Pre-release version alignment check |
| `.github/act/dry-run.json` | `act` event file for dry-run testing |
| `.github/act/push-feat.json` | `act` event file for feat commit testing |
| `.github/act/release-published.json` | `act` event file for release publish simulation |
| `.github/act/docker-version.json` | `act` event file for Docker workflow validation |
| `scripts/validate-workflows.sh` | Shared `actionlint` + `act -n` workflow validation script |
| `.actrc` | Default `act` flags (Ubuntu runner, reuse, quiet) |
| `.actrc.local.example` | Local `act` override template |
## Required GitHub Secrets
| Secret | Purpose | Where to Get |
|--------|---------|--------------|
| `NPM_TOKEN` | Publishing to npmjs.org | npmjs.com → Account → Access Tokens |
| `GITHUB_TOKEN` | GitHub Package Registry (auto-provided) | Automatically available in GitHub Actions |
The PyPI publish uses trusted publisher OIDC — no secret required, only the `pypi` GitHub Environment must be configured with your PyPI project.
## Release Cadence
Day to day:
1. Merge regular PRs to `main`.
2. Release Please updates the open release PR when releasable conventional commits land.
3. Review the release PR changelog and version bump.
4. Merge the release PR when ready to ship.
5. Watch `release.yml` publish PyPI, npm, GitHub Packages, Docker, and GitHub Release assets.
+62
View File
@@ -0,0 +1,62 @@
---
title: Savings Tracking
description: Durable, over-time compression savings — cost avoided plus Today / Last 7 days / All time and per-model / per-client breakdowns via `headroom savings`.
---
`headroom savings` shows how much Headroom has saved you over time — cost avoided, token counts, and breakdowns by model and client. Unlike `headroom_stats` (a single in-memory session snapshot), it reads a **durable ledger** that survives proxy and agent restarts.
## Usage
```bash
headroom savings # human-readable summary
headroom savings --json # machine-readable report
headroom savings --days 30 # restrict the lookback/retention window
headroom savings --reset # delete the ledger and start fresh
```
### Example
```text
Today ███████████░░░░░ 67.9% saved 19,000 / 28,000 tokens $0.0850
Last 7 days ███████████░░░░░ 67.1% saved 47,000 / 70,000 tokens $0.2250
All time ██████████░░░░░░ 65.0% saved 78,000 / 120,000 tokens $0.2680
Cost avoided per model:
claude-opus-4-8 $0.1750
gpt-5.5 $0.0350
unknown $0.0330
claude-haiku-4-5 $0.0250
Savings by client:
claude-code 4 calls · 60,000 tokens saved
codex 2 calls · 18,000 tokens saved
```
## How it works
Every compression appends one line to an **append-only, file-locked event ledger** at `~/.headroom/savings_events.jsonl`, and `headroom savings` aggregates it on read. This design is:
- **Durable** — the ledger is on disk, so totals survive proxy and agent restarts.
- **Accurate under concurrency** — Headroom's MCP server runs as multiple processes (the main agent plus each subagent), and the proxy is a separate process. An append-only, locked log lets every writer contribute without the lost-update races a single shared mutable file would suffer.
- **Self-pruning** — events older than the retention window (365 days by default) are dropped on read, and the file is compacted once it grows large.
Both compression paths feed the same ledger:
- **MCP tool** — each `headroom_compress` call records its client (the MCP client name) and tokens saved.
- **Proxy** — each request records its real upstream model, so cost is priced accurately.
### Cost basis
Cost avoided is the dollar value of the saved **input** tokens. Headroom uses [litellm](https://headroom-docs.vercel.app/docs/litellm) list pricing where the model is known (proxy traffic). MCP-tool compressions don't know the agent's upstream model, so they record `model="unknown"` and fall back to a blended per-token rate rather than reporting `$0`.
## Configuration
| Variable | Purpose |
| --- | --- |
| `HEADROOM_SAVINGS_EVENTS_PATH` | Override the ledger location (default `~/.headroom/savings_events.jsonl`). |
| `HEADROOM_MCP_CLIENT` | Override the client label recorded by the MCP tool path. |
| `HEADROOM_MCP_MODEL` | Optional model hint so MCP-tool compressions price against a known model instead of the blended fallback. |
<Callout type="info">
`headroom savings` is distinct from `headroom_stats` (a per-session, in-memory snapshot) and from the proxy's live `/stats` endpoint (backed by `proxy_savings.json`). The savings ledger is the durable, cross-process source of truth.
</Callout>
+227
View File
@@ -0,0 +1,227 @@
---
title: SharedContext
description: Compressed inter-agent context sharing. Reduce token usage by ~80% when agents hand off to each other.
---
When agents hand off to each other, context gets replayed in full. SharedContext compresses what moves between agents using Headroom's compression pipeline, typically saving **~80% of tokens** on agent handoffs.
## Quick Start
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { SharedContext } from "headroom-ai";
const ctx = new SharedContext();
// Agent A stores large output
const entry = await ctx.put("research", bigResearchOutput, {
agent: "researcher",
});
// Agent B gets compressed version (~80% smaller)
const summary = ctx.get("research");
// Agent B needs full details on demand
const full = ctx.get("research", { full: true });
```
</Tab>
<Tab value="Python">
```python
from headroom import SharedContext
ctx = SharedContext()
# Agent A stores large output
ctx.put("research", big_research_output, agent="researcher")
# Agent B gets compressed version (~80% smaller)
summary = ctx.get("research")
# Agent B needs full details on demand
full = ctx.get("research", full=True)
```
</Tab>
</Tabs>
## API
### `put(key, content, agent?)`
Store content under a key. Compresses automatically using Headroom's full pipeline (SmartCrusher for JSON, CodeCompressor for code, Kompress for text).
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { SharedContext } from "headroom-ai";
const ctx = new SharedContext();
// ---cut---
const entry = await ctx.put("findings", bigJsonOutput, {
agent: "researcher",
});
entry.originalTokens; // 20000
entry.compressedTokens; // 4000
entry.savingsPercent; // 80.0
entry.transforms; // ["router:json:0.20"]
```
</Tab>
<Tab value="Python">
```python
entry = ctx.put("findings", big_json_output, agent="researcher")
entry.original_tokens # 20,000
entry.compressed_tokens # 4,000
entry.savings_percent # 80.0
entry.transforms # ["router:json:0.20"]
```
</Tab>
</Tabs>
### `get(key, full?)`
Retrieve content. Returns the compressed version by default, or the original with `full=True`.
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { SharedContext } from "headroom-ai";
const ctx = new SharedContext();
// ---cut---
const compressed = ctx.get("findings"); // 4K tokens
const original = ctx.get("findings", { full: true }); // 20K tokens
const missing = ctx.get("nonexistent"); // null
```
</Tab>
<Tab value="Python">
```python
compressed = ctx.get("findings") # 4K tokens
original = ctx.get("findings", full=True) # 20K tokens
missing = ctx.get("nonexistent") # None
```
</Tab>
</Tabs>
### `stats()`
Aggregated statistics across all entries.
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { SharedContext } from "headroom-ai";
const ctx = new SharedContext();
// ---cut---
const stats = ctx.stats();
stats.entries; // 3
stats.totalOriginalTokens; // 60000
stats.totalCompressedTokens; // 12000
stats.totalTokensSaved; // 48000
stats.savingsPercent; // 80.0
```
</Tab>
<Tab value="Python">
```python
stats = ctx.stats()
stats.entries # 3
stats.total_original_tokens # 60000
stats.total_compressed_tokens # 12000
stats.total_tokens_saved # 48000
stats.savings_percent # 80.0
```
</Tab>
</Tabs>
### `keys()` and `clear()`
`keys()` lists all non-expired keys. `clear()` removes all entries.
## Configuration
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { SharedContext } from "headroom-ai";
// ---cut---
const ctx = new SharedContext({
model: "claude-sonnet-4-5-20250929", // For token counting
ttl: 3600, // 1 hour (default)
maxEntries: 100, // Evicts oldest when full
});
```
</Tab>
<Tab value="Python">
```python
ctx = SharedContext(
model="claude-sonnet-4-5-20250929", # For token counting
ttl=3600, # 1 hour (default)
max_entries=100, # Evicts oldest when full
)
```
</Tab>
</Tabs>
Entries expire after `ttl` seconds. When `maxEntries` is reached, the oldest entry is evicted.
## Framework Examples
SharedContext is framework-agnostic. It works anywhere context moves between agents.
### CrewAI
```python
from headroom import SharedContext
ctx = SharedContext()
# After researcher task completes
ctx.put("findings", researcher_task.output.raw)
# Coder task gets compressed context
coder_context = ctx.get("findings")
```
### LangGraph
```python
from headroom import SharedContext
ctx = SharedContext()
def researcher_node(state):
result = do_research()
ctx.put("research", result)
return {"research_summary": ctx.get("research")}
def coder_node(state):
# Compressed summary in state, full details on demand
full = ctx.get("research", full=True)
return {"code": write_code(full)}
```
### OpenAI Agents SDK
```python
from headroom import SharedContext
ctx = SharedContext()
def compress_handoff(messages):
for msg in messages:
if len(msg.content) > 1000:
ctx.put(msg.id, msg.content)
msg.content = ctx.get(msg.id)
return messages
handoff(agent=coder, input_filter=compress_handoff)
```
## How It Works
Under the hood, `put()` calls `headroom.compress()` -- the same pipeline used by the Headroom proxy -- and stores the original in memory. `get()` returns the compressed version. `get(full=True)` returns the original.
The compression pipeline routes content to the best compressor:
- **JSON arrays** -- SmartCrusher (70-95% compression)
- **Code** -- CodeCompressor (AST-aware)
- **Text** -- Kompress (ModernBERT-based) or passthrough
+149
View File
@@ -0,0 +1,149 @@
---
title: Simulation
description: Preview compression results without making an LLM call. Use simulation for cost estimation, debugging, and understanding waste signals.
---
Simulation mode lets you preview what Headroom would do to your messages without sending them to an LLM. This is useful for cost estimation, debugging compression behavior, and understanding where token waste comes from.
## Basic Usage
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from 'headroom-ai';
// compress() returns the same result structure —
// use it without sending to your LLM to simulate
const result = await compress(messages, { model: 'gpt-4o' });
console.log(`Would save: ${result.tokensSaved} tokens`);
console.log(`Compression ratio: ${(result.compressionRatio * 100).toFixed(1)}%`);
console.log(`Transforms: ${result.transformsApplied.join(', ')}`);
```
</Tab>
<Tab value="Python">
```python
plan = client.chat.completions.simulate(
model="gpt-4o",
messages=large_conversation,
)
print(f"Tokens before: {plan.tokens_before}")
print(f"Tokens after: {plan.tokens_after}")
print(f"Would save: {plan.tokens_saved} tokens ({plan.tokens_saved/plan.tokens_before*100:.1f}%)")
print(f"Transforms: {plan.transforms}")
```
</Tab>
</Tabs>
## Waste Signals
Simulation reports where token waste comes from in your messages:
```python
plan = client.chat.completions.simulate(
model="gpt-4o",
messages=messages,
)
waste = plan.waste_signals
print(f"JSON bloat: {waste.json_bloat_tokens} tokens")
print(f"HTML noise: {waste.html_noise_tokens} tokens")
print(f"Whitespace: {waste.whitespace_tokens} tokens")
print(f"Dynamic dates: {waste.dynamic_date_tokens} tokens")
print(f"Repetition: {waste.repetition_tokens} tokens")
```
Waste signals help you understand which parts of your input are contributing the most unnecessary tokens.
## Block Breakdown
The parser breaks your conversation into blocks so you can see where tokens are concentrated:
```python
# Block types: system, user, assistant, tool_call, tool_result, rag
# The breakdown shows token counts per block type
```
| Block Kind | Description |
|-----------|-------------|
| `system` | System prompt instructions |
| `user` | User messages |
| `assistant` | Model responses |
| `tool_call` | Function call requests |
| `tool_result` | Tool output (largest source of waste) |
| `rag` | Retrieved document context |
## Use Cases
### Cost Estimation
Run simulation on a representative sample of your workload to estimate savings before enabling `optimize` mode:
```python
import json
total_before = 0
total_after = 0
for messages in sample_conversations:
plan = client.chat.completions.simulate(
model="gpt-4o",
messages=messages,
)
total_before += plan.tokens_before
total_after += plan.tokens_after
savings_pct = (1 - total_after / total_before) * 100
print(f"Estimated savings: {savings_pct:.1f}%")
print(f"Tokens saved: {total_before - total_after:,}")
```
### Debugging Compression
Use simulation to understand why a particular conversation is or is not being compressed:
```python
plan = client.chat.completions.simulate(
model="gpt-4o",
messages=messages,
)
if plan.tokens_saved == 0:
print("No compression applied. Possible reasons:")
print("- Messages are too short (< 200 tokens per tool output)")
print("- No tool outputs with compressible JSON arrays")
print("- Content is already compact (code, grep results)")
else:
print(f"Transforms applied: {plan.transforms}")
# See the optimized messages
print(json.dumps(plan.messages_optimized, indent=2))
```
### Comparing Configurations
Test different configurations to find the best settings for your workload:
```python
from headroom import HeadroomClient, OpenAIProvider
from headroom.transforms import SmartCrusherConfig
configs = [
SmartCrusherConfig(max_items_after_crush=10),
SmartCrusherConfig(max_items_after_crush=25),
SmartCrusherConfig(max_items_after_crush=50),
]
for config in configs:
client = HeadroomClient(
original_client=OpenAI(),
provider=OpenAIProvider(),
smart_crusher_config=config,
)
plan = client.chat.completions.simulate(model="gpt-4o", messages=messages)
print(f"max_items={config.max_items_after_crush}: "
f"{plan.tokens_saved} tokens saved ({plan.tokens_saved/plan.tokens_before*100:.1f}%)")
```
<Callout type="info" title="No API call">
Simulation never calls the LLM API. It runs the full transform pipeline locally and returns the results, so there is no cost and no latency from the provider.
</Callout>
+146
View File
@@ -0,0 +1,146 @@
---
title: SmartCrusher
description: Statistical JSON and array compression that keeps important items and drops the rest, achieving 70-90% token reduction.
---
SmartCrusher is Headroom's compressor for JSON tool outputs. It analyzes arrays statistically, keeps the important items (errors, anomalies, relevant matches), and drops the rest. This is the compressor that fires automatically when ContentRouter detects JSON arrays.
## How It Works
SmartCrusher doesn't blindly truncate arrays. It scores each item across five dimensions:
1. **First/Last items** -- Context for pagination and recency
2. **Error items** -- 100% preservation of error states (never dropped)
3. **Anomalies** -- Statistical outliers (> 2 standard deviations from the mean)
4. **Relevant items** -- Matches to the user's query via BM25/embeddings
5. **Change points** -- Significant transitions in data
The result: a 1,000-item array becomes ~50 items with all the information the LLM actually needs.
## What Gets Preserved
| Category | Preserved | Why |
|---|---|---|
| Errors | 100% | Critical for debugging |
| First N | 100% | Context and pagination |
| Last N | 100% | Recency |
| Anomalies | All | Unusual values matter |
| Relevant | Top K | Match user's query |
| Others | Sampled | Statistical representation |
## Quick Start
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from "headroom-ai";
// SmartCrusher fires automatically for JSON tool outputs
const messages = [
{ role: "system" as const, content: "You are a helpful assistant." },
{ role: "user" as const, content: "Find errors in the last 24 hours" },
{
role: "tool" as const,
content: JSON.stringify({ results: new Array(1000).fill({ status: "ok" }) }),
tool_call_id: "call_1",
},
];
const result = await compress(messages);
console.log(`Tokens saved: ${result.tokensSaved}`);
// SmartCrusher keeps errors, anomalies, and relevant items
```
</Tab>
<Tab value="Python">
```python
from headroom import SmartCrusher
crusher = SmartCrusher()
# Before: 1000 search results (45,000 tokens)
tool_output = {"results": ["...1000 items..."]}
# After: ~50 important items (4,500 tokens) -- 90% reduction
compressed = crusher.crush(tool_output, query="user's question")
```
</Tab>
</Tabs>
## Configuration
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from "headroom-ai";
// Configure via the Headroom proxy or HeadroomClient
const result = await compress(messages, {
model: "gpt-4o",
tokenBudget: 10000, // SmartCrusher will reduce JSON to fit
});
console.log(`Transforms: ${result.transformsApplied}`);
// ["smart_crusher", "cache_aligner"]
```
</Tab>
<Tab value="Python">
```python
from headroom import SmartCrusher, SmartCrusherConfig
config = SmartCrusherConfig(
min_tokens_to_crush=200, # Only compress if > 200 tokens
max_items_after_crush=15, # Keep at most 15 items
first_fraction=0.3, # Keep first 30% of items
last_fraction=0.15, # Keep last 15% of items
variance_threshold=2.0, # Statistical variance threshold
preserve_change_points=True, # Keep significant transitions
)
crusher = SmartCrusher(config)
compressed = crusher.crush(tool_output, query="find payment failures")
```
</Tab>
</Tabs>
## Configuration Options
| Option | Default | Description |
|---|---|---|
| `min_tokens_to_crush` | `200` | Only compress arrays with more than this many tokens |
| `min_items_to_analyze` | `5` | Minimum items before analyzing for compression |
| `max_items_after_crush` | `15` | Maximum items to keep after compression |
| `variance_threshold` | `2.0` | Statistical variance threshold for analysis |
| `uniqueness_threshold` | `0.1` | Uniqueness threshold for deduplication |
| `similarity_threshold` | `0.8` | Similarity threshold for grouping |
| `preserve_change_points` | `True` | Preserve significant transitions in data |
| `first_fraction` | `0.3` | Fraction of items always kept from the start |
| `last_fraction` | `0.15` | Fraction of items always kept from the end |
| `dedup_identical_items` | `True` | Deduplicate identical items |
| `use_feedback_hints` | `True` | Use TOIN feedback hints for scoring |
## Example: Before and After
Consider a tool that returns 1,000 search results:
```python
# Before compression: 45,000 tokens
{
"results": [
{"id": 1, "status": "ok", "message": "Success", "timestamp": "..."},
{"id": 2, "status": "ok", "message": "Success", "timestamp": "..."},
# ... 995 more "ok" results ...
{"id": 998, "status": "error", "message": "Connection timeout", "timestamp": "..."},
{"id": 999, "status": "ok", "message": "Success", "timestamp": "..."},
{"id": 1000, "status": "ok", "message": "Success", "timestamp": "..."},
]
}
# After SmartCrusher: 4,500 tokens (90% reduction)
# Kept: first 3, last 2, the error at id=998, statistical sample
```
The LLM sees the structure, the error, and a representative sample -- everything it needs to answer "find errors in the last 24 hours" without wading through 1,000 identical success responses.
<Callout type="info" title="Automatic routing">
You don't need to call SmartCrusher directly. The ContentRouter detects JSON arrays and routes them to SmartCrusher automatically. Direct usage is available when you want fine-grained control over the configuration.
</Callout>
+137
View File
@@ -0,0 +1,137 @@
---
title: Strands
description: Context compression for Strands Agents via model wrapping and hook-based tool output compression.
---
Headroom integrates with [Strands Agents](https://github.com/strands-agents/sdk-python) through two patterns: wrap the model for full conversation compression, or hook into tool calls for targeted tool output compression.
## Installation
```bash
pip install headroom-ai strands-agents
```
## Quick start
```python
from strands import Agent
from strands.models.bedrock import BedrockModel
from headroom.integrations.strands import HeadroomStrandsModel
model = BedrockModel(model_id="us.anthropic.claude-sonnet-4-20250514-v1:0")
optimized = HeadroomStrandsModel(wrapped_model=model)
agent = Agent(model=optimized)
response = agent("Investigate the production incident")
print(f"Tokens saved: {optimized.total_tokens_saved}")
```
## Model wrapping
Wraps the Strands `Model` interface. Every call to `stream()` compresses messages before they reach the provider:
```python
from headroom import HeadroomConfig
from headroom.integrations.strands import HeadroomStrandsModel
optimized = HeadroomStrandsModel(
wrapped_model=model,
config=HeadroomConfig(),
)
agent = Agent(model=optimized)
response = agent("Analyze these logs")
```
## Hook provider (tool output compression)
Compresses tool call results via Strands' hook system. Uses SmartCrusher on JSON arrays returned by tools:
```python
from strands import Agent
from strands.models.bedrock import BedrockModel
from headroom.integrations.strands import HeadroomHookProvider
model = BedrockModel(model_id="us.anthropic.claude-sonnet-4-20250514-v1:0")
hooks = HeadroomHookProvider(
compress_tool_outputs=True,
min_tokens_to_compress=200,
preserve_errors=True,
)
agent = Agent(model=model, hooks=[hooks])
response = agent("Search the database for recent failures")
print(f"Tokens saved by hooks: {hooks.total_tokens_saved}")
```
The hook preserves error items, anomalous values (statistical outliers), items matching the query context, and boundary items (first/last).
## Both together
Model wrapping compresses conversation history. Hooks compress individual tool results. Use both for maximum savings:
```python
from headroom.integrations.strands import HeadroomStrandsModel, HeadroomHookProvider
optimized = HeadroomStrandsModel(wrapped_model=model)
hooks = HeadroomHookProvider(compress_tool_outputs=True)
agent = Agent(model=optimized, hooks=[hooks])
```
## How it works
```
Agent decides to call tool
|
v
Tool executes, returns result
|
v
HeadroomHookProvider (optional)
compresses tool result JSON
|
v
Agent builds next API request
|
v
HeadroomStrandsModel.stream()
compresses full message list
|
v
Provider API (Bedrock, etc.)
```
The model wrapper uses the full Headroom pipeline (CacheAligner, ContentRouter). The hook provider uses SmartCrusher directly for fast JSON compression.
## Structured output
```python
from pydantic import BaseModel
class Analysis(BaseModel):
severity: str
root_cause: str
recommendation: str
result = optimized.structured_output(Analysis, messages)
```
## Metrics
```python
for m in optimized.metrics_history:
print(f" {m.tokens_before} -> {m.tokens_after} ({m.tokens_saved} saved)")
print(f"Total saved: {optimized.total_tokens_saved}")
```
## Supported providers
| Strands Model | Provider Detected |
|--------------|-------------------|
| `BedrockModel` | Anthropic (via Bedrock) |
| `OllamaModel` | OpenAI-compatible |
| Custom `Model` | Falls back to estimation |
+192
View File
@@ -0,0 +1,192 @@
---
title: Text & Log Compression
description: Specialized compressors for search results, build logs, diffs, and general text. Each preserves what matters for its content type.
---
Headroom provides specialized compressors for text-based content that isn't JSON or source code. Each one understands the structure of its content type and preserves what the LLM needs while dropping the noise.
| Compressor | Input Type | What It Preserves | Typical Savings |
|---|---|---|---|
| `SearchCompressor` | grep/ripgrep output | Relevant matches, file diversity | 80-95% |
| `LogCompressor` | Build/test logs | Errors, stack traces, summaries | 85-95% |
| `DiffCompressor` | Unified diffs | Changed lines, context | 60-80% |
| `TextCompressor` | General text | Relevant paragraphs, anchors | 60-80% |
| `KompressCompressor` | General text fallback | Learned token scoring via ONNX | 30-50% |
## SearchCompressor
Compresses search results (grep, ripgrep, ag) while keeping the matches that matter.
```python
from headroom.transforms import SearchCompressor
search_results = """
src/utils.py:42:def process_data(items):
src/utils.py:43: \"\"\"Process items.\"\"\"
src/models.py:15:class DataProcessor:
src/models.py:89: def process(self, items):
... hundreds more matches ...
"""
compressor = SearchCompressor()
result = compressor.compress(search_results, context="find process")
print(f"Compressed {result.original_match_count} matches to {result.compressed_match_count}")
print(result.compressed)
```
**What gets preserved:**
- Exact query matches (lines containing the search term)
- High-relevance matches (scored by BM25 similarity)
- File diversity (results from different files are kept)
- First/last matches (context from start and end)
### Configuration
```python
from headroom.transforms import SearchCompressor, SearchCompressorConfig
config = SearchCompressorConfig(
max_results=50, # Keep up to 50 matches
preserve_file_diversity=True, # Ensure different files represented
relevance_threshold=0.3, # Minimum relevance score to keep
)
compressor = SearchCompressor(config)
```
## LogCompressor
Compresses build and test output while preserving errors, warnings, and summaries.
```python
from headroom.transforms import LogCompressor
build_output = """
===== test session starts =====
collected 500 items
tests/test_foo.py::test_1 PASSED
... hundreds of passed tests ...
tests/test_bar.py::test_fail FAILED
AssertionError: expected 5, got 3
===== 1 failed, 499 passed =====
"""
compressor = LogCompressor()
result = compressor.compress(build_output)
print(result.compressed)
print(f"Compression ratio: {result.compression_ratio:.1%}")
```
**What gets preserved:**
- Errors and failures (any line with ERROR, FAILED, Exception)
- Warnings
- Full stack traces for debugging
- Test/build summary lines
- Section headers (structural markers like `=====`)
**What gets dropped:**
- Hundreds of `PASSED` lines
- Verbose success output
- Repeated patterns
## DiffCompressor
Compresses unified diffs while keeping the actual changes and enough context to understand them.
```python
from headroom.transforms import DiffCompressor
diff_output = """
diff --git a/src/main.py b/src/main.py
--- a/src/main.py
+++ b/src/main.py
@@ -42,7 +42,7 @@
def process(items):
- return [x for x in items]
+ return [x.strip() for x in items if x]
"""
compressor = DiffCompressor()
result = compressor.compress(diff_output)
```
## TextCompressor
General-purpose text compression with anchor preservation. Best for documentation, README files, and prose content.
```python
from headroom.transforms import TextCompressor
long_text = """
... thousands of lines of documentation ...
"""
compressor = TextCompressor()
result = compressor.compress(long_text, context="authentication")
print(result.compressed)
```
**What gets preserved:**
- Paragraphs relevant to the context query
- Headers and section markers
- Document structure and organization
## Kompress
```python
from headroom.transforms.kompress_compressor import KompressCompressor
compressor = KompressCompressor()
result = compressor.compress(long_output)
print(f"Before: {result.original_tokens} tokens")
print(f"After: {result.compressed_tokens} tokens")
print(f"Saved: {result.savings_percentage:.1f}%")
```
<Callout type="info" title="LLMLingua was removed">
The old LLMLingua transform and helper functions are no longer exported. Use Kompress and ContentRouter for text compression.
</Callout>
## Content Type Detection
If you're building your own routing logic, you can use the content type detector directly:
```python
from headroom.transforms import detect_content_type, ContentType
content = "src/main.py:42:def process():"
detection = detect_content_type(content)
if detection.content_type == ContentType.SEARCH_RESULTS:
result = SearchCompressor().compress(content, context="process")
elif detection.content_type == ContentType.BUILD_OUTPUT:
result = LogCompressor().compress(content)
elif detection.content_type == ContentType.PLAIN_TEXT:
result = TextCompressor().compress(content, context="process")
```
## When Each Compressor Is Used
The ContentRouter selects the right compressor automatically. Here's when each fires:
| Content Pattern | Compressor | Detection Signal |
|---|---|---|
| `file:line:content` lines | SearchCompressor | grep/ripgrep output format |
| pytest, npm, cargo markers | LogCompressor | Build tool output patterns |
| `---/+++` and `@@` markers | DiffCompressor | Unified diff format |
| Prose, documentation | TextCompressor | Fallback for non-structured text |
| Long plain text | KompressCompressor | ContentRouter fallback |
## Performance
| Compressor | Typical Input | Output | Speed |
|---|---|---|---|
| SearchCompressor | 1,000 matches | 30-50 matches | ~2ms |
| LogCompressor | 5,000 lines | 100-200 lines | ~3ms |
| DiffCompressor | Large diff | Changed hunks only | ~2ms |
| TextCompressor | 10,000 chars | 2,000 chars | ~2ms |
| KompressCompressor | Plain text | 50-70% of original | model-dependent |
+467
View File
@@ -0,0 +1,467 @@
---
title: Troubleshooting
description: Solutions for common Headroom issues including proxy startup, connection errors, no token savings, high latency, and installation problems.
---
Solutions for common Headroom issues.
## Proxy Server Issues
### Proxy will not start
**Symptom**: `headroom proxy` fails or hangs.
```bash
# Check if port is already in use
lsof -i :8787
# Try a different port
headroom proxy --port 8788
# Check for missing dependencies
pip install "headroom-ai[proxy]"
# Run with debug logging
headroom proxy --log-file ~/.headroom/logs/proxy.jsonl --log-messages
```
### Connection refused when calling proxy
**Symptom**: `curl: (7) Failed to connect to localhost port 8787`
```bash
# Verify proxy is running
curl http://localhost:8787/health
# Check if proxy started on a different port
ps aux | grep headroom
```
### Proxy returns errors for some requests
**Symptom**: Some requests work, others fail with 502/503.
```bash
# Check proxy logs for the actual error
headroom proxy --log-file ~/.headroom/logs/proxy.jsonl --log-messages
# Verify API key is set
echo $OPENAI_API_KEY # or ANTHROPIC_API_KEY
# Test the underlying API directly
curl https://api.openai.com/v1/models \
-H "Authorization: Bearer $OPENAI_API_KEY"
```
### Windows: ML content detection hangs or silently falls back
**Symptom**: On Windows 11 24H2+, every proxied request stalls (historically
`Optimization failed: TimeoutError`), or the first detection in each process
burns ~5 seconds and compression quality drops because detection runs on the
non-ML fallback tiers. The proxy log may show
`magika ONNX session init timed out`.
**Cause**: The Rust core loads ONNX Runtime dynamically. Without
`ORT_DYLIB_PATH`, the bare Windows DLL search resolves `onnxruntime.dll` to
`C:\Windows\System32\onnxruntime.dll` — the Windows ML OS component (1.17.x),
which deadlocks ort session initialization instead of returning an error.
**Fix**: Headroom pins `ORT_DYLIB_PATH` automatically at import time to the
DLL inside the `onnxruntime` pip package (included in `headroom-ai[proxy]`).
Confirm in the startup log:
```
Pinned ORT_DYLIB_PATH to bundled ONNX Runtime: ...\onnxruntime\capi\onnxruntime.dll
```
If the pin is skipped (library install without `onnxruntime`), either install
it or point the variable at any modern ONNX Runtime yourself:
```powershell
pip install onnxruntime
# or
$env:ORT_DYLIB_PATH = "C:\path\to\onnxruntime.dll"
```
`HEADROOM_MAGIKA_INIT_TIMEOUT_SECS` (default `5`) bounds the init as a safety
net; on timeout detection degrades to non-ML tiers for the process lifetime.
## No Token Savings
**Symptom**: `stats['session']['tokens_saved_total']` is 0.
**Diagnosis**:
```python
stats = client.get_stats()
print(f"Mode: {stats['config']['mode']}") # Should be "optimize"
print(f"SmartCrusher: {stats['transforms']['smart_crusher_enabled']}")
```
**Common causes**:
- Mode is `audit` (observation only, no modifications)
- Messages do not contain tool outputs
- Tool outputs are below the 200-token threshold
- Data is not compressible (high uniqueness, code, grep results)
**Solutions**:
<Tabs groupId="lang" items={['TypeScript', 'Python']}>
<Tab value="TypeScript">
```ts twoslash
import { compress } from 'headroom-ai';
// Ensure the proxy is running in optimize mode
// (default, unless --no-optimize was passed)
const result = await compress(messages, { model: 'gpt-4o' });
console.log(`Saved: ${result.tokensSaved} tokens`);
console.log(`Compressed: ${result.compressed}`);
```
</Tab>
<Tab value="Python">
```python
# 1. Ensure mode is "optimize"
client = HeadroomClient(
original_client=OpenAI(),
provider=OpenAIProvider(),
default_mode="optimize", # NOT "audit"
)
# 2. Or override per-request
response = client.chat.completions.create(
model="gpt-4o",
messages=messages,
headroom_mode="optimize",
)
# 3. Lower the compression threshold
config = HeadroomConfig()
config.smart_crusher.min_tokens_to_crush = 100 # Default is 200
```
</Tab>
</Tabs>
## Claude Code context window is larger through the proxy
**Symptom**: After pointing Claude Code at Headroom (`ANTHROPIC_BASE_URL`), `/context all`
shows **more** tokens used than a direct session — the "System tools" and "MCP tools"
lines grow by tens of thousands of tokens, before you send any message.
**Cause**: Claude Code normally defers most tool schemas behind its server-side
**Tool Search Tool** (it sends only tool *names* and loads full schemas on demand).
It enables this only when it believes it is talking directly to `api.anthropic.com`.
The moment `ANTHROPIC_BASE_URL` is a custom host, Claude Code can't assume the endpoint
supports the feature, so it falls back to **eagerly** materializing every tool schema
into the local context window. This is a Claude Code client-side decision made before
the request reaches the proxy — no proxy header can reverse it.
**Solution**: set `ENABLE_TOOL_SEARCH` so Claude Code keeps deferring tools through the
proxy. The proxy forwards the `tool_reference` blocks correctly, so deferral works
end-to-end (both streaming and non-streaming, subscription and API-key auth).
```bash
# Easiest: `headroom wrap claude` sets ENABLE_TOOL_SEARCH=true automatically.
headroom wrap claude
# Choose the mode (true = always defer, the default; auto / auto:N = defer only
# when tool definitions exceed N% of the budget; false = off):
headroom wrap claude --tool-search auto
# Running `claude` manually instead of via wrap? Set it yourself:
ENABLE_TOOL_SEARCH=true ANTHROPIC_BASE_URL=http://localhost:8787 claude
```
**Verify (before / after)** with `/context all` in a fresh session, no messages sent:
| Section | Eager (no `ENABLE_TOOL_SEARCH`) | Deferred (`ENABLE_TOOL_SEARCH=true`) |
| --- | --- | --- |
| System tools | fully materialized | deferred subset |
| MCP tools | every tool shows a token cost | `(loaded on-demand)`, 0 tokens |
When deferral is off, the proxy log also prints a one-time hint naming the fix.
See [issue #746](https://github.com/chopratejas/headroom/issues/746) for the full analysis.
## Remote Control unavailable through custom ANTHROPIC_BASE_URL
**Symptom**: When Claude Code runs with `ANTHROPIC_BASE_URL` set to a custom host (for example, Headroom), the Remote Control menu is absent.
**Cause**: This is a Claude-side gate. Headroom only receives normal API traffic and can still compress it, but Claude evaluates Remote Control availability before proxy traffic reaches the server.
**Fix**: Use Headroom for normal proxied API sessions, and launch Claude directly (without `ANTHROPIC_BASE_URL`) when you need Claude Remote Control.
`ENABLE_TOOL_SEARCH` is unaffected and can stay enabled for context-window savings while routing through Headroom.
## Compression Too Aggressive
**Symptom**: LLM responses are missing information that was in tool outputs.
```python
# 1. Keep more items
config = HeadroomConfig()
config.smart_crusher.max_items_after_crush = 50 # Default: 15
# 2. Skip compression for specific tools
response = client.chat.completions.create(
model="gpt-4o",
messages=messages,
headroom_tool_profiles={
"important_tool": {"skip_compression": True},
},
)
# 3. Disable SmartCrusher entirely
config.smart_crusher.enabled = False
```
## High Latency
**Symptom**: Requests take longer than expected.
**Diagnosis**:
```python
import time
import logging
logging.basicConfig(level=logging.DEBUG)
start = time.time()
response = client.chat.completions.create(...)
print(f"Total time: {time.time() - start:.2f}s")
```
**Solutions**:
```python
# 1. Use BM25 instead of embeddings (faster)
config = HeadroomConfig()
config.smart_crusher.relevance.tier = "bm25"
# 2. Increase threshold to skip small payloads
config.smart_crusher.min_tokens_to_crush = 500
# 3. Disable transforms you don't need
config.cache_aligner.enabled = False
config.rolling_window.enabled = False
```
## Installation Issues
### pipx installs an older Headroom version
**Symptom**: PyPI shows a newer `headroom-ai` release, but `pipx install` or
`pipx upgrade` keeps an older version. A pinned install can also fail with
`No matching distribution found`.
**Cause**: `pipx` resolves packages inside its app virtual environment. If that
environment uses a Python version that Headroom does not publish wheels for yet,
pip may skip newer releases and choose the newest compatible build it can use.
Check the interpreter:
```bash
pipx list
```
Install with a supported Python explicitly:
```bash
pipx install --python python3.13 "headroom-ai[all]"
```
For a pinned release:
```bash
pipx install --python python3.13 "headroom-ai[all]==0.21.4"
```
If you already have Headroom installed under `pipx`, uninstall it first or
reinstall it with the supported interpreter.
### pip install fails with C++ compilation error
**Symptom**: `RuntimeError: Unsupported compiler -- at least C++11 support is needed!`
```bash
# Linux / Debian-based (including Docker)
apt-get install -y build-essential && pip install headroom-ai
# macOS (Xcode command line tools)
xcode-select --install && pip install headroom-ai
```
For Docker, install and remove build tools in one layer:
```dockerfile
FROM python:3.11-slim
RUN apt-get update && apt-get install -y --no-install-recommends build-essential \
&& pip install "headroom-ai[proxy]" \
&& apt-get purge -y build-essential && apt-get autoremove -y \
&& rm -rf /var/lib/apt/lists/*
```
### ModuleNotFoundError: No module named 'headroom'
```bash
# Check it is installed in the right environment
pip show headroom-ai
# If using virtual environment, ensure it is activated
source venv/bin/activate
# Reinstall
pip install --upgrade headroom-ai
```
### Missing optional dependency
```bash
# For proxy server
pip install "headroom-ai[proxy]"
# For embedding-based relevance scoring
pip install "headroom-ai[relevance]"
# For code compression (tree-sitter)
pip install "headroom-ai[code]"
# For everything
pip install "headroom-ai[all]"
```
## Provider-Specific Issues
### OpenAI: Invalid API key
```python
import os
from openai import OpenAI
api_key = os.environ.get("OPENAI_API_KEY")
if not api_key:
raise ValueError("OPENAI_API_KEY not set")
client = HeadroomClient(
original_client=OpenAI(api_key=api_key),
provider=OpenAIProvider(),
)
```
### Anthropic: Authentication error
```python
import os
from anthropic import Anthropic
api_key = os.environ.get("ANTHROPIC_API_KEY")
client = HeadroomClient(
original_client=Anthropic(api_key=api_key),
provider=AnthropicProvider(),
)
```
### Unknown model warnings
```python
# For custom/fine-tuned models, specify context limit
client = HeadroomClient(
original_client=OpenAI(),
provider=OpenAIProvider(),
model_context_limits={
"ft:gpt-4o-2024-08-06:my-org::abc123": 128000,
"my-custom-model": 32000,
},
)
```
## ValidationError on Setup
```python
result = client.validate_setup()
print(result)
# Common issues:
# {"provider": {"ok": False, "error": "No API key"}}
# -> Set OPENAI_API_KEY or pass api_key to OpenAI()
#
# {"storage": {"ok": False, "error": "unable to open database"}}
# -> Check path permissions, use :memory: for testing
#
# {"config": {"ok": False, "error": "Invalid mode"}}
# -> Use "audit" or "optimize" only
```
For testing, use in-memory storage:
```python
client = HeadroomClient(
original_client=OpenAI(),
provider=OpenAIProvider(),
store_url="sqlite:///:memory:",
)
```
## Debugging Techniques
### Enable Full Logging
```python
import logging
# See everything
logging.basicConfig(
level=logging.DEBUG,
format="%(asctime)s %(name)s %(levelname)s %(message)s",
)
# Or just Headroom logs
logging.getLogger("headroom").setLevel(logging.DEBUG)
```
### Use Simulation to Inspect Transforms
```python
plan = client.chat.completions.simulate(
model="gpt-4o",
messages=messages,
)
print(f"Tokens: {plan.tokens_before} -> {plan.tokens_after}")
print(f"Transforms: {plan.transforms_applied}")
print(f"Waste signals: {plan.waste_signals}")
import json
print(json.dumps(plan.messages_optimized, indent=2))
```
### Test Transforms Directly
```python
from headroom import SmartCrusher, Tokenizer
from headroom.config import SmartCrusherConfig
import json
config = SmartCrusherConfig()
crusher = SmartCrusher(config)
tokenizer = Tokenizer()
messages = [
{
"role": "tool",
"content": json.dumps({"items": list(range(100))}),
"tool_call_id": "1",
}
]
result = crusher.apply(messages, tokenizer)
print(f"Tokens: {result.tokens_before} -> {result.tokens_after}")
```
## Getting Help
1. Enable debug logging and check the output
2. Use `simulate()` to see what transforms would apply
3. Run `validate_setup()` for configuration issues
4. File an issue at [github.com/chopratejas/headroom](https://github.com/chopratejas/headroom/issues) with your Headroom version, Python version, provider, debug log output, and minimal reproduction code
+139
View File
@@ -0,0 +1,139 @@
---
title: Vercel AI SDK
description: Compress LLM context with the Vercel AI SDK using middleware, withHeadroom(), or standalone compression.
---
Headroom integrates with the [Vercel AI SDK](https://sdk.vercel.ai) through three patterns: a one-liner wrapper, composable middleware, and standalone message compression.
## Installation
```bash
npm install headroom-ai ai @ai-sdk/openai
```
<Callout type="info" title="Proxy required">
The TypeScript SDK sends messages to a local Headroom proxy for compression. Start the proxy before using the SDK:
```bash
pip install "headroom-ai[proxy]"
headroom proxy
```
</Callout>
## withHeadroom() one-liner
The simplest integration. Wraps any Vercel AI SDK language model with automatic compression:
```ts twoslash
import { withHeadroom } from 'headroom-ai/vercel-ai';
import { openai } from '@ai-sdk/openai';
import { generateText } from 'ai';
const model = withHeadroom(openai('gpt-4o'));
const { text } = await generateText({
model,
messages: [
{ role: 'user', content: 'Summarize these results...' },
],
});
```
`withHeadroom()` calls `wrapLanguageModel` + `headroomMiddleware()` under the hood. It works with any provider (`@ai-sdk/openai`, `@ai-sdk/anthropic`, `@ai-sdk/google`, etc.).
## headroomMiddleware() for composition
Use the middleware directly when you need to compose it with other middleware:
```ts twoslash
// @noErrors
import { headroomMiddleware } from 'headroom-ai/vercel-ai';
import { wrapLanguageModel } from 'ai';
import { openai } from '@ai-sdk/openai';
const model = wrapLanguageModel({
model: openai('gpt-4o'),
middleware: headroomMiddleware(),
});
```
Pass options to control compression behavior:
```ts twoslash
import { headroomMiddleware } from 'headroom-ai/vercel-ai';
const middleware = headroomMiddleware({
model: 'gpt-4o',
baseUrl: 'http://localhost:8787',
});
```
## compressVercelMessages() standalone
Compress Vercel-format messages directly without wrapping a model. Useful for custom pipelines:
```ts twoslash
import { compressVercelMessages } from 'headroom-ai/vercel-ai';
const result = await compressVercelMessages(messages, {
model: 'gpt-4o',
});
console.log(`Saved ${result.tokensSaved} tokens`);
// result.messages is in Vercel format, ready for the AI SDK
```
## Streaming with streamText
Compression happens before the request. Streaming responses are unaffected:
```ts twoslash
import { withHeadroom } from 'headroom-ai/vercel-ai';
import { openai } from '@ai-sdk/openai';
import { streamText } from 'ai';
const model = withHeadroom(openai('gpt-4o'));
const result = streamText({
model,
messages: longConversation,
});
for await (const chunk of result.textStream) {
process.stdout.write(chunk);
}
```
## generateObject with compressed context
Works with structured output:
```ts twoslash
// @noErrors
import { withHeadroom } from 'headroom-ai/vercel-ai';
import { openai } from '@ai-sdk/openai';
import { generateText, Output } from 'ai';
import { z } from 'zod';
const model = withHeadroom(openai('gpt-4o'));
const { output } = await generateText({
model,
output: Output.object({
schema: z.object({
summary: z.string(),
severity: z.enum(['low', 'medium', 'high']),
}),
}),
messages: largeConversationHistory,
});
```
## How it works
1. Messages are converted from Vercel format to OpenAI format
2. Headroom compresses them via the proxy's `/v1/compress` endpoint
3. Compressed messages are converted back to Vercel format
4. The original model receives the smaller prompt
All other model behavior (tool calling, structured output, streaming) is unchanged.
+11
View File
@@ -0,0 +1,11 @@
import { createMDX } from 'fumadocs-mdx/next';
const withMDX = createMDX();
/** @type {import('next').NextConfig} */
const config = {
reactStrictMode: true,
serverExternalPackages: ['typescript', 'twoslash'],
};
export default withMDX(config);
+274
View File
@@ -0,0 +1,274 @@
# Observability — proxy metrics
The Headroom Rust proxy exposes Prometheus-format metrics on the
`/metrics` endpoint of every running proxy instance. The metric
catalogue below covers Phase D (Bedrock route instrumentation) and
Phase G PR-G3 (per-invocation RTK + proxy-wide observability).
All metric names + label keys are constants in
`crates/headroom-proxy/src/observability/metric_names.rs`, so any
rename catches one file in code review.
## Metric catalogue
### Bedrock route (Phase D PR-D3)
| Name | Type | Labels | Purpose |
|------|------|--------|---------|
| `bedrock_invoke_count_total` | Counter | `model`, `region`, `auth_mode` | One increment per Bedrock `/invoke` or `/converse` request. |
| `bedrock_invoke_latency_seconds` | Histogram | `model`, `region` | Latency from proxy entry to upstream completion. Buckets target 50ms60s. |
| `bedrock_eventstream_message_count_total` | Counter | `model`, `region`, `event_type` | One increment per parsed binary EventStream message. |
### Proxy-wide (Phase G PR-G3)
#### Cache + compression
| Name | Type | Labels | Purpose |
|------|------|--------|---------|
| `proxy_cache_hit_rate_per_session` | Histogram | `provider` | Per-session cache hit rate. **Phase H canary gate.** |
| `proxy_compression_ratio_by_strategy` | Histogram | `strategy`, `content_type` | `compressed_tokens / original_tokens` per shrunk block. |
| `proxy_compression_rejected_by_token_check_total` | Counter | `strategy` | Compressor ran but failed the shrink check. |
#### Cache-safety alarm
| Name | Type | Labels | Purpose |
|------|------|--------|---------|
| `proxy_passthrough_bytes_modified_total` | Counter | `path` | Bytes mutated on a passthrough path. **Must stay 0 outside the compression hot path** — any non-zero rate fires the cache-safety alarm. |
The alarm metric is wired in `crates/headroom-proxy/src/proxy.rs`:
when the dispatcher returns `Outcome::NoCompression` or
`Outcome::Passthrough`, the post-dispatcher byte length is compared
to the original buffered length and any delta increments the
counter (by the byte delta) under the request's path label. The
PR-E4 prompt_cache_key injector runs AFTER the alarm check, so its
intentional byte mutations do not trip the alarm.
#### Upstream rate limits
| Name | Type | Labels | Purpose |
|------|------|--------|---------|
| `proxy_rate_limit_remaining_requests` | Gauge | `provider` | Last-seen remaining requests in the current window. |
| `proxy_rate_limit_remaining_tokens` | Gauge | `provider` | Last-seen remaining tokens in the current window. |
| `proxy_rate_limit_remaining_input_tokens` | Gauge | `provider` | Anthropic-only input-token bucket. |
| `proxy_rate_limit_remaining_output_tokens` | Gauge | `provider` | Anthropic-only output-token bucket. |
#### OpenAI Responses telemetry
| Name | Type | Labels | Purpose |
|------|------|--------|---------|
| `proxy_service_tier_count_total` | Counter | `tier` | Service-tier distribution observed at the proxy. |
| `proxy_response_status_count_total` | Counter | `status` | Terminal status distribution (`completed`, `incomplete`, `failed`, `cancelled`, `in_progress`). |
#### Wrap CLI / RTK (Python-side)
| Name | Type | Labels | Purpose |
|------|------|--------|---------|
| `wrap_rtk_invocations_total` | Counter | `tool` | RTK invocations observed via the wrap-CLI tail. Surfaced via the Python proxy's `/metrics` exporter; the wrap CLI bumps `headroom.cli.wrap_rtk_metrics.record_rtk_invocation(...)`. |
> **C4 remediation:** This counter is Python-side because RTK is
> wrapped by `headroom wrap` (Python CLI) and the wrap-side tail
> is the natural emit site. The Rust proxy previously held a dead
> counter for this metric; that has been removed.
#### Image log redaction (Python-side)
| Name | Type | Labels | Purpose |
|------|------|--------|---------|
| `proxy_image_generation_call_log_redacted_total` | Counter | _none_ | Base64-encoded image payloads redacted from request logs. Driven from `headroom.proxy.request_logger.redactions_total()`. |
> **C3 remediation:** Image redaction is purely a Python-proxy
> operation (the request logger walks JSON and replaces over-
> threshold image payloads with placeholders). The counter lives
> Python-side so we have one source of truth instead of two. The
> Rust proxy previously held a dead counter for this metric; that
> has been removed.
## How to query
The proxy renders Prometheus text-format on `GET /metrics`:
```bash
curl -s http://127.0.0.1:8787/metrics
```
### Phase H canary gate
The canary script that decides "ship Rust, retire Python" uses
**all four** of these queries against `proxy_cache_hit_rate_per_session`
to confirm parity vs the Python baseline. A single percentile is
not enough — a regression that only shows up at the tail (a small
class of long sessions losing cache hits) would slip through a
median-only check.
```promql
# p50, p95, p99 of cache hit rate over the last 5 minutes, per provider.
histogram_quantile(0.50, sum by (provider, le) (rate(proxy_cache_hit_rate_per_session_bucket{provider!="__init__"}[5m])))
histogram_quantile(0.95, sum by (provider, le) (rate(proxy_cache_hit_rate_per_session_bucket{provider!="__init__"}[5m])))
histogram_quantile(0.99, sum by (provider, le) (rate(proxy_cache_hit_rate_per_session_bucket{provider!="__init__"}[5m])))
# Mean cache hit rate over the last 5 minutes, per provider. The
# `sum / count` form is the cleanest "average without a quantile"
# query and is what the Python baseline reports.
sum by (provider) (rate(proxy_cache_hit_rate_per_session_sum{provider!="__init__"}[5m]))
/
sum by (provider) (rate(proxy_cache_hit_rate_per_session_count{provider!="__init__"}[5m]))
```
The canary fails if ANY of `p50`, `p95`, `p99`, or `mean` regresses
below the Python baseline for any provider over the canary window.
### Other common queries
```promql
# Cache-safety alarm. Should always be 0 (post-`__init__` row).
sum(rate(proxy_passthrough_bytes_modified_total{path!="__init__"}[5m]))
# Per-strategy compression value at p50 (post-H1 fix: each strategy
# reports its own before/after; pre-fix this was the same aggregate
# ratio repeated per strategy).
histogram_quantile(0.50, sum by (strategy, le) (rate(proxy_compression_ratio_by_strategy_bucket{strategy!="__init__"}[1h])))
# Per-strategy compression value at p95 and p99 (catch outlier
# strategies that fail to shrink at the tail).
histogram_quantile(0.95, sum by (strategy, le) (rate(proxy_compression_ratio_by_strategy_bucket{strategy!="__init__"}[1h])))
histogram_quantile(0.99, sum by (strategy, le) (rate(proxy_compression_ratio_by_strategy_bucket{strategy!="__init__"}[1h])))
# Strategies that ran but failed the token-check (compressor ran
# but its output was not strictly smaller, so the original was
# kept). High rate here means the compressor needs tuning.
sum by (strategy) (rate(proxy_compression_rejected_by_token_check_total{strategy!="__init__"}[1h]))
# Upstream rate-limit headroom (smaller = closer to throttle).
proxy_rate_limit_remaining_tokens{provider="anthropic"}
# RTK invocation rate (Python-side).
sum by (tool) (rate(wrap_rtk_invocations_total{tool!="__init__"}[5m]))
# Image-redaction rate (Python-side).
rate(proxy_image_generation_call_log_redacted_total[5m])
```
All queries above include a `{... != "__init__"}` filter so the
sentinel zero-rows the boot-touch contract emits do not skew the
result. See "Wiring → H3 force-zero" below.
## Wiring
Every metric registration is `OnceLock`-backed and lazy: the first
call to a `*_counter()` / `*_gauge()` / `*_histogram()` helper
registers the family with the shared registry. `handle_metrics`
force-touches every Phase G PR-G3 family before scraping.
### H3 force-zero
The `prometheus` crate v0.13 skips empty MetricVecs from `gather()`
entirely — neither HELP/TYPE lines nor rows appear until the
family has been incremented at least once with a label tuple.
Operators expect to see the catalogue from boot, so
`handle_metrics` increments each counter / gauge MetricVec by 0
under a sentinel `__init__` label tuple before the first scrape.
HELP/TYPE then surface from boot and dashboards/alarms see a
predictable scrape shape.
Counters with the `__init__` label increment by 0, so the
alarm-able "must stay 0" semantic of
`proxy_passthrough_bytes_modified_total` is preserved (the family
becomes visible, the rate stays 0). PromQL queries should filter
`{... != "__init__"}` so the sentinel rows are excluded from
aggregations (the catalogue above does this).
Histograms are NOT force-zeroed: a synthetic `observe(0.0)` would
contribute a real sample to the per-label distribution and pollute
percentile readings. The two histogram families
(`proxy_cache_hit_rate_per_session` and
`proxy_compression_ratio_by_strategy`) only surface in the scrape
after the first real session, by design.
### H4 prometheus crate version pin
The H3 contract above relies on the `prometheus` crate's v0.13
`gather()` semantics — empty MetricVec families are omitted from
the scrape. **This is implementation-defined behaviour.** If
`crates/headroom-proxy/Cargo.toml` ever bumps the `prometheus`
dependency, retest the alarm contract:
1. Start a fresh proxy.
2. `curl /metrics` and confirm every counter / gauge family has
HELP/TYPE + an `__init__` row.
3. Confirm histograms (`*_cache_hit_rate_per_session`,
`*_compression_ratio_by_strategy`) DO NOT appear (no
`observe()` calls yet).
4. Drive one cache-hit session, scrape again, confirm histograms
now appear.
5. Confirm `passthrough_bytes_modified_total` stays at 0 across
passthrough requests.
The crate version is pinned exactly (`= "0.13.4"`, no caret) in
`Cargo.toml` precisely so a silent semver bump cannot break the
contract without a code-review trigger.
### C2 alarm wiring
`proxy_passthrough_bytes_modified_total` fires from `proxy.rs` when
a dispatcher arm that promised byte-equal passthrough
(`Outcome::NoCompression` or `Outcome::Passthrough`) produces a
final body of a different byte length. The check runs BEFORE the
PR-E4 prompt_cache_key injector so the injector's intentional byte
mutations do not trip the alarm.
### H1 per-strategy ratio wiring
`proxy_compression_ratio_by_strategy` samples one observation per
strategy using the strategy's OWN before/after token counts
(plumbed through `Outcome::Compressed.per_strategy_tokens` from
the manifest in `live_zone_anthropic` / `live_zone_openai` /
`live_zone_responses`). Pre-H1 the same aggregate ratio was
emitted per strategy when multiple strategies ran on one body,
making Phase H per-strategy dashboards read garbage.
### H2 aborted-stream gate
The `proxy_cache_hit_rate_per_session` histogram observes ONLY
when the SSE stream completed:
* Anthropic: `state.status == StreamStatus::MessageStop` after the
channel closes.
* OpenAI Chat: `state.usage.is_some()` (the final usage chunk only
arrives at stream completion).
* OpenAI Responses: `state.terminal_status().is_some()`.
A client disconnect mid-stream closes the channel without setting
the terminal flag — under H2 we log + skip rather than observe a
garbage half-stream sample.
## Cardinality discipline
Every label vocabulary is bounded by code, not customer input:
- `model` / `region`: read from path params + `Config::bedrock_region`.
- `auth_mode`: 3-variant enum (`payg`, `oauth`, `subscription`).
- `provider`: 3 values (`anthropic`, `openai_chat`, `openai_responses`).
- `strategy`: `&'static str` from the compressor's `BlockAction::Compressed`.
- `content_type`: `&'static str` from `headroom_core::transforms::ContentType`.
- `tier`: validated through
`crate::observability::metric_names::service_tier::validate(raw: &str)`.
Returns one of `{auto, default, flex, on_demand, priority, scale}`
or the sentinel `"other"` for anything else. **The raw inbound
value is never used as a label.** A malicious client posting
`{"service_tier":"<random>"}` per request gets bucketed to
`"other"` and a `tracing::warn!` is emitted so wire-format drift
surfaces loudly in logs.
- `status`: 5-variant enum.
- `tool` (Python-side `wrap_rtk_invocations_total`): bounded by the
set of tools the wrap CLI rewrites, captured by
`headroom.cli.wrap_rtk_metrics`.
There is no code path where a malicious client can drive label
cardinality unbounded.
## See also
- `docs/rtk-architecture.md` — why RTK lives wrap-side, not proxy-side.
- `crates/headroom-proxy/src/observability/` — implementation.
- `REALIGNMENT/09-phase-G-rtk-observability.md` — spec.
- `REALIGNMENT/10-phase-H-python-retirement.md` — H1 acceptance gate.
+1
View File
@@ -0,0 +1 @@
+6576
View File
File diff suppressed because it is too large Load Diff
+47
View File
@@ -0,0 +1,47 @@
{
"name": "headroom-docs",
"version": "0.0.0",
"private": true,
"scripts": {
"build": "next build",
"dev": "next dev",
"start": "next start",
"types:check": "fumadocs-mdx && next typegen && tsc --noEmit",
"postinstall": "fumadocs-mdx"
},
"dependencies": {
"@radix-ui/react-slot": "1.3.0",
"class-variance-authority": "0.7.1",
"clsx": "2.1.1",
"dotted-map": "^3.1.0",
"fumadocs-core": "16.11.1",
"fumadocs-mdx": "15.1.0",
"fumadocs-twoslash": "^3.3.0",
"fumadocs-typescript": "^5.3.0",
"fumadocs-ui": "16.11.1",
"headroom-ai": "file:../sdk/typescript",
"lucide-react": "^1.7.0",
"next": "16.2.10",
"react": "^19.2.7",
"react-dom": "^19.2.7",
"recharts": "^3.9.2",
"tailwind-merge": "^3.5.0"
},
"devDependencies": {
"@ai-sdk/openai": "^3.0.51",
"@anthropic-ai/sdk": "^0.106.0",
"@tailwindcss/postcss": "^4.3.2",
"@types/mdx": "^2.0.14",
"@types/node": "^26.1.1",
"@types/react": "^19.2.17",
"@types/react-dom": "^19.2.3",
"ai": "^6.0.149",
"openai": "^6.33.0",
"postcss": "^8.5.16",
"tailwindcss": "^4.2.2",
"typescript": "^5.9.3"
},
"overrides": {
"postcss": "$postcss"
}
}
+329
View File
@@ -0,0 +1,329 @@
{
"schema_version": 1,
"updated": "2026-07-06",
"source_issues": [
"https://github.com/headroomlabs-ai/headroom/issues/1843"
],
"status_values": [
"covered",
"partial",
"gap",
"blocked"
],
"platforms": [
"linux",
"macos",
"windows"
],
"features": [
{
"id": "install_apply_python",
"name": "Persistent Python install applies and starts",
"risk": "install",
"platforms": {
"linux": {
"status": "covered",
"tests": [
"tests/test_cli/test_install_cli.py",
"tests/test_install/test_supervisors.py",
".github/workflows/install-native-e2e.yml"
]
},
"macos": {
"status": "covered",
"tests": [
"tests/test_install/test_supervisors.py",
".github/workflows/install-native-e2e.yml"
]
},
"windows": {
"status": "partial",
"tests": [
"tests/test_install/test_supervisors.py",
"tests/test_install/test_runtime.py",
".github/workflows/ci.yml#windows-native-wrapper"
],
"gap": "Native install smoke workflow is source-level until Windows wheel CRT conflicts are resolved."
}
}
},
{
"id": "install_windows_service",
"name": "Windows service install uses sc.exe safely",
"risk": "install",
"platforms": {
"linux": {
"status": "covered",
"tests": [
"tests/test_install/test_supervisors.py"
]
},
"macos": {
"status": "covered",
"tests": [
"tests/test_install/test_supervisors.py"
]
},
"windows": {
"status": "covered",
"tests": [
"tests/test_install/test_supervisors.py"
]
}
}
},
{
"id": "single_instance_start",
"name": "Persistent runtime start is single-instance by default",
"risk": "runtime",
"platforms": {
"linux": {
"status": "covered",
"tests": [
"tests/test_cli/test_install_cli.py",
"tests/test_install/test_runtime.py"
]
},
"macos": {
"status": "covered",
"tests": [
"tests/test_cli/test_install_cli.py",
"tests/test_install/test_runtime.py"
]
},
"windows": {
"status": "covered",
"tests": [
"tests/test_cli/test_install_cli.py",
"tests/test_install/test_runtime.py"
]
}
}
},
{
"id": "compression_fail_open",
"name": "Slow or saturated compression fails open",
"risk": "performance",
"platforms": {
"linux": {
"status": "covered",
"tests": [
"tests/test_platform_stabilization_functional.py",
"tests/test_kompress_request_nonblocking.py",
"tests/test_proxy_compression_executor.py",
"tests/test_openai_codex_ws_lifecycle.py"
]
},
"macos": {
"status": "partial",
"tests": [
"tests/test_platform_stabilization_functional.py",
"tests/test_kompress_request_nonblocking.py",
"tests/test_proxy_compression_executor.py"
],
"gap": "Native workflow coverage should add focused performance-gate tests without full model downloads."
},
"windows": {
"status": "partial",
"tests": [
"tests/test_platform_stabilization_functional.py",
"tests/test_kompress_request_nonblocking.py",
"tests/test_proxy_compression_executor.py"
],
"gap": "Source-level tests validate fail-open semantics; full native proxy e2e waits on Windows wheel availability."
}
}
},
{
"id": "proxy_functional_smoke",
"name": "Proxy health and /v1/compress work through the app surface",
"risk": "proxy",
"platforms": {
"linux": {
"status": "covered",
"tests": [
"tests/test_platform_stabilization_functional.py",
"tests/test_ccr_row_drop_store_bridge.py"
]
},
"macos": {
"status": "partial",
"tests": [
"tests/test_platform_stabilization_functional.py",
"tests/test_ccr_row_drop_store_bridge.py"
],
"gap": "FastAPI route coverage exists; native persistent proxy process smoke should be added."
},
"windows": {
"status": "partial",
"tests": [
"tests/test_platform_stabilization_functional.py",
"tests/test_ccr_row_drop_store_bridge.py"
],
"gap": "FastAPI route coverage exists; native persistent proxy process smoke waits on Windows wheel availability."
}
}
},
{
"id": "ccr_persistence",
"name": "CCR survives restart when persistent storage is enabled",
"risk": "cache",
"platforms": {
"linux": {
"status": "covered",
"tests": [
"crates/headroom-core/tests/ccr_backends.rs",
"tests/test_storage_backends.py"
]
},
"macos": {
"status": "partial",
"tests": [
"crates/headroom-core/tests/ccr_backends.rs",
"tests/test_storage_backends.py"
],
"gap": "Rust backend tests run in the Rust workflow; native macOS restart e2e is not yet present."
},
"windows": {
"status": "partial",
"tests": [
"crates/headroom-core/tests/ccr_backends.rs",
"tests/test_storage_backends.py"
],
"gap": "Windows restart e2e is blocked by the native wheel CRT conflict."
}
}
},
{
"id": "init_cli",
"name": "headroom init configures supported agents",
"risk": "install",
"platforms": {
"linux": {
"status": "covered",
"tests": [
".github/workflows/init-native-e2e.yml"
]
},
"macos": {
"status": "covered",
"tests": [
".github/workflows/init-native-e2e.yml"
]
},
"windows": {
"status": "blocked",
"tests": [
".github/workflows/init-native-e2e.yml"
],
"gap": "Matrix entry is intentionally excluded until Windows wheel CRT conflicts are resolved."
}
}
},
{
"id": "wrap_prepare_only",
"name": "headroom wrap prepare-only mutates config safely",
"risk": "install",
"platforms": {
"linux": {
"status": "covered",
"tests": [
".github/workflows/wrap-native-e2e.yml"
]
},
"macos": {
"status": "covered",
"tests": [
".github/workflows/wrap-native-e2e.yml"
]
},
"windows": {
"status": "blocked",
"tests": [
".github/workflows/wrap-native-e2e.yml"
],
"gap": "Matrix entry is intentionally excluded until Windows wheel CRT conflicts are resolved."
}
}
},
{
"id": "toin_skip_recommendations",
"name": "TOIN skip-compression recommendations are exposed",
"risk": "cache",
"platforms": {
"linux": {
"status": "covered",
"tests": [
"tests/test_toin.py",
"tests/test_proxy_ccr.py",
"tests/test_compression_policy_toin_gate.py"
]
},
"macos": {
"status": "partial",
"tests": [
"tests/test_toin.py",
"tests/test_compression_policy_toin_gate.py"
],
"gap": "Native end-to-end verification of served recommendations is still needed."
},
"windows": {
"status": "partial",
"tests": [
"tests/test_toin.py",
"tests/test_compression_policy_toin_gate.py"
],
"gap": "Native end-to-end verification is blocked by Windows wheel availability."
}
}
}
],
"sanity_tests": [
{
"id": "cli_help",
"description": "Every public CLI command renders help without importing optional heavy runtimes.",
"tests": [
"tests/test_cli"
]
},
{
"id": "install_paths",
"description": "Install paths resolve under the active user profile and never require administrator paths for user scope.",
"tests": [
"tests/test_install/test_paths.py"
]
},
{
"id": "runtime_selection",
"description": "Python, Docker, service, and task runtime commands are generated deterministically for each platform.",
"tests": [
"tests/test_install/test_runtime.py",
"tests/test_install/test_supervisors.py"
]
},
{
"id": "health_startup",
"description": "Persistent starts wait for /readyz and surface startup failure instead of silently succeeding.",
"tests": [
"tests/test_cli/test_install_cli.py",
"tests/test_install/test_health.py"
]
},
{
"id": "compression_backpressure",
"description": "Compression queue saturation and model cold-start fail open without hanging request paths.",
"tests": [
"tests/test_platform_stabilization_functional.py",
"tests/test_kompress_request_nonblocking.py",
"tests/test_proxy_compression_executor.py"
]
},
{
"id": "proxy_route_smoke",
"description": "Health and /v1/compress routes return functional metrics and fail open on timeout.",
"tests": [
"tests/test_platform_stabilization_functional.py"
]
}
]
}
+28
View File
@@ -0,0 +1,28 @@
# Platform Stabilization Matrix
This matrix is the hardening source of truth for install, startup, runtime, cache, and compression behavior across Linux, macOS, and Windows. The machine-readable matrix lives in `docs/platform-feature-matrix.json`; CI tests validate its shape so every feature has explicit platform status and test evidence.
## Status Values
- `covered`: unit, integration, or native e2e coverage exists for the platform.
- `partial`: coverage exists, but a named gap remains.
- `gap`: no meaningful coverage exists yet.
- `blocked`: coverage is intentionally excluded by a known external blocker.
## Current Priorities
1. Keep install/setup/run idempotent. Persistent starts must not create duplicate proxy instances unless a future explicit opt-in exists.
2. Keep compression fail-open. Cold model downloads, saturated executors, or timeout paths must pass through unchanged traffic instead of hanging agents.
3. Keep cache behavior visible. CCR persistence and TOIN skip recommendations must have restart and served-recommendation coverage.
4. Keep Windows honest. Windows-specific tests should run locally and in CI whenever they do not require the currently blocked native wheel build.
## Issue 1843 Coverage Map
- Windows service quoting: `install_windows_service`
- Duplicate startup processes: `single_instance_start`
- Slow compression and hangs: `compression_fail_open`
- CCR restart persistence: `ccr_persistence`
- TOIN recommendation wiring: `toin_skip_recommendations`
- Install/setup/run sanity: `install_apply_python`, `init_cli`, `wrap_prepare_only`
When adding or closing a hardening item, update `docs/platform-feature-matrix.json` in the same PR as the implementation or test change.
+7
View File
@@ -0,0 +1,7 @@
const config = {
plugins: {
'@tailwindcss/postcss': {},
},
};
export default config;
+29
View File
@@ -0,0 +1,29 @@
import { NextRequest, NextResponse } from 'next/server';
import { isMarkdownPreferred, rewritePath } from 'fumadocs-core/negotiation';
import { docsContentRoute, docsRoute } from '@/lib/shared';
const { rewrite: rewriteDocs } = rewritePath(
`${docsRoute}{/*path}`,
`${docsContentRoute}{/*path}/content.md`,
);
const { rewrite: rewriteSuffix } = rewritePath(
`${docsRoute}{/*path}.mdx`,
`${docsContentRoute}{/*path}/content.md`,
);
export default function proxy(request: NextRequest) {
const result = rewriteSuffix(request.nextUrl.pathname);
if (result) {
return NextResponse.rewrite(new URL(result, request.nextUrl));
}
if (isMarkdownPreferred(request)) {
const result = rewriteDocs(request.nextUrl.pathname);
if (result) {
return NextResponse.rewrite(new URL(result, request.nextUrl));
}
}
return NextResponse.next();
}
+122
View File
@@ -0,0 +1,122 @@
# RTK architecture — why wrap-CLI only
**Status:** decided. Locked at Phase G PR-G3 (2026-05).
**Owner:** Headroom realignment.
## TL;DR
**RTK is a wrap-CLI hook, not a proxy-side compressor.** The Headroom
proxy does NOT invoke RTK on tool-result content. Future contributors
who consider moving RTK into the proxy hot path: read this doc first.
## Background
RTK (Realtime Token Kompress) rewrites shell **commands** at exec
time so that a `git diff` or `grep` invocation emits a more
compressed output before the agent ever ingests it. RTK runs in the
wrap-CLI tail — `headroom wrap claude`, `headroom wrap codex`, etc.
— where it installs a `~/.rtk/bin/rtk` shim ahead of the agent CLI
and intercepts shelled-out subprocesses.
It surfaces value in two places:
1. **Tokens saved per invocation** — measured by `rtk gain --format json`.
2. **Tokens saved per session** — aggregated at wrap-session end.
Both signals feed `wrap_rtk_invocations_total` and
`wrap_rtk_tokens_saved_per_session` (registered by the Rust proxy's
observability surface so a single `/metrics` scrape exposes the full
picture).
## Proxy-side RTK was considered and rejected
At Phase G scoping, three reviewers floated the idea of invoking
RTK on the **proxy** side: when a `tool_result` block flows
upstream, dispatch it through RTK to shrink the content before it
hits the model.
**Decision: rejected.** Three load-bearing reasons.
### 1. Cache hot zone risk
The proxy's Phase B cache-safety contract pins `tool_result`
content as part of the cache hot zone. Compression there bursts
the prompt cache because the rewritten bytes diverge from the
canonical wire bytes the upstream cached. Phase B PR-B2 → PR-B7
spent ~3000 LOC carving the live-zone-only surface specifically
to prevent this class of cache-invalidation. Inserting RTK
proxy-side would re-introduce it.
### 2. Parallel implementation with `log_compressor.rs`
The Rust proxy already has a `crates/headroom-core/src/transforms/log_compressor.rs`
that compresses **tool output text** in the live zone. It uses the
same heuristics RTK uses (whitespace de-dup, line de-dup,
file-listing collapse) but invoked at the proxy's per-block
dispatcher rather than at the shell exec boundary. Adding RTK
proxy-side would mean two implementations of the same compression
in the same hot path; "no silent fallbacks, no parallel impls" is
explicit project policy.
### 3. Command-rewrite vs output-rewrite — different value propositions
RTK rewrites **commands** before they execute. The
`git log --oneline` you typed becomes `git log --oneline -n 50`
because RTK has learned that the first 50 commits are usually
enough context. That's a fundamentally different mechanism from
compressing the **output** of an unmodified command. A proxy-side
invocation would skip the command-rewrite half — the half that
generates the largest savings on heavy shell workloads — and only
catch the output side, which is already covered by
`log_compressor` and `code_compressor`.
## What the proxy does provide
Per Phase G PR-G3, the proxy exposes RTK-derived metrics via its
registry:
- `wrap_rtk_invocations_total{tool}` — driven by the wrap-CLI
polling `rtk gain --format json` and incrementing the registered
counter by the delta since last poll.
- `wrap_rtk_tokens_saved_per_session` — emitted at wrap-session
close.
This keeps the operator dashboard single-pane-of-glass without
re-implementing RTK inside the proxy.
## What the wrap CLI does
Every `headroom wrap <agent>` subcommand:
1. Ensures the RTK binary is installed via `_ensure_rtk_binary()`.
2. Injects the `<!-- headroom:rtk-instructions -->` block into the
agent's instruction file (e.g. `AGENTS.md`, `.cursorrules`).
3. Spawns the proxy and the agent CLI side-by-side.
4. Polls `rtk gain --format json` on a 5-second memoization window
and feeds the delta into the proxy's metric registry.
See `headroom/cli/wrap/` for the per-agent shims.
## Re-litigation policy
A change to this architecture should:
1. Quote the live-zone-only contract from
`REALIGNMENT/04-phase-B-live-zone.md` and explain why the
cache-burst risk is acceptable.
2. Show measurements (not estimates) that proxy-side RTK adds value
beyond `log_compressor.rs` on real production traffic.
3. Have an exit ramp: a CLI flag to disable proxy-side RTK without
reverting the wrap-CLI integration.
Without all three, treat the proposal as a regression and link this
doc.
## References
- `REALIGNMENT/09-phase-G-rtk-observability.md` — Phase G plan.
- `REALIGNMENT/04-phase-B-live-zone.md` — cache hot-zone contract.
- `headroom/cli/wrap/` — wrap-CLI implementation.
- `crates/headroom-core/src/transforms/log_compressor.rs` — the
proxy-side log compressor RTK would parallel.
- 2026-05-01 user direction message archived in
`project_compression_realignment_2026_05` memory note.
Binary file not shown.

After

Width:  |  Height:  |  Size: 100 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 38 KiB

+44
View File
@@ -0,0 +1,44 @@
import { defineConfig, defineDocs } from 'fumadocs-mdx/config';
import { metaSchema, pageSchema } from 'fumadocs-core/source/schema';
import { transformerTwoslash } from 'fumadocs-twoslash';
import { rehypeCodeDefaultOptions } from 'fumadocs-core/mdx-plugins';
export const docs = defineDocs({
dir: 'content/docs',
docs: {
schema: pageSchema,
postprocess: {
includeProcessedMarkdown: true,
},
},
meta: {
schema: metaSchema,
},
});
export default defineConfig({
mdxOptions: {
rehypeCodeOptions: {
themes: {
light: 'github-light',
dark: 'github-dark',
},
transformers: [
...(rehypeCodeDefaultOptions.transformers ?? []),
transformerTwoslash({
twoslashOptions: {
compilerOptions: {
target: 9, // ES2022
lib: ['lib.es2022.d.ts', 'lib.dom.d.ts', 'lib.dom.iterable.d.ts'],
},
// Documentation code snippets are illustrative — don't require full type validity
handbookOptions: {
noErrors: true,
},
},
}),
],
langs: ['js', 'jsx', 'ts', 'tsx', 'python', 'bash', 'json', 'yaml', 'toml', 'css', 'mermaid'],
},
},
});
+35
View File
@@ -0,0 +1,35 @@
{
"compilerOptions": {
"target": "ESNext",
"lib": ["dom", "dom.iterable", "esnext"],
"allowJs": true,
"skipLibCheck": true,
"strict": true,
"forceConsistentCasingInFileNames": true,
"noEmit": true,
"esModuleInterop": true,
"module": "esnext",
"moduleResolution": "bundler",
"resolveJsonModule": true,
"isolatedModules": true,
"jsx": "react-jsx",
"incremental": true,
"paths": {
"@/*": ["./*"],
"collections/*": ["./.source/*"]
},
"plugins": [
{
"name": "next"
}
]
},
"include": [
"next-env.d.ts",
"**/*.ts",
"**/*.tsx",
".next/types/**/*.ts",
".next/dev/types/**/*.ts"
],
"exclude": ["node_modules"]
}