Files
wehub-resource-sync 426e9eeabd
Voice Workbench / headless workbench (mocked backends) (push) Has been cancelled
Voice Workbench / real acoustic lane (nightly, provisioned only) (push) Has been cancelled
ci / test (push) Has been cancelled
ci / lint-and-format (push) Has been cancelled
ci / build (push) Has been cancelled
ci / dev-startup (push) Has been cancelled
gitleaks / gitleaks (push) Has been cancelled
Markdown Links / Relative Markdown Links (push) Has been cancelled
Quality (Extended) / Homepage Build (PR smoke) (push) Has been cancelled
Quality (Extended) / Comment-only diff guard (push) Has been cancelled
Quality (Extended) / Format + Type Safety Ratchet (push) Has been cancelled
Quality (Extended) / Develop Gate (secret scan + UI determinism) (push) Has been cancelled
Quality (Extended) / Develop Gate (lint) (push) Has been cancelled
Chat shell gestures / Chat shell gesture + parity e2e (push) Has been cancelled
Cloud Gateway Discord / Test (push) Has been cancelled
Benchmark Bridge Tests / benchmark (bunx @biomejs/biome check packages/lifeops-bench/src, benchmark-lint) (push) Has been cancelled
Benchmark Bridge Tests / benchmark (bunx vitest run --config packages/lifeops-bench/vitest.config.ts --root packages/lifeops-bench --passWithNoTests, benchmark-tests) (push) Has been cancelled
Build Agent Image / build-and-push (push) Has been cancelled
Dev Smoke / bun run dev onboarding chat (push) Has been cancelled
Dev Smoke / Vite HMR dependency-level smoke (push) Has been cancelled
Electrobun Submodule Guard / electrobun gitlink is fetchable (push) Has been cancelled
Publish @elizaos/example-code / check_npm (push) Has been cancelled
Publish @elizaos/example-code / publish_npm (push) Has been cancelled
Publish @elizaos/plugin-elizacloud / verify_version (push) Has been cancelled
Publish @elizaos/plugin-elizacloud / publish_npm (push) Has been cancelled
Sandbox Live Smoke / Sandbox live smoke (push) Has been cancelled
Snap Build & Test / Build Snap (amd64) (push) Has been cancelled
Snap Build & Test / Build Snap (arm64) (push) Has been cancelled
Test Packaging / elizaos CLI global-install smoke (node + bun) (push) Has been cancelled
Cloud Gateway Webhook / Test (push) Has been cancelled
Cloud Tests / lint-and-types (push) Has been cancelled
Cloud Tests / unit-tests (push) Has been cancelled
Cloud Tests / integration-tests (push) Has been cancelled
Cloud Tests / e2e-tests (push) Has been cancelled
CodeQL Advanced / Analyze (javascript-typescript) (push) Has been cancelled
Deploy Apps Worker (Product 2) / Determine environment (push) Has been cancelled
Deploy Apps Worker (Product 2) / Deploy apps worker to apps-control host (${{ needs.determine-env.outputs.environment }}) (push) Has been cancelled
Deploy Eliza Provisioning Worker / Determine environment (push) Has been cancelled
Deploy Eliza Provisioning Worker / Deploy worker to Hetzner host (${{ needs.determine-env.outputs.environment }} @ ${{ needs.determine-env.outputs.deployment_sha }}) (push) Has been cancelled
Dev Smoke / Classify changed paths (push) Has been cancelled
supply-chain / sbom (push) Has been cancelled
supply-chain / vulnerability-scan (push) Has been cancelled
Build, Push & Deploy to Phala Cloud / build-and-push (push) Has been cancelled
Test Packaging / Validate Packaging Configs (push) Has been cancelled
Test Packaging / Build & Test PyPI Package (push) Has been cancelled
Test Packaging / PyPI on Python ${{ matrix.python }} (push) Has been cancelled
Test Packaging / Pack & Test JS Tarballs (push) Has been cancelled
UI Fixture E2E / ui-fixture-e2e (push) Has been cancelled
UI Fixture E2E / fixture-e2e (push) Has been cancelled
UI Story Gate / story-gate (push) Has been cancelled
vault-ci / test (macos-latest) (push) Has been cancelled
vault-ci / test (ubuntu-latest) (push) Has been cancelled
vault-ci / test (windows-latest) (push) Has been cancelled
vault-ci / app-core wiring tests (push) Has been cancelled
verify-patches / verify patches/CHECKSUMS.sha256 (push) Has been cancelled
Voice Benchmark Smoke / voice-emotion fixture smoke (push) Has been cancelled
Voice Benchmark Smoke / voiceagentbench fixture smoke (push) Has been cancelled
Voice Benchmark Smoke / voicebench-quality unit smoke (push) Has been cancelled
Voice Benchmark Smoke / voicebench TypeScript unit (no audio) (push) Has been cancelled
Voice Benchmark Smoke / voice bench smoke summary (push) Has been cancelled
Windows CI / windows ([bun run --cwd packages/app-core test bun run --cwd packages/elizaos test bun run --cwd packages/cloud/shared test], app-and-cli) (push) Has been cancelled
Windows CI / windows ([bun run --cwd packages/scenario-runner test bun run --cwd packages/vault test bun run --cwd packages/security test bun run --cwd plugins/plugin-coding-tools test], framework-packages) (push) Has been cancelled
Windows CI / windows ([bun run --cwd plugins/plugin-elizacloud test bun run --cwd plugins/plugin-discord test bun run --cwd plugins/plugin-anthropic test bun run --cwd plugins/plugin-openai test bun run --cwd plugins/plugin-app-control test bun run --cwd plugins/pl… (push) Has been cancelled
Windows CI / windows ([node packages/scripts/run-turbo.mjs run build --filter=@elizaos/core --filter=@elizaos/shared --filter=@elizaos/agent --concurrency=4 node packages/scripts/run-bash-linux-only.mjs scripts/verify-riscv64-buildpaths.sh node packages/scripts/run… (push) Has been cancelled
Windows CI / windows ([node packages/scripts/run-turbo.mjs run typecheck --filter=@elizaos/core --filter=@elizaos/shared --filter=@elizaos/cloud-shared --concurrency=4 bun run --cwd packages/core test bun run --cwd packages/shared test], core-runtime, 75) (push) Has been cancelled
chore: import upstream snapshot with attribution
2026-07-13 12:43:05 +08:00

231 lines
7.0 KiB
TypeScript

/**
* CLI harness that runs the LifeOps, executive-assistant, and self-care prompt
* benchmark suites against a live provider and writes a markdown report plus
* serialized Ax optimization rows. Invoked with `--report <path>`; the case
* builders and runner live under `test/helpers/`.
*/
import { mkdir, writeFile } from "node:fs/promises";
import path from "node:path";
import process from "node:process";
import {
buildExecutiveAssistantPromptBenchmarkCases,
buildLifeOpsPromptBenchmarkCases,
buildSelfCarePromptBenchmarkCases,
type PromptBenchmarkCase,
} from "../test/helpers/lifeops-prompt-benchmark-cases.ts";
import {
buildAxOptimizationRows,
formatPromptBenchmarkReportMarkdown,
runLifeOpsPromptBenchmark,
serializeAxOptimizationRows,
} from "../test/helpers/lifeops-prompt-benchmark-runner.ts";
type RunLifeOpsPromptBenchmarkOptions = NonNullable<
Parameters<typeof runLifeOpsPromptBenchmark>[0]
>;
type LiveProviderName = RunLifeOpsPromptBenchmarkOptions["preferredProvider"];
type CliOptions = {
axPath?: string;
compress: boolean;
isolate: "shared" | "per-case";
listOnly: boolean;
markdownPath?: string;
preferredProvider?: LiveProviderName;
reportPath?: string;
suite: "all" | "executive-assistant" | "self-care";
variantIds: string[];
};
function parseArgs(argv: string[]): CliOptions {
const options: CliOptions = {
compress: process.env.ELIZA_PROMPT_COMPRESS === "1",
isolate: "shared",
listOnly: false,
suite: "all",
variantIds: [],
};
for (let index = 0; index < argv.length; index += 1) {
const arg = argv[index];
if (arg === "--list") {
options.listOnly = true;
continue;
}
if (arg === "--compress") {
// Wave 2-D: Cerebras compress mode. Drops few-shot examples from the
// resolved optimized prompt, skips routing-hint rendering, and caps
// retrieval top-K at 8. One-flag escape hatch for token-budget-pressed
// runs against the Cerebras large tier.
options.compress = true;
continue;
}
if (arg === "--suite") {
const value = String(argv[index + 1] ?? "").trim();
if (
value === "all" ||
value === "self-care" ||
value === "executive-assistant"
) {
options.suite = value;
index += 1;
continue;
}
throw new Error(`Unsupported --suite value: ${value}`);
}
if (arg === "--variant") {
const value = String(argv[index + 1] ?? "").trim();
options.variantIds.push(
...value
.split(",")
.map((entry) => entry.trim())
.filter(Boolean),
);
index += 1;
continue;
}
if (arg === "--report") {
options.reportPath = argv[index + 1];
index += 1;
continue;
}
if (arg === "--markdown") {
options.markdownPath = argv[index + 1];
index += 1;
continue;
}
if (arg === "--ax") {
options.axPath = argv[index + 1];
index += 1;
continue;
}
if (arg === "--provider") {
options.preferredProvider = argv[index + 1] as
| LiveProviderName
| undefined;
index += 1;
continue;
}
if (arg === "--isolate") {
const value = String(argv[index + 1] ?? "").trim();
if (value === "shared" || value === "per-case") {
options.isolate = value;
index += 1;
continue;
}
throw new Error(`Unsupported --isolate value: ${value}`);
}
throw new Error(`Unknown argument: ${arg}`);
}
return options;
}
async function loadCases(
suite: CliOptions["suite"],
): Promise<PromptBenchmarkCase[]> {
if (suite === "self-care") {
return buildSelfCarePromptBenchmarkCases();
}
if (suite === "executive-assistant") {
return buildExecutiveAssistantPromptBenchmarkCases();
}
return buildLifeOpsPromptBenchmarkCases();
}
function filterCases(
cases: PromptBenchmarkCase[],
options: CliOptions,
): PromptBenchmarkCase[] {
const selectedVariants = new Set(options.variantIds);
if (selectedVariants.size === 0) {
return cases;
}
return cases.filter((testCase) => selectedVariants.has(testCase.variantId));
}
function defaultArtifactBase(): string {
return path.join(
process.cwd(),
".tmp",
`lifeops-prompt-benchmark-${Date.now()}`,
);
}
async function main(): Promise<void> {
const options = parseArgs(process.argv.slice(2));
const basePath = defaultArtifactBase();
const reportPath = options.reportPath
? path.resolve(process.cwd(), options.reportPath)
: `${basePath}.json`;
const markdownPath = options.markdownPath
? path.resolve(process.cwd(), options.markdownPath)
: `${basePath}.md`;
const axPath = options.axPath
? path.resolve(process.cwd(), options.axPath)
: `${basePath}.jsonl`;
const allCases = filterCases(await loadCases(options.suite), options);
if (allCases.length === 0) {
throw new Error("No prompt benchmark cases matched the requested filters.");
}
if (options.listOnly) {
const bySuite = new Map<string, number>();
const byVariant = new Map<string, number>();
for (const testCase of allCases) {
bySuite.set(testCase.suiteId, (bySuite.get(testCase.suiteId) ?? 0) + 1);
byVariant.set(
testCase.variantId,
(byVariant.get(testCase.variantId) ?? 0) + 1,
);
}
console.log(`[lifeops-prompt-benchmark] total=${allCases.length}`);
for (const [suiteId, count] of Array.from(bySuite.entries()).sort()) {
console.log(`[lifeops-prompt-benchmark] suite ${suiteId}: ${count}`);
}
for (const [variantId, count] of Array.from(byVariant.entries()).sort()) {
console.log(`[lifeops-prompt-benchmark] variant ${variantId}: ${count}`);
}
return;
}
if (options.compress) {
process.env.ELIZA_PROMPT_COMPRESS = "1";
}
const report = await runLifeOpsPromptBenchmark({
cases: allCases,
isolate: options.isolate,
preferredProvider: options.preferredProvider,
});
const markdown = formatPromptBenchmarkReportMarkdown(report);
const axRows = buildAxOptimizationRows(report);
await mkdir(path.dirname(reportPath), { recursive: true });
await mkdir(path.dirname(markdownPath), { recursive: true });
await mkdir(path.dirname(axPath), { recursive: true });
await writeFile(reportPath, `${JSON.stringify(report, null, 2)}\n`, "utf8");
await writeFile(markdownPath, `${markdown}\n`, "utf8");
await writeFile(axPath, serializeAxOptimizationRows(axRows), "utf8");
console.log(
`[lifeops-prompt-benchmark] provider=${report.providerName} total=${report.total} passed=${report.passed} failed=${report.failed} report=${reportPath}`,
);
console.log(
`[lifeops-prompt-benchmark] markdown=${markdownPath} ax=${axPath}`,
);
for (const failure of report.failures.slice(0, 10)) {
console.log(
`[lifeops-prompt-benchmark] FAIL ${failure.case.caseId} expected=${failure.case.expectedAction ?? "null/REPLY"} actual=${failure.actualPrimaryAction ?? "null"}`,
);
}
if (report.failed > 0) {
process.exitCode = 1;
}
}
await main();