From b215884021491772a1eb7a9f92c6653a2a52a69d Mon Sep 17 00:00:00 2001 From: Christian Klotz Date: Fri, 11 Sep 2026 17:29:34 +0200 Subject: [PATCH] feat(coding-agent): add customization documentation evals (#9491) * fix(coding-agent): isolate extension eval documentation treatment * feat(coding-agent): add custom provider documentation eval * feat(coding-agent): add customization documentation evals * feat(coding-agent): add repeatable eval reports --- packages/evals/README.md | 51 ++- packages/evals/scripts/run-evals.mjs | 23 + packages/evals/src/docs.eval.ts | 2 +- packages/evals/src/extensions.eval.ts | 35 +- packages/evals/src/models.eval.ts | 151 +++++++ packages/evals/src/pi-harness.ts | 71 ++- packages/evals/src/providers.eval.ts | 410 ++++++++++++++++++ packages/evals/src/smoke.eval.ts | 4 +- .../evals/src/vitest-evals/harness-table.ts | 13 +- packages/evals/src/vitest-evals/reporter.ts | 19 +- .../test/vitest-evals/harness-table.test.ts | 13 + vitest.base.ts | 5 + 12 files changed, 747 insertions(+), 50 deletions(-) create mode 100644 packages/evals/src/models.eval.ts create mode 100644 packages/evals/src/providers.eval.ts diff --git a/packages/evals/README.md b/packages/evals/README.md index 893e6f998..334f71551 100644 --- a/packages/evals/README.md +++ b/packages/evals/README.md @@ -26,13 +26,56 @@ Additional arguments are forwarded to Vitest: ```bash npm run eval -- src/extensions.eval.ts -npm run eval -- -t "creates, reloads, and uses" +npm run eval -- -t "creates and uses the extension" npm run eval -- src/docs.eval.ts -t "session-format\.md" ``` -Each invocation prints an ignored `.eval/` artifact directory. `runs.jsonl` indexes completed harness runs and their -native Pi session JSONL attachments under `sessions/`. These files may contain prompts, responses, source code, and tool -output. +Run all comparative customization evals five times in one invocation: + +```bash +npm run eval -- \ + src/extensions.eval.ts src/models.eval.ts src/providers.eval.ts \ + --provider openai --model gpt-5.6-sol \ + --repetitions 5 +``` + +`--repetitions` applies to suites declared with `evalHarnessTable(...)`. An explicit `repetitions` value in a suite +overrides the command-line default. `PI_EVAL_REPETITIONS=5` is equivalent to the command-line option. Use one repetition +while developing an eval and five when reporting lift. + +## Reports and artifacts + +Each invocation prints a compound `Eval Comparisons` report after the Vitest results. When several comparative files run +in the same invocation, this report contains one section for every eval set. For example, with illustrative values: + +```text +Eval Comparisons + Add model to existing provider + Baseline system-prompt-without-docs + Candidate default-system-prompt (5/5 pairs) + Pass rate +60.0 pp (candidate 80.0%, baseline 20.0%) + Tokens +1200.0 (candidate 24000.0, baseline 22800.0) + Latency -850.0ms (candidate 14000.0ms, baseline 14850.0ms) + Est. cost +$0.0100 (candidate $0.1200, baseline $0.1100) + + Add OpenAI-compatible provider + ... + + Add custom streaming provider + ... +``` + +The runner prints the ignored `.eval/` artifact directory at startup. It contains: + +- `report.txt`: the terminal comparison report without color codes. +- `report.json`: the same aggregate comparison data as structured JSON. +- `runs.jsonl`: one record for every completed harness run. +- `sessions/`: native Pi session JSONL attachments. +- `sources/`: source attachments recorded by individual evals. + +The report covers comparative suites using `evalHarnessTable(...)`. Ordinary evals still appear in the Vitest summary and +in `runs.jsonl`, but not in the baseline-versus-candidate comparison report. Artifacts may contain prompts, responses, +source code, and tool output. ## Writing evals diff --git a/packages/evals/scripts/run-evals.mjs b/packages/evals/scripts/run-evals.mjs index fa41b2390..7dd68489a 100644 --- a/packages/evals/scripts/run-evals.mjs +++ b/packages/evals/scripts/run-evals.mjs @@ -16,6 +16,7 @@ const artifactDirectory = process.env.PI_EVAL_ARTIFACT_DIR const args = process.argv.slice(2); let provider; let model; +let repetitions; let hasCliModelSelection = false; const vitestArgs = []; @@ -33,6 +34,16 @@ for (let index = 0; index < args.length; index += 1) { index += 1; continue; } + if (arg === "--repetitions") { + const value = args[index + 1]; + if (!value || value.startsWith("-")) { + console.error("Missing value for --repetitions"); + process.exit(1); + } + repetitions = value; + index += 1; + continue; + } if (arg.startsWith("--provider=")) { provider = arg.slice("--provider=".length); hasCliModelSelection = true; @@ -43,11 +54,21 @@ for (let index = 0; index < args.length; index += 1) { hasCliModelSelection = true; continue; } + if (arg.startsWith("--repetitions=")) { + repetitions = arg.slice("--repetitions=".length); + continue; + } vitestArgs.push(arg); } provider = provider?.trim() || undefined; model = model?.trim() || undefined; +const repetitionsText = (repetitions ?? process.env.PI_EVAL_REPETITIONS)?.trim(); +const repetitionCount = repetitionsText ? Number(repetitionsText) : 1; +if (!Number.isSafeInteger(repetitionCount) || repetitionCount < 1) { + console.error("Repetitions must be a positive integer."); + process.exit(1); +} if (hasCliModelSelection) { if (!provider || !model) { console.error("CLI model selection requires both --provider and --model."); @@ -68,10 +89,12 @@ const vitestCliPath = resolve(dirname(vitestPackagePath), "vitest.mjs"); mkdirSync(artifactDirectory, { recursive: true, mode: 0o700 }); console.error(`[eval] default-model=${provider && model ? `${provider}/${model}` : "none"}`); +console.error(`[eval] repetitions=${repetitionCount}`); console.error(`[eval] artifacts=${artifactDirectory}`); const childEnvironment = { ...process.env, PI_EVAL_ARTIFACT_DIR: artifactDirectory, + PI_EVAL_REPETITIONS: String(repetitionCount), }; if (provider && model) { childEnvironment.PI_PROVIDER = provider; diff --git a/packages/evals/src/docs.eval.ts b/packages/evals/src/docs.eval.ts index d02d03990..949c5e009 100644 --- a/packages/evals/src/docs.eval.ts +++ b/packages/evals/src/docs.eval.ts @@ -43,7 +43,7 @@ const documentationAuditHarness = createPiCodingAgentHarness({ customTools: [submitDocumentationAuditTool], }); -describeEval("Coding agent documentation", { harness: documentationAuditHarness }, (it) => { +describeEval("Audit documentation against implementation", { harness: documentationAuditHarness }, (it) => { it.for(documentationPages)("$path matches the implementation", { timeout: 300_000 }, async ({ path }, { run }) => { const documentationPath = resolve(docsRoot, path); const result = await run(`Audit one Pi documentation page against the repository implementation. diff --git a/packages/evals/src/extensions.eval.ts b/packages/evals/src/extensions.eval.ts index 577b998d6..d5a919f08 100644 --- a/packages/evals/src/extensions.eval.ts +++ b/packages/evals/src/extensions.eval.ts @@ -2,7 +2,7 @@ import { existsSync, readFileSync } from "node:fs"; import { join } from "node:path"; import { describe, expect } from "vitest"; import { createJudge, describeEval } from "vitest-evals"; -import { createPiCodingAgentHarness, type PiCodingAgentInput } from "./pi-harness.ts"; +import { createPiCodingAgentHarness, excludePiDocumentation, type PiCodingAgentInput } from "./pi-harness.ts"; import { recordEvalSourceArtifact } from "./vitest-evals/artifacts.ts"; import { evalHarnessTable } from "./vitest-evals/harness-table.ts"; @@ -19,14 +19,14 @@ function createExtensionAuthoringHarness(name: string, transformSystemPrompt?: ( return createPiCodingAgentHarness({ name, ...(transformSystemPrompt ? { transformSystemPrompt } : {}), - output: ({ response, session }) => { + output: ({ response, session, systemPrompt }) => { const extensions = session.resourceLoader.getExtensions(); const extensionPath = join(session.sessionManager.getCwd(), ".pi", "extensions", "hello.ts"); const extensionSource = existsSync(extensionPath) ? readFileSync(extensionPath, "utf8") : null; return { response, - systemPromptHasGuidelines: session.systemPrompt.includes("\nGuidelines:\n"), - systemPromptHasPiDocs: session.systemPrompt.includes("\nPi documentation (read only"), + systemPromptHasGuidelines: systemPrompt.includes("\nGuidelines:\n"), + systemPromptHasPiDocs: systemPrompt.includes("\nPi documentation (read only"), extensionErrors: extensions.errors, loadedExtensions: extensions.extensions.map(({ path, tools }) => ({ path, @@ -38,18 +38,6 @@ function createExtensionAuthoringHarness(name: string, transformSystemPrompt?: ( }); } -function excludeGuidelinesAndDocumentation(defaultPrompt: string): string { - const guidelinesStart = defaultPrompt.indexOf("\nGuidelines:\n"); - if (guidelinesStart === -1) throw new Error("Default Pi system prompt has no Guidelines section."); - return defaultPrompt.slice(0, guidelinesStart); -} - -function prepareDefaultPromptOverride(defaultPrompt: string): string { - const cwdStart = defaultPrompt.lastIndexOf("\nCurrent working directory: "); - if (cwdStart === -1) throw new Error("Default Pi system prompt has no working-directory section."); - return defaultPrompt.slice(0, cwdStart); -} - const ExtensionAuthoringJudge = createJudge( "ExtensionAuthoringJudge", ({ output, toolCalls }) => { @@ -97,17 +85,17 @@ const ExtensionAuthoringJudge = createJudge { describeEval( - "Pi extension authoring system prompt", + "Create and use a tool extension", { harness, judges: [ExtensionAuthoringJudge], judgeThreshold: null }, (it) => { - it("creates, reloads, and uses a hello extension", async ({ run, task }) => { + it("creates and uses the extension", async ({ run, task }) => { const result = await run([ { type: "prompt", @@ -131,9 +119,8 @@ describe.for(extensionHarnessTable)("$name", ({ harness }) => { bodyEncoding: "utf-8", }); } - const expectsFullPrompt = harness.name === "default-system-prompt"; - expect(result.output.systemPromptHasGuidelines).toBe(expectsFullPrompt); - expect(result.output.systemPromptHasPiDocs).toBe(expectsFullPrompt); + expect(result.output.systemPromptHasGuidelines).toBe(true); + expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt"); }); }, ); diff --git a/packages/evals/src/models.eval.ts b/packages/evals/src/models.eval.ts new file mode 100644 index 000000000..419cfaa4d --- /dev/null +++ b/packages/evals/src/models.eval.ts @@ -0,0 +1,151 @@ +import { deepStrictEqual } from "node:assert/strict"; +import { join } from "node:path"; +import type { Api, Model } from "@earendil-works/pi-ai"; +import { ModelRuntime } from "@earendil-works/pi-coding-agent"; +import { describe, expect } from "vitest"; +import { createJudge, describeEval } from "vitest-evals"; +import { createPiCodingAgentHarness, excludePiDocumentation, type PiCodingAgentInput } from "./pi-harness.ts"; +import { evalHarnessTable } from "./vitest-evals/harness-table.ts"; + +const PROVIDER_ID = "openai"; +const MODEL_ID = "fixture-chat"; +const MODEL_NAME = "Fixture Chat"; + +type ModelSummary = { + id: string; + name: string; + provider: string; + reasoning: boolean; + input: Array<"text" | "image">; + cost: { input: number; output: number; cacheRead: number; cacheWrite: number }; + contextWindow: number; + maxTokens: number; +}; + +type ModelAuthoringResult = { model: ModelSummary; existingModelsPreserved: boolean } | { error: string }; + +type ModelAuthoringOutput = { + systemPromptHasGuidelines: boolean; + systemPromptHasPiDocs: boolean; + result: ModelAuthoringResult; +}; + +function summarizeModel(model: Model): ModelSummary { + return { + id: model.id, + name: model.name, + provider: model.provider, + reasoning: model.reasoning, + input: [...model.input], + cost: { + input: model.cost.input, + output: model.cost.output, + cacheRead: model.cost.cacheRead, + cacheWrite: model.cost.cacheWrite, + }, + contextWindow: model.contextWindow, + maxTokens: model.maxTokens, + }; +} + +function errorMessage(error: unknown): string { + return error instanceof Error ? error.message : String(error); +} + +function createModelAuthoringHarness(name: string, transformSystemPrompt?: (defaultPrompt: string) => string) { + return createPiCodingAgentHarness({ + name, + ...(transformSystemPrompt ? { transformSystemPrompt } : {}), + output: async ({ session, systemPrompt, agentDir }) => { + let result: ModelAuthoringResult; + try { + const pristineRuntime = await ModelRuntime.create({ modelsPath: null, allowModelNetwork: false }); + const existingModelIds = + pristineRuntime + .getProvider(PROVIDER_ID) + ?.getModels() + .map(({ id }) => id) ?? []; + let runtime = session.modelRuntime; + if (!runtime.getModel(PROVIDER_ID, MODEL_ID)) { + runtime = await ModelRuntime.create({ + modelsPath: join(agentDir, "models.json"), + authPath: join(agentDir, "auth.json"), + modelsStorePath: join(agentDir, "models-store.json"), + allowModelNetwork: false, + }); + } + const configurationError = runtime.getError(); + if (configurationError) throw new Error(configurationError); + const model = runtime.getModel(PROVIDER_ID, MODEL_ID); + if (!model) throw new Error(`Model ${PROVIDER_ID}/${MODEL_ID} is unavailable after reload.`); + result = { + model: summarizeModel(model), + existingModelsPreserved: + existingModelIds.length > 0 && + existingModelIds.every((id) => runtime.getModel(PROVIDER_ID, id) !== undefined), + }; + } catch (error) { + result = { error: errorMessage(error) }; + } + return { + systemPromptHasGuidelines: systemPrompt.includes("\nGuidelines:\n"), + systemPromptHasPiDocs: systemPrompt.includes("\nPi documentation (read only"), + result, + }; + }, + }); +} + +const expectedResult: Exclude = { + model: { + id: MODEL_ID, + name: MODEL_NAME, + provider: PROVIDER_ID, + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 32768, + maxTokens: 4096, + }, + existingModelsPreserved: true, +}; + +const ModelAuthoringJudge = createJudge( + "ModelAuthoringJudge", + ({ output }) => { + if ("error" in output.result) { + return { score: 0, metadata: { rationale: output.result.error } }; + } + try { + deepStrictEqual(output.result, expectedResult); + return { score: 1, metadata: { rationale: "Model was added to the existing provider." } }; + } catch (error) { + return { score: 0, metadata: { rationale: errorMessage(error) } }; + } + }, +); + +const modelHarnessTable = evalHarnessTable("Add model to existing provider", { + baseline: createModelAuthoringHarness("system-prompt-without-docs", excludePiDocumentation), + candidate: createModelAuthoringHarness("default-system-prompt"), +}); + +describe.for(modelHarnessTable)("$name", ({ harness }) => { + describeEval( + "Add model to existing provider", + { harness, judges: [ModelAuthoringJudge], judgeThreshold: null }, + (it) => { + it("adds the model", { timeout: 300_000 }, async ({ run }) => { + const result = await run([ + { + type: "prompt", + content: `Configure Pi with a new \`${PROVIDER_ID}/${MODEL_ID}\` model. Show it as “${MODEL_NAME}”. It accepts text, supports reasoning, has a 32,768-token context window and a 4,096-token maximum output, and has no usage cost.`, + }, + { type: "reload" }, + ]); + expect(result.output.systemPromptHasGuidelines).toBe(true); + expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt"); + }); + }, + ); +}); diff --git a/packages/evals/src/pi-harness.ts b/packages/evals/src/pi-harness.ts index d1a54830a..9942e18f2 100644 --- a/packages/evals/src/pi-harness.ts +++ b/packages/evals/src/pi-harness.ts @@ -9,6 +9,7 @@ import { type CreateAgentSessionOptions, createAgentSessionFromServices, createAgentSessionServices, + type InlineExtension, ModelRuntime, SessionManager, SettingsManager, @@ -42,9 +43,25 @@ type PiCodingAgentHarnessOptions = { }; type PiCodingAgentHarnessWithOutput = PiCodingAgentHarnessOptions & { - output: (args: { response: string; session: AgentSession }) => TOutput | Promise; + output: (args: { + response: string; + session: AgentSession; + systemPrompt: string; + agentDir: string; + }) => TOutput | Promise; }; +// Comparative evals intentionally remove the documentation block using stable prompt markers instead of changing Pi's +// production prompt builder. The isolated eval prompt has no project context or skills between these markers. If +// that setup changes, this transform must be updated so baseline and candidate still differ only by documentation. +export function excludePiDocumentation(defaultPrompt: string): string { + const documentationStart = defaultPrompt.indexOf("\nPi documentation (read only"); + if (documentationStart === -1) throw new Error("Default Pi system prompt has no Pi documentation section."); + const cwdStart = defaultPrompt.lastIndexOf("\nCurrent working directory: "); + if (cwdStart === -1) throw new Error("Default Pi system prompt has no working-directory section."); + return defaultPrompt.slice(0, documentationStart) + defaultPrompt.slice(cwdStart); +} + export function resolveModelSelection( explicitModel: PiCodingAgentModelSelection | undefined, environment: { PI_PROVIDER?: string; PI_MODEL?: string } = process.env, @@ -123,21 +140,36 @@ async function runPiCodingAgent( const root = await mkdtemp(join(tmpdir(), "pi-eval-")); const cwd = join(root, "workspace"); - const agentDir = join(root, "agent"); - let transformedSystemPrompt: string | undefined; + const isolatedHome = join(root, "home"); + const agentDir = join(isolatedHome, ".pi", "agent"); + const transformSystemPrompt = options.transformSystemPrompt; + let evaluatedSystemPrompt: string | undefined; + const extensionFactories: InlineExtension[] = []; + if (transformSystemPrompt) { + extensionFactories.push({ + name: "eval-system-prompt-transform", + hidden: true, + factory: (pi) => { + pi.on("before_agent_start", (event) => { + evaluatedSystemPrompt = transformSystemPrompt(event.systemPrompt); + return { systemPrompt: evaluatedSystemPrompt }; + }); + }, + }); + } let sessionManager: SessionManager | undefined; let session: AgentSession | undefined; let outcome: { success: true; result: SimpleHarnessResult } | { success: false; error: unknown }; try { - await Promise.all([mkdir(cwd), mkdir(agentDir)]); + await Promise.all([mkdir(cwd), mkdir(agentDir, { recursive: true })]); const services = await createAgentSessionServices({ cwd, agentDir, modelRuntime, - settingsManager: SettingsManager.inMemory(), - ...(options.transformSystemPrompt - ? { resourceLoaderOptions: { systemPromptOverride: () => transformedSystemPrompt } } - : {}), + settingsManager: SettingsManager.inMemory({ + shellCommandPrefix: `export HOME=${JSON.stringify(isolatedHome)}; unset PI_CODING_AGENT_DIR PI_EVAL_ARTIFACT_DIR PI_MODEL PI_PROVIDER PI_REASONING_LEVEL PI_SESSION_FILE PI_SESSION_ID;`, + }), + ...(extensionFactories.length > 0 ? { resourceLoaderOptions: { extensionFactories } } : {}), }); signal?.throwIfAborted(); sessionManager = SessionManager.create(cwd, join(root, "sessions")); @@ -155,11 +187,6 @@ async function runPiCodingAgent( ).session; const evalSession = session; - if (options.transformSystemPrompt) { - transformedSystemPrompt = options.transformSystemPrompt(evalSession.systemPrompt); - if (!transformedSystemPrompt.trim()) throw new Error("Transformed eval system prompt must not be empty."); - await evalSession.reload(); - } let abortPromise: Promise | undefined; const abort = () => { abortPromise ??= evalSession.abort(); @@ -167,7 +194,10 @@ async function runPiCodingAgent( signal?.addEventListener("abort", abort, { once: true }); try { signal?.throwIfAborted(); - if (evalSession.extensionRunner.getExtensionPaths().length !== 0) { + const unexpectedExtensionPaths = evalSession.extensionRunner + .getExtensionPaths() + .filter((path) => path !== ""); + if (unexpectedExtensionPaths.length !== 0) { throw new Error("Expected an isolated eval session to start without extensions."); } const steps = typeof input === "string" ? [{ type: "prompt" as const, content: input }] : input; @@ -175,12 +205,23 @@ async function runPiCodingAgent( for (const step of steps) { if (step.type === "prompt") { response = await promptAgent(evalSession, step.content, signal); + if (transformSystemPrompt && !evaluatedSystemPrompt?.trim()) { + throw new Error("System-prompt transform did not produce a non-empty prompt."); + } } else { await evalSession.reload(); } } if (response === undefined) throw new Error("Pi eval input must include at least one prompt step."); - const output = "output" in options ? await options.output({ response, session: evalSession }) : response; + const output = + "output" in options + ? await options.output({ + response, + session: evalSession, + systemPrompt: evaluatedSystemPrompt ?? evalSession.systemPrompt, + agentDir, + }) + : response; const stats = evalSession.getSessionStats(); const hasPricing = [model.cost, ...(model.cost.tiers ?? [])].some( ({ input, output, cacheRead, cacheWrite }) => input > 0 || output > 0 || cacheRead > 0 || cacheWrite > 0, diff --git a/packages/evals/src/providers.eval.ts b/packages/evals/src/providers.eval.ts new file mode 100644 index 000000000..c6101631b --- /dev/null +++ b/packages/evals/src/providers.eval.ts @@ -0,0 +1,410 @@ +import { deepStrictEqual } from "node:assert/strict"; +import { createServer, type IncomingMessage, type Server, type ServerResponse } from "node:http"; +import type { AddressInfo } from "node:net"; +import { join } from "node:path"; +import { type Api, type Context, contentText, type Model, type ModelsSimpleStreamOptions } from "@earendil-works/pi-ai"; +import { type AgentSession, ModelRuntime } from "@earendil-works/pi-coding-agent"; +import { afterAll, beforeAll, beforeEach, describe, expect } from "vitest"; +import { createJudge, describeEval } from "vitest-evals"; +import { createPiCodingAgentHarness, excludePiDocumentation, type PiCodingAgentInput } from "./pi-harness.ts"; +import { evalHarnessTable } from "./vitest-evals/harness-table.ts"; + +const PROVIDER_ID = "acme"; +const MODEL_ID = "acme-chat"; +const PROBE_PROMPT = "Reply with ACME_OK."; +const PROBE_RESPONSE = "ACME_OK"; +const CUSTOM_PROVIDER_ID = "acme-stream"; +const CUSTOM_MODEL_ID = "acme-stream-chat"; +const CUSTOM_PROBE_PROMPT = "Reply with ACME_STREAM_OK."; +const CUSTOM_PROBE_RESPONSE = "ACME_STREAM_OK"; + +let acmeServer: Server | undefined; +let acmeOrigin = ""; +let acmeBaseUrl = ""; +let validAcmeRequestReceived = false; +let validAcmeStreamRequestReceived = false; + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function rejectRequest(response: ServerResponse, status: number, message: string): void { + response.writeHead(status, { "content-type": "text/plain" }); + response.end(message); +} + +async function handleAcmeRequest(request: IncomingMessage, response: ServerResponse): Promise { + if (request.method === "GET" && request.url === "/docs") { + response.writeHead(200, { "content-type": "application/json" }); + response.end( + JSON.stringify({ + name: "Acme Streaming API", + request: { + method: "POST", + path: "/generate", + headers: { "content-type": "application/json", "x-acme-key": "resolved credential" }, + body: { + model: CUSTOM_MODEL_ID, + messages: [{ role: "user", content: "Hello" }], + stream: true, + }, + }, + response: { + contentType: "application/x-ndjson", + events: [ + { type: "text_delta", text: "Hello" }, + { type: "usage", input_tokens: 3, output_tokens: 2 }, + { type: "done", reason: "stop" }, + ], + }, + }), + ); + return; + } + + if (request.url !== "/generate" && request.url !== "/v1/chat/completions") { + rejectRequest(response, 404, "Unknown endpoint"); + return; + } + if (request.method !== "POST") { + rejectRequest(response, 405, "Expected POST"); + return; + } + if (!request.headers["content-type"]?.startsWith("application/json")) { + rejectRequest(response, 415, "Expected application/json"); + return; + } + + let body = ""; + for await (const chunk of request) body += chunk.toString(); + let payload: unknown; + try { + payload = JSON.parse(body); + } catch { + rejectRequest(response, 400, "Invalid JSON"); + return; + } + if (!isRecord(payload)) { + rejectRequest(response, 422, "Expected a JSON object"); + return; + } + const messages: unknown[] = Array.isArray(payload.messages) ? payload.messages : []; + const userMessage = messages.find( + (message): message is Record => isRecord(message) && message.role === "user", + ); + const userPrompt = typeof userMessage?.content === "string" ? userMessage.content : null; + + if (request.url === "/generate") { + if (request.headers["x-acme-key"] !== "resolved-stream-key") { + rejectRequest(response, 401, "Invalid Acme Stream credential"); + return; + } + if (payload.model !== CUSTOM_MODEL_ID || userPrompt === null || payload.stream !== true) { + rejectRequest(response, 422, "Invalid Acme Stream request"); + return; + } + validAcmeStreamRequestReceived = userPrompt === CUSTOM_PROBE_PROMPT; + response.writeHead(200, { "content-type": "application/x-ndjson" }); + response.write(`${JSON.stringify({ type: "text_delta", text: "ACME_" })}\n`); + response.write(`${JSON.stringify({ type: "text_delta", text: "STREAM_OK" })}\n`); + response.write(`${JSON.stringify({ type: "usage", input_tokens: 4, output_tokens: 3 })}\n`); + response.end(`${JSON.stringify({ type: "done", reason: "stop" })}\n`); + return; + } + + if (request.headers.authorization !== "Bearer resolved-acme-key") { + rejectRequest(response, 401, "Invalid Acme credential"); + return; + } + if (payload.model !== MODEL_ID || userPrompt === null || payload.stream !== true) { + rejectRequest(response, 422, "Invalid OpenAI-compatible request"); + return; + } + validAcmeRequestReceived = userPrompt === PROBE_PROMPT; + response.writeHead(200, { "content-type": "text/event-stream", "cache-control": "no-cache" }); + response.write( + `data: ${JSON.stringify({ + id: "chatcmpl-acme", + object: "chat.completion.chunk", + created: 0, + model: MODEL_ID, + choices: [{ index: 0, delta: { role: "assistant", content: PROBE_RESPONSE }, finish_reason: null }], + })}\n\n`, + ); + response.write( + `data: ${JSON.stringify({ + id: "chatcmpl-acme", + object: "chat.completion.chunk", + created: 0, + model: MODEL_ID, + choices: [{ index: 0, delta: {}, finish_reason: "stop" }], + usage: { prompt_tokens: 3, completion_tokens: 2 }, + })}\n\n`, + ); + response.end("data: [DONE]\n\n"); +} + +beforeAll(async () => { + acmeServer = createServer((request, response) => { + void handleAcmeRequest(request, response); + }); + await new Promise((resolve, reject) => { + acmeServer!.once("error", reject); + acmeServer!.listen(0, "127.0.0.1", resolve); + }); + const address = acmeServer.address(); + if (!address || typeof address === "string") throw new Error("Fake Acme server did not bind a TCP port."); + acmeOrigin = `http://127.0.0.1:${(address as AddressInfo).port}`; + acmeBaseUrl = `${acmeOrigin}/v1`; +}); + +beforeEach(() => { + validAcmeRequestReceived = false; + validAcmeStreamRequestReceived = false; +}); + +afterAll(async () => { + if (!acmeServer) return; + await new Promise((resolve, reject) => { + acmeServer!.close((error) => (error ? reject(error) : resolve())); + }); +}); + +type ProviderRuntimeSuccess = { + validRequestReceived: boolean; + provider: { id: string; name: string }; + model: { + id: string; + name: string; + provider: string; + reasoning: boolean; + input: Array<"text" | "image">; + cost: { input: number; output: number; cacheRead: number; cacheWrite: number }; + contextWindow: number; + maxTokens: number; + }; + response: { + text: string; + stopReason: string; + inputTokens: number; + outputTokens: number; + }; +}; + +type ProviderRuntimeResult = ProviderRuntimeSuccess | { error: string }; + +type ProviderRuntimeOutput = { + systemPromptHasGuidelines: boolean; + systemPromptHasPiDocs: boolean; + result: ProviderRuntimeResult; +}; + +type ProviderScenario = { + providerId: string; + modelId: string; + createContext: () => Context; + options?: ModelsSimpleStreamOptions; + validRequestReceived: () => boolean; +}; + +type RuntimeResolver = (session: AgentSession, agentDir: string) => Promise; + +function summarizeModel(model: Model): ProviderRuntimeSuccess["model"] { + return { + id: model.id, + name: model.name, + provider: model.provider, + reasoning: model.reasoning, + input: [...model.input], + cost: { + input: model.cost.input, + output: model.cost.output, + cacheRead: model.cost.cacheRead, + cacheWrite: model.cost.cacheWrite, + }, + contextWindow: model.contextWindow, + maxTokens: model.maxTokens, + }; +} + +function errorMessage(error: unknown): string { + return error instanceof Error ? error.message : String(error); +} + +async function probeProvider(runtime: ModelRuntime, scenario: ProviderScenario): Promise { + const configurationError = runtime.getError(); + if (configurationError) throw new Error(configurationError); + const provider = runtime.getProvider(scenario.providerId); + const model = runtime.getModel(scenario.providerId, scenario.modelId); + if (!provider || !model) { + throw new Error(`Model ${scenario.providerId}/${scenario.modelId} is unavailable after reload.`); + } + const response = await runtime.completeSimple(model, scenario.createContext(), scenario.options); + return { + validRequestReceived: scenario.validRequestReceived(), + provider: { id: provider.id, name: provider.name }, + model: summarizeModel(model), + response: { + text: contentText(response.content), + stopReason: response.stopReason, + inputTokens: response.usage.input, + outputTokens: response.usage.output, + }, + }; +} + +function createProviderHarness( + name: string, + scenario: ProviderScenario, + transformSystemPrompt?: (defaultPrompt: string) => string, + resolveRuntime: RuntimeResolver = async (session) => session.modelRuntime, +) { + return createPiCodingAgentHarness({ + name, + ...(transformSystemPrompt ? { transformSystemPrompt } : {}), + output: async ({ session, systemPrompt, agentDir }) => { + let result: ProviderRuntimeResult; + try { + result = await probeProvider(await resolveRuntime(session, agentDir), scenario); + } catch (error) { + result = { error: errorMessage(error) }; + } + return { + systemPromptHasGuidelines: systemPrompt.includes("\nGuidelines:\n"), + systemPromptHasPiDocs: systemPrompt.includes("\nPi documentation (read only"), + result, + }; + }, + }); +} + +function createProviderRuntimeJudge(expected: ProviderRuntimeSuccess) { + return createJudge("ProviderRuntimeJudge", ({ output }) => { + if ("error" in output.result) { + return { score: 0, metadata: { rationale: output.result.error } }; + } + try { + deepStrictEqual(output.result, expected); + return { score: 1, metadata: { rationale: "Provider works through Pi." } }; + } catch (error) { + return { score: 0, metadata: { rationale: errorMessage(error) } }; + } + }); +} + +const providerScenario: ProviderScenario = { + providerId: PROVIDER_ID, + modelId: MODEL_ID, + createContext: () => ({ messages: [{ role: "user", content: PROBE_PROMPT, timestamp: Date.now() }] }), + options: { env: { ACME_API_KEY: "resolved-acme-key" }, maxTokens: 32 }, + validRequestReceived: () => validAcmeRequestReceived, +}; + +const customProviderScenario: ProviderScenario = { + providerId: CUSTOM_PROVIDER_ID, + modelId: CUSTOM_MODEL_ID, + createContext: () => ({ messages: [{ role: "user", content: CUSTOM_PROBE_PROMPT, timestamp: Date.now() }] }), + options: { env: { ACME_STREAM_API_KEY: "resolved-stream-key" }, maxTokens: 32 }, + validRequestReceived: () => validAcmeStreamRequestReceived, +}; + +const ProviderAuthoringJudge = createProviderRuntimeJudge({ + validRequestReceived: true, + provider: { id: PROVIDER_ID, name: "Acme" }, + model: { + id: MODEL_ID, + name: "Acme Chat", + provider: PROVIDER_ID, + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 32768, + maxTokens: 4096, + }, + response: { text: PROBE_RESPONSE, stopReason: "stop", inputTokens: 3, outputTokens: 2 }, +}); + +const CustomProviderJudge = createProviderRuntimeJudge({ + validRequestReceived: true, + provider: { id: CUSTOM_PROVIDER_ID, name: "Acme Stream" }, + model: { + id: CUSTOM_MODEL_ID, + name: "Acme Stream Chat", + provider: CUSTOM_PROVIDER_ID, + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 16384, + maxTokens: 2048, + }, + response: { text: CUSTOM_PROBE_RESPONSE, stopReason: "stop", inputTokens: 4, outputTokens: 3 }, +}); + +const resolveProviderRuntime: RuntimeResolver = async (session, agentDir) => { + if (session.modelRuntime.getModel(PROVIDER_ID, MODEL_ID)) return session.modelRuntime; + return ModelRuntime.create({ + modelsPath: join(agentDir, "models.json"), + authPath: join(agentDir, "auth.json"), + modelsStorePath: join(agentDir, "models-store.json"), + allowModelNetwork: false, + }); +}; + +const providerHarnessTable = evalHarnessTable("Add OpenAI-compatible provider", { + baseline: createProviderHarness( + "system-prompt-without-docs", + providerScenario, + excludePiDocumentation, + resolveProviderRuntime, + ), + candidate: createProviderHarness("default-system-prompt", providerScenario, undefined, resolveProviderRuntime), +}); + +describe.for(providerHarnessTable)("$name", ({ harness }) => { + describeEval( + "Add OpenAI-compatible provider", + { harness, judges: [ProviderAuthoringJudge], judgeThreshold: null }, + (it) => { + it("adds the provider", { timeout: 300_000 }, async ({ run }) => { + const result = await run([ + { + type: "prompt", + content: `Can you add Acme to Pi as a provider? Its provider ID is ${PROVIDER_ID}, its API is at ${acmeBaseUrl}, and it uses OpenAI Chat Completions. Read its API key from the ACME_API_KEY environment variable. + +The provider offers one model, ${MODEL_ID}, shown as “Acme Chat”. It accepts text, does not support reasoning, has a 32,768-token context window and a 4,096-token maximum output, and has no usage cost.`, + }, + { type: "reload" }, + ]); + expect(result.output.systemPromptHasGuidelines).toBe(true); + expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt"); + }); + }, + ); +}); + +const customProviderHarnessTable = evalHarnessTable("Add custom streaming provider", { + baseline: createProviderHarness("system-prompt-without-docs", customProviderScenario, excludePiDocumentation), + candidate: createProviderHarness("default-system-prompt", customProviderScenario), +}); + +describe.for(customProviderHarnessTable)("$name", ({ harness }) => { + describeEval( + "Add custom streaming provider", + { harness, judges: [CustomProviderJudge], judgeThreshold: null }, + (it) => { + it("adds the provider", { timeout: 300_000 }, async ({ run }) => { + const result = await run([ + { + type: "prompt", + content: `Can you add Acme Stream to Pi as a provider? Its provider ID is ${CUSTOM_PROVIDER_ID}, its API is at ${acmeOrigin}, and its documentation is available at ${acmeOrigin}/docs. Read its credential from the ACME_STREAM_API_KEY environment variable. + +It offers one model, ${CUSTOM_MODEL_ID}, shown as “Acme Stream Chat”. The model accepts text, does not support reasoning, has a 16,384-token context window and a 2,048-token maximum output, and has no usage cost.`, + }, + { type: "reload" }, + ]); + expect(result.output.systemPromptHasGuidelines).toBe(true); + expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt"); + }); + }, + ); +}); diff --git a/packages/evals/src/smoke.eval.ts b/packages/evals/src/smoke.eval.ts index 3dd8920a9..f98da549f 100644 --- a/packages/evals/src/smoke.eval.ts +++ b/packages/evals/src/smoke.eval.ts @@ -4,8 +4,8 @@ import { createPiCodingAgentHarness } from "./pi-harness.ts"; const piCodingAgentHarness = createPiCodingAgentHarness({ noTools: "all" }); -describeEval("Pi Coding Agent smoke", { harness: piCodingAgentHarness }, (it) => { - it("runs a basic prompt end to end", async ({ run }) => { +describeEval("Answer a basic prompt", { harness: piCodingAgentHarness }, (it) => { + it("returns the expected answer", async ({ run }) => { const result = await run("What's the capital of France? Respond with only the city name."); expect(result.output.trim()).toBe("Paris"); diff --git a/packages/evals/src/vitest-evals/harness-table.ts b/packages/evals/src/vitest-evals/harness-table.ts index 4551cf9d8..27f005538 100644 --- a/packages/evals/src/vitest-evals/harness-table.ts +++ b/packages/evals/src/vitest-evals/harness-table.ts @@ -111,6 +111,17 @@ export function deriveEvalGroupKey(input: unknown, repetition: number): string { return JSON.stringify([deriveInputKey(input), repetition]); } +export function resolveEvalRepetitions( + explicit: number | undefined, + environmentValue: string | undefined = process.env.PI_EVAL_REPETITIONS, +): number { + const repetitions = explicit ?? (environmentValue === undefined ? 1 : Number(environmentValue)); + if (!Number.isSafeInteger(repetitions) || repetitions < 1) { + throw new TypeError("repetitions must be a positive integer."); + } + return repetitions; +} + function validateOptions( evalSet: string, baseline: Harness, @@ -166,7 +177,7 @@ export function evalHarnessTable( evalSet: string, options: EvalHarnessTableOptions, ): EvalHarnessTableRow[] { - const repetitions = options.repetitions ?? 1; + const repetitions = resolveEvalRepetitions(options.repetitions); const candidates = "candidate" in options ? [options.candidate] : options.candidates; validateOptions(evalSet, options.baseline, candidates, repetitions); diff --git a/packages/evals/src/vitest-evals/reporter.ts b/packages/evals/src/vitest-evals/reporter.ts index b1d279aa7..da0fccbdd 100644 --- a/packages/evals/src/vitest-evals/reporter.ts +++ b/packages/evals/src/vitest-evals/reporter.ts @@ -1,6 +1,7 @@ import { randomUUID } from "node:crypto"; -import { appendFile, mkdir } from "node:fs/promises"; +import { appendFile, mkdir, writeFile } from "node:fs/promises"; import { join } from "node:path"; +import { stripVTControlCharacters } from "node:util"; import type { Reporter, SerializedError, TestCase, TestModule, TestRunEndReason, Vitest } from "vitest/node"; import { isHarnessRun } from "vitest-evals/harness"; import { PI_SESSION_SNAPSHOT_ARTIFACT, persistEvalArtifactReferences } from "./artifacts.ts"; @@ -95,11 +96,11 @@ export default class EvalHarnessReporter implements Reporter { await appendHarnessRunReport(test); } - onTestRunEnd( + async onTestRunEnd( modules: ReadonlyArray, _errors: ReadonlyArray, reason: TestRunEndReason, - ): void { + ): Promise { if (reason === "interrupted") { this.vitest?.logger.log("\nEval comparisons unavailable: test run interrupted."); return; @@ -107,5 +108,17 @@ export default class EvalHarnessReporter implements Reporter { const report = summarizeHarnessComparisons(collectHarnessObservations(modules)); const formatted = formatHarnessComparisonReport(report); if (formatted) this.vitest?.logger.log(`\n${formatted}`); + const artifactDirectory = process.env.PI_EVAL_ARTIFACT_DIR?.trim(); + if (artifactDirectory) { + await mkdir(artifactDirectory, { recursive: true, mode: 0o700 }); + await Promise.all([ + writeFile(join(artifactDirectory, "report.json"), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 }), + writeFile( + join(artifactDirectory, "report.txt"), + formatted ? `${stripVTControlCharacters(formatted)}\n` : "", + { mode: 0o600 }, + ), + ]); + } } } diff --git a/packages/evals/test/vitest-evals/harness-table.test.ts b/packages/evals/test/vitest-evals/harness-table.test.ts index 13ad6ab8d..52275fa7d 100644 --- a/packages/evals/test/vitest-evals/harness-table.test.ts +++ b/packages/evals/test/vitest-evals/harness-table.test.ts @@ -5,6 +5,7 @@ import { EVAL_HARNESS_ITERATION_ARTIFACT, evalHarnessTable, parseEvalHarnessIterationArtifact, + resolveEvalRepetitions, } from "../../src/vitest-evals/harness-table.ts"; describe("deriveEvalGroupKey", () => { @@ -30,6 +31,18 @@ describe("deriveEvalGroupKey", () => { }); }); +describe("resolveEvalRepetitions", () => { + it("uses an explicit value before the environment default", () => { + expect(resolveEvalRepetitions(3, "5")).toBe(3); + expect(resolveEvalRepetitions(undefined, "5")).toBe(5); + expect(resolveEvalRepetitions(undefined, undefined)).toBe(1); + }); + + it.each(["0", "-1", "1.5", "nope"])("rejects invalid environment value %s", (value) => { + expect(() => resolveEvalRepetitions(undefined, value)).toThrow("positive integer"); + }); +}); + function createFakeHarness(name: string) { return createHarness<{ id: string }, { harness: string; inputId: string }>({ name, diff --git a/vitest.base.ts b/vitest.base.ts index 9394bcb1e..67285273a 100644 --- a/vitest.base.ts +++ b/vitest.base.ts @@ -13,6 +13,7 @@ export const workspaceSourcePaths = { aiCompat: fileURLToPath(new URL("./packages/ai/src/compat.ts", import.meta.url)), aiOAuth: fileURLToPath(new URL("./packages/ai/src/oauth.ts", import.meta.url)), aiProviders: fileURLToPath(new URL("./packages/ai/src/providers", import.meta.url)), + aiUtils: fileURLToPath(new URL("./packages/ai/src/utils", import.meta.url)), agentIndex: fileURLToPath(new URL("./packages/agent/src/index.ts", import.meta.url)), agentNode: fileURLToPath(new URL("./packages/agent/src/node.ts", import.meta.url)), protocolIndex: fileURLToPath(new URL("./packages/protocol/src/index.ts", import.meta.url)), @@ -37,6 +38,10 @@ export default defineConfig({ { find: /^@earendil-works\/pi-ai$/, replacement: workspaceSourcePaths.aiIndex }, { find: /^@earendil-works\/pi-ai\/compat$/, replacement: workspaceSourcePaths.aiCompat }, { find: /^@earendil-works\/pi-ai\/oauth$/, replacement: workspaceSourcePaths.aiOAuth }, + { + find: /^@earendil-works\/pi-ai\/utils\/(.+)$/, + replacement: `${workspaceSourcePaths.aiUtils}/$1.ts`, + }, { find: /^@earendil-works\/pi-ai\/providers\/(.+)$/, replacement: `${workspaceSourcePaths.aiProviders}/$1.ts`,