feat(coding-agent): add customization documentation evals (#9491)

* fix(coding-agent): isolate extension eval documentation treatment

* feat(coding-agent): add custom provider documentation eval

* feat(coding-agent): add customization documentation evals

* feat(coding-agent): add repeatable eval reports
This commit is contained in:
Christian Klotz
2026-09-11 17:29:34 +02:00
committed by GitHub
parent 713bdf38d5
commit b215884021
12 changed files with 747 additions and 50 deletions
+47 -4
View File
@@ -26,13 +26,56 @@ Additional arguments are forwarded to Vitest:
```bash
npm run eval -- src/extensions.eval.ts
npm run eval -- -t "creates, reloads, and uses"
npm run eval -- -t "creates and uses the extension"
npm run eval -- src/docs.eval.ts -t "session-format\.md"
```
Each invocation prints an ignored `.eval/` artifact directory. `runs.jsonl` indexes completed harness runs and their
native Pi session JSONL attachments under `sessions/`. These files may contain prompts, responses, source code, and tool
output.
Run all comparative customization evals five times in one invocation:
```bash
npm run eval -- \
src/extensions.eval.ts src/models.eval.ts src/providers.eval.ts \
--provider openai --model gpt-5.6-sol \
--repetitions 5
```
`--repetitions` applies to suites declared with `evalHarnessTable(...)`. An explicit `repetitions` value in a suite
overrides the command-line default. `PI_EVAL_REPETITIONS=5` is equivalent to the command-line option. Use one repetition
while developing an eval and five when reporting lift.
## Reports and artifacts
Each invocation prints a compound `Eval Comparisons` report after the Vitest results. When several comparative files run
in the same invocation, this report contains one section for every eval set. For example, with illustrative values:
```text
Eval Comparisons
Add model to existing provider
Baseline system-prompt-without-docs
Candidate default-system-prompt (5/5 pairs)
Pass rate +60.0 pp (candidate 80.0%, baseline 20.0%)
Tokens +1200.0 (candidate 24000.0, baseline 22800.0)
Latency -850.0ms (candidate 14000.0ms, baseline 14850.0ms)
Est. cost +$0.0100 (candidate $0.1200, baseline $0.1100)
Add OpenAI-compatible provider
...
Add custom streaming provider
...
```
The runner prints the ignored `.eval/` artifact directory at startup. It contains:
- `report.txt`: the terminal comparison report without color codes.
- `report.json`: the same aggregate comparison data as structured JSON.
- `runs.jsonl`: one record for every completed harness run.
- `sessions/`: native Pi session JSONL attachments.
- `sources/`: source attachments recorded by individual evals.
The report covers comparative suites using `evalHarnessTable(...)`. Ordinary evals still appear in the Vitest summary and
in `runs.jsonl`, but not in the baseline-versus-candidate comparison report. Artifacts may contain prompts, responses,
source code, and tool output.
## Writing evals
+23
View File
@@ -16,6 +16,7 @@ const artifactDirectory = process.env.PI_EVAL_ARTIFACT_DIR
const args = process.argv.slice(2);
let provider;
let model;
let repetitions;
let hasCliModelSelection = false;
const vitestArgs = [];
@@ -33,6 +34,16 @@ for (let index = 0; index < args.length; index += 1) {
index += 1;
continue;
}
if (arg === "--repetitions") {
const value = args[index + 1];
if (!value || value.startsWith("-")) {
console.error("Missing value for --repetitions");
process.exit(1);
}
repetitions = value;
index += 1;
continue;
}
if (arg.startsWith("--provider=")) {
provider = arg.slice("--provider=".length);
hasCliModelSelection = true;
@@ -43,11 +54,21 @@ for (let index = 0; index < args.length; index += 1) {
hasCliModelSelection = true;
continue;
}
if (arg.startsWith("--repetitions=")) {
repetitions = arg.slice("--repetitions=".length);
continue;
}
vitestArgs.push(arg);
}
provider = provider?.trim() || undefined;
model = model?.trim() || undefined;
const repetitionsText = (repetitions ?? process.env.PI_EVAL_REPETITIONS)?.trim();
const repetitionCount = repetitionsText ? Number(repetitionsText) : 1;
if (!Number.isSafeInteger(repetitionCount) || repetitionCount < 1) {
console.error("Repetitions must be a positive integer.");
process.exit(1);
}
if (hasCliModelSelection) {
if (!provider || !model) {
console.error("CLI model selection requires both --provider and --model.");
@@ -68,10 +89,12 @@ const vitestCliPath = resolve(dirname(vitestPackagePath), "vitest.mjs");
mkdirSync(artifactDirectory, { recursive: true, mode: 0o700 });
console.error(`[eval] default-model=${provider && model ? `${provider}/${model}` : "none"}`);
console.error(`[eval] repetitions=${repetitionCount}`);
console.error(`[eval] artifacts=${artifactDirectory}`);
const childEnvironment = {
...process.env,
PI_EVAL_ARTIFACT_DIR: artifactDirectory,
PI_EVAL_REPETITIONS: String(repetitionCount),
};
if (provider && model) {
childEnvironment.PI_PROVIDER = provider;
+1 -1
View File
@@ -43,7 +43,7 @@ const documentationAuditHarness = createPiCodingAgentHarness({
customTools: [submitDocumentationAuditTool],
});
describeEval("Coding agent documentation", { harness: documentationAuditHarness }, (it) => {
describeEval("Audit documentation against implementation", { harness: documentationAuditHarness }, (it) => {
it.for(documentationPages)("$path matches the implementation", { timeout: 300_000 }, async ({ path }, { run }) => {
const documentationPath = resolve(docsRoot, path);
const result = await run(`Audit one Pi documentation page against the repository implementation.
+11 -24
View File
@@ -2,7 +2,7 @@ import { existsSync, readFileSync } from "node:fs";
import { join } from "node:path";
import { describe, expect } from "vitest";
import { createJudge, describeEval } from "vitest-evals";
import { createPiCodingAgentHarness, type PiCodingAgentInput } from "./pi-harness.ts";
import { createPiCodingAgentHarness, excludePiDocumentation, type PiCodingAgentInput } from "./pi-harness.ts";
import { recordEvalSourceArtifact } from "./vitest-evals/artifacts.ts";
import { evalHarnessTable } from "./vitest-evals/harness-table.ts";
@@ -19,14 +19,14 @@ function createExtensionAuthoringHarness(name: string, transformSystemPrompt?: (
return createPiCodingAgentHarness({
name,
...(transformSystemPrompt ? { transformSystemPrompt } : {}),
output: ({ response, session }) => {
output: ({ response, session, systemPrompt }) => {
const extensions = session.resourceLoader.getExtensions();
const extensionPath = join(session.sessionManager.getCwd(), ".pi", "extensions", "hello.ts");
const extensionSource = existsSync(extensionPath) ? readFileSync(extensionPath, "utf8") : null;
return {
response,
systemPromptHasGuidelines: session.systemPrompt.includes("\nGuidelines:\n"),
systemPromptHasPiDocs: session.systemPrompt.includes("\nPi documentation (read only"),
systemPromptHasGuidelines: systemPrompt.includes("\nGuidelines:\n"),
systemPromptHasPiDocs: systemPrompt.includes("\nPi documentation (read only"),
extensionErrors: extensions.errors,
loadedExtensions: extensions.extensions.map(({ path, tools }) => ({
path,
@@ -38,18 +38,6 @@ function createExtensionAuthoringHarness(name: string, transformSystemPrompt?: (
});
}
function excludeGuidelinesAndDocumentation(defaultPrompt: string): string {
const guidelinesStart = defaultPrompt.indexOf("\nGuidelines:\n");
if (guidelinesStart === -1) throw new Error("Default Pi system prompt has no Guidelines section.");
return defaultPrompt.slice(0, guidelinesStart);
}
function prepareDefaultPromptOverride(defaultPrompt: string): string {
const cwdStart = defaultPrompt.lastIndexOf("\nCurrent working directory: ");
if (cwdStart === -1) throw new Error("Default Pi system prompt has no working-directory section.");
return defaultPrompt.slice(0, cwdStart);
}
const ExtensionAuthoringJudge = createJudge<PiCodingAgentInput, ExtensionAuthoringOutput>(
"ExtensionAuthoringJudge",
({ output, toolCalls }) => {
@@ -97,17 +85,17 @@ const ExtensionAuthoringJudge = createJudge<PiCodingAgentInput, ExtensionAuthori
},
);
const extensionHarnessTable = evalHarnessTable("Pi extension authoring system prompt", {
baseline: createExtensionAuthoringHarness("system-prompt-without-docs", excludeGuidelinesAndDocumentation),
candidate: createExtensionAuthoringHarness("default-system-prompt", prepareDefaultPromptOverride),
const extensionHarnessTable = evalHarnessTable("Create and use a tool extension", {
baseline: createExtensionAuthoringHarness("system-prompt-without-docs", excludePiDocumentation),
candidate: createExtensionAuthoringHarness("default-system-prompt"),
});
describe.for(extensionHarnessTable)("$name", ({ harness }) => {
describeEval(
"Pi extension authoring system prompt",
"Create and use a tool extension",
{ harness, judges: [ExtensionAuthoringJudge], judgeThreshold: null },
(it) => {
it("creates, reloads, and uses a hello extension", async ({ run, task }) => {
it("creates and uses the extension", async ({ run, task }) => {
const result = await run([
{
type: "prompt",
@@ -131,9 +119,8 @@ describe.for(extensionHarnessTable)("$name", ({ harness }) => {
bodyEncoding: "utf-8",
});
}
const expectsFullPrompt = harness.name === "default-system-prompt";
expect(result.output.systemPromptHasGuidelines).toBe(expectsFullPrompt);
expect(result.output.systemPromptHasPiDocs).toBe(expectsFullPrompt);
expect(result.output.systemPromptHasGuidelines).toBe(true);
expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt");
});
},
);
+151
View File
@@ -0,0 +1,151 @@
import { deepStrictEqual } from "node:assert/strict";
import { join } from "node:path";
import type { Api, Model } from "@earendil-works/pi-ai";
import { ModelRuntime } from "@earendil-works/pi-coding-agent";
import { describe, expect } from "vitest";
import { createJudge, describeEval } from "vitest-evals";
import { createPiCodingAgentHarness, excludePiDocumentation, type PiCodingAgentInput } from "./pi-harness.ts";
import { evalHarnessTable } from "./vitest-evals/harness-table.ts";
const PROVIDER_ID = "openai";
const MODEL_ID = "fixture-chat";
const MODEL_NAME = "Fixture Chat";
type ModelSummary = {
id: string;
name: string;
provider: string;
reasoning: boolean;
input: Array<"text" | "image">;
cost: { input: number; output: number; cacheRead: number; cacheWrite: number };
contextWindow: number;
maxTokens: number;
};
type ModelAuthoringResult = { model: ModelSummary; existingModelsPreserved: boolean } | { error: string };
type ModelAuthoringOutput = {
systemPromptHasGuidelines: boolean;
systemPromptHasPiDocs: boolean;
result: ModelAuthoringResult;
};
function summarizeModel(model: Model<Api>): ModelSummary {
return {
id: model.id,
name: model.name,
provider: model.provider,
reasoning: model.reasoning,
input: [...model.input],
cost: {
input: model.cost.input,
output: model.cost.output,
cacheRead: model.cost.cacheRead,
cacheWrite: model.cost.cacheWrite,
},
contextWindow: model.contextWindow,
maxTokens: model.maxTokens,
};
}
function errorMessage(error: unknown): string {
return error instanceof Error ? error.message : String(error);
}
function createModelAuthoringHarness(name: string, transformSystemPrompt?: (defaultPrompt: string) => string) {
return createPiCodingAgentHarness({
name,
...(transformSystemPrompt ? { transformSystemPrompt } : {}),
output: async ({ session, systemPrompt, agentDir }) => {
let result: ModelAuthoringResult;
try {
const pristineRuntime = await ModelRuntime.create({ modelsPath: null, allowModelNetwork: false });
const existingModelIds =
pristineRuntime
.getProvider(PROVIDER_ID)
?.getModels()
.map(({ id }) => id) ?? [];
let runtime = session.modelRuntime;
if (!runtime.getModel(PROVIDER_ID, MODEL_ID)) {
runtime = await ModelRuntime.create({
modelsPath: join(agentDir, "models.json"),
authPath: join(agentDir, "auth.json"),
modelsStorePath: join(agentDir, "models-store.json"),
allowModelNetwork: false,
});
}
const configurationError = runtime.getError();
if (configurationError) throw new Error(configurationError);
const model = runtime.getModel(PROVIDER_ID, MODEL_ID);
if (!model) throw new Error(`Model ${PROVIDER_ID}/${MODEL_ID} is unavailable after reload.`);
result = {
model: summarizeModel(model),
existingModelsPreserved:
existingModelIds.length > 0 &&
existingModelIds.every((id) => runtime.getModel(PROVIDER_ID, id) !== undefined),
};
} catch (error) {
result = { error: errorMessage(error) };
}
return {
systemPromptHasGuidelines: systemPrompt.includes("\nGuidelines:\n"),
systemPromptHasPiDocs: systemPrompt.includes("\nPi documentation (read only"),
result,
};
},
});
}
const expectedResult: Exclude<ModelAuthoringResult, { error: string }> = {
model: {
id: MODEL_ID,
name: MODEL_NAME,
provider: PROVIDER_ID,
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 32768,
maxTokens: 4096,
},
existingModelsPreserved: true,
};
const ModelAuthoringJudge = createJudge<PiCodingAgentInput, ModelAuthoringOutput>(
"ModelAuthoringJudge",
({ output }) => {
if ("error" in output.result) {
return { score: 0, metadata: { rationale: output.result.error } };
}
try {
deepStrictEqual(output.result, expectedResult);
return { score: 1, metadata: { rationale: "Model was added to the existing provider." } };
} catch (error) {
return { score: 0, metadata: { rationale: errorMessage(error) } };
}
},
);
const modelHarnessTable = evalHarnessTable("Add model to existing provider", {
baseline: createModelAuthoringHarness("system-prompt-without-docs", excludePiDocumentation),
candidate: createModelAuthoringHarness("default-system-prompt"),
});
describe.for(modelHarnessTable)("$name", ({ harness }) => {
describeEval(
"Add model to existing provider",
{ harness, judges: [ModelAuthoringJudge], judgeThreshold: null },
(it) => {
it("adds the model", { timeout: 300_000 }, async ({ run }) => {
const result = await run([
{
type: "prompt",
content: `Configure Pi with a new \`${PROVIDER_ID}/${MODEL_ID}\` model. Show it as “${MODEL_NAME}”. It accepts text, supports reasoning, has a 32,768-token context window and a 4,096-token maximum output, and has no usage cost.`,
},
{ type: "reload" },
]);
expect(result.output.systemPromptHasGuidelines).toBe(true);
expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt");
});
},
);
});
+56 -15
View File
@@ -9,6 +9,7 @@ import {
type CreateAgentSessionOptions,
createAgentSessionFromServices,
createAgentSessionServices,
type InlineExtension,
ModelRuntime,
SessionManager,
SettingsManager,
@@ -42,9 +43,25 @@ type PiCodingAgentHarnessOptions = {
};
type PiCodingAgentHarnessWithOutput<TOutput extends JsonValue> = PiCodingAgentHarnessOptions & {
output: (args: { response: string; session: AgentSession }) => TOutput | Promise<TOutput>;
output: (args: {
response: string;
session: AgentSession;
systemPrompt: string;
agentDir: string;
}) => TOutput | Promise<TOutput>;
};
// Comparative evals intentionally remove the documentation block using stable prompt markers instead of changing Pi's
// production prompt builder. The isolated eval prompt has no project context or skills between these markers. If
// that setup changes, this transform must be updated so baseline and candidate still differ only by documentation.
export function excludePiDocumentation(defaultPrompt: string): string {
const documentationStart = defaultPrompt.indexOf("\nPi documentation (read only");
if (documentationStart === -1) throw new Error("Default Pi system prompt has no Pi documentation section.");
const cwdStart = defaultPrompt.lastIndexOf("\nCurrent working directory: ");
if (cwdStart === -1) throw new Error("Default Pi system prompt has no working-directory section.");
return defaultPrompt.slice(0, documentationStart) + defaultPrompt.slice(cwdStart);
}
export function resolveModelSelection(
explicitModel: PiCodingAgentModelSelection | undefined,
environment: { PI_PROVIDER?: string; PI_MODEL?: string } = process.env,
@@ -123,21 +140,36 @@ async function runPiCodingAgent<TOutput extends JsonValue>(
const root = await mkdtemp(join(tmpdir(), "pi-eval-"));
const cwd = join(root, "workspace");
const agentDir = join(root, "agent");
let transformedSystemPrompt: string | undefined;
const isolatedHome = join(root, "home");
const agentDir = join(isolatedHome, ".pi", "agent");
const transformSystemPrompt = options.transformSystemPrompt;
let evaluatedSystemPrompt: string | undefined;
const extensionFactories: InlineExtension[] = [];
if (transformSystemPrompt) {
extensionFactories.push({
name: "eval-system-prompt-transform",
hidden: true,
factory: (pi) => {
pi.on("before_agent_start", (event) => {
evaluatedSystemPrompt = transformSystemPrompt(event.systemPrompt);
return { systemPrompt: evaluatedSystemPrompt };
});
},
});
}
let sessionManager: SessionManager | undefined;
let session: AgentSession | undefined;
let outcome: { success: true; result: SimpleHarnessResult<string | TOutput> } | { success: false; error: unknown };
try {
await Promise.all([mkdir(cwd), mkdir(agentDir)]);
await Promise.all([mkdir(cwd), mkdir(agentDir, { recursive: true })]);
const services = await createAgentSessionServices({
cwd,
agentDir,
modelRuntime,
settingsManager: SettingsManager.inMemory(),
...(options.transformSystemPrompt
? { resourceLoaderOptions: { systemPromptOverride: () => transformedSystemPrompt } }
: {}),
settingsManager: SettingsManager.inMemory({
shellCommandPrefix: `export HOME=${JSON.stringify(isolatedHome)}; unset PI_CODING_AGENT_DIR PI_EVAL_ARTIFACT_DIR PI_MODEL PI_PROVIDER PI_REASONING_LEVEL PI_SESSION_FILE PI_SESSION_ID;`,
}),
...(extensionFactories.length > 0 ? { resourceLoaderOptions: { extensionFactories } } : {}),
});
signal?.throwIfAborted();
sessionManager = SessionManager.create(cwd, join(root, "sessions"));
@@ -155,11 +187,6 @@ async function runPiCodingAgent<TOutput extends JsonValue>(
).session;
const evalSession = session;
if (options.transformSystemPrompt) {
transformedSystemPrompt = options.transformSystemPrompt(evalSession.systemPrompt);
if (!transformedSystemPrompt.trim()) throw new Error("Transformed eval system prompt must not be empty.");
await evalSession.reload();
}
let abortPromise: Promise<void> | undefined;
const abort = () => {
abortPromise ??= evalSession.abort();
@@ -167,7 +194,10 @@ async function runPiCodingAgent<TOutput extends JsonValue>(
signal?.addEventListener("abort", abort, { once: true });
try {
signal?.throwIfAborted();
if (evalSession.extensionRunner.getExtensionPaths().length !== 0) {
const unexpectedExtensionPaths = evalSession.extensionRunner
.getExtensionPaths()
.filter((path) => path !== "<inline:eval-system-prompt-transform>");
if (unexpectedExtensionPaths.length !== 0) {
throw new Error("Expected an isolated eval session to start without extensions.");
}
const steps = typeof input === "string" ? [{ type: "prompt" as const, content: input }] : input;
@@ -175,12 +205,23 @@ async function runPiCodingAgent<TOutput extends JsonValue>(
for (const step of steps) {
if (step.type === "prompt") {
response = await promptAgent(evalSession, step.content, signal);
if (transformSystemPrompt && !evaluatedSystemPrompt?.trim()) {
throw new Error("System-prompt transform did not produce a non-empty prompt.");
}
} else {
await evalSession.reload();
}
}
if (response === undefined) throw new Error("Pi eval input must include at least one prompt step.");
const output = "output" in options ? await options.output({ response, session: evalSession }) : response;
const output =
"output" in options
? await options.output({
response,
session: evalSession,
systemPrompt: evaluatedSystemPrompt ?? evalSession.systemPrompt,
agentDir,
})
: response;
const stats = evalSession.getSessionStats();
const hasPricing = [model.cost, ...(model.cost.tiers ?? [])].some(
({ input, output, cacheRead, cacheWrite }) => input > 0 || output > 0 || cacheRead > 0 || cacheWrite > 0,
+410
View File
@@ -0,0 +1,410 @@
import { deepStrictEqual } from "node:assert/strict";
import { createServer, type IncomingMessage, type Server, type ServerResponse } from "node:http";
import type { AddressInfo } from "node:net";
import { join } from "node:path";
import { type Api, type Context, contentText, type Model, type ModelsSimpleStreamOptions } from "@earendil-works/pi-ai";
import { type AgentSession, ModelRuntime } from "@earendil-works/pi-coding-agent";
import { afterAll, beforeAll, beforeEach, describe, expect } from "vitest";
import { createJudge, describeEval } from "vitest-evals";
import { createPiCodingAgentHarness, excludePiDocumentation, type PiCodingAgentInput } from "./pi-harness.ts";
import { evalHarnessTable } from "./vitest-evals/harness-table.ts";
const PROVIDER_ID = "acme";
const MODEL_ID = "acme-chat";
const PROBE_PROMPT = "Reply with ACME_OK.";
const PROBE_RESPONSE = "ACME_OK";
const CUSTOM_PROVIDER_ID = "acme-stream";
const CUSTOM_MODEL_ID = "acme-stream-chat";
const CUSTOM_PROBE_PROMPT = "Reply with ACME_STREAM_OK.";
const CUSTOM_PROBE_RESPONSE = "ACME_STREAM_OK";
let acmeServer: Server | undefined;
let acmeOrigin = "";
let acmeBaseUrl = "";
let validAcmeRequestReceived = false;
let validAcmeStreamRequestReceived = false;
function isRecord(value: unknown): value is Record<string, unknown> {
return typeof value === "object" && value !== null && !Array.isArray(value);
}
function rejectRequest(response: ServerResponse, status: number, message: string): void {
response.writeHead(status, { "content-type": "text/plain" });
response.end(message);
}
async function handleAcmeRequest(request: IncomingMessage, response: ServerResponse): Promise<void> {
if (request.method === "GET" && request.url === "/docs") {
response.writeHead(200, { "content-type": "application/json" });
response.end(
JSON.stringify({
name: "Acme Streaming API",
request: {
method: "POST",
path: "/generate",
headers: { "content-type": "application/json", "x-acme-key": "resolved credential" },
body: {
model: CUSTOM_MODEL_ID,
messages: [{ role: "user", content: "Hello" }],
stream: true,
},
},
response: {
contentType: "application/x-ndjson",
events: [
{ type: "text_delta", text: "Hello" },
{ type: "usage", input_tokens: 3, output_tokens: 2 },
{ type: "done", reason: "stop" },
],
},
}),
);
return;
}
if (request.url !== "/generate" && request.url !== "/v1/chat/completions") {
rejectRequest(response, 404, "Unknown endpoint");
return;
}
if (request.method !== "POST") {
rejectRequest(response, 405, "Expected POST");
return;
}
if (!request.headers["content-type"]?.startsWith("application/json")) {
rejectRequest(response, 415, "Expected application/json");
return;
}
let body = "";
for await (const chunk of request) body += chunk.toString();
let payload: unknown;
try {
payload = JSON.parse(body);
} catch {
rejectRequest(response, 400, "Invalid JSON");
return;
}
if (!isRecord(payload)) {
rejectRequest(response, 422, "Expected a JSON object");
return;
}
const messages: unknown[] = Array.isArray(payload.messages) ? payload.messages : [];
const userMessage = messages.find(
(message): message is Record<string, unknown> => isRecord(message) && message.role === "user",
);
const userPrompt = typeof userMessage?.content === "string" ? userMessage.content : null;
if (request.url === "/generate") {
if (request.headers["x-acme-key"] !== "resolved-stream-key") {
rejectRequest(response, 401, "Invalid Acme Stream credential");
return;
}
if (payload.model !== CUSTOM_MODEL_ID || userPrompt === null || payload.stream !== true) {
rejectRequest(response, 422, "Invalid Acme Stream request");
return;
}
validAcmeStreamRequestReceived = userPrompt === CUSTOM_PROBE_PROMPT;
response.writeHead(200, { "content-type": "application/x-ndjson" });
response.write(`${JSON.stringify({ type: "text_delta", text: "ACME_" })}\n`);
response.write(`${JSON.stringify({ type: "text_delta", text: "STREAM_OK" })}\n`);
response.write(`${JSON.stringify({ type: "usage", input_tokens: 4, output_tokens: 3 })}\n`);
response.end(`${JSON.stringify({ type: "done", reason: "stop" })}\n`);
return;
}
if (request.headers.authorization !== "Bearer resolved-acme-key") {
rejectRequest(response, 401, "Invalid Acme credential");
return;
}
if (payload.model !== MODEL_ID || userPrompt === null || payload.stream !== true) {
rejectRequest(response, 422, "Invalid OpenAI-compatible request");
return;
}
validAcmeRequestReceived = userPrompt === PROBE_PROMPT;
response.writeHead(200, { "content-type": "text/event-stream", "cache-control": "no-cache" });
response.write(
`data: ${JSON.stringify({
id: "chatcmpl-acme",
object: "chat.completion.chunk",
created: 0,
model: MODEL_ID,
choices: [{ index: 0, delta: { role: "assistant", content: PROBE_RESPONSE }, finish_reason: null }],
})}\n\n`,
);
response.write(
`data: ${JSON.stringify({
id: "chatcmpl-acme",
object: "chat.completion.chunk",
created: 0,
model: MODEL_ID,
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
usage: { prompt_tokens: 3, completion_tokens: 2 },
})}\n\n`,
);
response.end("data: [DONE]\n\n");
}
beforeAll(async () => {
acmeServer = createServer((request, response) => {
void handleAcmeRequest(request, response);
});
await new Promise<void>((resolve, reject) => {
acmeServer!.once("error", reject);
acmeServer!.listen(0, "127.0.0.1", resolve);
});
const address = acmeServer.address();
if (!address || typeof address === "string") throw new Error("Fake Acme server did not bind a TCP port.");
acmeOrigin = `http://127.0.0.1:${(address as AddressInfo).port}`;
acmeBaseUrl = `${acmeOrigin}/v1`;
});
beforeEach(() => {
validAcmeRequestReceived = false;
validAcmeStreamRequestReceived = false;
});
afterAll(async () => {
if (!acmeServer) return;
await new Promise<void>((resolve, reject) => {
acmeServer!.close((error) => (error ? reject(error) : resolve()));
});
});
type ProviderRuntimeSuccess = {
validRequestReceived: boolean;
provider: { id: string; name: string };
model: {
id: string;
name: string;
provider: string;
reasoning: boolean;
input: Array<"text" | "image">;
cost: { input: number; output: number; cacheRead: number; cacheWrite: number };
contextWindow: number;
maxTokens: number;
};
response: {
text: string;
stopReason: string;
inputTokens: number;
outputTokens: number;
};
};
type ProviderRuntimeResult = ProviderRuntimeSuccess | { error: string };
type ProviderRuntimeOutput = {
systemPromptHasGuidelines: boolean;
systemPromptHasPiDocs: boolean;
result: ProviderRuntimeResult;
};
type ProviderScenario = {
providerId: string;
modelId: string;
createContext: () => Context;
options?: ModelsSimpleStreamOptions;
validRequestReceived: () => boolean;
};
type RuntimeResolver = (session: AgentSession, agentDir: string) => Promise<ModelRuntime>;
function summarizeModel(model: Model<Api>): ProviderRuntimeSuccess["model"] {
return {
id: model.id,
name: model.name,
provider: model.provider,
reasoning: model.reasoning,
input: [...model.input],
cost: {
input: model.cost.input,
output: model.cost.output,
cacheRead: model.cost.cacheRead,
cacheWrite: model.cost.cacheWrite,
},
contextWindow: model.contextWindow,
maxTokens: model.maxTokens,
};
}
function errorMessage(error: unknown): string {
return error instanceof Error ? error.message : String(error);
}
async function probeProvider(runtime: ModelRuntime, scenario: ProviderScenario): Promise<ProviderRuntimeSuccess> {
const configurationError = runtime.getError();
if (configurationError) throw new Error(configurationError);
const provider = runtime.getProvider(scenario.providerId);
const model = runtime.getModel(scenario.providerId, scenario.modelId);
if (!provider || !model) {
throw new Error(`Model ${scenario.providerId}/${scenario.modelId} is unavailable after reload.`);
}
const response = await runtime.completeSimple(model, scenario.createContext(), scenario.options);
return {
validRequestReceived: scenario.validRequestReceived(),
provider: { id: provider.id, name: provider.name },
model: summarizeModel(model),
response: {
text: contentText(response.content),
stopReason: response.stopReason,
inputTokens: response.usage.input,
outputTokens: response.usage.output,
},
};
}
function createProviderHarness(
name: string,
scenario: ProviderScenario,
transformSystemPrompt?: (defaultPrompt: string) => string,
resolveRuntime: RuntimeResolver = async (session) => session.modelRuntime,
) {
return createPiCodingAgentHarness({
name,
...(transformSystemPrompt ? { transformSystemPrompt } : {}),
output: async ({ session, systemPrompt, agentDir }) => {
let result: ProviderRuntimeResult;
try {
result = await probeProvider(await resolveRuntime(session, agentDir), scenario);
} catch (error) {
result = { error: errorMessage(error) };
}
return {
systemPromptHasGuidelines: systemPrompt.includes("\nGuidelines:\n"),
systemPromptHasPiDocs: systemPrompt.includes("\nPi documentation (read only"),
result,
};
},
});
}
function createProviderRuntimeJudge(expected: ProviderRuntimeSuccess) {
return createJudge<PiCodingAgentInput, ProviderRuntimeOutput>("ProviderRuntimeJudge", ({ output }) => {
if ("error" in output.result) {
return { score: 0, metadata: { rationale: output.result.error } };
}
try {
deepStrictEqual(output.result, expected);
return { score: 1, metadata: { rationale: "Provider works through Pi." } };
} catch (error) {
return { score: 0, metadata: { rationale: errorMessage(error) } };
}
});
}
const providerScenario: ProviderScenario = {
providerId: PROVIDER_ID,
modelId: MODEL_ID,
createContext: () => ({ messages: [{ role: "user", content: PROBE_PROMPT, timestamp: Date.now() }] }),
options: { env: { ACME_API_KEY: "resolved-acme-key" }, maxTokens: 32 },
validRequestReceived: () => validAcmeRequestReceived,
};
const customProviderScenario: ProviderScenario = {
providerId: CUSTOM_PROVIDER_ID,
modelId: CUSTOM_MODEL_ID,
createContext: () => ({ messages: [{ role: "user", content: CUSTOM_PROBE_PROMPT, timestamp: Date.now() }] }),
options: { env: { ACME_STREAM_API_KEY: "resolved-stream-key" }, maxTokens: 32 },
validRequestReceived: () => validAcmeStreamRequestReceived,
};
const ProviderAuthoringJudge = createProviderRuntimeJudge({
validRequestReceived: true,
provider: { id: PROVIDER_ID, name: "Acme" },
model: {
id: MODEL_ID,
name: "Acme Chat",
provider: PROVIDER_ID,
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 32768,
maxTokens: 4096,
},
response: { text: PROBE_RESPONSE, stopReason: "stop", inputTokens: 3, outputTokens: 2 },
});
const CustomProviderJudge = createProviderRuntimeJudge({
validRequestReceived: true,
provider: { id: CUSTOM_PROVIDER_ID, name: "Acme Stream" },
model: {
id: CUSTOM_MODEL_ID,
name: "Acme Stream Chat",
provider: CUSTOM_PROVIDER_ID,
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 16384,
maxTokens: 2048,
},
response: { text: CUSTOM_PROBE_RESPONSE, stopReason: "stop", inputTokens: 4, outputTokens: 3 },
});
const resolveProviderRuntime: RuntimeResolver = async (session, agentDir) => {
if (session.modelRuntime.getModel(PROVIDER_ID, MODEL_ID)) return session.modelRuntime;
return ModelRuntime.create({
modelsPath: join(agentDir, "models.json"),
authPath: join(agentDir, "auth.json"),
modelsStorePath: join(agentDir, "models-store.json"),
allowModelNetwork: false,
});
};
const providerHarnessTable = evalHarnessTable("Add OpenAI-compatible provider", {
baseline: createProviderHarness(
"system-prompt-without-docs",
providerScenario,
excludePiDocumentation,
resolveProviderRuntime,
),
candidate: createProviderHarness("default-system-prompt", providerScenario, undefined, resolveProviderRuntime),
});
describe.for(providerHarnessTable)("$name", ({ harness }) => {
describeEval(
"Add OpenAI-compatible provider",
{ harness, judges: [ProviderAuthoringJudge], judgeThreshold: null },
(it) => {
it("adds the provider", { timeout: 300_000 }, async ({ run }) => {
const result = await run([
{
type: "prompt",
content: `Can you add Acme to Pi as a provider? Its provider ID is ${PROVIDER_ID}, its API is at ${acmeBaseUrl}, and it uses OpenAI Chat Completions. Read its API key from the ACME_API_KEY environment variable.
The provider offers one model, ${MODEL_ID}, shown as “Acme Chat”. It accepts text, does not support reasoning, has a 32,768-token context window and a 4,096-token maximum output, and has no usage cost.`,
},
{ type: "reload" },
]);
expect(result.output.systemPromptHasGuidelines).toBe(true);
expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt");
});
},
);
});
const customProviderHarnessTable = evalHarnessTable("Add custom streaming provider", {
baseline: createProviderHarness("system-prompt-without-docs", customProviderScenario, excludePiDocumentation),
candidate: createProviderHarness("default-system-prompt", customProviderScenario),
});
describe.for(customProviderHarnessTable)("$name", ({ harness }) => {
describeEval(
"Add custom streaming provider",
{ harness, judges: [CustomProviderJudge], judgeThreshold: null },
(it) => {
it("adds the provider", { timeout: 300_000 }, async ({ run }) => {
const result = await run([
{
type: "prompt",
content: `Can you add Acme Stream to Pi as a provider? Its provider ID is ${CUSTOM_PROVIDER_ID}, its API is at ${acmeOrigin}, and its documentation is available at ${acmeOrigin}/docs. Read its credential from the ACME_STREAM_API_KEY environment variable.
It offers one model, ${CUSTOM_MODEL_ID}, shown as “Acme Stream Chat”. The model accepts text, does not support reasoning, has a 16,384-token context window and a 2,048-token maximum output, and has no usage cost.`,
},
{ type: "reload" },
]);
expect(result.output.systemPromptHasGuidelines).toBe(true);
expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt");
});
},
);
});
+2 -2
View File
@@ -4,8 +4,8 @@ import { createPiCodingAgentHarness } from "./pi-harness.ts";
const piCodingAgentHarness = createPiCodingAgentHarness({ noTools: "all" });
describeEval("Pi Coding Agent smoke", { harness: piCodingAgentHarness }, (it) => {
it("runs a basic prompt end to end", async ({ run }) => {
describeEval("Answer a basic prompt", { harness: piCodingAgentHarness }, (it) => {
it("returns the expected answer", async ({ run }) => {
const result = await run("What's the capital of France? Respond with only the city name.");
expect(result.output.trim()).toBe("Paris");
@@ -111,6 +111,17 @@ export function deriveEvalGroupKey(input: unknown, repetition: number): string {
return JSON.stringify([deriveInputKey(input), repetition]);
}
export function resolveEvalRepetitions(
explicit: number | undefined,
environmentValue: string | undefined = process.env.PI_EVAL_REPETITIONS,
): number {
const repetitions = explicit ?? (environmentValue === undefined ? 1 : Number(environmentValue));
if (!Number.isSafeInteger(repetitions) || repetitions < 1) {
throw new TypeError("repetitions must be a positive integer.");
}
return repetitions;
}
function validateOptions<TInput, TOutput extends JsonValue | undefined>(
evalSet: string,
baseline: Harness<TInput, TOutput>,
@@ -166,7 +177,7 @@ export function evalHarnessTable<TInput, TOutput extends JsonValue | undefined>(
evalSet: string,
options: EvalHarnessTableOptions<TInput, TOutput>,
): EvalHarnessTableRow<TInput, TOutput>[] {
const repetitions = options.repetitions ?? 1;
const repetitions = resolveEvalRepetitions(options.repetitions);
const candidates = "candidate" in options ? [options.candidate] : options.candidates;
validateOptions(evalSet, options.baseline, candidates, repetitions);
+16 -3
View File
@@ -1,6 +1,7 @@
import { randomUUID } from "node:crypto";
import { appendFile, mkdir } from "node:fs/promises";
import { appendFile, mkdir, writeFile } from "node:fs/promises";
import { join } from "node:path";
import { stripVTControlCharacters } from "node:util";
import type { Reporter, SerializedError, TestCase, TestModule, TestRunEndReason, Vitest } from "vitest/node";
import { isHarnessRun } from "vitest-evals/harness";
import { PI_SESSION_SNAPSHOT_ARTIFACT, persistEvalArtifactReferences } from "./artifacts.ts";
@@ -95,11 +96,11 @@ export default class EvalHarnessReporter implements Reporter {
await appendHarnessRunReport(test);
}
onTestRunEnd(
async onTestRunEnd(
modules: ReadonlyArray<TestModule>,
_errors: ReadonlyArray<SerializedError>,
reason: TestRunEndReason,
): void {
): Promise<void> {
if (reason === "interrupted") {
this.vitest?.logger.log("\nEval comparisons unavailable: test run interrupted.");
return;
@@ -107,5 +108,17 @@ export default class EvalHarnessReporter implements Reporter {
const report = summarizeHarnessComparisons(collectHarnessObservations(modules));
const formatted = formatHarnessComparisonReport(report);
if (formatted) this.vitest?.logger.log(`\n${formatted}`);
const artifactDirectory = process.env.PI_EVAL_ARTIFACT_DIR?.trim();
if (artifactDirectory) {
await mkdir(artifactDirectory, { recursive: true, mode: 0o700 });
await Promise.all([
writeFile(join(artifactDirectory, "report.json"), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 }),
writeFile(
join(artifactDirectory, "report.txt"),
formatted ? `${stripVTControlCharacters(formatted)}\n` : "",
{ mode: 0o600 },
),
]);
}
}
}
@@ -5,6 +5,7 @@ import {
EVAL_HARNESS_ITERATION_ARTIFACT,
evalHarnessTable,
parseEvalHarnessIterationArtifact,
resolveEvalRepetitions,
} from "../../src/vitest-evals/harness-table.ts";
describe("deriveEvalGroupKey", () => {
@@ -30,6 +31,18 @@ describe("deriveEvalGroupKey", () => {
});
});
describe("resolveEvalRepetitions", () => {
it("uses an explicit value before the environment default", () => {
expect(resolveEvalRepetitions(3, "5")).toBe(3);
expect(resolveEvalRepetitions(undefined, "5")).toBe(5);
expect(resolveEvalRepetitions(undefined, undefined)).toBe(1);
});
it.each(["0", "-1", "1.5", "nope"])("rejects invalid environment value %s", (value) => {
expect(() => resolveEvalRepetitions(undefined, value)).toThrow("positive integer");
});
});
function createFakeHarness(name: string) {
return createHarness<{ id: string }, { harness: string; inputId: string }>({
name,
+5
View File
@@ -13,6 +13,7 @@ export const workspaceSourcePaths = {
aiCompat: fileURLToPath(new URL("./packages/ai/src/compat.ts", import.meta.url)),
aiOAuth: fileURLToPath(new URL("./packages/ai/src/oauth.ts", import.meta.url)),
aiProviders: fileURLToPath(new URL("./packages/ai/src/providers", import.meta.url)),
aiUtils: fileURLToPath(new URL("./packages/ai/src/utils", import.meta.url)),
agentIndex: fileURLToPath(new URL("./packages/agent/src/index.ts", import.meta.url)),
agentNode: fileURLToPath(new URL("./packages/agent/src/node.ts", import.meta.url)),
protocolIndex: fileURLToPath(new URL("./packages/protocol/src/index.ts", import.meta.url)),
@@ -37,6 +38,10 @@ export default defineConfig({
{ find: /^@earendil-works\/pi-ai$/, replacement: workspaceSourcePaths.aiIndex },
{ find: /^@earendil-works\/pi-ai\/compat$/, replacement: workspaceSourcePaths.aiCompat },
{ find: /^@earendil-works\/pi-ai\/oauth$/, replacement: workspaceSourcePaths.aiOAuth },
{
find: /^@earendil-works\/pi-ai\/utils\/(.+)$/,
replacement: `${workspaceSourcePaths.aiUtils}/$1.ts`,
},
{
find: /^@earendil-works\/pi-ai\/providers\/(.+)$/,
replacement: `${workspaceSourcePaths.aiProviders}/$1.ts`,