mirror of
https://github.com/earendil-works/pi.git
synced 2026-09-28 05:54:43 +08:00
feat(coding-agent): add customization documentation evals (#9491)
* fix(coding-agent): isolate extension eval documentation treatment * feat(coding-agent): add custom provider documentation eval * feat(coding-agent): add customization documentation evals * feat(coding-agent): add repeatable eval reports
This commit is contained in:
@@ -26,13 +26,56 @@ Additional arguments are forwarded to Vitest:
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
npm run eval -- src/extensions.eval.ts
|
npm run eval -- src/extensions.eval.ts
|
||||||
npm run eval -- -t "creates, reloads, and uses"
|
npm run eval -- -t "creates and uses the extension"
|
||||||
npm run eval -- src/docs.eval.ts -t "session-format\.md"
|
npm run eval -- src/docs.eval.ts -t "session-format\.md"
|
||||||
```
|
```
|
||||||
|
|
||||||
Each invocation prints an ignored `.eval/` artifact directory. `runs.jsonl` indexes completed harness runs and their
|
Run all comparative customization evals five times in one invocation:
|
||||||
native Pi session JSONL attachments under `sessions/`. These files may contain prompts, responses, source code, and tool
|
|
||||||
output.
|
```bash
|
||||||
|
npm run eval -- \
|
||||||
|
src/extensions.eval.ts src/models.eval.ts src/providers.eval.ts \
|
||||||
|
--provider openai --model gpt-5.6-sol \
|
||||||
|
--repetitions 5
|
||||||
|
```
|
||||||
|
|
||||||
|
`--repetitions` applies to suites declared with `evalHarnessTable(...)`. An explicit `repetitions` value in a suite
|
||||||
|
overrides the command-line default. `PI_EVAL_REPETITIONS=5` is equivalent to the command-line option. Use one repetition
|
||||||
|
while developing an eval and five when reporting lift.
|
||||||
|
|
||||||
|
## Reports and artifacts
|
||||||
|
|
||||||
|
Each invocation prints a compound `Eval Comparisons` report after the Vitest results. When several comparative files run
|
||||||
|
in the same invocation, this report contains one section for every eval set. For example, with illustrative values:
|
||||||
|
|
||||||
|
```text
|
||||||
|
Eval Comparisons
|
||||||
|
Add model to existing provider
|
||||||
|
Baseline system-prompt-without-docs
|
||||||
|
Candidate default-system-prompt (5/5 pairs)
|
||||||
|
Pass rate +60.0 pp (candidate 80.0%, baseline 20.0%)
|
||||||
|
Tokens +1200.0 (candidate 24000.0, baseline 22800.0)
|
||||||
|
Latency -850.0ms (candidate 14000.0ms, baseline 14850.0ms)
|
||||||
|
Est. cost +$0.0100 (candidate $0.1200, baseline $0.1100)
|
||||||
|
|
||||||
|
Add OpenAI-compatible provider
|
||||||
|
...
|
||||||
|
|
||||||
|
Add custom streaming provider
|
||||||
|
...
|
||||||
|
```
|
||||||
|
|
||||||
|
The runner prints the ignored `.eval/` artifact directory at startup. It contains:
|
||||||
|
|
||||||
|
- `report.txt`: the terminal comparison report without color codes.
|
||||||
|
- `report.json`: the same aggregate comparison data as structured JSON.
|
||||||
|
- `runs.jsonl`: one record for every completed harness run.
|
||||||
|
- `sessions/`: native Pi session JSONL attachments.
|
||||||
|
- `sources/`: source attachments recorded by individual evals.
|
||||||
|
|
||||||
|
The report covers comparative suites using `evalHarnessTable(...)`. Ordinary evals still appear in the Vitest summary and
|
||||||
|
in `runs.jsonl`, but not in the baseline-versus-candidate comparison report. Artifacts may contain prompts, responses,
|
||||||
|
source code, and tool output.
|
||||||
|
|
||||||
## Writing evals
|
## Writing evals
|
||||||
|
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ const artifactDirectory = process.env.PI_EVAL_ARTIFACT_DIR
|
|||||||
const args = process.argv.slice(2);
|
const args = process.argv.slice(2);
|
||||||
let provider;
|
let provider;
|
||||||
let model;
|
let model;
|
||||||
|
let repetitions;
|
||||||
let hasCliModelSelection = false;
|
let hasCliModelSelection = false;
|
||||||
const vitestArgs = [];
|
const vitestArgs = [];
|
||||||
|
|
||||||
@@ -33,6 +34,16 @@ for (let index = 0; index < args.length; index += 1) {
|
|||||||
index += 1;
|
index += 1;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
if (arg === "--repetitions") {
|
||||||
|
const value = args[index + 1];
|
||||||
|
if (!value || value.startsWith("-")) {
|
||||||
|
console.error("Missing value for --repetitions");
|
||||||
|
process.exit(1);
|
||||||
|
}
|
||||||
|
repetitions = value;
|
||||||
|
index += 1;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
if (arg.startsWith("--provider=")) {
|
if (arg.startsWith("--provider=")) {
|
||||||
provider = arg.slice("--provider=".length);
|
provider = arg.slice("--provider=".length);
|
||||||
hasCliModelSelection = true;
|
hasCliModelSelection = true;
|
||||||
@@ -43,11 +54,21 @@ for (let index = 0; index < args.length; index += 1) {
|
|||||||
hasCliModelSelection = true;
|
hasCliModelSelection = true;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
if (arg.startsWith("--repetitions=")) {
|
||||||
|
repetitions = arg.slice("--repetitions=".length);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
vitestArgs.push(arg);
|
vitestArgs.push(arg);
|
||||||
}
|
}
|
||||||
|
|
||||||
provider = provider?.trim() || undefined;
|
provider = provider?.trim() || undefined;
|
||||||
model = model?.trim() || undefined;
|
model = model?.trim() || undefined;
|
||||||
|
const repetitionsText = (repetitions ?? process.env.PI_EVAL_REPETITIONS)?.trim();
|
||||||
|
const repetitionCount = repetitionsText ? Number(repetitionsText) : 1;
|
||||||
|
if (!Number.isSafeInteger(repetitionCount) || repetitionCount < 1) {
|
||||||
|
console.error("Repetitions must be a positive integer.");
|
||||||
|
process.exit(1);
|
||||||
|
}
|
||||||
if (hasCliModelSelection) {
|
if (hasCliModelSelection) {
|
||||||
if (!provider || !model) {
|
if (!provider || !model) {
|
||||||
console.error("CLI model selection requires both --provider and --model.");
|
console.error("CLI model selection requires both --provider and --model.");
|
||||||
@@ -68,10 +89,12 @@ const vitestCliPath = resolve(dirname(vitestPackagePath), "vitest.mjs");
|
|||||||
|
|
||||||
mkdirSync(artifactDirectory, { recursive: true, mode: 0o700 });
|
mkdirSync(artifactDirectory, { recursive: true, mode: 0o700 });
|
||||||
console.error(`[eval] default-model=${provider && model ? `${provider}/${model}` : "none"}`);
|
console.error(`[eval] default-model=${provider && model ? `${provider}/${model}` : "none"}`);
|
||||||
|
console.error(`[eval] repetitions=${repetitionCount}`);
|
||||||
console.error(`[eval] artifacts=${artifactDirectory}`);
|
console.error(`[eval] artifacts=${artifactDirectory}`);
|
||||||
const childEnvironment = {
|
const childEnvironment = {
|
||||||
...process.env,
|
...process.env,
|
||||||
PI_EVAL_ARTIFACT_DIR: artifactDirectory,
|
PI_EVAL_ARTIFACT_DIR: artifactDirectory,
|
||||||
|
PI_EVAL_REPETITIONS: String(repetitionCount),
|
||||||
};
|
};
|
||||||
if (provider && model) {
|
if (provider && model) {
|
||||||
childEnvironment.PI_PROVIDER = provider;
|
childEnvironment.PI_PROVIDER = provider;
|
||||||
|
|||||||
@@ -43,7 +43,7 @@ const documentationAuditHarness = createPiCodingAgentHarness({
|
|||||||
customTools: [submitDocumentationAuditTool],
|
customTools: [submitDocumentationAuditTool],
|
||||||
});
|
});
|
||||||
|
|
||||||
describeEval("Coding agent documentation", { harness: documentationAuditHarness }, (it) => {
|
describeEval("Audit documentation against implementation", { harness: documentationAuditHarness }, (it) => {
|
||||||
it.for(documentationPages)("$path matches the implementation", { timeout: 300_000 }, async ({ path }, { run }) => {
|
it.for(documentationPages)("$path matches the implementation", { timeout: 300_000 }, async ({ path }, { run }) => {
|
||||||
const documentationPath = resolve(docsRoot, path);
|
const documentationPath = resolve(docsRoot, path);
|
||||||
const result = await run(`Audit one Pi documentation page against the repository implementation.
|
const result = await run(`Audit one Pi documentation page against the repository implementation.
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ import { existsSync, readFileSync } from "node:fs";
|
|||||||
import { join } from "node:path";
|
import { join } from "node:path";
|
||||||
import { describe, expect } from "vitest";
|
import { describe, expect } from "vitest";
|
||||||
import { createJudge, describeEval } from "vitest-evals";
|
import { createJudge, describeEval } from "vitest-evals";
|
||||||
import { createPiCodingAgentHarness, type PiCodingAgentInput } from "./pi-harness.ts";
|
import { createPiCodingAgentHarness, excludePiDocumentation, type PiCodingAgentInput } from "./pi-harness.ts";
|
||||||
import { recordEvalSourceArtifact } from "./vitest-evals/artifacts.ts";
|
import { recordEvalSourceArtifact } from "./vitest-evals/artifacts.ts";
|
||||||
import { evalHarnessTable } from "./vitest-evals/harness-table.ts";
|
import { evalHarnessTable } from "./vitest-evals/harness-table.ts";
|
||||||
|
|
||||||
@@ -19,14 +19,14 @@ function createExtensionAuthoringHarness(name: string, transformSystemPrompt?: (
|
|||||||
return createPiCodingAgentHarness({
|
return createPiCodingAgentHarness({
|
||||||
name,
|
name,
|
||||||
...(transformSystemPrompt ? { transformSystemPrompt } : {}),
|
...(transformSystemPrompt ? { transformSystemPrompt } : {}),
|
||||||
output: ({ response, session }) => {
|
output: ({ response, session, systemPrompt }) => {
|
||||||
const extensions = session.resourceLoader.getExtensions();
|
const extensions = session.resourceLoader.getExtensions();
|
||||||
const extensionPath = join(session.sessionManager.getCwd(), ".pi", "extensions", "hello.ts");
|
const extensionPath = join(session.sessionManager.getCwd(), ".pi", "extensions", "hello.ts");
|
||||||
const extensionSource = existsSync(extensionPath) ? readFileSync(extensionPath, "utf8") : null;
|
const extensionSource = existsSync(extensionPath) ? readFileSync(extensionPath, "utf8") : null;
|
||||||
return {
|
return {
|
||||||
response,
|
response,
|
||||||
systemPromptHasGuidelines: session.systemPrompt.includes("\nGuidelines:\n"),
|
systemPromptHasGuidelines: systemPrompt.includes("\nGuidelines:\n"),
|
||||||
systemPromptHasPiDocs: session.systemPrompt.includes("\nPi documentation (read only"),
|
systemPromptHasPiDocs: systemPrompt.includes("\nPi documentation (read only"),
|
||||||
extensionErrors: extensions.errors,
|
extensionErrors: extensions.errors,
|
||||||
loadedExtensions: extensions.extensions.map(({ path, tools }) => ({
|
loadedExtensions: extensions.extensions.map(({ path, tools }) => ({
|
||||||
path,
|
path,
|
||||||
@@ -38,18 +38,6 @@ function createExtensionAuthoringHarness(name: string, transformSystemPrompt?: (
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
function excludeGuidelinesAndDocumentation(defaultPrompt: string): string {
|
|
||||||
const guidelinesStart = defaultPrompt.indexOf("\nGuidelines:\n");
|
|
||||||
if (guidelinesStart === -1) throw new Error("Default Pi system prompt has no Guidelines section.");
|
|
||||||
return defaultPrompt.slice(0, guidelinesStart);
|
|
||||||
}
|
|
||||||
|
|
||||||
function prepareDefaultPromptOverride(defaultPrompt: string): string {
|
|
||||||
const cwdStart = defaultPrompt.lastIndexOf("\nCurrent working directory: ");
|
|
||||||
if (cwdStart === -1) throw new Error("Default Pi system prompt has no working-directory section.");
|
|
||||||
return defaultPrompt.slice(0, cwdStart);
|
|
||||||
}
|
|
||||||
|
|
||||||
const ExtensionAuthoringJudge = createJudge<PiCodingAgentInput, ExtensionAuthoringOutput>(
|
const ExtensionAuthoringJudge = createJudge<PiCodingAgentInput, ExtensionAuthoringOutput>(
|
||||||
"ExtensionAuthoringJudge",
|
"ExtensionAuthoringJudge",
|
||||||
({ output, toolCalls }) => {
|
({ output, toolCalls }) => {
|
||||||
@@ -97,17 +85,17 @@ const ExtensionAuthoringJudge = createJudge<PiCodingAgentInput, ExtensionAuthori
|
|||||||
},
|
},
|
||||||
);
|
);
|
||||||
|
|
||||||
const extensionHarnessTable = evalHarnessTable("Pi extension authoring system prompt", {
|
const extensionHarnessTable = evalHarnessTable("Create and use a tool extension", {
|
||||||
baseline: createExtensionAuthoringHarness("system-prompt-without-docs", excludeGuidelinesAndDocumentation),
|
baseline: createExtensionAuthoringHarness("system-prompt-without-docs", excludePiDocumentation),
|
||||||
candidate: createExtensionAuthoringHarness("default-system-prompt", prepareDefaultPromptOverride),
|
candidate: createExtensionAuthoringHarness("default-system-prompt"),
|
||||||
});
|
});
|
||||||
|
|
||||||
describe.for(extensionHarnessTable)("$name", ({ harness }) => {
|
describe.for(extensionHarnessTable)("$name", ({ harness }) => {
|
||||||
describeEval(
|
describeEval(
|
||||||
"Pi extension authoring system prompt",
|
"Create and use a tool extension",
|
||||||
{ harness, judges: [ExtensionAuthoringJudge], judgeThreshold: null },
|
{ harness, judges: [ExtensionAuthoringJudge], judgeThreshold: null },
|
||||||
(it) => {
|
(it) => {
|
||||||
it("creates, reloads, and uses a hello extension", async ({ run, task }) => {
|
it("creates and uses the extension", async ({ run, task }) => {
|
||||||
const result = await run([
|
const result = await run([
|
||||||
{
|
{
|
||||||
type: "prompt",
|
type: "prompt",
|
||||||
@@ -131,9 +119,8 @@ describe.for(extensionHarnessTable)("$name", ({ harness }) => {
|
|||||||
bodyEncoding: "utf-8",
|
bodyEncoding: "utf-8",
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
const expectsFullPrompt = harness.name === "default-system-prompt";
|
expect(result.output.systemPromptHasGuidelines).toBe(true);
|
||||||
expect(result.output.systemPromptHasGuidelines).toBe(expectsFullPrompt);
|
expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt");
|
||||||
expect(result.output.systemPromptHasPiDocs).toBe(expectsFullPrompt);
|
|
||||||
});
|
});
|
||||||
},
|
},
|
||||||
);
|
);
|
||||||
|
|||||||
@@ -0,0 +1,151 @@
|
|||||||
|
import { deepStrictEqual } from "node:assert/strict";
|
||||||
|
import { join } from "node:path";
|
||||||
|
import type { Api, Model } from "@earendil-works/pi-ai";
|
||||||
|
import { ModelRuntime } from "@earendil-works/pi-coding-agent";
|
||||||
|
import { describe, expect } from "vitest";
|
||||||
|
import { createJudge, describeEval } from "vitest-evals";
|
||||||
|
import { createPiCodingAgentHarness, excludePiDocumentation, type PiCodingAgentInput } from "./pi-harness.ts";
|
||||||
|
import { evalHarnessTable } from "./vitest-evals/harness-table.ts";
|
||||||
|
|
||||||
|
const PROVIDER_ID = "openai";
|
||||||
|
const MODEL_ID = "fixture-chat";
|
||||||
|
const MODEL_NAME = "Fixture Chat";
|
||||||
|
|
||||||
|
type ModelSummary = {
|
||||||
|
id: string;
|
||||||
|
name: string;
|
||||||
|
provider: string;
|
||||||
|
reasoning: boolean;
|
||||||
|
input: Array<"text" | "image">;
|
||||||
|
cost: { input: number; output: number; cacheRead: number; cacheWrite: number };
|
||||||
|
contextWindow: number;
|
||||||
|
maxTokens: number;
|
||||||
|
};
|
||||||
|
|
||||||
|
type ModelAuthoringResult = { model: ModelSummary; existingModelsPreserved: boolean } | { error: string };
|
||||||
|
|
||||||
|
type ModelAuthoringOutput = {
|
||||||
|
systemPromptHasGuidelines: boolean;
|
||||||
|
systemPromptHasPiDocs: boolean;
|
||||||
|
result: ModelAuthoringResult;
|
||||||
|
};
|
||||||
|
|
||||||
|
function summarizeModel(model: Model<Api>): ModelSummary {
|
||||||
|
return {
|
||||||
|
id: model.id,
|
||||||
|
name: model.name,
|
||||||
|
provider: model.provider,
|
||||||
|
reasoning: model.reasoning,
|
||||||
|
input: [...model.input],
|
||||||
|
cost: {
|
||||||
|
input: model.cost.input,
|
||||||
|
output: model.cost.output,
|
||||||
|
cacheRead: model.cost.cacheRead,
|
||||||
|
cacheWrite: model.cost.cacheWrite,
|
||||||
|
},
|
||||||
|
contextWindow: model.contextWindow,
|
||||||
|
maxTokens: model.maxTokens,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function errorMessage(error: unknown): string {
|
||||||
|
return error instanceof Error ? error.message : String(error);
|
||||||
|
}
|
||||||
|
|
||||||
|
function createModelAuthoringHarness(name: string, transformSystemPrompt?: (defaultPrompt: string) => string) {
|
||||||
|
return createPiCodingAgentHarness({
|
||||||
|
name,
|
||||||
|
...(transformSystemPrompt ? { transformSystemPrompt } : {}),
|
||||||
|
output: async ({ session, systemPrompt, agentDir }) => {
|
||||||
|
let result: ModelAuthoringResult;
|
||||||
|
try {
|
||||||
|
const pristineRuntime = await ModelRuntime.create({ modelsPath: null, allowModelNetwork: false });
|
||||||
|
const existingModelIds =
|
||||||
|
pristineRuntime
|
||||||
|
.getProvider(PROVIDER_ID)
|
||||||
|
?.getModels()
|
||||||
|
.map(({ id }) => id) ?? [];
|
||||||
|
let runtime = session.modelRuntime;
|
||||||
|
if (!runtime.getModel(PROVIDER_ID, MODEL_ID)) {
|
||||||
|
runtime = await ModelRuntime.create({
|
||||||
|
modelsPath: join(agentDir, "models.json"),
|
||||||
|
authPath: join(agentDir, "auth.json"),
|
||||||
|
modelsStorePath: join(agentDir, "models-store.json"),
|
||||||
|
allowModelNetwork: false,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
const configurationError = runtime.getError();
|
||||||
|
if (configurationError) throw new Error(configurationError);
|
||||||
|
const model = runtime.getModel(PROVIDER_ID, MODEL_ID);
|
||||||
|
if (!model) throw new Error(`Model ${PROVIDER_ID}/${MODEL_ID} is unavailable after reload.`);
|
||||||
|
result = {
|
||||||
|
model: summarizeModel(model),
|
||||||
|
existingModelsPreserved:
|
||||||
|
existingModelIds.length > 0 &&
|
||||||
|
existingModelIds.every((id) => runtime.getModel(PROVIDER_ID, id) !== undefined),
|
||||||
|
};
|
||||||
|
} catch (error) {
|
||||||
|
result = { error: errorMessage(error) };
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
systemPromptHasGuidelines: systemPrompt.includes("\nGuidelines:\n"),
|
||||||
|
systemPromptHasPiDocs: systemPrompt.includes("\nPi documentation (read only"),
|
||||||
|
result,
|
||||||
|
};
|
||||||
|
},
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
const expectedResult: Exclude<ModelAuthoringResult, { error: string }> = {
|
||||||
|
model: {
|
||||||
|
id: MODEL_ID,
|
||||||
|
name: MODEL_NAME,
|
||||||
|
provider: PROVIDER_ID,
|
||||||
|
reasoning: true,
|
||||||
|
input: ["text"],
|
||||||
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||||
|
contextWindow: 32768,
|
||||||
|
maxTokens: 4096,
|
||||||
|
},
|
||||||
|
existingModelsPreserved: true,
|
||||||
|
};
|
||||||
|
|
||||||
|
const ModelAuthoringJudge = createJudge<PiCodingAgentInput, ModelAuthoringOutput>(
|
||||||
|
"ModelAuthoringJudge",
|
||||||
|
({ output }) => {
|
||||||
|
if ("error" in output.result) {
|
||||||
|
return { score: 0, metadata: { rationale: output.result.error } };
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
deepStrictEqual(output.result, expectedResult);
|
||||||
|
return { score: 1, metadata: { rationale: "Model was added to the existing provider." } };
|
||||||
|
} catch (error) {
|
||||||
|
return { score: 0, metadata: { rationale: errorMessage(error) } };
|
||||||
|
}
|
||||||
|
},
|
||||||
|
);
|
||||||
|
|
||||||
|
const modelHarnessTable = evalHarnessTable("Add model to existing provider", {
|
||||||
|
baseline: createModelAuthoringHarness("system-prompt-without-docs", excludePiDocumentation),
|
||||||
|
candidate: createModelAuthoringHarness("default-system-prompt"),
|
||||||
|
});
|
||||||
|
|
||||||
|
describe.for(modelHarnessTable)("$name", ({ harness }) => {
|
||||||
|
describeEval(
|
||||||
|
"Add model to existing provider",
|
||||||
|
{ harness, judges: [ModelAuthoringJudge], judgeThreshold: null },
|
||||||
|
(it) => {
|
||||||
|
it("adds the model", { timeout: 300_000 }, async ({ run }) => {
|
||||||
|
const result = await run([
|
||||||
|
{
|
||||||
|
type: "prompt",
|
||||||
|
content: `Configure Pi with a new \`${PROVIDER_ID}/${MODEL_ID}\` model. Show it as “${MODEL_NAME}”. It accepts text, supports reasoning, has a 32,768-token context window and a 4,096-token maximum output, and has no usage cost.`,
|
||||||
|
},
|
||||||
|
{ type: "reload" },
|
||||||
|
]);
|
||||||
|
expect(result.output.systemPromptHasGuidelines).toBe(true);
|
||||||
|
expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt");
|
||||||
|
});
|
||||||
|
},
|
||||||
|
);
|
||||||
|
});
|
||||||
@@ -9,6 +9,7 @@ import {
|
|||||||
type CreateAgentSessionOptions,
|
type CreateAgentSessionOptions,
|
||||||
createAgentSessionFromServices,
|
createAgentSessionFromServices,
|
||||||
createAgentSessionServices,
|
createAgentSessionServices,
|
||||||
|
type InlineExtension,
|
||||||
ModelRuntime,
|
ModelRuntime,
|
||||||
SessionManager,
|
SessionManager,
|
||||||
SettingsManager,
|
SettingsManager,
|
||||||
@@ -42,9 +43,25 @@ type PiCodingAgentHarnessOptions = {
|
|||||||
};
|
};
|
||||||
|
|
||||||
type PiCodingAgentHarnessWithOutput<TOutput extends JsonValue> = PiCodingAgentHarnessOptions & {
|
type PiCodingAgentHarnessWithOutput<TOutput extends JsonValue> = PiCodingAgentHarnessOptions & {
|
||||||
output: (args: { response: string; session: AgentSession }) => TOutput | Promise<TOutput>;
|
output: (args: {
|
||||||
|
response: string;
|
||||||
|
session: AgentSession;
|
||||||
|
systemPrompt: string;
|
||||||
|
agentDir: string;
|
||||||
|
}) => TOutput | Promise<TOutput>;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Comparative evals intentionally remove the documentation block using stable prompt markers instead of changing Pi's
|
||||||
|
// production prompt builder. The isolated eval prompt has no project context or skills between these markers. If
|
||||||
|
// that setup changes, this transform must be updated so baseline and candidate still differ only by documentation.
|
||||||
|
export function excludePiDocumentation(defaultPrompt: string): string {
|
||||||
|
const documentationStart = defaultPrompt.indexOf("\nPi documentation (read only");
|
||||||
|
if (documentationStart === -1) throw new Error("Default Pi system prompt has no Pi documentation section.");
|
||||||
|
const cwdStart = defaultPrompt.lastIndexOf("\nCurrent working directory: ");
|
||||||
|
if (cwdStart === -1) throw new Error("Default Pi system prompt has no working-directory section.");
|
||||||
|
return defaultPrompt.slice(0, documentationStart) + defaultPrompt.slice(cwdStart);
|
||||||
|
}
|
||||||
|
|
||||||
export function resolveModelSelection(
|
export function resolveModelSelection(
|
||||||
explicitModel: PiCodingAgentModelSelection | undefined,
|
explicitModel: PiCodingAgentModelSelection | undefined,
|
||||||
environment: { PI_PROVIDER?: string; PI_MODEL?: string } = process.env,
|
environment: { PI_PROVIDER?: string; PI_MODEL?: string } = process.env,
|
||||||
@@ -123,21 +140,36 @@ async function runPiCodingAgent<TOutput extends JsonValue>(
|
|||||||
|
|
||||||
const root = await mkdtemp(join(tmpdir(), "pi-eval-"));
|
const root = await mkdtemp(join(tmpdir(), "pi-eval-"));
|
||||||
const cwd = join(root, "workspace");
|
const cwd = join(root, "workspace");
|
||||||
const agentDir = join(root, "agent");
|
const isolatedHome = join(root, "home");
|
||||||
let transformedSystemPrompt: string | undefined;
|
const agentDir = join(isolatedHome, ".pi", "agent");
|
||||||
|
const transformSystemPrompt = options.transformSystemPrompt;
|
||||||
|
let evaluatedSystemPrompt: string | undefined;
|
||||||
|
const extensionFactories: InlineExtension[] = [];
|
||||||
|
if (transformSystemPrompt) {
|
||||||
|
extensionFactories.push({
|
||||||
|
name: "eval-system-prompt-transform",
|
||||||
|
hidden: true,
|
||||||
|
factory: (pi) => {
|
||||||
|
pi.on("before_agent_start", (event) => {
|
||||||
|
evaluatedSystemPrompt = transformSystemPrompt(event.systemPrompt);
|
||||||
|
return { systemPrompt: evaluatedSystemPrompt };
|
||||||
|
});
|
||||||
|
},
|
||||||
|
});
|
||||||
|
}
|
||||||
let sessionManager: SessionManager | undefined;
|
let sessionManager: SessionManager | undefined;
|
||||||
let session: AgentSession | undefined;
|
let session: AgentSession | undefined;
|
||||||
let outcome: { success: true; result: SimpleHarnessResult<string | TOutput> } | { success: false; error: unknown };
|
let outcome: { success: true; result: SimpleHarnessResult<string | TOutput> } | { success: false; error: unknown };
|
||||||
try {
|
try {
|
||||||
await Promise.all([mkdir(cwd), mkdir(agentDir)]);
|
await Promise.all([mkdir(cwd), mkdir(agentDir, { recursive: true })]);
|
||||||
const services = await createAgentSessionServices({
|
const services = await createAgentSessionServices({
|
||||||
cwd,
|
cwd,
|
||||||
agentDir,
|
agentDir,
|
||||||
modelRuntime,
|
modelRuntime,
|
||||||
settingsManager: SettingsManager.inMemory(),
|
settingsManager: SettingsManager.inMemory({
|
||||||
...(options.transformSystemPrompt
|
shellCommandPrefix: `export HOME=${JSON.stringify(isolatedHome)}; unset PI_CODING_AGENT_DIR PI_EVAL_ARTIFACT_DIR PI_MODEL PI_PROVIDER PI_REASONING_LEVEL PI_SESSION_FILE PI_SESSION_ID;`,
|
||||||
? { resourceLoaderOptions: { systemPromptOverride: () => transformedSystemPrompt } }
|
}),
|
||||||
: {}),
|
...(extensionFactories.length > 0 ? { resourceLoaderOptions: { extensionFactories } } : {}),
|
||||||
});
|
});
|
||||||
signal?.throwIfAborted();
|
signal?.throwIfAborted();
|
||||||
sessionManager = SessionManager.create(cwd, join(root, "sessions"));
|
sessionManager = SessionManager.create(cwd, join(root, "sessions"));
|
||||||
@@ -155,11 +187,6 @@ async function runPiCodingAgent<TOutput extends JsonValue>(
|
|||||||
).session;
|
).session;
|
||||||
|
|
||||||
const evalSession = session;
|
const evalSession = session;
|
||||||
if (options.transformSystemPrompt) {
|
|
||||||
transformedSystemPrompt = options.transformSystemPrompt(evalSession.systemPrompt);
|
|
||||||
if (!transformedSystemPrompt.trim()) throw new Error("Transformed eval system prompt must not be empty.");
|
|
||||||
await evalSession.reload();
|
|
||||||
}
|
|
||||||
let abortPromise: Promise<void> | undefined;
|
let abortPromise: Promise<void> | undefined;
|
||||||
const abort = () => {
|
const abort = () => {
|
||||||
abortPromise ??= evalSession.abort();
|
abortPromise ??= evalSession.abort();
|
||||||
@@ -167,7 +194,10 @@ async function runPiCodingAgent<TOutput extends JsonValue>(
|
|||||||
signal?.addEventListener("abort", abort, { once: true });
|
signal?.addEventListener("abort", abort, { once: true });
|
||||||
try {
|
try {
|
||||||
signal?.throwIfAborted();
|
signal?.throwIfAborted();
|
||||||
if (evalSession.extensionRunner.getExtensionPaths().length !== 0) {
|
const unexpectedExtensionPaths = evalSession.extensionRunner
|
||||||
|
.getExtensionPaths()
|
||||||
|
.filter((path) => path !== "<inline:eval-system-prompt-transform>");
|
||||||
|
if (unexpectedExtensionPaths.length !== 0) {
|
||||||
throw new Error("Expected an isolated eval session to start without extensions.");
|
throw new Error("Expected an isolated eval session to start without extensions.");
|
||||||
}
|
}
|
||||||
const steps = typeof input === "string" ? [{ type: "prompt" as const, content: input }] : input;
|
const steps = typeof input === "string" ? [{ type: "prompt" as const, content: input }] : input;
|
||||||
@@ -175,12 +205,23 @@ async function runPiCodingAgent<TOutput extends JsonValue>(
|
|||||||
for (const step of steps) {
|
for (const step of steps) {
|
||||||
if (step.type === "prompt") {
|
if (step.type === "prompt") {
|
||||||
response = await promptAgent(evalSession, step.content, signal);
|
response = await promptAgent(evalSession, step.content, signal);
|
||||||
|
if (transformSystemPrompt && !evaluatedSystemPrompt?.trim()) {
|
||||||
|
throw new Error("System-prompt transform did not produce a non-empty prompt.");
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
await evalSession.reload();
|
await evalSession.reload();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (response === undefined) throw new Error("Pi eval input must include at least one prompt step.");
|
if (response === undefined) throw new Error("Pi eval input must include at least one prompt step.");
|
||||||
const output = "output" in options ? await options.output({ response, session: evalSession }) : response;
|
const output =
|
||||||
|
"output" in options
|
||||||
|
? await options.output({
|
||||||
|
response,
|
||||||
|
session: evalSession,
|
||||||
|
systemPrompt: evaluatedSystemPrompt ?? evalSession.systemPrompt,
|
||||||
|
agentDir,
|
||||||
|
})
|
||||||
|
: response;
|
||||||
const stats = evalSession.getSessionStats();
|
const stats = evalSession.getSessionStats();
|
||||||
const hasPricing = [model.cost, ...(model.cost.tiers ?? [])].some(
|
const hasPricing = [model.cost, ...(model.cost.tiers ?? [])].some(
|
||||||
({ input, output, cacheRead, cacheWrite }) => input > 0 || output > 0 || cacheRead > 0 || cacheWrite > 0,
|
({ input, output, cacheRead, cacheWrite }) => input > 0 || output > 0 || cacheRead > 0 || cacheWrite > 0,
|
||||||
|
|||||||
@@ -0,0 +1,410 @@
|
|||||||
|
import { deepStrictEqual } from "node:assert/strict";
|
||||||
|
import { createServer, type IncomingMessage, type Server, type ServerResponse } from "node:http";
|
||||||
|
import type { AddressInfo } from "node:net";
|
||||||
|
import { join } from "node:path";
|
||||||
|
import { type Api, type Context, contentText, type Model, type ModelsSimpleStreamOptions } from "@earendil-works/pi-ai";
|
||||||
|
import { type AgentSession, ModelRuntime } from "@earendil-works/pi-coding-agent";
|
||||||
|
import { afterAll, beforeAll, beforeEach, describe, expect } from "vitest";
|
||||||
|
import { createJudge, describeEval } from "vitest-evals";
|
||||||
|
import { createPiCodingAgentHarness, excludePiDocumentation, type PiCodingAgentInput } from "./pi-harness.ts";
|
||||||
|
import { evalHarnessTable } from "./vitest-evals/harness-table.ts";
|
||||||
|
|
||||||
|
const PROVIDER_ID = "acme";
|
||||||
|
const MODEL_ID = "acme-chat";
|
||||||
|
const PROBE_PROMPT = "Reply with ACME_OK.";
|
||||||
|
const PROBE_RESPONSE = "ACME_OK";
|
||||||
|
const CUSTOM_PROVIDER_ID = "acme-stream";
|
||||||
|
const CUSTOM_MODEL_ID = "acme-stream-chat";
|
||||||
|
const CUSTOM_PROBE_PROMPT = "Reply with ACME_STREAM_OK.";
|
||||||
|
const CUSTOM_PROBE_RESPONSE = "ACME_STREAM_OK";
|
||||||
|
|
||||||
|
let acmeServer: Server | undefined;
|
||||||
|
let acmeOrigin = "";
|
||||||
|
let acmeBaseUrl = "";
|
||||||
|
let validAcmeRequestReceived = false;
|
||||||
|
let validAcmeStreamRequestReceived = false;
|
||||||
|
|
||||||
|
function isRecord(value: unknown): value is Record<string, unknown> {
|
||||||
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
||||||
|
}
|
||||||
|
|
||||||
|
function rejectRequest(response: ServerResponse, status: number, message: string): void {
|
||||||
|
response.writeHead(status, { "content-type": "text/plain" });
|
||||||
|
response.end(message);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function handleAcmeRequest(request: IncomingMessage, response: ServerResponse): Promise<void> {
|
||||||
|
if (request.method === "GET" && request.url === "/docs") {
|
||||||
|
response.writeHead(200, { "content-type": "application/json" });
|
||||||
|
response.end(
|
||||||
|
JSON.stringify({
|
||||||
|
name: "Acme Streaming API",
|
||||||
|
request: {
|
||||||
|
method: "POST",
|
||||||
|
path: "/generate",
|
||||||
|
headers: { "content-type": "application/json", "x-acme-key": "resolved credential" },
|
||||||
|
body: {
|
||||||
|
model: CUSTOM_MODEL_ID,
|
||||||
|
messages: [{ role: "user", content: "Hello" }],
|
||||||
|
stream: true,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
response: {
|
||||||
|
contentType: "application/x-ndjson",
|
||||||
|
events: [
|
||||||
|
{ type: "text_delta", text: "Hello" },
|
||||||
|
{ type: "usage", input_tokens: 3, output_tokens: 2 },
|
||||||
|
{ type: "done", reason: "stop" },
|
||||||
|
],
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (request.url !== "/generate" && request.url !== "/v1/chat/completions") {
|
||||||
|
rejectRequest(response, 404, "Unknown endpoint");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (request.method !== "POST") {
|
||||||
|
rejectRequest(response, 405, "Expected POST");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!request.headers["content-type"]?.startsWith("application/json")) {
|
||||||
|
rejectRequest(response, 415, "Expected application/json");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
let body = "";
|
||||||
|
for await (const chunk of request) body += chunk.toString();
|
||||||
|
let payload: unknown;
|
||||||
|
try {
|
||||||
|
payload = JSON.parse(body);
|
||||||
|
} catch {
|
||||||
|
rejectRequest(response, 400, "Invalid JSON");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!isRecord(payload)) {
|
||||||
|
rejectRequest(response, 422, "Expected a JSON object");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const messages: unknown[] = Array.isArray(payload.messages) ? payload.messages : [];
|
||||||
|
const userMessage = messages.find(
|
||||||
|
(message): message is Record<string, unknown> => isRecord(message) && message.role === "user",
|
||||||
|
);
|
||||||
|
const userPrompt = typeof userMessage?.content === "string" ? userMessage.content : null;
|
||||||
|
|
||||||
|
if (request.url === "/generate") {
|
||||||
|
if (request.headers["x-acme-key"] !== "resolved-stream-key") {
|
||||||
|
rejectRequest(response, 401, "Invalid Acme Stream credential");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (payload.model !== CUSTOM_MODEL_ID || userPrompt === null || payload.stream !== true) {
|
||||||
|
rejectRequest(response, 422, "Invalid Acme Stream request");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
validAcmeStreamRequestReceived = userPrompt === CUSTOM_PROBE_PROMPT;
|
||||||
|
response.writeHead(200, { "content-type": "application/x-ndjson" });
|
||||||
|
response.write(`${JSON.stringify({ type: "text_delta", text: "ACME_" })}\n`);
|
||||||
|
response.write(`${JSON.stringify({ type: "text_delta", text: "STREAM_OK" })}\n`);
|
||||||
|
response.write(`${JSON.stringify({ type: "usage", input_tokens: 4, output_tokens: 3 })}\n`);
|
||||||
|
response.end(`${JSON.stringify({ type: "done", reason: "stop" })}\n`);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (request.headers.authorization !== "Bearer resolved-acme-key") {
|
||||||
|
rejectRequest(response, 401, "Invalid Acme credential");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (payload.model !== MODEL_ID || userPrompt === null || payload.stream !== true) {
|
||||||
|
rejectRequest(response, 422, "Invalid OpenAI-compatible request");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
validAcmeRequestReceived = userPrompt === PROBE_PROMPT;
|
||||||
|
response.writeHead(200, { "content-type": "text/event-stream", "cache-control": "no-cache" });
|
||||||
|
response.write(
|
||||||
|
`data: ${JSON.stringify({
|
||||||
|
id: "chatcmpl-acme",
|
||||||
|
object: "chat.completion.chunk",
|
||||||
|
created: 0,
|
||||||
|
model: MODEL_ID,
|
||||||
|
choices: [{ index: 0, delta: { role: "assistant", content: PROBE_RESPONSE }, finish_reason: null }],
|
||||||
|
})}\n\n`,
|
||||||
|
);
|
||||||
|
response.write(
|
||||||
|
`data: ${JSON.stringify({
|
||||||
|
id: "chatcmpl-acme",
|
||||||
|
object: "chat.completion.chunk",
|
||||||
|
created: 0,
|
||||||
|
model: MODEL_ID,
|
||||||
|
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
||||||
|
usage: { prompt_tokens: 3, completion_tokens: 2 },
|
||||||
|
})}\n\n`,
|
||||||
|
);
|
||||||
|
response.end("data: [DONE]\n\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
beforeAll(async () => {
|
||||||
|
acmeServer = createServer((request, response) => {
|
||||||
|
void handleAcmeRequest(request, response);
|
||||||
|
});
|
||||||
|
await new Promise<void>((resolve, reject) => {
|
||||||
|
acmeServer!.once("error", reject);
|
||||||
|
acmeServer!.listen(0, "127.0.0.1", resolve);
|
||||||
|
});
|
||||||
|
const address = acmeServer.address();
|
||||||
|
if (!address || typeof address === "string") throw new Error("Fake Acme server did not bind a TCP port.");
|
||||||
|
acmeOrigin = `http://127.0.0.1:${(address as AddressInfo).port}`;
|
||||||
|
acmeBaseUrl = `${acmeOrigin}/v1`;
|
||||||
|
});
|
||||||
|
|
||||||
|
beforeEach(() => {
|
||||||
|
validAcmeRequestReceived = false;
|
||||||
|
validAcmeStreamRequestReceived = false;
|
||||||
|
});
|
||||||
|
|
||||||
|
afterAll(async () => {
|
||||||
|
if (!acmeServer) return;
|
||||||
|
await new Promise<void>((resolve, reject) => {
|
||||||
|
acmeServer!.close((error) => (error ? reject(error) : resolve()));
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
type ProviderRuntimeSuccess = {
|
||||||
|
validRequestReceived: boolean;
|
||||||
|
provider: { id: string; name: string };
|
||||||
|
model: {
|
||||||
|
id: string;
|
||||||
|
name: string;
|
||||||
|
provider: string;
|
||||||
|
reasoning: boolean;
|
||||||
|
input: Array<"text" | "image">;
|
||||||
|
cost: { input: number; output: number; cacheRead: number; cacheWrite: number };
|
||||||
|
contextWindow: number;
|
||||||
|
maxTokens: number;
|
||||||
|
};
|
||||||
|
response: {
|
||||||
|
text: string;
|
||||||
|
stopReason: string;
|
||||||
|
inputTokens: number;
|
||||||
|
outputTokens: number;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
type ProviderRuntimeResult = ProviderRuntimeSuccess | { error: string };
|
||||||
|
|
||||||
|
type ProviderRuntimeOutput = {
|
||||||
|
systemPromptHasGuidelines: boolean;
|
||||||
|
systemPromptHasPiDocs: boolean;
|
||||||
|
result: ProviderRuntimeResult;
|
||||||
|
};
|
||||||
|
|
||||||
|
type ProviderScenario = {
|
||||||
|
providerId: string;
|
||||||
|
modelId: string;
|
||||||
|
createContext: () => Context;
|
||||||
|
options?: ModelsSimpleStreamOptions;
|
||||||
|
validRequestReceived: () => boolean;
|
||||||
|
};
|
||||||
|
|
||||||
|
type RuntimeResolver = (session: AgentSession, agentDir: string) => Promise<ModelRuntime>;
|
||||||
|
|
||||||
|
function summarizeModel(model: Model<Api>): ProviderRuntimeSuccess["model"] {
|
||||||
|
return {
|
||||||
|
id: model.id,
|
||||||
|
name: model.name,
|
||||||
|
provider: model.provider,
|
||||||
|
reasoning: model.reasoning,
|
||||||
|
input: [...model.input],
|
||||||
|
cost: {
|
||||||
|
input: model.cost.input,
|
||||||
|
output: model.cost.output,
|
||||||
|
cacheRead: model.cost.cacheRead,
|
||||||
|
cacheWrite: model.cost.cacheWrite,
|
||||||
|
},
|
||||||
|
contextWindow: model.contextWindow,
|
||||||
|
maxTokens: model.maxTokens,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function errorMessage(error: unknown): string {
|
||||||
|
return error instanceof Error ? error.message : String(error);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function probeProvider(runtime: ModelRuntime, scenario: ProviderScenario): Promise<ProviderRuntimeSuccess> {
|
||||||
|
const configurationError = runtime.getError();
|
||||||
|
if (configurationError) throw new Error(configurationError);
|
||||||
|
const provider = runtime.getProvider(scenario.providerId);
|
||||||
|
const model = runtime.getModel(scenario.providerId, scenario.modelId);
|
||||||
|
if (!provider || !model) {
|
||||||
|
throw new Error(`Model ${scenario.providerId}/${scenario.modelId} is unavailable after reload.`);
|
||||||
|
}
|
||||||
|
const response = await runtime.completeSimple(model, scenario.createContext(), scenario.options);
|
||||||
|
return {
|
||||||
|
validRequestReceived: scenario.validRequestReceived(),
|
||||||
|
provider: { id: provider.id, name: provider.name },
|
||||||
|
model: summarizeModel(model),
|
||||||
|
response: {
|
||||||
|
text: contentText(response.content),
|
||||||
|
stopReason: response.stopReason,
|
||||||
|
inputTokens: response.usage.input,
|
||||||
|
outputTokens: response.usage.output,
|
||||||
|
},
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function createProviderHarness(
|
||||||
|
name: string,
|
||||||
|
scenario: ProviderScenario,
|
||||||
|
transformSystemPrompt?: (defaultPrompt: string) => string,
|
||||||
|
resolveRuntime: RuntimeResolver = async (session) => session.modelRuntime,
|
||||||
|
) {
|
||||||
|
return createPiCodingAgentHarness({
|
||||||
|
name,
|
||||||
|
...(transformSystemPrompt ? { transformSystemPrompt } : {}),
|
||||||
|
output: async ({ session, systemPrompt, agentDir }) => {
|
||||||
|
let result: ProviderRuntimeResult;
|
||||||
|
try {
|
||||||
|
result = await probeProvider(await resolveRuntime(session, agentDir), scenario);
|
||||||
|
} catch (error) {
|
||||||
|
result = { error: errorMessage(error) };
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
systemPromptHasGuidelines: systemPrompt.includes("\nGuidelines:\n"),
|
||||||
|
systemPromptHasPiDocs: systemPrompt.includes("\nPi documentation (read only"),
|
||||||
|
result,
|
||||||
|
};
|
||||||
|
},
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
function createProviderRuntimeJudge(expected: ProviderRuntimeSuccess) {
|
||||||
|
return createJudge<PiCodingAgentInput, ProviderRuntimeOutput>("ProviderRuntimeJudge", ({ output }) => {
|
||||||
|
if ("error" in output.result) {
|
||||||
|
return { score: 0, metadata: { rationale: output.result.error } };
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
deepStrictEqual(output.result, expected);
|
||||||
|
return { score: 1, metadata: { rationale: "Provider works through Pi." } };
|
||||||
|
} catch (error) {
|
||||||
|
return { score: 0, metadata: { rationale: errorMessage(error) } };
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
const providerScenario: ProviderScenario = {
|
||||||
|
providerId: PROVIDER_ID,
|
||||||
|
modelId: MODEL_ID,
|
||||||
|
createContext: () => ({ messages: [{ role: "user", content: PROBE_PROMPT, timestamp: Date.now() }] }),
|
||||||
|
options: { env: { ACME_API_KEY: "resolved-acme-key" }, maxTokens: 32 },
|
||||||
|
validRequestReceived: () => validAcmeRequestReceived,
|
||||||
|
};
|
||||||
|
|
||||||
|
const customProviderScenario: ProviderScenario = {
|
||||||
|
providerId: CUSTOM_PROVIDER_ID,
|
||||||
|
modelId: CUSTOM_MODEL_ID,
|
||||||
|
createContext: () => ({ messages: [{ role: "user", content: CUSTOM_PROBE_PROMPT, timestamp: Date.now() }] }),
|
||||||
|
options: { env: { ACME_STREAM_API_KEY: "resolved-stream-key" }, maxTokens: 32 },
|
||||||
|
validRequestReceived: () => validAcmeStreamRequestReceived,
|
||||||
|
};
|
||||||
|
|
||||||
|
const ProviderAuthoringJudge = createProviderRuntimeJudge({
|
||||||
|
validRequestReceived: true,
|
||||||
|
provider: { id: PROVIDER_ID, name: "Acme" },
|
||||||
|
model: {
|
||||||
|
id: MODEL_ID,
|
||||||
|
name: "Acme Chat",
|
||||||
|
provider: PROVIDER_ID,
|
||||||
|
reasoning: false,
|
||||||
|
input: ["text"],
|
||||||
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||||
|
contextWindow: 32768,
|
||||||
|
maxTokens: 4096,
|
||||||
|
},
|
||||||
|
response: { text: PROBE_RESPONSE, stopReason: "stop", inputTokens: 3, outputTokens: 2 },
|
||||||
|
});
|
||||||
|
|
||||||
|
const CustomProviderJudge = createProviderRuntimeJudge({
|
||||||
|
validRequestReceived: true,
|
||||||
|
provider: { id: CUSTOM_PROVIDER_ID, name: "Acme Stream" },
|
||||||
|
model: {
|
||||||
|
id: CUSTOM_MODEL_ID,
|
||||||
|
name: "Acme Stream Chat",
|
||||||
|
provider: CUSTOM_PROVIDER_ID,
|
||||||
|
reasoning: false,
|
||||||
|
input: ["text"],
|
||||||
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||||
|
contextWindow: 16384,
|
||||||
|
maxTokens: 2048,
|
||||||
|
},
|
||||||
|
response: { text: CUSTOM_PROBE_RESPONSE, stopReason: "stop", inputTokens: 4, outputTokens: 3 },
|
||||||
|
});
|
||||||
|
|
||||||
|
const resolveProviderRuntime: RuntimeResolver = async (session, agentDir) => {
|
||||||
|
if (session.modelRuntime.getModel(PROVIDER_ID, MODEL_ID)) return session.modelRuntime;
|
||||||
|
return ModelRuntime.create({
|
||||||
|
modelsPath: join(agentDir, "models.json"),
|
||||||
|
authPath: join(agentDir, "auth.json"),
|
||||||
|
modelsStorePath: join(agentDir, "models-store.json"),
|
||||||
|
allowModelNetwork: false,
|
||||||
|
});
|
||||||
|
};
|
||||||
|
|
||||||
|
const providerHarnessTable = evalHarnessTable("Add OpenAI-compatible provider", {
|
||||||
|
baseline: createProviderHarness(
|
||||||
|
"system-prompt-without-docs",
|
||||||
|
providerScenario,
|
||||||
|
excludePiDocumentation,
|
||||||
|
resolveProviderRuntime,
|
||||||
|
),
|
||||||
|
candidate: createProviderHarness("default-system-prompt", providerScenario, undefined, resolveProviderRuntime),
|
||||||
|
});
|
||||||
|
|
||||||
|
describe.for(providerHarnessTable)("$name", ({ harness }) => {
|
||||||
|
describeEval(
|
||||||
|
"Add OpenAI-compatible provider",
|
||||||
|
{ harness, judges: [ProviderAuthoringJudge], judgeThreshold: null },
|
||||||
|
(it) => {
|
||||||
|
it("adds the provider", { timeout: 300_000 }, async ({ run }) => {
|
||||||
|
const result = await run([
|
||||||
|
{
|
||||||
|
type: "prompt",
|
||||||
|
content: `Can you add Acme to Pi as a provider? Its provider ID is ${PROVIDER_ID}, its API is at ${acmeBaseUrl}, and it uses OpenAI Chat Completions. Read its API key from the ACME_API_KEY environment variable.
|
||||||
|
|
||||||
|
The provider offers one model, ${MODEL_ID}, shown as “Acme Chat”. It accepts text, does not support reasoning, has a 32,768-token context window and a 4,096-token maximum output, and has no usage cost.`,
|
||||||
|
},
|
||||||
|
{ type: "reload" },
|
||||||
|
]);
|
||||||
|
expect(result.output.systemPromptHasGuidelines).toBe(true);
|
||||||
|
expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt");
|
||||||
|
});
|
||||||
|
},
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
const customProviderHarnessTable = evalHarnessTable("Add custom streaming provider", {
|
||||||
|
baseline: createProviderHarness("system-prompt-without-docs", customProviderScenario, excludePiDocumentation),
|
||||||
|
candidate: createProviderHarness("default-system-prompt", customProviderScenario),
|
||||||
|
});
|
||||||
|
|
||||||
|
describe.for(customProviderHarnessTable)("$name", ({ harness }) => {
|
||||||
|
describeEval(
|
||||||
|
"Add custom streaming provider",
|
||||||
|
{ harness, judges: [CustomProviderJudge], judgeThreshold: null },
|
||||||
|
(it) => {
|
||||||
|
it("adds the provider", { timeout: 300_000 }, async ({ run }) => {
|
||||||
|
const result = await run([
|
||||||
|
{
|
||||||
|
type: "prompt",
|
||||||
|
content: `Can you add Acme Stream to Pi as a provider? Its provider ID is ${CUSTOM_PROVIDER_ID}, its API is at ${acmeOrigin}, and its documentation is available at ${acmeOrigin}/docs. Read its credential from the ACME_STREAM_API_KEY environment variable.
|
||||||
|
|
||||||
|
It offers one model, ${CUSTOM_MODEL_ID}, shown as “Acme Stream Chat”. The model accepts text, does not support reasoning, has a 16,384-token context window and a 2,048-token maximum output, and has no usage cost.`,
|
||||||
|
},
|
||||||
|
{ type: "reload" },
|
||||||
|
]);
|
||||||
|
expect(result.output.systemPromptHasGuidelines).toBe(true);
|
||||||
|
expect(result.output.systemPromptHasPiDocs).toBe(harness.name === "default-system-prompt");
|
||||||
|
});
|
||||||
|
},
|
||||||
|
);
|
||||||
|
});
|
||||||
@@ -4,8 +4,8 @@ import { createPiCodingAgentHarness } from "./pi-harness.ts";
|
|||||||
|
|
||||||
const piCodingAgentHarness = createPiCodingAgentHarness({ noTools: "all" });
|
const piCodingAgentHarness = createPiCodingAgentHarness({ noTools: "all" });
|
||||||
|
|
||||||
describeEval("Pi Coding Agent smoke", { harness: piCodingAgentHarness }, (it) => {
|
describeEval("Answer a basic prompt", { harness: piCodingAgentHarness }, (it) => {
|
||||||
it("runs a basic prompt end to end", async ({ run }) => {
|
it("returns the expected answer", async ({ run }) => {
|
||||||
const result = await run("What's the capital of France? Respond with only the city name.");
|
const result = await run("What's the capital of France? Respond with only the city name.");
|
||||||
|
|
||||||
expect(result.output.trim()).toBe("Paris");
|
expect(result.output.trim()).toBe("Paris");
|
||||||
|
|||||||
@@ -111,6 +111,17 @@ export function deriveEvalGroupKey(input: unknown, repetition: number): string {
|
|||||||
return JSON.stringify([deriveInputKey(input), repetition]);
|
return JSON.stringify([deriveInputKey(input), repetition]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export function resolveEvalRepetitions(
|
||||||
|
explicit: number | undefined,
|
||||||
|
environmentValue: string | undefined = process.env.PI_EVAL_REPETITIONS,
|
||||||
|
): number {
|
||||||
|
const repetitions = explicit ?? (environmentValue === undefined ? 1 : Number(environmentValue));
|
||||||
|
if (!Number.isSafeInteger(repetitions) || repetitions < 1) {
|
||||||
|
throw new TypeError("repetitions must be a positive integer.");
|
||||||
|
}
|
||||||
|
return repetitions;
|
||||||
|
}
|
||||||
|
|
||||||
function validateOptions<TInput, TOutput extends JsonValue | undefined>(
|
function validateOptions<TInput, TOutput extends JsonValue | undefined>(
|
||||||
evalSet: string,
|
evalSet: string,
|
||||||
baseline: Harness<TInput, TOutput>,
|
baseline: Harness<TInput, TOutput>,
|
||||||
@@ -166,7 +177,7 @@ export function evalHarnessTable<TInput, TOutput extends JsonValue | undefined>(
|
|||||||
evalSet: string,
|
evalSet: string,
|
||||||
options: EvalHarnessTableOptions<TInput, TOutput>,
|
options: EvalHarnessTableOptions<TInput, TOutput>,
|
||||||
): EvalHarnessTableRow<TInput, TOutput>[] {
|
): EvalHarnessTableRow<TInput, TOutput>[] {
|
||||||
const repetitions = options.repetitions ?? 1;
|
const repetitions = resolveEvalRepetitions(options.repetitions);
|
||||||
const candidates = "candidate" in options ? [options.candidate] : options.candidates;
|
const candidates = "candidate" in options ? [options.candidate] : options.candidates;
|
||||||
validateOptions(evalSet, options.baseline, candidates, repetitions);
|
validateOptions(evalSet, options.baseline, candidates, repetitions);
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
import { randomUUID } from "node:crypto";
|
import { randomUUID } from "node:crypto";
|
||||||
import { appendFile, mkdir } from "node:fs/promises";
|
import { appendFile, mkdir, writeFile } from "node:fs/promises";
|
||||||
import { join } from "node:path";
|
import { join } from "node:path";
|
||||||
|
import { stripVTControlCharacters } from "node:util";
|
||||||
import type { Reporter, SerializedError, TestCase, TestModule, TestRunEndReason, Vitest } from "vitest/node";
|
import type { Reporter, SerializedError, TestCase, TestModule, TestRunEndReason, Vitest } from "vitest/node";
|
||||||
import { isHarnessRun } from "vitest-evals/harness";
|
import { isHarnessRun } from "vitest-evals/harness";
|
||||||
import { PI_SESSION_SNAPSHOT_ARTIFACT, persistEvalArtifactReferences } from "./artifacts.ts";
|
import { PI_SESSION_SNAPSHOT_ARTIFACT, persistEvalArtifactReferences } from "./artifacts.ts";
|
||||||
@@ -95,11 +96,11 @@ export default class EvalHarnessReporter implements Reporter {
|
|||||||
await appendHarnessRunReport(test);
|
await appendHarnessRunReport(test);
|
||||||
}
|
}
|
||||||
|
|
||||||
onTestRunEnd(
|
async onTestRunEnd(
|
||||||
modules: ReadonlyArray<TestModule>,
|
modules: ReadonlyArray<TestModule>,
|
||||||
_errors: ReadonlyArray<SerializedError>,
|
_errors: ReadonlyArray<SerializedError>,
|
||||||
reason: TestRunEndReason,
|
reason: TestRunEndReason,
|
||||||
): void {
|
): Promise<void> {
|
||||||
if (reason === "interrupted") {
|
if (reason === "interrupted") {
|
||||||
this.vitest?.logger.log("\nEval comparisons unavailable: test run interrupted.");
|
this.vitest?.logger.log("\nEval comparisons unavailable: test run interrupted.");
|
||||||
return;
|
return;
|
||||||
@@ -107,5 +108,17 @@ export default class EvalHarnessReporter implements Reporter {
|
|||||||
const report = summarizeHarnessComparisons(collectHarnessObservations(modules));
|
const report = summarizeHarnessComparisons(collectHarnessObservations(modules));
|
||||||
const formatted = formatHarnessComparisonReport(report);
|
const formatted = formatHarnessComparisonReport(report);
|
||||||
if (formatted) this.vitest?.logger.log(`\n${formatted}`);
|
if (formatted) this.vitest?.logger.log(`\n${formatted}`);
|
||||||
|
const artifactDirectory = process.env.PI_EVAL_ARTIFACT_DIR?.trim();
|
||||||
|
if (artifactDirectory) {
|
||||||
|
await mkdir(artifactDirectory, { recursive: true, mode: 0o700 });
|
||||||
|
await Promise.all([
|
||||||
|
writeFile(join(artifactDirectory, "report.json"), `${JSON.stringify(report, null, 2)}\n`, { mode: 0o600 }),
|
||||||
|
writeFile(
|
||||||
|
join(artifactDirectory, "report.txt"),
|
||||||
|
formatted ? `${stripVTControlCharacters(formatted)}\n` : "",
|
||||||
|
{ mode: 0o600 },
|
||||||
|
),
|
||||||
|
]);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,6 +5,7 @@ import {
|
|||||||
EVAL_HARNESS_ITERATION_ARTIFACT,
|
EVAL_HARNESS_ITERATION_ARTIFACT,
|
||||||
evalHarnessTable,
|
evalHarnessTable,
|
||||||
parseEvalHarnessIterationArtifact,
|
parseEvalHarnessIterationArtifact,
|
||||||
|
resolveEvalRepetitions,
|
||||||
} from "../../src/vitest-evals/harness-table.ts";
|
} from "../../src/vitest-evals/harness-table.ts";
|
||||||
|
|
||||||
describe("deriveEvalGroupKey", () => {
|
describe("deriveEvalGroupKey", () => {
|
||||||
@@ -30,6 +31,18 @@ describe("deriveEvalGroupKey", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("resolveEvalRepetitions", () => {
|
||||||
|
it("uses an explicit value before the environment default", () => {
|
||||||
|
expect(resolveEvalRepetitions(3, "5")).toBe(3);
|
||||||
|
expect(resolveEvalRepetitions(undefined, "5")).toBe(5);
|
||||||
|
expect(resolveEvalRepetitions(undefined, undefined)).toBe(1);
|
||||||
|
});
|
||||||
|
|
||||||
|
it.each(["0", "-1", "1.5", "nope"])("rejects invalid environment value %s", (value) => {
|
||||||
|
expect(() => resolveEvalRepetitions(undefined, value)).toThrow("positive integer");
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
function createFakeHarness(name: string) {
|
function createFakeHarness(name: string) {
|
||||||
return createHarness<{ id: string }, { harness: string; inputId: string }>({
|
return createHarness<{ id: string }, { harness: string; inputId: string }>({
|
||||||
name,
|
name,
|
||||||
|
|||||||
@@ -13,6 +13,7 @@ export const workspaceSourcePaths = {
|
|||||||
aiCompat: fileURLToPath(new URL("./packages/ai/src/compat.ts", import.meta.url)),
|
aiCompat: fileURLToPath(new URL("./packages/ai/src/compat.ts", import.meta.url)),
|
||||||
aiOAuth: fileURLToPath(new URL("./packages/ai/src/oauth.ts", import.meta.url)),
|
aiOAuth: fileURLToPath(new URL("./packages/ai/src/oauth.ts", import.meta.url)),
|
||||||
aiProviders: fileURLToPath(new URL("./packages/ai/src/providers", import.meta.url)),
|
aiProviders: fileURLToPath(new URL("./packages/ai/src/providers", import.meta.url)),
|
||||||
|
aiUtils: fileURLToPath(new URL("./packages/ai/src/utils", import.meta.url)),
|
||||||
agentIndex: fileURLToPath(new URL("./packages/agent/src/index.ts", import.meta.url)),
|
agentIndex: fileURLToPath(new URL("./packages/agent/src/index.ts", import.meta.url)),
|
||||||
agentNode: fileURLToPath(new URL("./packages/agent/src/node.ts", import.meta.url)),
|
agentNode: fileURLToPath(new URL("./packages/agent/src/node.ts", import.meta.url)),
|
||||||
protocolIndex: fileURLToPath(new URL("./packages/protocol/src/index.ts", import.meta.url)),
|
protocolIndex: fileURLToPath(new URL("./packages/protocol/src/index.ts", import.meta.url)),
|
||||||
@@ -37,6 +38,10 @@ export default defineConfig({
|
|||||||
{ find: /^@earendil-works\/pi-ai$/, replacement: workspaceSourcePaths.aiIndex },
|
{ find: /^@earendil-works\/pi-ai$/, replacement: workspaceSourcePaths.aiIndex },
|
||||||
{ find: /^@earendil-works\/pi-ai\/compat$/, replacement: workspaceSourcePaths.aiCompat },
|
{ find: /^@earendil-works\/pi-ai\/compat$/, replacement: workspaceSourcePaths.aiCompat },
|
||||||
{ find: /^@earendil-works\/pi-ai\/oauth$/, replacement: workspaceSourcePaths.aiOAuth },
|
{ find: /^@earendil-works\/pi-ai\/oauth$/, replacement: workspaceSourcePaths.aiOAuth },
|
||||||
|
{
|
||||||
|
find: /^@earendil-works\/pi-ai\/utils\/(.+)$/,
|
||||||
|
replacement: `${workspaceSourcePaths.aiUtils}/$1.ts`,
|
||||||
|
},
|
||||||
{
|
{
|
||||||
find: /^@earendil-works\/pi-ai\/providers\/(.+)$/,
|
find: /^@earendil-works\/pi-ai\/providers\/(.+)$/,
|
||||||
replacement: `${workspaceSourcePaths.aiProviders}/$1.ts`,
|
replacement: `${workspaceSourcePaths.aiProviders}/$1.ts`,
|
||||||
|
|||||||
Reference in New Issue
Block a user