mirror of
https://github.com/luckyyzh/pi-agent-integrated.git
synced 2026-10-03 11:09:34 +00:00
feat: integrate Pi backend and Pi Web
This commit is contained in:
@@ -0,0 +1 @@
|
||||
.eval/
|
||||
@@ -0,0 +1,153 @@
|
||||
# Pi evals
|
||||
|
||||
Pi evals are behavioral, model-backed checks for Pi workflows. They adapt a real `AgentSession` to `vitest-evals`, run
|
||||
it in isolated temporary project and agent directories, and attach native Pi session artifacts.
|
||||
Use them to measure end-to-end behavior and compare prompts, tools, skills, models, or other harness configurations.
|
||||
|
||||
## Running evals
|
||||
|
||||
Run from the repository root with a default provider and model:
|
||||
|
||||
```bash
|
||||
npm run eval -- --provider openai --model gpt-5.6-sol
|
||||
```
|
||||
|
||||
The equivalent environment variables are:
|
||||
|
||||
```bash
|
||||
PI_PROVIDER=openai PI_MODEL=gpt-5.6-sol npm run eval
|
||||
```
|
||||
|
||||
CLI values take precedence and become defaults for harnesses that do not select a model explicitly. Provider and model must be supplied together. The runner also allows no default when every executed harness configures its own model.
|
||||
Authentication comes from Pi's normal `ModelRuntime`, including Pi subscription credentials and provider API-key
|
||||
environment variables.
|
||||
|
||||
Additional arguments are forwarded to Vitest:
|
||||
|
||||
```bash
|
||||
npm run eval -- src/extensions.eval.ts
|
||||
npm run eval -- -t "creates, reloads, and uses"
|
||||
```
|
||||
|
||||
Each invocation prints an ignored `.eval/` artifact directory. `runs.jsonl` indexes completed harness runs and their
|
||||
native Pi session JSONL attachments under `sessions/`. These files may contain prompts, responses, source code, and tool
|
||||
output.
|
||||
|
||||
## Writing evals
|
||||
|
||||
Follow [`vitest-evals`](https://github.com/getsentry/vitest-evals) for general suite, judge, assertion, and normalized
|
||||
trace guidance. Pi-specific evals use `createPiCodingAgentHarness(...)` from `src/pi-harness.ts`, with one harness bound
|
||||
to each `describeEval(...)` suite:
|
||||
|
||||
```ts
|
||||
import { expect } from "vitest";
|
||||
import { describeEval } from "vitest-evals";
|
||||
import { createPiCodingAgentHarness } from "./pi-harness.ts";
|
||||
|
||||
const harness = createPiCodingAgentHarness({ noTools: "all" });
|
||||
|
||||
describeEval("Pi smoke", { harness }, (it) => {
|
||||
it("answers a factual question", async ({ run }) => {
|
||||
const result = await run("What is the capital of France? Reply with only the city name.");
|
||||
expect(result.output).toBe("Paris");
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
### Configuring the Pi harness
|
||||
|
||||
`createPiCodingAgentHarness(...)` accepts:
|
||||
|
||||
- `name`: stable harness identity used by reports and comparisons.
|
||||
- `model`: optional `{ provider, id }` selection. It overrides the runner's default model.
|
||||
- `noTools`: Pi's tool-disable configuration.
|
||||
- `transformSystemPrompt`: transforms the complete default prompt before the eval starts.
|
||||
- `output`: transforms the final response and `AgentSession` into a JSON-safe domain result.
|
||||
|
||||
An explicitly selected model makes model-comparison harnesses independent of the runner default:
|
||||
|
||||
```ts
|
||||
const harness = createPiCodingAgentHarness({
|
||||
name: "claude-opus-4-6",
|
||||
model: { provider: "anthropic", id: "claude-opus-4-6" },
|
||||
});
|
||||
```
|
||||
|
||||
A run accepts either one prompt or a sequence of prompt and reload steps. Reload steps are useful when the preceding
|
||||
prompt creates or changes Pi resources:
|
||||
|
||||
```ts
|
||||
const result = await run([
|
||||
{ type: "prompt", content: "Create a Pi extension." },
|
||||
{ type: "reload" },
|
||||
{ type: "prompt", content: "Use the extension." },
|
||||
]);
|
||||
```
|
||||
|
||||
### Transforming harness output
|
||||
|
||||
Use `output` to expose scenario-specific, JSON-safe behavior without adding that behavior to the generic Pi adapter:
|
||||
|
||||
```ts
|
||||
const harness = createPiCodingAgentHarness({
|
||||
output: ({ response, session }) => ({
|
||||
response,
|
||||
activeTools: session.getActiveToolNames(),
|
||||
extensionErrors: session.resourceLoader.getExtensions().errors,
|
||||
}),
|
||||
});
|
||||
```
|
||||
|
||||
Assert application behavior on `result.output`. Assert model and tool traces on `result.session`, using
|
||||
`vitest-evals` helpers such as `toolCalls(...)`.
|
||||
|
||||
### Writing comparative eval sets
|
||||
|
||||
Use `evalHarnessTable(...)` with Vitest's native `describe.for(...)` to run the same inputs against multiple harnesses.
|
||||
Harnesses may differ by prompt, tools, skills, model, or any other Pi configuration:
|
||||
|
||||
```ts
|
||||
import { describe } from "vitest";
|
||||
import { createJudge, describeEval } from "vitest-evals";
|
||||
import { evalHarnessTable } from "./vitest-evals/harness-table.ts";
|
||||
|
||||
const TargetTaskJudge = createJudge<string, string>("TargetTaskJudge", ({ output }) => ({
|
||||
score: output === "expected result" ? 1 : 0,
|
||||
}));
|
||||
|
||||
const harnessTable = evalHarnessTable(
|
||||
"target skill effectiveness",
|
||||
{
|
||||
baseline: withoutTargetSkillHarness,
|
||||
candidate: withTargetSkillHarness,
|
||||
repetitions: 6,
|
||||
},
|
||||
);
|
||||
|
||||
describe.for(harnessTable)("$name repetition $repetition", ({ harness }) => {
|
||||
describeEval("target skill effectiveness", { harness, judges: [TargetTaskJudge], judgeThreshold: null }, (it) => {
|
||||
it("completes the target task", async ({ run }) => {
|
||||
await run("Complete the target task.");
|
||||
});
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
Comparative suites should record correctness with deterministic or model-backed judges and set `judgeThreshold: null`.
|
||||
This keeps a low score as an observation instead of making the Vitest invocation fail. Use hard assertions only for
|
||||
suite invariants and infrastructure contracts. `expect.soft(...)` still fails the test and is not a scoring mechanism.
|
||||
|
||||
The Pi harness snapshots native session JSONL before deleting its temporary workspace. An eval-only `afterEach` hook
|
||||
registers that snapshot against the explicit Vitest test task before reporters run.
|
||||
|
||||
Harness names must be stable and unique within an eval set. The grouping key combines repetition with a non-empty string
|
||||
`input.id` when available, otherwise with a SHA-256 hash of strict canonical JSON input. Use `candidate` for one treatment
|
||||
or `candidates` for multiple treatments. Each candidate is compared only with the declared baseline. For each matched
|
||||
input and repetition, the reporter computes pass-rate lift from each run's recorded average judge score, treating a score
|
||||
of at least `1` as passing. Lift is the candidate pass rate minus the baseline pass rate, in percentage points. Missing
|
||||
judge scores are reported as incomplete observations. Tokens, latency, and estimated cost remain separate
|
||||
candidate-minus-baseline paired deltas; missing telemetry remains unavailable. If execution-order randomization becomes
|
||||
necessary, use Vitest's built-in sequence shuffling.
|
||||
|
||||
See the [`skill-eval-harness`](https://github.com/adewale/skill-eval-harness/) guidance for comparative-eval methodology,
|
||||
repetition strategy, trustworthy judges, and telemetry interpretation.
|
||||
@@ -0,0 +1,20 @@
|
||||
{
|
||||
"name": "@earendil-works/pi-evals",
|
||||
"version": "0.83.0",
|
||||
"private": true,
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"clean": "shx rm -rf .eval",
|
||||
"eval": "node scripts/run-evals.mjs",
|
||||
"test": "vitest run --config vitest.test.config.ts"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@earendil-works/pi-ai": "^0.83.0",
|
||||
"@earendil-works/pi-coding-agent": "^0.83.0",
|
||||
"@types/node": "24.12.4",
|
||||
"shx": "0.4.0",
|
||||
"typescript": "5.9.3",
|
||||
"vitest-evals": "0.15.0",
|
||||
"vitest": "4.1.9"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { mkdirSync } from "node:fs";
|
||||
import { spawnSync } from "node:child_process";
|
||||
import { createRequire } from "node:module";
|
||||
import { dirname, resolve } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
|
||||
const packageRoot = resolve(dirname(fileURLToPath(import.meta.url)), "..");
|
||||
const artifactDirectory = process.env.PI_EVAL_ARTIFACT_DIR
|
||||
? resolve(packageRoot, process.env.PI_EVAL_ARTIFACT_DIR)
|
||||
: resolve(
|
||||
packageRoot,
|
||||
".eval",
|
||||
`${new Date().toISOString().replaceAll(":", "-")}_${randomUUID()}`,
|
||||
);
|
||||
const args = process.argv.slice(2);
|
||||
let provider;
|
||||
let model;
|
||||
let hasCliModelSelection = false;
|
||||
const vitestArgs = [];
|
||||
|
||||
for (let index = 0; index < args.length; index += 1) {
|
||||
const arg = args[index];
|
||||
if (arg === "--provider" || arg === "--model") {
|
||||
const value = args[index + 1];
|
||||
if (!value || value.startsWith("-")) {
|
||||
console.error(`Missing value for ${arg}`);
|
||||
process.exit(1);
|
||||
}
|
||||
if (arg === "--provider") provider = value;
|
||||
else model = value;
|
||||
hasCliModelSelection = true;
|
||||
index += 1;
|
||||
continue;
|
||||
}
|
||||
if (arg.startsWith("--provider=")) {
|
||||
provider = arg.slice("--provider=".length);
|
||||
hasCliModelSelection = true;
|
||||
continue;
|
||||
}
|
||||
if (arg.startsWith("--model=")) {
|
||||
model = arg.slice("--model=".length);
|
||||
hasCliModelSelection = true;
|
||||
continue;
|
||||
}
|
||||
vitestArgs.push(arg);
|
||||
}
|
||||
|
||||
provider = provider?.trim() || undefined;
|
||||
model = model?.trim() || undefined;
|
||||
if (hasCliModelSelection) {
|
||||
if (!provider || !model) {
|
||||
console.error("CLI model selection requires both --provider and --model.");
|
||||
process.exit(1);
|
||||
}
|
||||
} else {
|
||||
provider = process.env.PI_PROVIDER?.trim() || undefined;
|
||||
model = process.env.PI_MODEL?.trim() || undefined;
|
||||
if (Boolean(provider) !== Boolean(model)) {
|
||||
console.error("Default model selection requires both PI_PROVIDER and PI_MODEL.");
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const vitestPackagePath = require.resolve("vitest/package.json");
|
||||
const vitestCliPath = resolve(dirname(vitestPackagePath), "vitest.mjs");
|
||||
|
||||
mkdirSync(artifactDirectory, { recursive: true, mode: 0o700 });
|
||||
console.error(`[eval] default-model=${provider && model ? `${provider}/${model}` : "none"}`);
|
||||
console.error(`[eval] artifacts=${artifactDirectory}`);
|
||||
const childEnvironment = {
|
||||
...process.env,
|
||||
PI_EVAL_ARTIFACT_DIR: artifactDirectory,
|
||||
};
|
||||
if (provider && model) {
|
||||
childEnvironment.PI_PROVIDER = provider;
|
||||
childEnvironment.PI_MODEL = model;
|
||||
} else {
|
||||
delete childEnvironment.PI_PROVIDER;
|
||||
delete childEnvironment.PI_MODEL;
|
||||
}
|
||||
const result = spawnSync(
|
||||
process.execPath,
|
||||
[vitestCliPath, "run", "--config", "vitest.config.ts", ...vitestArgs],
|
||||
{
|
||||
cwd: packageRoot,
|
||||
stdio: "inherit",
|
||||
env: childEnvironment,
|
||||
},
|
||||
);
|
||||
|
||||
if (result.error) {
|
||||
throw result.error;
|
||||
}
|
||||
|
||||
process.exit(result.status ?? 1);
|
||||
@@ -0,0 +1,140 @@
|
||||
import { existsSync, readFileSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
import { describe, expect } from "vitest";
|
||||
import { createJudge, describeEval } from "vitest-evals";
|
||||
import { createPiCodingAgentHarness, type PiCodingAgentInput } from "./pi-harness.ts";
|
||||
import { recordEvalSourceArtifact } from "./vitest-evals/artifacts.ts";
|
||||
import { evalHarnessTable } from "./vitest-evals/harness-table.ts";
|
||||
|
||||
type ExtensionAuthoringOutput = {
|
||||
response: string;
|
||||
systemPromptHasGuidelines: boolean;
|
||||
systemPromptHasPiDocs: boolean;
|
||||
extensionErrors: Array<{ path: string; error: string }>;
|
||||
loadedExtensions: Array<{ path: string; tools: string[] }>;
|
||||
extensionSource: string | null;
|
||||
};
|
||||
|
||||
function createExtensionAuthoringHarness(name: string, transformSystemPrompt?: (defaultPrompt: string) => string) {
|
||||
return createPiCodingAgentHarness({
|
||||
name,
|
||||
...(transformSystemPrompt ? { transformSystemPrompt } : {}),
|
||||
output: ({ response, session }) => {
|
||||
const extensions = session.resourceLoader.getExtensions();
|
||||
const extensionPath = join(session.sessionManager.getCwd(), ".pi", "extensions", "hello.ts");
|
||||
const extensionSource = existsSync(extensionPath) ? readFileSync(extensionPath, "utf8") : null;
|
||||
return {
|
||||
response,
|
||||
systemPromptHasGuidelines: session.systemPrompt.includes("\nGuidelines:\n"),
|
||||
systemPromptHasPiDocs: session.systemPrompt.includes("\nPi documentation (read only"),
|
||||
extensionErrors: extensions.errors,
|
||||
loadedExtensions: extensions.extensions.map(({ path, tools }) => ({
|
||||
path,
|
||||
tools: [...tools.keys()],
|
||||
})),
|
||||
extensionSource,
|
||||
};
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
function excludeGuidelinesAndDocumentation(defaultPrompt: string): string {
|
||||
const guidelinesStart = defaultPrompt.indexOf("\nGuidelines:\n");
|
||||
if (guidelinesStart === -1) throw new Error("Default Pi system prompt has no Guidelines section.");
|
||||
return defaultPrompt.slice(0, guidelinesStart);
|
||||
}
|
||||
|
||||
function prepareDefaultPromptOverride(defaultPrompt: string): string {
|
||||
const cwdStart = defaultPrompt.lastIndexOf("\nCurrent working directory: ");
|
||||
if (cwdStart === -1) throw new Error("Default Pi system prompt has no working-directory section.");
|
||||
return defaultPrompt.slice(0, cwdStart);
|
||||
}
|
||||
|
||||
const ExtensionAuthoringJudge = createJudge<PiCodingAgentInput, ExtensionAuthoringOutput>(
|
||||
"ExtensionAuthoringJudge",
|
||||
({ output, toolCalls }) => {
|
||||
const failures: string[] = [];
|
||||
if (output.extensionSource === null) {
|
||||
failures.push("generated extension source is unavailable");
|
||||
} else {
|
||||
const imports = Array.from(
|
||||
output.extensionSource.matchAll(/\b(?:from|import)\s+["']([^"']+)["']/g),
|
||||
(match) => match[1],
|
||||
);
|
||||
if (!imports.includes("@earendil-works/pi-coding-agent")) {
|
||||
failures.push("extension does not import the canonical @earendil-works/pi-coding-agent package");
|
||||
}
|
||||
if (imports.some((specifier) => specifier.startsWith("@mariozechner/"))) {
|
||||
failures.push("extension imports a legacy @mariozechner package");
|
||||
}
|
||||
if (imports.some((specifier) => specifier.startsWith("@sinclair/typebox"))) {
|
||||
failures.push('extension imports legacy "@sinclair/typebox" instead of "typebox"');
|
||||
}
|
||||
}
|
||||
if (output.extensionErrors.length > 0) failures.push("extension loader reported errors");
|
||||
if (!output.loadedExtensions.some(({ tools }) => tools.includes("hello"))) {
|
||||
failures.push('no loaded extension registered the "hello" tool');
|
||||
}
|
||||
if (
|
||||
!toolCalls.some(
|
||||
(call) =>
|
||||
call.name === "hello" &&
|
||||
call.status === "ok" &&
|
||||
call.arguments?.name === "Bob" &&
|
||||
call.result === "Hello, Bob!",
|
||||
)
|
||||
) {
|
||||
failures.push('no successful hello({ name: "Bob" }) call returned "Hello, Bob!"');
|
||||
}
|
||||
if (output.response !== "Hello, Bob!") failures.push('final response was not exactly "Hello, Bob!"');
|
||||
|
||||
return {
|
||||
score: failures.length === 0 ? 1 : 0,
|
||||
metadata: {
|
||||
rationale: failures.length === 0 ? "Extension authoring workflow completed." : failures.join("; "),
|
||||
},
|
||||
};
|
||||
},
|
||||
);
|
||||
|
||||
const extensionHarnessTable = evalHarnessTable("Pi extension authoring system prompt", {
|
||||
baseline: createExtensionAuthoringHarness("system-prompt-without-docs", excludeGuidelinesAndDocumentation),
|
||||
candidate: createExtensionAuthoringHarness("default-system-prompt", prepareDefaultPromptOverride),
|
||||
});
|
||||
|
||||
describe.for(extensionHarnessTable)("$name", ({ harness }) => {
|
||||
describeEval(
|
||||
"Pi extension authoring system prompt",
|
||||
{ harness, judges: [ExtensionAuthoringJudge], judgeThreshold: null },
|
||||
(it) => {
|
||||
it("creates, reloads, and uses a hello extension", async ({ run, task }) => {
|
||||
const result = await run([
|
||||
{
|
||||
type: "prompt",
|
||||
content:
|
||||
"Create a Pi extension with a hello tool that takes a name and returns a greeting. For example, passing Bob should return `Hello, Bob!`.",
|
||||
},
|
||||
{ type: "reload" },
|
||||
{
|
||||
type: "prompt",
|
||||
content:
|
||||
"Use the hello tool to greet Bob. Respond with exactly the tool's greeting and nothing else.",
|
||||
},
|
||||
]);
|
||||
if (result.output.extensionSource !== null) {
|
||||
const runId = result.artifacts?.runId;
|
||||
if (typeof runId !== "string") throw new Error("Pi eval run did not record a run ID.");
|
||||
await recordEvalSourceArtifact(task, runId, {
|
||||
name: "hello.ts",
|
||||
contentType: "text/typescript",
|
||||
body: result.output.extensionSource,
|
||||
bodyEncoding: "utf-8",
|
||||
});
|
||||
}
|
||||
const expectsFullPrompt = harness.name === "default-system-prompt";
|
||||
expect(result.output.systemPromptHasGuidelines).toBe(expectsFullPrompt);
|
||||
expect(result.output.systemPromptHasPiDocs).toBe(expectsFullPrompt);
|
||||
});
|
||||
},
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,257 @@
|
||||
import { existsSync } from "node:fs";
|
||||
import { mkdir, mkdtemp, readFile, rm } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { performance } from "node:perf_hooks";
|
||||
import { contentText } from "@earendil-works/pi-ai";
|
||||
import {
|
||||
type AgentSession,
|
||||
type CreateAgentSessionOptions,
|
||||
createAgentSessionFromServices,
|
||||
createAgentSessionServices,
|
||||
ModelRuntime,
|
||||
SessionManager,
|
||||
SettingsManager,
|
||||
} from "@earendil-works/pi-coding-agent";
|
||||
import {
|
||||
createHarness,
|
||||
type Harness,
|
||||
type HarnessContext,
|
||||
type JsonValue,
|
||||
normalizeRecord,
|
||||
type SimpleHarnessResult,
|
||||
type TranscriptEvent,
|
||||
toJsonValue,
|
||||
} from "vitest-evals/harness";
|
||||
import { PI_SESSION_SNAPSHOT_ARTIFACT } from "./vitest-evals/artifacts.ts";
|
||||
|
||||
export type PiCodingAgentInput = string | Array<{ type: "prompt"; content: string } | { type: "reload" }>;
|
||||
|
||||
type PiCodingAgentModelSelection = {
|
||||
provider: string;
|
||||
id: string;
|
||||
};
|
||||
|
||||
type PiCodingAgentHarnessOptions = {
|
||||
name?: string;
|
||||
model?: PiCodingAgentModelSelection;
|
||||
noTools?: CreateAgentSessionOptions["noTools"];
|
||||
transformSystemPrompt?: (defaultPrompt: string) => string;
|
||||
};
|
||||
|
||||
type PiCodingAgentHarnessWithOutput<TOutput extends JsonValue> = PiCodingAgentHarnessOptions & {
|
||||
output: (args: { response: string; session: AgentSession }) => TOutput | Promise<TOutput>;
|
||||
};
|
||||
|
||||
export function resolveModelSelection(
|
||||
explicitModel: PiCodingAgentModelSelection | undefined,
|
||||
environment: { PI_PROVIDER?: string; PI_MODEL?: string } = process.env,
|
||||
): PiCodingAgentModelSelection {
|
||||
const provider = (explicitModel?.provider ?? environment.PI_PROVIDER)?.trim();
|
||||
const id = (explicitModel?.id ?? environment.PI_MODEL)?.trim();
|
||||
if (!provider || !id) {
|
||||
throw new Error("Select a harness model explicitly or set both PI_PROVIDER and PI_MODEL as defaults.");
|
||||
}
|
||||
return { provider, id };
|
||||
}
|
||||
|
||||
function toTranscriptEvents(messages: AgentSession["messages"]): TranscriptEvent[] {
|
||||
const events: TranscriptEvent[] = [];
|
||||
for (const message of messages) {
|
||||
if (message.role === "user") {
|
||||
events.push({ type: "message", role: "user", content: contentText(message.content) });
|
||||
} else if (message.role === "assistant") {
|
||||
const text = contentText(message.content);
|
||||
if (text) events.push({ type: "message", role: "assistant", content: text });
|
||||
for (const part of message.content) {
|
||||
if (part.type === "toolCall") {
|
||||
events.push({
|
||||
type: "tool_call",
|
||||
id: part.id,
|
||||
name: part.name,
|
||||
arguments: normalizeRecord(part.arguments),
|
||||
});
|
||||
}
|
||||
}
|
||||
} else if (message.role === "toolResult") {
|
||||
const text = contentText(message.content);
|
||||
events.push({
|
||||
type: "tool_result",
|
||||
toolCallId: message.toolCallId,
|
||||
name: message.toolName,
|
||||
content: message.content.every((part) => part.type === "text") ? text : toJsonValue(message.content),
|
||||
...(message.isError ? { error: { message: text || "Tool failed" } } : {}),
|
||||
});
|
||||
}
|
||||
}
|
||||
return events;
|
||||
}
|
||||
|
||||
async function promptAgent(session: AgentSession, input: string, signal: AbortSignal | undefined): Promise<string> {
|
||||
signal?.throwIfAborted();
|
||||
const previousMessageCount = session.messages.length;
|
||||
await session.prompt(input);
|
||||
const assistant = session.messages
|
||||
.slice(previousMessageCount)
|
||||
.reverse()
|
||||
.find((message) => message.role === "assistant");
|
||||
if (!assistant) throw new Error("Agent run completed without an assistant message.");
|
||||
if (assistant.stopReason !== "stop") {
|
||||
throw new Error(
|
||||
assistant.errorMessage ?? `Agent run ended with unexpected stop reason: ${assistant.stopReason}.`,
|
||||
);
|
||||
}
|
||||
const output = session.getLastAssistantText();
|
||||
if (!output) throw new Error("Agent run produced no assistant text.");
|
||||
return output;
|
||||
}
|
||||
|
||||
async function runPiCodingAgent<TOutput extends JsonValue>(
|
||||
input: PiCodingAgentInput,
|
||||
signal: AbortSignal | undefined,
|
||||
setArtifact: HarnessContext["setArtifact"],
|
||||
options: PiCodingAgentHarnessOptions | PiCodingAgentHarnessWithOutput<TOutput>,
|
||||
): Promise<SimpleHarnessResult<string | TOutput>> {
|
||||
const startedAt = performance.now();
|
||||
signal?.throwIfAborted();
|
||||
const selection = resolveModelSelection(options.model);
|
||||
const modelRuntime = await ModelRuntime.create();
|
||||
const model = modelRuntime.getModel(selection.provider, selection.id);
|
||||
if (!model) throw new Error(`Eval model not found: ${selection.provider}/${selection.id}`);
|
||||
|
||||
const root = await mkdtemp(join(tmpdir(), "pi-eval-"));
|
||||
const cwd = join(root, "workspace");
|
||||
const agentDir = join(root, "agent");
|
||||
let transformedSystemPrompt: string | undefined;
|
||||
let sessionManager: SessionManager | undefined;
|
||||
let session: AgentSession | undefined;
|
||||
let outcome: { success: true; result: SimpleHarnessResult<string | TOutput> } | { success: false; error: unknown };
|
||||
try {
|
||||
await Promise.all([mkdir(cwd), mkdir(agentDir)]);
|
||||
const services = await createAgentSessionServices({
|
||||
cwd,
|
||||
agentDir,
|
||||
modelRuntime,
|
||||
settingsManager: SettingsManager.inMemory(),
|
||||
...(options.transformSystemPrompt
|
||||
? { resourceLoaderOptions: { systemPromptOverride: () => transformedSystemPrompt } }
|
||||
: {}),
|
||||
});
|
||||
signal?.throwIfAborted();
|
||||
sessionManager = SessionManager.create(cwd, join(root, "sessions"));
|
||||
setArtifact("runId", sessionManager.getSessionId());
|
||||
session = (
|
||||
await createAgentSessionFromServices({
|
||||
services,
|
||||
sessionManager,
|
||||
model,
|
||||
thinkingLevel: "off",
|
||||
noTools: options.noTools,
|
||||
})
|
||||
).session;
|
||||
|
||||
const evalSession = session;
|
||||
if (options.transformSystemPrompt) {
|
||||
transformedSystemPrompt = options.transformSystemPrompt(evalSession.systemPrompt);
|
||||
if (!transformedSystemPrompt.trim()) throw new Error("Transformed eval system prompt must not be empty.");
|
||||
await evalSession.reload();
|
||||
}
|
||||
let abortPromise: Promise<void> | undefined;
|
||||
const abort = () => {
|
||||
abortPromise ??= evalSession.abort();
|
||||
};
|
||||
signal?.addEventListener("abort", abort, { once: true });
|
||||
try {
|
||||
signal?.throwIfAborted();
|
||||
if (evalSession.extensionRunner.getExtensionPaths().length !== 0) {
|
||||
throw new Error("Expected an isolated eval session to start without extensions.");
|
||||
}
|
||||
const steps = typeof input === "string" ? [{ type: "prompt" as const, content: input }] : input;
|
||||
let response: string | undefined;
|
||||
for (const step of steps) {
|
||||
if (step.type === "prompt") {
|
||||
response = await promptAgent(evalSession, step.content, signal);
|
||||
} else {
|
||||
await evalSession.reload();
|
||||
}
|
||||
}
|
||||
if (response === undefined) throw new Error("Pi eval input must include at least one prompt step.");
|
||||
const output = "output" in options ? await options.output({ response, session: evalSession }) : response;
|
||||
const stats = evalSession.getSessionStats();
|
||||
const hasPricing = [model.cost, ...(model.cost.tiers ?? [])].some(
|
||||
({ input, output, cacheRead, cacheWrite }) => input > 0 || output > 0 || cacheRead > 0 || cacheWrite > 0,
|
||||
);
|
||||
outcome = {
|
||||
success: true,
|
||||
result: {
|
||||
output,
|
||||
events: toTranscriptEvents(evalSession.messages),
|
||||
usage: {
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
inputTokens: stats.tokens.input,
|
||||
outputTokens: stats.tokens.output,
|
||||
totalTokens: stats.tokens.total,
|
||||
toolCalls: stats.toolCalls,
|
||||
metadata: {
|
||||
cacheReadTokens: stats.tokens.cacheRead,
|
||||
cacheWriteTokens: stats.tokens.cacheWrite,
|
||||
...(hasPricing ? { estimatedCostUsd: stats.cost } : {}),
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
} finally {
|
||||
signal?.removeEventListener("abort", abort);
|
||||
if (abortPromise) await abortPromise;
|
||||
}
|
||||
} catch (error) {
|
||||
outcome = { success: false, error };
|
||||
}
|
||||
|
||||
const cleanupErrors: unknown[] = [];
|
||||
if (sessionManager) {
|
||||
try {
|
||||
const sessionPath = sessionManager.getSessionFile();
|
||||
if (sessionPath && existsSync(sessionPath)) {
|
||||
setArtifact(PI_SESSION_SNAPSHOT_ARTIFACT, await readFile(sessionPath, "utf8"));
|
||||
}
|
||||
} catch (error) {
|
||||
cleanupErrors.push(error);
|
||||
}
|
||||
}
|
||||
try {
|
||||
session?.dispose();
|
||||
} catch (error) {
|
||||
cleanupErrors.push(error);
|
||||
}
|
||||
try {
|
||||
await rm(root, { recursive: true, force: true });
|
||||
} catch (error) {
|
||||
cleanupErrors.push(error);
|
||||
}
|
||||
|
||||
if (!outcome.success) {
|
||||
if (cleanupErrors.length === 0) throw outcome.error;
|
||||
throw new AggregateError([outcome.error, ...cleanupErrors], "Agent run failed and cleanup also failed.");
|
||||
}
|
||||
if (cleanupErrors.length === 1) throw cleanupErrors[0];
|
||||
if (cleanupErrors.length > 1) throw new AggregateError(cleanupErrors, "Agent cleanup failed.");
|
||||
return {
|
||||
...outcome.result,
|
||||
timings: { totalMs: performance.now() - startedAt },
|
||||
};
|
||||
}
|
||||
|
||||
export function createPiCodingAgentHarness<TOutput extends JsonValue>(
|
||||
options: PiCodingAgentHarnessWithOutput<TOutput>,
|
||||
): Harness<PiCodingAgentInput, TOutput>;
|
||||
export function createPiCodingAgentHarness(options?: PiCodingAgentHarnessOptions): Harness<PiCodingAgentInput, string>;
|
||||
export function createPiCodingAgentHarness<TOutput extends JsonValue>(
|
||||
options: PiCodingAgentHarnessOptions | PiCodingAgentHarnessWithOutput<TOutput> = {},
|
||||
) {
|
||||
return createHarness<PiCodingAgentInput, string | TOutput>({
|
||||
name: options.name ?? "pi-coding-agent",
|
||||
run: ({ input, signal, setArtifact }) => runPiCodingAgent(input, signal, setArtifact, options),
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
import { expect } from "vitest";
|
||||
import { describeEval } from "vitest-evals";
|
||||
import { createPiCodingAgentHarness } from "./pi-harness.ts";
|
||||
|
||||
const piCodingAgentHarness = createPiCodingAgentHarness({ noTools: "all" });
|
||||
|
||||
describeEval("Pi Coding Agent smoke", { harness: piCodingAgentHarness }, (it) => {
|
||||
it("runs a basic prompt end to end", async ({ run }) => {
|
||||
const result = await run("What's the capital of France? Respond with only the city name.");
|
||||
|
||||
expect(result.output.trim()).toBe("Paris");
|
||||
expect(result.errors).toEqual([]);
|
||||
expect(result.usage.provider).toBe(process.env.PI_PROVIDER);
|
||||
expect(result.usage.model).toBe(process.env.PI_MODEL);
|
||||
expect(result.usage.totalTokens).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,113 @@
|
||||
import { createHash } from "node:crypto";
|
||||
import { mkdir, writeFile } from "node:fs/promises";
|
||||
import { basename, join, relative } from "node:path";
|
||||
import {
|
||||
type RunnerTestCase,
|
||||
recordArtifact,
|
||||
type TestArtifact,
|
||||
type TestArtifactBase,
|
||||
type TestAttachment,
|
||||
} from "vitest";
|
||||
import type { HarnessRun } from "vitest-evals/harness";
|
||||
|
||||
export const PI_SESSION_SNAPSHOT_ARTIFACT = "piSessionJsonl";
|
||||
|
||||
const evalSessionArtifactKey = Symbol("pi-evals-session-artifact");
|
||||
const evalSourceArtifactKey = Symbol("pi-evals-source-artifact");
|
||||
|
||||
interface PiSessionAttachment extends TestAttachment {
|
||||
name: "session.jsonl";
|
||||
contentType: "application/jsonl";
|
||||
body: string;
|
||||
bodyEncoding: "utf-8";
|
||||
}
|
||||
|
||||
export interface SourceAttachment extends TestAttachment {
|
||||
name: string;
|
||||
contentType: string;
|
||||
body: string;
|
||||
bodyEncoding: "utf-8";
|
||||
}
|
||||
|
||||
interface PiSessionArtifact extends TestArtifactBase {
|
||||
type: "@earendil-works/pi-evals:session";
|
||||
runId: string;
|
||||
attachments: [PiSessionAttachment] | [];
|
||||
}
|
||||
|
||||
interface SourceArtifact extends TestArtifactBase {
|
||||
type: "@earendil-works/pi-evals:source";
|
||||
runId: string;
|
||||
attachments: [SourceAttachment] | [];
|
||||
}
|
||||
|
||||
declare module "vitest" {
|
||||
interface TestArtifactRegistry {
|
||||
[evalSessionArtifactKey]: PiSessionArtifact;
|
||||
[evalSourceArtifactKey]: SourceArtifact;
|
||||
}
|
||||
}
|
||||
|
||||
export async function recordEvalSessionArtifact(
|
||||
task: Readonly<RunnerTestCase>,
|
||||
run: Pick<HarnessRun, "artifacts">,
|
||||
): Promise<void> {
|
||||
const runId = run.artifacts?.runId;
|
||||
const session = run.artifacts?.[PI_SESSION_SNAPSHOT_ARTIFACT];
|
||||
if (session === undefined) return;
|
||||
if (typeof runId !== "string" || typeof session !== "string") {
|
||||
throw new TypeError("Pi eval session artifact metadata is invalid.");
|
||||
}
|
||||
await recordArtifact(task, {
|
||||
type: "@earendil-works/pi-evals:session",
|
||||
runId,
|
||||
attachments: [
|
||||
{
|
||||
name: "session.jsonl",
|
||||
contentType: "application/jsonl",
|
||||
body: session,
|
||||
bodyEncoding: "utf-8",
|
||||
},
|
||||
],
|
||||
});
|
||||
}
|
||||
|
||||
export async function recordEvalSourceArtifact(
|
||||
task: Readonly<RunnerTestCase>,
|
||||
runId: string,
|
||||
attachment: SourceAttachment,
|
||||
): Promise<void> {
|
||||
await recordArtifact(task, {
|
||||
type: "@earendil-works/pi-evals:source",
|
||||
runId,
|
||||
attachments: [attachment],
|
||||
});
|
||||
}
|
||||
|
||||
export async function persistEvalArtifactReferences(
|
||||
artifacts: ReadonlyArray<TestArtifact>,
|
||||
runId: string,
|
||||
artifactDirectory: string,
|
||||
): Promise<Array<{ name: string; path: string }>> {
|
||||
const references: Array<{ name: string; path: string }> = [];
|
||||
for (const artifact of artifacts) {
|
||||
if (
|
||||
(artifact.type !== "@earendil-works/pi-evals:session" &&
|
||||
artifact.type !== "@earendil-works/pi-evals:source") ||
|
||||
artifact.runId !== runId
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
const category = artifact.type === "@earendil-works/pi-evals:session" ? "sessions" : "sources";
|
||||
for (const attachment of artifact.attachments) {
|
||||
const name = basename(attachment.name);
|
||||
if (name !== attachment.name) throw new TypeError(`Invalid eval artifact name: ${attachment.name}`);
|
||||
const directory = join(artifactDirectory, category, createHash("sha256").update(runId).digest("hex"));
|
||||
await mkdir(directory, { recursive: true, mode: 0o700 });
|
||||
const path = join(directory, name);
|
||||
await writeFile(path, attachment.body, { encoding: "utf8", mode: 0o600 });
|
||||
references.push({ name, path: relative(artifactDirectory, path) });
|
||||
}
|
||||
}
|
||||
return references;
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
import { createHash } from "node:crypto";
|
||||
import {
|
||||
attachHarnessRunToError,
|
||||
getHarnessRunFromError,
|
||||
type Harness,
|
||||
type HarnessRun,
|
||||
type JsonValue,
|
||||
} from "vitest-evals/harness";
|
||||
|
||||
export const EVAL_HARNESS_ITERATION_ARTIFACT = "vitestEvalsHarnessIteration";
|
||||
|
||||
export type EvalHarnessIterationArtifact = {
|
||||
schemaVersion: 1;
|
||||
evalSet: string;
|
||||
groupKey: string;
|
||||
harness: string;
|
||||
baseline: string;
|
||||
candidates: string[];
|
||||
repetition: number;
|
||||
};
|
||||
|
||||
export type EvalHarnessTableRow<TInput, TOutput extends JsonValue | undefined> = {
|
||||
harness: Harness<TInput, TOutput>;
|
||||
name: string;
|
||||
repetition: number;
|
||||
};
|
||||
|
||||
export type EvalHarnessTablePairOptions<TInput, TOutput extends JsonValue | undefined> = {
|
||||
baseline: Harness<TInput, TOutput>;
|
||||
candidate: Harness<TInput, TOutput>;
|
||||
repetitions?: number;
|
||||
};
|
||||
|
||||
export type EvalHarnessTableCandidatesOptions<TInput, TOutput extends JsonValue | undefined> = {
|
||||
baseline: Harness<TInput, TOutput>;
|
||||
candidates: readonly Harness<TInput, TOutput>[];
|
||||
repetitions?: number;
|
||||
};
|
||||
|
||||
export type EvalHarnessTableOptions<TInput, TOutput extends JsonValue | undefined> =
|
||||
| EvalHarnessTablePairOptions<TInput, TOutput>
|
||||
| EvalHarnessTableCandidatesOptions<TInput, TOutput>;
|
||||
|
||||
type EvalHarnessIterationPlan = Omit<EvalHarnessIterationArtifact, "groupKey">;
|
||||
|
||||
export function parseEvalHarnessIterationArtifact(
|
||||
value: JsonValue | undefined,
|
||||
): EvalHarnessIterationArtifact | undefined {
|
||||
if (value === null || value === undefined || typeof value !== "object" || Array.isArray(value)) return undefined;
|
||||
const { schemaVersion, evalSet, groupKey, harness, baseline, candidates, repetition } = value;
|
||||
if (
|
||||
schemaVersion !== 1 ||
|
||||
typeof evalSet !== "string" ||
|
||||
typeof groupKey !== "string" ||
|
||||
typeof harness !== "string" ||
|
||||
typeof baseline !== "string" ||
|
||||
!Array.isArray(candidates) ||
|
||||
!candidates.every((name): name is string => typeof name === "string") ||
|
||||
typeof repetition !== "number"
|
||||
) {
|
||||
return undefined;
|
||||
}
|
||||
return { schemaVersion, evalSet, groupKey, harness, baseline, candidates, repetition };
|
||||
}
|
||||
|
||||
function canonicalizeJson(value: unknown, ancestors: WeakSet<object>): JsonValue {
|
||||
if (value === null || typeof value === "string" || typeof value === "boolean") return value;
|
||||
if (typeof value === "number") {
|
||||
if (!Number.isFinite(value)) throw new TypeError("Eval input must contain only finite numbers.");
|
||||
return value;
|
||||
}
|
||||
if (typeof value !== "object") throw new TypeError("Eval input must be JSON-serializable.");
|
||||
if (ancestors.has(value)) throw new TypeError("Eval input must not contain circular references.");
|
||||
|
||||
ancestors.add(value);
|
||||
try {
|
||||
if (Array.isArray(value)) {
|
||||
const result: JsonValue[] = [];
|
||||
for (let index = 0; index < value.length; index += 1) {
|
||||
if (!Object.hasOwn(value, index)) throw new TypeError("Eval input arrays must not be sparse.");
|
||||
result.push(canonicalizeJson(value[index], ancestors));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
const prototype = Object.getPrototypeOf(value);
|
||||
if (prototype !== Object.prototype && prototype !== null) {
|
||||
throw new TypeError("Eval input must contain only plain objects and arrays.");
|
||||
}
|
||||
const entries: Array<[string, unknown]> = Object.entries(value);
|
||||
return Object.fromEntries(
|
||||
entries
|
||||
.sort(([left], [right]) => (left < right ? -1 : left > right ? 1 : 0))
|
||||
.map(([key, item]): [string, JsonValue] => [key, canonicalizeJson(item, ancestors)]),
|
||||
);
|
||||
} finally {
|
||||
ancestors.delete(value);
|
||||
}
|
||||
}
|
||||
|
||||
function deriveInputKey(input: unknown): string {
|
||||
if (typeof input === "object" && input !== null && !Array.isArray(input) && "id" in input) {
|
||||
const id = input.id;
|
||||
if (typeof id === "string" && id.trim()) return id.trim();
|
||||
}
|
||||
const canonicalInput = JSON.stringify(canonicalizeJson(input, new WeakSet()));
|
||||
if (canonicalInput === undefined) throw new TypeError("Eval input must be JSON-serializable.");
|
||||
return createHash("sha256").update(canonicalInput).digest("hex");
|
||||
}
|
||||
|
||||
export function deriveEvalGroupKey(input: unknown, repetition: number): string {
|
||||
return JSON.stringify([deriveInputKey(input), repetition]);
|
||||
}
|
||||
|
||||
function validateOptions<TInput, TOutput extends JsonValue | undefined>(
|
||||
evalSet: string,
|
||||
baseline: Harness<TInput, TOutput>,
|
||||
candidates: readonly Harness<TInput, TOutput>[],
|
||||
repetitions: number,
|
||||
): void {
|
||||
if (!evalSet.trim()) throw new TypeError("evalSet must not be empty.");
|
||||
if (candidates.length === 0) throw new TypeError("At least one candidate harness is required.");
|
||||
const harnesses = [baseline, ...candidates];
|
||||
const names = new Set(harnesses.map((harness) => harness.name));
|
||||
if (names.size !== harnesses.length) throw new TypeError("Harness names must be unique within an eval set.");
|
||||
if (!Number.isSafeInteger(repetitions) || repetitions < 1) {
|
||||
throw new TypeError("repetitions must be a positive integer.");
|
||||
}
|
||||
}
|
||||
|
||||
function withIterationArtifact<TInput, TOutput extends JsonValue | undefined>(
|
||||
harness: Harness<TInput, TOutput>,
|
||||
plan: EvalHarnessIterationPlan,
|
||||
): Harness<TInput, TOutput> {
|
||||
return {
|
||||
name: harness.name,
|
||||
async run(input, context) {
|
||||
const groupKey = deriveEvalGroupKey(input, plan.repetition);
|
||||
const artifact: EvalHarnessIterationArtifact = { ...plan, groupKey };
|
||||
context.setArtifact(EVAL_HARNESS_ITERATION_ARTIFACT, artifact);
|
||||
const attachIterationArtifact = <TRun extends HarnessRun>(run: TRun): TRun => {
|
||||
run.artifacts = { ...context.artifacts, ...run.artifacts, [EVAL_HARNESS_ITERATION_ARTIFACT]: artifact };
|
||||
return run;
|
||||
};
|
||||
try {
|
||||
return attachIterationArtifact(await harness.run(input, context));
|
||||
} catch (error) {
|
||||
const partialRun = getHarnessRunFromError(error);
|
||||
if (partialRun) {
|
||||
throw attachHarnessRunToError(error, attachIterationArtifact(partialRun));
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
export function evalHarnessTable<TInput, TOutput extends JsonValue | undefined>(
|
||||
evalSet: string,
|
||||
options: EvalHarnessTablePairOptions<TInput, TOutput>,
|
||||
): EvalHarnessTableRow<TInput, TOutput>[];
|
||||
export function evalHarnessTable<TInput, TOutput extends JsonValue | undefined>(
|
||||
evalSet: string,
|
||||
options: EvalHarnessTableCandidatesOptions<TInput, TOutput>,
|
||||
): EvalHarnessTableRow<TInput, TOutput>[];
|
||||
export function evalHarnessTable<TInput, TOutput extends JsonValue | undefined>(
|
||||
evalSet: string,
|
||||
options: EvalHarnessTableOptions<TInput, TOutput>,
|
||||
): EvalHarnessTableRow<TInput, TOutput>[] {
|
||||
const repetitions = options.repetitions ?? 1;
|
||||
const candidates = "candidate" in options ? [options.candidate] : options.candidates;
|
||||
validateOptions(evalSet, options.baseline, candidates, repetitions);
|
||||
|
||||
const rows: EvalHarnessTableRow<TInput, TOutput>[] = [];
|
||||
const harnesses = [options.baseline, ...candidates];
|
||||
for (let repetition = 1; repetition <= repetitions; repetition += 1) {
|
||||
for (const harness of harnesses) {
|
||||
const plan: EvalHarnessIterationPlan = {
|
||||
schemaVersion: 1,
|
||||
evalSet,
|
||||
harness: harness.name,
|
||||
baseline: options.baseline.name,
|
||||
candidates: candidates.map(({ name }) => name),
|
||||
repetition,
|
||||
};
|
||||
rows.push({
|
||||
harness: withIterationArtifact(harness, plan),
|
||||
name: harness.name,
|
||||
repetition,
|
||||
});
|
||||
}
|
||||
}
|
||||
return rows;
|
||||
}
|
||||
@@ -0,0 +1,111 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { appendFile, mkdir } from "node:fs/promises";
|
||||
import { join } from "node:path";
|
||||
import type { Reporter, SerializedError, TestCase, TestModule, TestRunEndReason, Vitest } from "vitest/node";
|
||||
import { isHarnessRun } from "vitest-evals/harness";
|
||||
import { PI_SESSION_SNAPSHOT_ARTIFACT, persistEvalArtifactReferences } from "./artifacts.ts";
|
||||
import { EVAL_HARNESS_ITERATION_ARTIFACT, parseEvalHarnessIterationArtifact } from "./harness-table.ts";
|
||||
import { formatHarnessComparisonReport, type HarnessObservation, summarizeHarnessComparisons } from "./summary.ts";
|
||||
|
||||
function readFiniteNumber(value: unknown): number | undefined {
|
||||
return typeof value === "number" && Number.isFinite(value) ? value : undefined;
|
||||
}
|
||||
|
||||
async function appendHarnessRunReport(test: TestCase): Promise<void> {
|
||||
const artifactDirectory = process.env.PI_EVAL_ARTIFACT_DIR?.trim();
|
||||
if (!artifactDirectory) return;
|
||||
const harness = test.meta().harness;
|
||||
if (!harness || !isHarnessRun(harness.run)) return;
|
||||
|
||||
const run = harness.run;
|
||||
const artifactRunId = run.artifacts?.runId;
|
||||
const runId = typeof artifactRunId === "string" ? artifactRunId : randomUUID();
|
||||
const metadata = Object.fromEntries(
|
||||
Object.entries(run.artifacts ?? {}).filter(([name]) => name !== "runId" && name !== PI_SESSION_SNAPSHOT_ARTIFACT),
|
||||
);
|
||||
const record = {
|
||||
schemaVersion: 1,
|
||||
runId,
|
||||
test: {
|
||||
id: test.id,
|
||||
file: test.module.relativeModuleId,
|
||||
name: test.name,
|
||||
fullName: test.fullName,
|
||||
status: test.result().state,
|
||||
},
|
||||
harness: harness.name,
|
||||
usage: run.usage,
|
||||
...(run.timings ? { timings: run.timings } : {}),
|
||||
...(run.errors.length > 0 ? { errors: run.errors } : {}),
|
||||
artifacts: await persistEvalArtifactReferences(test.artifacts(), runId, artifactDirectory),
|
||||
...(Object.keys(metadata).length > 0 ? { metadata } : {}),
|
||||
};
|
||||
await mkdir(artifactDirectory, { recursive: true, mode: 0o700 });
|
||||
await appendFile(join(artifactDirectory, "runs.jsonl"), `${JSON.stringify(record)}\n`, {
|
||||
encoding: "utf8",
|
||||
flag: "a",
|
||||
mode: 0o600,
|
||||
});
|
||||
}
|
||||
|
||||
function collectHarnessObservations(modules: ReadonlyArray<TestModule>): HarnessObservation[] {
|
||||
const observations: HarnessObservation[] = [];
|
||||
for (const module of modules) {
|
||||
for (const test of module.children.allTests()) {
|
||||
const harness = test.meta().harness;
|
||||
if (!harness || !isHarnessRun(harness.run)) continue;
|
||||
const run = harness.run;
|
||||
const iteration = parseEvalHarnessIterationArtifact(run.artifacts?.[EVAL_HARNESS_ITERATION_ARTIFACT]);
|
||||
if (!iteration) continue;
|
||||
const score = readFiniteNumber(test.meta().eval?.avgScore);
|
||||
const estimatedCostUsd = readFiniteNumber(run.usage.metadata?.estimatedCostUsd);
|
||||
const observation = {
|
||||
evalSet: iteration.evalSet,
|
||||
groupKey: iteration.groupKey,
|
||||
testName: test.name,
|
||||
file: module.relativeModuleId,
|
||||
harness: iteration.harness,
|
||||
baseline: iteration.baseline,
|
||||
candidates: iteration.candidates,
|
||||
repetition: iteration.repetition,
|
||||
...(run.usage.totalTokens === undefined ? {} : { totalTokens: run.usage.totalTokens }),
|
||||
...(run.timings?.totalMs === undefined ? {} : { totalMs: run.timings.totalMs }),
|
||||
...(estimatedCostUsd === undefined ? {} : { estimatedCostUsd }),
|
||||
};
|
||||
if (run.errors.length > 0) observations.push({ ...observation, outcome: "errored" });
|
||||
else if (score !== undefined) observations.push({ ...observation, outcome: "scored", score });
|
||||
else {
|
||||
const state = test.result().state;
|
||||
const outcome = state === "passed" ? "unscored" : state === "failed" ? "errored" : state;
|
||||
observations.push({ ...observation, outcome });
|
||||
}
|
||||
}
|
||||
}
|
||||
return observations;
|
||||
}
|
||||
|
||||
export default class EvalHarnessReporter implements Reporter {
|
||||
private vitest: Vitest | undefined;
|
||||
|
||||
onInit(vitest: Vitest): void {
|
||||
this.vitest = vitest;
|
||||
}
|
||||
|
||||
async onTestCaseResult(test: TestCase): Promise<void> {
|
||||
await appendHarnessRunReport(test);
|
||||
}
|
||||
|
||||
onTestRunEnd(
|
||||
modules: ReadonlyArray<TestModule>,
|
||||
_errors: ReadonlyArray<SerializedError>,
|
||||
reason: TestRunEndReason,
|
||||
): void {
|
||||
if (reason === "interrupted") {
|
||||
this.vitest?.logger.log("\nEval comparisons unavailable: test run interrupted.");
|
||||
return;
|
||||
}
|
||||
const report = summarizeHarnessComparisons(collectHarnessObservations(modules));
|
||||
const formatted = formatHarnessComparisonReport(report);
|
||||
if (formatted) this.vitest?.logger.log(`\n${formatted}`);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
import { afterEach } from "vitest";
|
||||
import type {} from "vitest-evals";
|
||||
import { recordEvalSessionArtifact } from "./artifacts.ts";
|
||||
|
||||
afterEach(async ({ task }) => {
|
||||
const run = task.meta.harness?.run;
|
||||
if (run) await recordEvalSessionArtifact(task, run);
|
||||
});
|
||||
@@ -0,0 +1,438 @@
|
||||
import { styleText } from "node:util";
|
||||
|
||||
type HarnessObservationOutcome = "scored" | "unscored" | "skipped" | "pending" | "errored";
|
||||
|
||||
type HarnessObservationBase = {
|
||||
evalSet: string;
|
||||
groupKey: string;
|
||||
testName: string;
|
||||
file: string;
|
||||
harness: string;
|
||||
baseline: string;
|
||||
candidates: string[];
|
||||
repetition: number;
|
||||
totalTokens?: number;
|
||||
totalMs?: number;
|
||||
estimatedCostUsd?: number;
|
||||
};
|
||||
|
||||
export type HarnessObservation = HarnessObservationBase &
|
||||
({ outcome: "scored"; score: number } | { outcome: Exclude<HarnessObservationOutcome, "scored">; score?: never });
|
||||
|
||||
export type PairedMetricSummary = {
|
||||
totalPairs: number;
|
||||
eligiblePairs: number;
|
||||
baselineMean: number | null;
|
||||
candidateMean: number | null;
|
||||
meanDelta: number | null;
|
||||
};
|
||||
|
||||
export type CorrectnessLiftSummary = {
|
||||
totalPairs: number;
|
||||
eligiblePairs: number;
|
||||
baselinePassRate: number | null;
|
||||
candidatePassRate: number | null;
|
||||
lift: number | null;
|
||||
baselineWins: number;
|
||||
candidateWins: number;
|
||||
ties: number;
|
||||
};
|
||||
|
||||
export type HarnessPairComparison = {
|
||||
baseline: string;
|
||||
candidate: string;
|
||||
correctness: CorrectnessLiftSummary;
|
||||
totalTokens: PairedMetricSummary;
|
||||
totalMs: PairedMetricSummary;
|
||||
estimatedCostUsd: PairedMetricSummary;
|
||||
};
|
||||
|
||||
export type HarnessComparisonDiagnostic = {
|
||||
evalSet: string;
|
||||
groupKey: string;
|
||||
testName: string;
|
||||
file: string;
|
||||
repetition: number;
|
||||
harness: string;
|
||||
reason: "missing-observation" | "duplicate-observation" | "harness-error" | "missing-score" | "unscorable-outcome";
|
||||
};
|
||||
|
||||
export type HarnessEvalSetReport = {
|
||||
evalSet: string;
|
||||
comparisons: HarnessPairComparison[];
|
||||
};
|
||||
|
||||
export type HarnessComparisonReport = {
|
||||
schemaVersion: 1;
|
||||
evalSets: HarnessEvalSetReport[];
|
||||
diagnostics: HarnessComparisonDiagnostic[];
|
||||
};
|
||||
|
||||
type HarnessDescriptor = {
|
||||
name: string;
|
||||
index: number;
|
||||
};
|
||||
|
||||
type ObservationGroup = {
|
||||
evalSet: string;
|
||||
groupKey: string;
|
||||
testName: string;
|
||||
file: string;
|
||||
repetition: number;
|
||||
observationsByHarness: Map<string, HarnessObservation[]>;
|
||||
};
|
||||
|
||||
type EvalSetData = {
|
||||
baseline: HarnessDescriptor;
|
||||
candidatesByName: Map<string, HarnessDescriptor>;
|
||||
groupsByKey: Map<string, ObservationGroup>;
|
||||
};
|
||||
|
||||
type ObservationPair = {
|
||||
baseline: HarnessObservation;
|
||||
candidate: HarnessObservation;
|
||||
};
|
||||
|
||||
function getOrCreate<K, V extends object>(map: Map<K, V>, key: K, create: () => V): V {
|
||||
const existing = map.get(key);
|
||||
if (existing !== undefined) return existing;
|
||||
const value = create();
|
||||
map.set(key, value);
|
||||
return value;
|
||||
}
|
||||
|
||||
function mean(values: readonly number[]): number | null {
|
||||
return values.length === 0 ? null : values.reduce((sum, value) => sum + value, 0) / values.length;
|
||||
}
|
||||
|
||||
function preciseDifference(left: number, right: number): number {
|
||||
return Number((left - right).toPrecision(15));
|
||||
}
|
||||
|
||||
function groupObservations(observations: readonly HarnessObservation[]): Map<string, EvalSetData> {
|
||||
const evalSets = new Map<string, EvalSetData>();
|
||||
for (const observation of observations) {
|
||||
const evalSet = getOrCreate(evalSets, observation.evalSet, () => ({
|
||||
baseline: { name: observation.baseline, index: 0 },
|
||||
candidatesByName: new Map(),
|
||||
groupsByKey: new Map(),
|
||||
}));
|
||||
|
||||
for (const [index, name] of observation.candidates.entries()) {
|
||||
const existing = evalSet.candidatesByName.get(name);
|
||||
if (!existing || index < existing.index) evalSet.candidatesByName.set(name, { name, index });
|
||||
}
|
||||
|
||||
const group = getOrCreate(
|
||||
evalSet.groupsByKey,
|
||||
JSON.stringify([observation.file, observation.testName, observation.groupKey]),
|
||||
() => ({
|
||||
evalSet: observation.evalSet,
|
||||
groupKey: observation.groupKey,
|
||||
testName: observation.testName,
|
||||
file: observation.file,
|
||||
repetition: observation.repetition,
|
||||
observationsByHarness: new Map(),
|
||||
}),
|
||||
);
|
||||
getOrCreate(group.observationsByHarness, observation.harness, (): HarnessObservation[] => []).push(observation);
|
||||
}
|
||||
return evalSets;
|
||||
}
|
||||
|
||||
function orderedHarnesses(evalSet: EvalSetData): HarnessDescriptor[] {
|
||||
return [
|
||||
evalSet.baseline,
|
||||
...[...evalSet.candidatesByName.values()].sort(
|
||||
(left, right) => left.index - right.index || left.name.localeCompare(right.name),
|
||||
),
|
||||
];
|
||||
}
|
||||
|
||||
function orderedCandidates(evalSet: EvalSetData): HarnessDescriptor[] {
|
||||
return [...evalSet.candidatesByName.values()].sort(
|
||||
(left, right) => left.index - right.index || left.name.localeCompare(right.name),
|
||||
);
|
||||
}
|
||||
|
||||
function orderedGroups(evalSet: EvalSetData): ObservationGroup[] {
|
||||
return [...evalSet.groupsByKey.values()].sort(
|
||||
(left, right) => left.groupKey.localeCompare(right.groupKey) || left.repetition - right.repetition,
|
||||
);
|
||||
}
|
||||
|
||||
function collectDiagnostics(
|
||||
harnesses: readonly HarnessDescriptor[],
|
||||
groups: readonly ObservationGroup[],
|
||||
): HarnessComparisonDiagnostic[] {
|
||||
const diagnostics: HarnessComparisonDiagnostic[] = [];
|
||||
for (const group of groups) {
|
||||
for (const { name: harness } of harnesses) {
|
||||
const observations = group.observationsByHarness.get(harness) ?? [];
|
||||
let reason: HarnessComparisonDiagnostic["reason"] | undefined;
|
||||
if (observations.length === 0) reason = "missing-observation";
|
||||
else if (observations.length > 1) reason = "duplicate-observation";
|
||||
else if (observations[0].outcome === "errored") reason = "harness-error";
|
||||
else if (observations[0].outcome === "unscored") {
|
||||
reason = "missing-score";
|
||||
} else if (observations[0].outcome !== "scored") {
|
||||
reason = "unscorable-outcome";
|
||||
}
|
||||
if (!reason) continue;
|
||||
diagnostics.push({
|
||||
evalSet: group.evalSet,
|
||||
groupKey: group.groupKey,
|
||||
testName: group.testName,
|
||||
file: group.file,
|
||||
repetition: group.repetition,
|
||||
harness,
|
||||
reason,
|
||||
});
|
||||
}
|
||||
}
|
||||
return diagnostics;
|
||||
}
|
||||
|
||||
function pairObservations(
|
||||
groups: readonly ObservationGroup[],
|
||||
baselineHarness: string,
|
||||
candidateHarness: string,
|
||||
): ObservationPair[] {
|
||||
const pairs: ObservationPair[] = [];
|
||||
for (const group of groups) {
|
||||
const baseline = group.observationsByHarness.get(baselineHarness) ?? [];
|
||||
const candidate = group.observationsByHarness.get(candidateHarness) ?? [];
|
||||
if (baseline.length === 1 && candidate.length === 1) {
|
||||
pairs.push({ baseline: baseline[0], candidate: candidate[0] });
|
||||
}
|
||||
}
|
||||
return pairs;
|
||||
}
|
||||
|
||||
function summarizeMetric(
|
||||
pairs: readonly ObservationPair[],
|
||||
select: (observation: HarnessObservation) => number | undefined,
|
||||
totalPairs: number,
|
||||
): PairedMetricSummary {
|
||||
const baselineValues: number[] = [];
|
||||
const candidateValues: number[] = [];
|
||||
for (const { baseline, candidate } of pairs) {
|
||||
if (baseline.outcome !== "scored" || candidate.outcome !== "scored") continue;
|
||||
const baselineValue = select(baseline);
|
||||
const candidateValue = select(candidate);
|
||||
if (
|
||||
baselineValue === undefined ||
|
||||
candidateValue === undefined ||
|
||||
!Number.isFinite(baselineValue) ||
|
||||
!Number.isFinite(candidateValue)
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
baselineValues.push(baselineValue);
|
||||
candidateValues.push(candidateValue);
|
||||
}
|
||||
|
||||
const baselineMean = mean(baselineValues);
|
||||
const candidateMean = mean(candidateValues);
|
||||
return {
|
||||
totalPairs,
|
||||
eligiblePairs: baselineValues.length,
|
||||
baselineMean,
|
||||
candidateMean,
|
||||
meanDelta:
|
||||
baselineMean === null || candidateMean === null ? null : preciseDifference(candidateMean, baselineMean),
|
||||
};
|
||||
}
|
||||
|
||||
function summarizeCorrectness(pairs: readonly ObservationPair[], totalPairs: number): CorrectnessLiftSummary {
|
||||
let eligiblePairs = 0;
|
||||
let baselinePasses = 0;
|
||||
let candidatePasses = 0;
|
||||
let baselineWins = 0;
|
||||
let candidateWins = 0;
|
||||
let ties = 0;
|
||||
|
||||
for (const { baseline, candidate } of pairs) {
|
||||
if (baseline.outcome !== "scored" || candidate.outcome !== "scored") continue;
|
||||
eligiblePairs += 1;
|
||||
const baselinePassed = baseline.score >= 1;
|
||||
const candidatePassed = candidate.score >= 1;
|
||||
if (baselinePassed) baselinePasses += 1;
|
||||
if (candidatePassed) candidatePasses += 1;
|
||||
if (baselinePassed === candidatePassed) ties += 1;
|
||||
else if (baselinePassed) baselineWins += 1;
|
||||
else candidateWins += 1;
|
||||
}
|
||||
|
||||
const baselinePassRate = eligiblePairs === 0 ? null : baselinePasses / eligiblePairs;
|
||||
const candidatePassRate = eligiblePairs === 0 ? null : candidatePasses / eligiblePairs;
|
||||
return {
|
||||
totalPairs,
|
||||
eligiblePairs,
|
||||
baselinePassRate,
|
||||
candidatePassRate,
|
||||
lift:
|
||||
baselinePassRate === null || candidatePassRate === null
|
||||
? null
|
||||
: preciseDifference(candidatePassRate, baselinePassRate),
|
||||
baselineWins,
|
||||
candidateWins,
|
||||
ties,
|
||||
};
|
||||
}
|
||||
|
||||
function compareHarnesses(
|
||||
baseline: HarnessDescriptor,
|
||||
candidate: HarnessDescriptor,
|
||||
groups: readonly ObservationGroup[],
|
||||
): HarnessPairComparison {
|
||||
const pairs = pairObservations(groups, baseline.name, candidate.name);
|
||||
return {
|
||||
baseline: baseline.name,
|
||||
candidate: candidate.name,
|
||||
correctness: summarizeCorrectness(pairs, groups.length),
|
||||
totalTokens: summarizeMetric(pairs, ({ totalTokens }) => totalTokens, groups.length),
|
||||
totalMs: summarizeMetric(pairs, ({ totalMs }) => totalMs, groups.length),
|
||||
estimatedCostUsd: summarizeMetric(pairs, ({ estimatedCostUsd }) => estimatedCostUsd, groups.length),
|
||||
};
|
||||
}
|
||||
|
||||
export function summarizeHarnessComparisons(observations: readonly HarnessObservation[]): HarnessComparisonReport {
|
||||
const evalSets: HarnessEvalSetReport[] = [];
|
||||
const diagnostics: HarnessComparisonDiagnostic[] = [];
|
||||
for (const [evalSet, data] of [...groupObservations(observations)].sort(([left], [right]) =>
|
||||
left.localeCompare(right),
|
||||
)) {
|
||||
const harnesses = orderedHarnesses(data);
|
||||
const candidates = orderedCandidates(data);
|
||||
const groups = orderedGroups(data);
|
||||
evalSets.push({
|
||||
evalSet,
|
||||
comparisons: candidates.map((candidate) => compareHarnesses(data.baseline, candidate, groups)),
|
||||
});
|
||||
diagnostics.push(...collectDiagnostics(harnesses, groups));
|
||||
}
|
||||
|
||||
return {
|
||||
schemaVersion: 1,
|
||||
evalSets,
|
||||
diagnostics: diagnostics.sort(
|
||||
(left, right) =>
|
||||
left.evalSet.localeCompare(right.evalSet) ||
|
||||
left.file.localeCompare(right.file) ||
|
||||
left.groupKey.localeCompare(right.groupKey) ||
|
||||
left.repetition - right.repetition ||
|
||||
left.harness.localeCompare(right.harness),
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
function formatPercentage(value: number | null): string {
|
||||
return value === null ? "unavailable" : `${(value * 100).toFixed(1)}%`;
|
||||
}
|
||||
|
||||
function formatSigned(value: number, fractionDigits: number): string {
|
||||
return `${value >= 0 ? "+" : ""}${value.toFixed(fractionDigits)}`;
|
||||
}
|
||||
|
||||
function formatCoverage(eligiblePairs: number, totalPairs: number): string {
|
||||
return styleText("gray", `(${eligiblePairs}/${totalPairs} pairs)`);
|
||||
}
|
||||
|
||||
function formatReportLine(label: string, value: string): string {
|
||||
return ` ${styleText("gray", label.padStart(9))} ${value}`;
|
||||
}
|
||||
|
||||
function colorDelta(value: number, formatted: string, positiveIsBetter: boolean): string {
|
||||
if (value === 0) return styleText("gray", formatted);
|
||||
const improved = positiveIsBetter ? value > 0 : value < 0;
|
||||
return styleText(improved ? "green" : "red", formatted);
|
||||
}
|
||||
|
||||
function formatMetric(
|
||||
label: string,
|
||||
metric: PairedMetricSummary,
|
||||
formatValue: (value: number) => string,
|
||||
formatDelta: (value: number) => string,
|
||||
comparisonPairs: number,
|
||||
): string {
|
||||
const coverage =
|
||||
metric.eligiblePairs === 0 || metric.eligiblePairs === comparisonPairs
|
||||
? ""
|
||||
: ` ${formatCoverage(metric.eligiblePairs, metric.totalPairs)}`;
|
||||
if (metric.baselineMean === null || metric.candidateMean === null || metric.meanDelta === null) {
|
||||
return formatReportLine(label, `${styleText("yellow", "unavailable")}${coverage}`);
|
||||
}
|
||||
const delta = colorDelta(metric.meanDelta, formatDelta(metric.meanDelta), false);
|
||||
const values = styleText(
|
||||
"gray",
|
||||
`(candidate ${formatValue(metric.candidateMean)}, baseline ${formatValue(metric.baselineMean)})`,
|
||||
);
|
||||
return formatReportLine(label, `${delta} ${values}${coverage}`);
|
||||
}
|
||||
|
||||
export function formatHarnessComparisonReport(report: HarnessComparisonReport): string {
|
||||
if (report.evalSets.every(({ comparisons }) => comparisons.length === 0)) return "";
|
||||
const lines = [styleText("bold", "Eval Comparisons")];
|
||||
for (const evalSet of report.evalSets) {
|
||||
lines.push(` ${evalSet.evalSet}`);
|
||||
for (const [index, comparison] of evalSet.comparisons.entries()) {
|
||||
if (index > 0) lines.push("");
|
||||
const { correctness } = comparison;
|
||||
lines.push(formatReportLine("Baseline", comparison.baseline));
|
||||
lines.push(
|
||||
formatReportLine(
|
||||
"Candidate",
|
||||
`${comparison.candidate} ${formatCoverage(correctness.eligiblePairs, correctness.totalPairs)}`,
|
||||
),
|
||||
);
|
||||
if (correctness.lift === null) {
|
||||
lines.push(formatReportLine("Pass rate", styleText("yellow", "unavailable")));
|
||||
} else {
|
||||
const lift = correctness.lift * 100;
|
||||
const delta = colorDelta(lift, `${formatSigned(lift, 1)} pp`, true);
|
||||
const values = styleText(
|
||||
"gray",
|
||||
`(candidate ${formatPercentage(correctness.candidatePassRate)}, baseline ${formatPercentage(correctness.baselinePassRate)})`,
|
||||
);
|
||||
lines.push(formatReportLine("Pass rate", `${delta} ${values}`));
|
||||
}
|
||||
lines.push(
|
||||
formatMetric(
|
||||
"Tokens",
|
||||
comparison.totalTokens,
|
||||
(value) => value.toFixed(1),
|
||||
(value) => formatSigned(value, 1),
|
||||
correctness.eligiblePairs,
|
||||
),
|
||||
);
|
||||
lines.push(
|
||||
formatMetric(
|
||||
"Latency",
|
||||
comparison.totalMs,
|
||||
(value) => `${value.toFixed(1)}ms`,
|
||||
(value) => `${formatSigned(value, 1)}ms`,
|
||||
correctness.eligiblePairs,
|
||||
),
|
||||
);
|
||||
lines.push(
|
||||
formatMetric(
|
||||
"Est. cost",
|
||||
comparison.estimatedCostUsd,
|
||||
(value) => `$${value.toFixed(4)}`,
|
||||
(value) => `${value >= 0 ? "+" : "-"}$${Math.abs(value).toFixed(4)}`,
|
||||
correctness.eligiblePairs,
|
||||
),
|
||||
);
|
||||
}
|
||||
}
|
||||
if (report.diagnostics.length > 0) {
|
||||
lines.push(` ${styleText("yellow", "Incomplete observations")}`);
|
||||
for (const diagnostic of report.diagnostics) {
|
||||
lines.push(
|
||||
` ${diagnostic.reason}: ${diagnostic.file}/${diagnostic.testName} repetition ${diagnostic.repetition}, harness ${diagnostic.harness}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
return lines.join("\n");
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { resolveModelSelection } from "../src/pi-harness.ts";
|
||||
|
||||
describe("resolveModelSelection", () => {
|
||||
it("prefers an explicit harness model over environment defaults", () => {
|
||||
expect(
|
||||
resolveModelSelection(
|
||||
{ provider: "anthropic", id: "claude-opus-4-6" },
|
||||
{ PI_PROVIDER: "openai-codex", PI_MODEL: "gpt-5.6-sol" },
|
||||
),
|
||||
).toEqual({ provider: "anthropic", id: "claude-opus-4-6" });
|
||||
});
|
||||
|
||||
it("uses trimmed environment defaults when the harness has no explicit model", () => {
|
||||
expect(resolveModelSelection(undefined, { PI_PROVIDER: " openai-codex ", PI_MODEL: " gpt-5.6-sol " })).toEqual({
|
||||
provider: "openai-codex",
|
||||
id: "gpt-5.6-sol",
|
||||
});
|
||||
});
|
||||
|
||||
it.each([
|
||||
[undefined, {}],
|
||||
[undefined, { PI_PROVIDER: "openai-codex" }],
|
||||
[undefined, { PI_MODEL: "gpt-5.6-sol" }],
|
||||
[
|
||||
{ provider: "", id: "gpt-5.6-sol" },
|
||||
{ PI_PROVIDER: "openai-codex", PI_MODEL: "gpt-5.6-sol" },
|
||||
],
|
||||
] as const)("rejects an incomplete model selection", (explicitModel, environment) => {
|
||||
expect(() => resolveModelSelection(explicitModel, environment)).toThrow(
|
||||
"Select a harness model explicitly or set both PI_PROVIDER and PI_MODEL as defaults.",
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,103 @@
|
||||
import { mkdtemp, readFile, rm } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { expect, it } from "vitest";
|
||||
import {
|
||||
persistEvalArtifactReferences,
|
||||
recordEvalSessionArtifact,
|
||||
recordEvalSourceArtifact,
|
||||
} from "../../src/vitest-evals/artifacts.ts";
|
||||
|
||||
it("records session and source artifacts against the explicit test task", async ({ task }) => {
|
||||
const runId = "run-1";
|
||||
await recordEvalSessionArtifact(task, {
|
||||
artifacts: { runId, piSessionJsonl: '{"type":"session"}\n' },
|
||||
});
|
||||
await recordEvalSourceArtifact(task, runId, {
|
||||
name: "hello.ts",
|
||||
contentType: "text/typescript",
|
||||
body: "export default function () {}\n",
|
||||
bodyEncoding: "utf-8",
|
||||
});
|
||||
|
||||
expect(task.artifacts).toContainEqual(
|
||||
expect.objectContaining({
|
||||
type: "@earendil-works/pi-evals:session",
|
||||
runId,
|
||||
attachments: [
|
||||
expect.objectContaining({
|
||||
name: "session.jsonl",
|
||||
body: '{"type":"session"}\n',
|
||||
bodyEncoding: "utf-8",
|
||||
contentType: "application/jsonl",
|
||||
}),
|
||||
],
|
||||
}),
|
||||
);
|
||||
expect(task.artifacts).toContainEqual(
|
||||
expect.objectContaining({
|
||||
type: "@earendil-works/pi-evals:source",
|
||||
runId,
|
||||
attachments: [
|
||||
expect.objectContaining({
|
||||
name: "hello.ts",
|
||||
body: "export default function () {}\n",
|
||||
bodyEncoding: "utf-8",
|
||||
contentType: "text/typescript",
|
||||
}),
|
||||
],
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("persists and selects attachments belonging to the reported run", async () => {
|
||||
const root = await mkdtemp(join(tmpdir(), "pi-eval-artifact-report-test-"));
|
||||
try {
|
||||
const references = await persistEvalArtifactReferences(
|
||||
[
|
||||
{
|
||||
type: "@earendil-works/pi-evals:session",
|
||||
runId: "run-1",
|
||||
attachments: [
|
||||
{
|
||||
name: "session.jsonl",
|
||||
body: '{"type":"session"}\n',
|
||||
bodyEncoding: "utf-8",
|
||||
contentType: "application/jsonl",
|
||||
},
|
||||
],
|
||||
},
|
||||
{
|
||||
type: "@earendil-works/pi-evals:session",
|
||||
runId: "run-2",
|
||||
attachments: [],
|
||||
},
|
||||
{
|
||||
type: "@earendil-works/pi-evals:source",
|
||||
runId: "run-1",
|
||||
attachments: [
|
||||
{
|
||||
name: "hello.ts",
|
||||
body: "export default function () {}\n",
|
||||
bodyEncoding: "utf-8",
|
||||
contentType: "text/typescript",
|
||||
},
|
||||
],
|
||||
},
|
||||
{ type: "internal:annotation", annotation: { message: "other", type: "info" } },
|
||||
],
|
||||
"run-1",
|
||||
root,
|
||||
);
|
||||
expect(references).toEqual([
|
||||
{ name: "session.jsonl", path: expect.stringMatching(/^sessions\/[a-f0-9]{64}\/session\.jsonl$/) },
|
||||
{ name: "hello.ts", path: expect.stringMatching(/^sources\/[a-f0-9]{64}\/hello\.ts$/) },
|
||||
]);
|
||||
for (const { name, path } of references) {
|
||||
const expected = name === "session.jsonl" ? '{"type":"session"}\n' : "export default function () {}\n";
|
||||
expect(await readFile(join(root, path), "utf8")).toBe(expected);
|
||||
}
|
||||
} finally {
|
||||
await rm(root, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
@@ -0,0 +1,94 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { createHarness, type HarnessContext } from "vitest-evals/harness";
|
||||
import {
|
||||
deriveEvalGroupKey,
|
||||
EVAL_HARNESS_ITERATION_ARTIFACT,
|
||||
evalHarnessTable,
|
||||
parseEvalHarnessIterationArtifact,
|
||||
} from "../../src/vitest-evals/harness-table.ts";
|
||||
|
||||
describe("deriveEvalGroupKey", () => {
|
||||
it("combines a trimmed string input ID with repetition", () => {
|
||||
expect(deriveEvalGroupKey({ id: " input-1 ", prompt: "hello" }, 2)).toBe(JSON.stringify(["input-1", 2]));
|
||||
});
|
||||
|
||||
it("hashes canonical JSON independently of object key order", () => {
|
||||
expect(deriveEvalGroupKey({ first: 1, second: [true, "value"] }, 1)).toBe(
|
||||
deriveEvalGroupKey({ second: [true, "value"], first: 1 }, 1),
|
||||
);
|
||||
expect(deriveEvalGroupKey({ first: 1 }, 1)).not.toBe(deriveEvalGroupKey({ first: 2 }, 1));
|
||||
expect(deriveEvalGroupKey({ first: 1 }, 1)).not.toBe(deriveEvalGroupKey({ first: 1 }, 2));
|
||||
expect(deriveEvalGroupKey(["first", "second"], 1)).not.toBe(deriveEvalGroupKey(["second", "first"], 1));
|
||||
});
|
||||
|
||||
it("rejects non-JSON and circular input", () => {
|
||||
const circular: { self?: unknown } = {};
|
||||
circular.self = circular;
|
||||
expect(() => deriveEvalGroupKey(new Date(0), 1)).toThrow("only plain objects and arrays");
|
||||
expect(() => deriveEvalGroupKey(Array(1), 1)).toThrow("must not be sparse");
|
||||
expect(() => deriveEvalGroupKey(circular, 1)).toThrow("must not contain circular references");
|
||||
});
|
||||
});
|
||||
|
||||
function createFakeHarness(name: string) {
|
||||
return createHarness<{ id: string }, { harness: string; inputId: string }>({
|
||||
name,
|
||||
run: ({ input }) => ({
|
||||
output: { harness: name, inputId: input.id },
|
||||
events: [
|
||||
{ type: "message", role: "user", content: input.id },
|
||||
{ type: "message", role: "assistant", content: name },
|
||||
],
|
||||
}),
|
||||
});
|
||||
}
|
||||
|
||||
const harnessTable = evalHarnessTable("local multi-harness eval", {
|
||||
baseline: createFakeHarness("withoutSkill"),
|
||||
candidates: [createFakeHarness("withSkill")],
|
||||
repetitions: 2,
|
||||
});
|
||||
|
||||
describe("evalHarnessTable", () => {
|
||||
it("plans repetitions in declaration order", () => {
|
||||
expect(harnessTable.map(({ name, repetition }) => ({ name, repetition }))).toEqual([
|
||||
{ name: "withoutSkill", repetition: 1 },
|
||||
{ name: "withSkill", repetition: 1 },
|
||||
{ name: "withoutSkill", repetition: 2 },
|
||||
{ name: "withSkill", repetition: 2 },
|
||||
]);
|
||||
});
|
||||
|
||||
it("accepts a singular candidate", () => {
|
||||
const rows = evalHarnessTable("singular candidate", {
|
||||
baseline: createFakeHarness("baseline"),
|
||||
candidate: createFakeHarness("candidate"),
|
||||
});
|
||||
|
||||
expect(rows.map(({ name }) => name)).toEqual(["baseline", "candidate"]);
|
||||
});
|
||||
|
||||
it("attaches iteration metadata to every wrapped harness run", async () => {
|
||||
for (const row of harnessTable) {
|
||||
const artifacts: HarnessContext["artifacts"] = {};
|
||||
const context: HarnessContext = {
|
||||
artifacts,
|
||||
setArtifact(name, value) {
|
||||
artifacts[name] = value;
|
||||
},
|
||||
};
|
||||
const result = await row.harness.run({ id: "first" }, context);
|
||||
|
||||
expect(result.output).toEqual({ harness: row.name, inputId: "first" });
|
||||
expect(parseEvalHarnessIterationArtifact(result.artifacts?.[EVAL_HARNESS_ITERATION_ARTIFACT])).toEqual({
|
||||
schemaVersion: 1,
|
||||
evalSet: "local multi-harness eval",
|
||||
groupKey: deriveEvalGroupKey({ id: "first" }, row.repetition),
|
||||
harness: row.name,
|
||||
baseline: "withoutSkill",
|
||||
candidates: ["withSkill"],
|
||||
repetition: row.repetition,
|
||||
});
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,232 @@
|
||||
import { stripVTControlCharacters } from "node:util";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
formatHarnessComparisonReport,
|
||||
type HarnessObservation,
|
||||
summarizeHarnessComparisons,
|
||||
} from "../../src/vitest-evals/summary.ts";
|
||||
|
||||
type ObservationResult = "passed" | "failed" | Exclude<HarnessObservation["outcome"], "scored">;
|
||||
|
||||
function observation(
|
||||
harness: string,
|
||||
testName: string,
|
||||
result: ObservationResult,
|
||||
metrics: Pick<HarnessObservation, "totalTokens" | "totalMs" | "estimatedCostUsd"> = {},
|
||||
baseline = "without-tools",
|
||||
candidates: string[] = ["with-tools"],
|
||||
): HarnessObservation {
|
||||
const base = {
|
||||
evalSet: "tool access",
|
||||
groupKey: JSON.stringify([testName, 1]),
|
||||
testName,
|
||||
file: "src/tool-access.eval.ts",
|
||||
harness,
|
||||
baseline,
|
||||
candidates,
|
||||
repetition: 1,
|
||||
...metrics,
|
||||
};
|
||||
if (result === "passed" || result === "failed") {
|
||||
return { ...base, outcome: "scored", score: result === "passed" ? 1 : 0 };
|
||||
}
|
||||
return { ...base, outcome: result };
|
||||
}
|
||||
|
||||
describe("summarizeHarnessComparisons", () => {
|
||||
it("computes paired correctness lift separately from efficiency deltas", () => {
|
||||
const report = summarizeHarnessComparisons([
|
||||
observation("without-tools", "create", "failed", {
|
||||
totalTokens: 100,
|
||||
totalMs: 1000,
|
||||
estimatedCostUsd: 0.01,
|
||||
}),
|
||||
observation("with-tools", "create", "passed", {
|
||||
totalTokens: 120,
|
||||
totalMs: 800,
|
||||
estimatedCostUsd: 0.02,
|
||||
}),
|
||||
observation("without-tools", "inspect", "passed", { totalTokens: 200 }),
|
||||
observation("with-tools", "inspect", "passed", { totalTokens: 180 }),
|
||||
]);
|
||||
|
||||
expect(report.evalSets).toHaveLength(1);
|
||||
expect(report.evalSets[0]?.comparisons).toEqual([
|
||||
expect.objectContaining({
|
||||
baseline: "without-tools",
|
||||
candidate: "with-tools",
|
||||
correctness: {
|
||||
totalPairs: 2,
|
||||
eligiblePairs: 2,
|
||||
baselinePassRate: 0.5,
|
||||
candidatePassRate: 1,
|
||||
lift: 0.5,
|
||||
baselineWins: 0,
|
||||
candidateWins: 1,
|
||||
ties: 1,
|
||||
},
|
||||
totalTokens: {
|
||||
totalPairs: 2,
|
||||
eligiblePairs: 2,
|
||||
baselineMean: 150,
|
||||
candidateMean: 150,
|
||||
meanDelta: 0,
|
||||
},
|
||||
totalMs: {
|
||||
totalPairs: 2,
|
||||
eligiblePairs: 1,
|
||||
baselineMean: 1000,
|
||||
candidateMean: 800,
|
||||
meanDelta: -200,
|
||||
},
|
||||
estimatedCostUsd: {
|
||||
totalPairs: 2,
|
||||
eligiblePairs: 1,
|
||||
baselineMean: 0.01,
|
||||
candidateMean: 0.02,
|
||||
meanDelta: 0.01,
|
||||
},
|
||||
}),
|
||||
]);
|
||||
expect(report.diagnostics).toEqual([]);
|
||||
});
|
||||
|
||||
it("reports missing observations without coercing them to failures or zero telemetry", () => {
|
||||
const report = summarizeHarnessComparisons([
|
||||
observation("without-tools", "create", "failed"),
|
||||
observation("with-tools", "create", "passed"),
|
||||
observation("without-tools", "inspect", "passed"),
|
||||
]);
|
||||
const comparison = report.evalSets[0]?.comparisons[0];
|
||||
|
||||
expect(comparison?.correctness).toEqual({
|
||||
totalPairs: 2,
|
||||
eligiblePairs: 1,
|
||||
baselinePassRate: 0,
|
||||
candidatePassRate: 1,
|
||||
lift: 1,
|
||||
baselineWins: 0,
|
||||
candidateWins: 1,
|
||||
ties: 0,
|
||||
});
|
||||
expect(comparison?.totalTokens).toEqual({
|
||||
totalPairs: 2,
|
||||
eligiblePairs: 0,
|
||||
baselineMean: null,
|
||||
candidateMean: null,
|
||||
meanDelta: null,
|
||||
});
|
||||
expect(report.diagnostics).toContainEqual(
|
||||
expect.objectContaining({
|
||||
testName: "inspect",
|
||||
harness: "with-tools",
|
||||
reason: "missing-observation",
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("keeps identical inputs in different test files separate", () => {
|
||||
const report = summarizeHarnessComparisons([
|
||||
observation("without-tools", "shared", "failed"),
|
||||
observation("with-tools", "shared", "passed"),
|
||||
{ ...observation("without-tools", "shared", "passed"), file: "src/other.eval.ts" },
|
||||
{ ...observation("with-tools", "shared", "passed"), file: "src/other.eval.ts" },
|
||||
]);
|
||||
|
||||
expect(report.evalSets[0]?.comparisons[0]?.correctness).toEqual(
|
||||
expect.objectContaining({ totalPairs: 2, eligiblePairs: 2 }),
|
||||
);
|
||||
expect(report.diagnostics).toEqual([]);
|
||||
});
|
||||
|
||||
it("does not score harness errors as correctness failures", () => {
|
||||
const report = summarizeHarnessComparisons([
|
||||
observation("without-tools", "create", "errored", { totalTokens: 100 }),
|
||||
observation("with-tools", "create", "passed", { totalTokens: 100 }),
|
||||
]);
|
||||
|
||||
expect(report.evalSets[0]?.comparisons[0]?.correctness).toEqual(
|
||||
expect.objectContaining({ totalPairs: 1, eligiblePairs: 0 }),
|
||||
);
|
||||
expect(report.evalSets[0]?.comparisons[0]?.totalTokens.eligiblePairs).toBe(0);
|
||||
expect(report.diagnostics).toContainEqual(
|
||||
expect.objectContaining({ harness: "without-tools", reason: "harness-error" }),
|
||||
);
|
||||
});
|
||||
|
||||
it("does not derive correctness from completed Vitest tests without judge scores", () => {
|
||||
const withoutScore = observation("without-tools", "create", "unscored");
|
||||
const withScore = observation("with-tools", "create", "unscored");
|
||||
|
||||
const report = summarizeHarnessComparisons([withoutScore, withScore]);
|
||||
|
||||
expect(report.evalSets[0]?.comparisons[0]?.correctness.eligiblePairs).toBe(0);
|
||||
expect(report.diagnostics).toEqual([
|
||||
expect.objectContaining({ harness: "with-tools", reason: "missing-score" }),
|
||||
expect.objectContaining({ harness: "without-tools", reason: "missing-score" }),
|
||||
]);
|
||||
});
|
||||
|
||||
it("compares each candidate with the declared baseline", () => {
|
||||
const candidates = ["second", "third"];
|
||||
const report = summarizeHarnessComparisons([
|
||||
observation("first", "input", "passed", {}, "first", candidates),
|
||||
observation("second", "input", "passed", {}, "first", candidates),
|
||||
observation("third", "input", "passed", {}, "first", candidates),
|
||||
]);
|
||||
|
||||
expect(report.evalSets[0]?.comparisons.map(({ baseline, candidate }) => [baseline, candidate])).toEqual([
|
||||
["first", "second"],
|
||||
["first", "third"],
|
||||
]);
|
||||
});
|
||||
|
||||
it("retains a declared harness with no completed observations", () => {
|
||||
const report = summarizeHarnessComparisons([observation("without-tools", "create", "failed")]);
|
||||
|
||||
expect(report.evalSets[0]?.comparisons).toHaveLength(1);
|
||||
expect(report.evalSets[0]?.comparisons[0]?.correctness.eligiblePairs).toBe(0);
|
||||
expect(report.diagnostics).toContainEqual(
|
||||
expect.objectContaining({
|
||||
testName: "create",
|
||||
harness: "with-tools",
|
||||
reason: "missing-observation",
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("reports duplicate and unscorable observations once across multiple harness pairs", () => {
|
||||
const candidates = ["second", "third"];
|
||||
const report = summarizeHarnessComparisons([
|
||||
observation("first", "duplicate", "passed", {}, "first", candidates),
|
||||
observation("first", "duplicate", "failed", {}, "first", candidates),
|
||||
observation("second", "duplicate", "passed", {}, "first", candidates),
|
||||
observation("third", "duplicate", "passed", {}, "first", candidates),
|
||||
observation("first", "skipped", "skipped", {}, "first", candidates),
|
||||
observation("second", "skipped", "passed", {}, "first", candidates),
|
||||
observation("third", "skipped", "passed", {}, "first", candidates),
|
||||
]);
|
||||
|
||||
expect(report.diagnostics.filter(({ reason }) => reason === "duplicate-observation")).toEqual([
|
||||
expect.objectContaining({ testName: "duplicate", harness: "first" }),
|
||||
]);
|
||||
expect(report.diagnostics.filter(({ reason }) => reason === "unscorable-outcome")).toEqual([
|
||||
expect.objectContaining({ testName: "skipped", harness: "first" }),
|
||||
]);
|
||||
});
|
||||
|
||||
it("formats lift and telemetry availability for the terminal report", () => {
|
||||
const report = summarizeHarnessComparisons([
|
||||
observation("without-tools", "create", "failed", { totalMs: 34853.7 }),
|
||||
observation("with-tools", "create", "passed", { totalMs: 30694.2 }),
|
||||
]);
|
||||
|
||||
const formatted = stripVTControlCharacters(formatHarnessComparisonReport(report));
|
||||
expect(formatted).toContain("Eval Comparisons");
|
||||
expect(formatted).toContain(" Baseline without-tools");
|
||||
expect(formatted).toContain("Candidate with-tools (1/1 pairs)");
|
||||
expect(formatted).toContain("Pass rate +100.0 pp (candidate 100.0%, baseline 0.0%)");
|
||||
expect(formatted).toContain(" Tokens unavailable");
|
||||
expect(formatted).toContain(" Latency -4159.5ms (candidate 30694.2ms, baseline 34853.7ms)");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,10 @@
|
||||
{
|
||||
"extends": "../../tsconfig.json",
|
||||
"compilerOptions": {
|
||||
"noEmit": true,
|
||||
"module": "NodeNext",
|
||||
"moduleResolution": "NodeNext",
|
||||
"types": ["node", "vitest"]
|
||||
},
|
||||
"include": ["src/**/*.ts"]
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
import { defineConfig, mergeConfig } from "vitest/config";
|
||||
import baseConfig, { workspaceSourcePaths } from "../../vitest.base.ts";
|
||||
|
||||
export default mergeConfig(
|
||||
baseConfig,
|
||||
defineConfig({
|
||||
test: {
|
||||
environment: "node",
|
||||
fileParallelism: false,
|
||||
include: ["src/**/*.eval.ts"],
|
||||
testTimeout: 120000,
|
||||
hookTimeout: 30000,
|
||||
setupFiles: ["./src/vitest-evals/setup.ts"],
|
||||
reporters: ["vitest-evals/reporter", "./src/vitest-evals/reporter.ts"],
|
||||
},
|
||||
resolve: {
|
||||
alias: [{ find: /^@earendil-works\/pi-coding-agent$/, replacement: workspaceSourcePaths.codingAgentIndex }],
|
||||
},
|
||||
}),
|
||||
);
|
||||
@@ -0,0 +1,14 @@
|
||||
import { defineConfig, mergeConfig } from "vitest/config";
|
||||
import baseConfig, { workspaceSourcePaths } from "../../vitest.base.ts";
|
||||
|
||||
export default mergeConfig(
|
||||
baseConfig,
|
||||
defineConfig({
|
||||
test: {
|
||||
include: ["test/**/*.test.ts"],
|
||||
},
|
||||
resolve: {
|
||||
alias: [{ find: /^@earendil-works\/pi-coding-agent$/, replacement: workspaceSourcePaths.codingAgentIndex }],
|
||||
},
|
||||
}),
|
||||
);
|
||||
Reference in New Issue
Block a user