feat: integrate Pi backend and Pi Web

This commit is contained in:
luckyyzh
2026-07-30 19:37:53 +08:00
commit 7392ab9dd7
1390 changed files with 337197 additions and 0 deletions
+34
View File
@@ -0,0 +1,34 @@
import { describe, expect, it } from "vitest";
import { resolveModelSelection } from "../src/pi-harness.ts";
describe("resolveModelSelection", () => {
it("prefers an explicit harness model over environment defaults", () => {
expect(
resolveModelSelection(
{ provider: "anthropic", id: "claude-opus-4-6" },
{ PI_PROVIDER: "openai-codex", PI_MODEL: "gpt-5.6-sol" },
),
).toEqual({ provider: "anthropic", id: "claude-opus-4-6" });
});
it("uses trimmed environment defaults when the harness has no explicit model", () => {
expect(resolveModelSelection(undefined, { PI_PROVIDER: " openai-codex ", PI_MODEL: " gpt-5.6-sol " })).toEqual({
provider: "openai-codex",
id: "gpt-5.6-sol",
});
});
it.each([
[undefined, {}],
[undefined, { PI_PROVIDER: "openai-codex" }],
[undefined, { PI_MODEL: "gpt-5.6-sol" }],
[
{ provider: "", id: "gpt-5.6-sol" },
{ PI_PROVIDER: "openai-codex", PI_MODEL: "gpt-5.6-sol" },
],
] as const)("rejects an incomplete model selection", (explicitModel, environment) => {
expect(() => resolveModelSelection(explicitModel, environment)).toThrow(
"Select a harness model explicitly or set both PI_PROVIDER and PI_MODEL as defaults.",
);
});
});
@@ -0,0 +1,103 @@
import { mkdtemp, readFile, rm } from "node:fs/promises";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { expect, it } from "vitest";
import {
persistEvalArtifactReferences,
recordEvalSessionArtifact,
recordEvalSourceArtifact,
} from "../../src/vitest-evals/artifacts.ts";
it("records session and source artifacts against the explicit test task", async ({ task }) => {
const runId = "run-1";
await recordEvalSessionArtifact(task, {
artifacts: { runId, piSessionJsonl: '{"type":"session"}\n' },
});
await recordEvalSourceArtifact(task, runId, {
name: "hello.ts",
contentType: "text/typescript",
body: "export default function () {}\n",
bodyEncoding: "utf-8",
});
expect(task.artifacts).toContainEqual(
expect.objectContaining({
type: "@earendil-works/pi-evals:session",
runId,
attachments: [
expect.objectContaining({
name: "session.jsonl",
body: '{"type":"session"}\n',
bodyEncoding: "utf-8",
contentType: "application/jsonl",
}),
],
}),
);
expect(task.artifacts).toContainEqual(
expect.objectContaining({
type: "@earendil-works/pi-evals:source",
runId,
attachments: [
expect.objectContaining({
name: "hello.ts",
body: "export default function () {}\n",
bodyEncoding: "utf-8",
contentType: "text/typescript",
}),
],
}),
);
});
it("persists and selects attachments belonging to the reported run", async () => {
const root = await mkdtemp(join(tmpdir(), "pi-eval-artifact-report-test-"));
try {
const references = await persistEvalArtifactReferences(
[
{
type: "@earendil-works/pi-evals:session",
runId: "run-1",
attachments: [
{
name: "session.jsonl",
body: '{"type":"session"}\n',
bodyEncoding: "utf-8",
contentType: "application/jsonl",
},
],
},
{
type: "@earendil-works/pi-evals:session",
runId: "run-2",
attachments: [],
},
{
type: "@earendil-works/pi-evals:source",
runId: "run-1",
attachments: [
{
name: "hello.ts",
body: "export default function () {}\n",
bodyEncoding: "utf-8",
contentType: "text/typescript",
},
],
},
{ type: "internal:annotation", annotation: { message: "other", type: "info" } },
],
"run-1",
root,
);
expect(references).toEqual([
{ name: "session.jsonl", path: expect.stringMatching(/^sessions\/[a-f0-9]{64}\/session\.jsonl$/) },
{ name: "hello.ts", path: expect.stringMatching(/^sources\/[a-f0-9]{64}\/hello\.ts$/) },
]);
for (const { name, path } of references) {
const expected = name === "session.jsonl" ? '{"type":"session"}\n' : "export default function () {}\n";
expect(await readFile(join(root, path), "utf8")).toBe(expected);
}
} finally {
await rm(root, { recursive: true, force: true });
}
});
@@ -0,0 +1,94 @@
import { describe, expect, it } from "vitest";
import { createHarness, type HarnessContext } from "vitest-evals/harness";
import {
deriveEvalGroupKey,
EVAL_HARNESS_ITERATION_ARTIFACT,
evalHarnessTable,
parseEvalHarnessIterationArtifact,
} from "../../src/vitest-evals/harness-table.ts";
describe("deriveEvalGroupKey", () => {
it("combines a trimmed string input ID with repetition", () => {
expect(deriveEvalGroupKey({ id: " input-1 ", prompt: "hello" }, 2)).toBe(JSON.stringify(["input-1", 2]));
});
it("hashes canonical JSON independently of object key order", () => {
expect(deriveEvalGroupKey({ first: 1, second: [true, "value"] }, 1)).toBe(
deriveEvalGroupKey({ second: [true, "value"], first: 1 }, 1),
);
expect(deriveEvalGroupKey({ first: 1 }, 1)).not.toBe(deriveEvalGroupKey({ first: 2 }, 1));
expect(deriveEvalGroupKey({ first: 1 }, 1)).not.toBe(deriveEvalGroupKey({ first: 1 }, 2));
expect(deriveEvalGroupKey(["first", "second"], 1)).not.toBe(deriveEvalGroupKey(["second", "first"], 1));
});
it("rejects non-JSON and circular input", () => {
const circular: { self?: unknown } = {};
circular.self = circular;
expect(() => deriveEvalGroupKey(new Date(0), 1)).toThrow("only plain objects and arrays");
expect(() => deriveEvalGroupKey(Array(1), 1)).toThrow("must not be sparse");
expect(() => deriveEvalGroupKey(circular, 1)).toThrow("must not contain circular references");
});
});
function createFakeHarness(name: string) {
return createHarness<{ id: string }, { harness: string; inputId: string }>({
name,
run: ({ input }) => ({
output: { harness: name, inputId: input.id },
events: [
{ type: "message", role: "user", content: input.id },
{ type: "message", role: "assistant", content: name },
],
}),
});
}
const harnessTable = evalHarnessTable("local multi-harness eval", {
baseline: createFakeHarness("withoutSkill"),
candidates: [createFakeHarness("withSkill")],
repetitions: 2,
});
describe("evalHarnessTable", () => {
it("plans repetitions in declaration order", () => {
expect(harnessTable.map(({ name, repetition }) => ({ name, repetition }))).toEqual([
{ name: "withoutSkill", repetition: 1 },
{ name: "withSkill", repetition: 1 },
{ name: "withoutSkill", repetition: 2 },
{ name: "withSkill", repetition: 2 },
]);
});
it("accepts a singular candidate", () => {
const rows = evalHarnessTable("singular candidate", {
baseline: createFakeHarness("baseline"),
candidate: createFakeHarness("candidate"),
});
expect(rows.map(({ name }) => name)).toEqual(["baseline", "candidate"]);
});
it("attaches iteration metadata to every wrapped harness run", async () => {
for (const row of harnessTable) {
const artifacts: HarnessContext["artifacts"] = {};
const context: HarnessContext = {
artifacts,
setArtifact(name, value) {
artifacts[name] = value;
},
};
const result = await row.harness.run({ id: "first" }, context);
expect(result.output).toEqual({ harness: row.name, inputId: "first" });
expect(parseEvalHarnessIterationArtifact(result.artifacts?.[EVAL_HARNESS_ITERATION_ARTIFACT])).toEqual({
schemaVersion: 1,
evalSet: "local multi-harness eval",
groupKey: deriveEvalGroupKey({ id: "first" }, row.repetition),
harness: row.name,
baseline: "withoutSkill",
candidates: ["withSkill"],
repetition: row.repetition,
});
}
});
});
@@ -0,0 +1,232 @@
import { stripVTControlCharacters } from "node:util";
import { describe, expect, it } from "vitest";
import {
formatHarnessComparisonReport,
type HarnessObservation,
summarizeHarnessComparisons,
} from "../../src/vitest-evals/summary.ts";
type ObservationResult = "passed" | "failed" | Exclude<HarnessObservation["outcome"], "scored">;
function observation(
harness: string,
testName: string,
result: ObservationResult,
metrics: Pick<HarnessObservation, "totalTokens" | "totalMs" | "estimatedCostUsd"> = {},
baseline = "without-tools",
candidates: string[] = ["with-tools"],
): HarnessObservation {
const base = {
evalSet: "tool access",
groupKey: JSON.stringify([testName, 1]),
testName,
file: "src/tool-access.eval.ts",
harness,
baseline,
candidates,
repetition: 1,
...metrics,
};
if (result === "passed" || result === "failed") {
return { ...base, outcome: "scored", score: result === "passed" ? 1 : 0 };
}
return { ...base, outcome: result };
}
describe("summarizeHarnessComparisons", () => {
it("computes paired correctness lift separately from efficiency deltas", () => {
const report = summarizeHarnessComparisons([
observation("without-tools", "create", "failed", {
totalTokens: 100,
totalMs: 1000,
estimatedCostUsd: 0.01,
}),
observation("with-tools", "create", "passed", {
totalTokens: 120,
totalMs: 800,
estimatedCostUsd: 0.02,
}),
observation("without-tools", "inspect", "passed", { totalTokens: 200 }),
observation("with-tools", "inspect", "passed", { totalTokens: 180 }),
]);
expect(report.evalSets).toHaveLength(1);
expect(report.evalSets[0]?.comparisons).toEqual([
expect.objectContaining({
baseline: "without-tools",
candidate: "with-tools",
correctness: {
totalPairs: 2,
eligiblePairs: 2,
baselinePassRate: 0.5,
candidatePassRate: 1,
lift: 0.5,
baselineWins: 0,
candidateWins: 1,
ties: 1,
},
totalTokens: {
totalPairs: 2,
eligiblePairs: 2,
baselineMean: 150,
candidateMean: 150,
meanDelta: 0,
},
totalMs: {
totalPairs: 2,
eligiblePairs: 1,
baselineMean: 1000,
candidateMean: 800,
meanDelta: -200,
},
estimatedCostUsd: {
totalPairs: 2,
eligiblePairs: 1,
baselineMean: 0.01,
candidateMean: 0.02,
meanDelta: 0.01,
},
}),
]);
expect(report.diagnostics).toEqual([]);
});
it("reports missing observations without coercing them to failures or zero telemetry", () => {
const report = summarizeHarnessComparisons([
observation("without-tools", "create", "failed"),
observation("with-tools", "create", "passed"),
observation("without-tools", "inspect", "passed"),
]);
const comparison = report.evalSets[0]?.comparisons[0];
expect(comparison?.correctness).toEqual({
totalPairs: 2,
eligiblePairs: 1,
baselinePassRate: 0,
candidatePassRate: 1,
lift: 1,
baselineWins: 0,
candidateWins: 1,
ties: 0,
});
expect(comparison?.totalTokens).toEqual({
totalPairs: 2,
eligiblePairs: 0,
baselineMean: null,
candidateMean: null,
meanDelta: null,
});
expect(report.diagnostics).toContainEqual(
expect.objectContaining({
testName: "inspect",
harness: "with-tools",
reason: "missing-observation",
}),
);
});
it("keeps identical inputs in different test files separate", () => {
const report = summarizeHarnessComparisons([
observation("without-tools", "shared", "failed"),
observation("with-tools", "shared", "passed"),
{ ...observation("without-tools", "shared", "passed"), file: "src/other.eval.ts" },
{ ...observation("with-tools", "shared", "passed"), file: "src/other.eval.ts" },
]);
expect(report.evalSets[0]?.comparisons[0]?.correctness).toEqual(
expect.objectContaining({ totalPairs: 2, eligiblePairs: 2 }),
);
expect(report.diagnostics).toEqual([]);
});
it("does not score harness errors as correctness failures", () => {
const report = summarizeHarnessComparisons([
observation("without-tools", "create", "errored", { totalTokens: 100 }),
observation("with-tools", "create", "passed", { totalTokens: 100 }),
]);
expect(report.evalSets[0]?.comparisons[0]?.correctness).toEqual(
expect.objectContaining({ totalPairs: 1, eligiblePairs: 0 }),
);
expect(report.evalSets[0]?.comparisons[0]?.totalTokens.eligiblePairs).toBe(0);
expect(report.diagnostics).toContainEqual(
expect.objectContaining({ harness: "without-tools", reason: "harness-error" }),
);
});
it("does not derive correctness from completed Vitest tests without judge scores", () => {
const withoutScore = observation("without-tools", "create", "unscored");
const withScore = observation("with-tools", "create", "unscored");
const report = summarizeHarnessComparisons([withoutScore, withScore]);
expect(report.evalSets[0]?.comparisons[0]?.correctness.eligiblePairs).toBe(0);
expect(report.diagnostics).toEqual([
expect.objectContaining({ harness: "with-tools", reason: "missing-score" }),
expect.objectContaining({ harness: "without-tools", reason: "missing-score" }),
]);
});
it("compares each candidate with the declared baseline", () => {
const candidates = ["second", "third"];
const report = summarizeHarnessComparisons([
observation("first", "input", "passed", {}, "first", candidates),
observation("second", "input", "passed", {}, "first", candidates),
observation("third", "input", "passed", {}, "first", candidates),
]);
expect(report.evalSets[0]?.comparisons.map(({ baseline, candidate }) => [baseline, candidate])).toEqual([
["first", "second"],
["first", "third"],
]);
});
it("retains a declared harness with no completed observations", () => {
const report = summarizeHarnessComparisons([observation("without-tools", "create", "failed")]);
expect(report.evalSets[0]?.comparisons).toHaveLength(1);
expect(report.evalSets[0]?.comparisons[0]?.correctness.eligiblePairs).toBe(0);
expect(report.diagnostics).toContainEqual(
expect.objectContaining({
testName: "create",
harness: "with-tools",
reason: "missing-observation",
}),
);
});
it("reports duplicate and unscorable observations once across multiple harness pairs", () => {
const candidates = ["second", "third"];
const report = summarizeHarnessComparisons([
observation("first", "duplicate", "passed", {}, "first", candidates),
observation("first", "duplicate", "failed", {}, "first", candidates),
observation("second", "duplicate", "passed", {}, "first", candidates),
observation("third", "duplicate", "passed", {}, "first", candidates),
observation("first", "skipped", "skipped", {}, "first", candidates),
observation("second", "skipped", "passed", {}, "first", candidates),
observation("third", "skipped", "passed", {}, "first", candidates),
]);
expect(report.diagnostics.filter(({ reason }) => reason === "duplicate-observation")).toEqual([
expect.objectContaining({ testName: "duplicate", harness: "first" }),
]);
expect(report.diagnostics.filter(({ reason }) => reason === "unscorable-outcome")).toEqual([
expect.objectContaining({ testName: "skipped", harness: "first" }),
]);
});
it("formats lift and telemetry availability for the terminal report", () => {
const report = summarizeHarnessComparisons([
observation("without-tools", "create", "failed", { totalMs: 34853.7 }),
observation("with-tools", "create", "passed", { totalMs: 30694.2 }),
]);
const formatted = stripVTControlCharacters(formatHarnessComparisonReport(report));
expect(formatted).toContain("Eval Comparisons");
expect(formatted).toContain(" Baseline without-tools");
expect(formatted).toContain("Candidate with-tools (1/1 pairs)");
expect(formatted).toContain("Pass rate +100.0 pp (candidate 100.0%, baseline 0.0%)");
expect(formatted).toContain(" Tokens unavailable");
expect(formatted).toContain(" Latency -4159.5ms (candidate 30694.2ms, baseline 34853.7ms)");
});
});