mirror of
https://github.com/SikongJueluo/pi-extensions.git
synced 2026-10-05 11:52:55 +08:00
refactor(ai-judge): split remaining hotspots and cover CRAP branches
- stage-ify judgeAuthorize into auditEnrollment, runPreflightGates, prepareModelCall, and enforceAndEmit - extract tallyJoinedRows and tallyAttributable from computeMetrics, removing three dead locals - extract validateVerdictResponse from requestStructuredVerdict - extract collectUserTexts and charBudgetStart from buildConversationEvidence - table-drive corpus-replay parseArgs and split main into resolveReplayModel, selectCorpusCases, and replayCorpus - split analyzer cli main into loadReviewEvents, withinWindow, loadAuditEnrolled, and printReport with a run-as-script guard - add 81 tests covering parseEntry, extractBashCommandEvidence, validateVerdictResponse, forcedToolChoice, classifyGit dry-run paths, CLI arg/window/report rendering, and the infra-failure result path - add @vitest/coverage-istanbul for exact per-function CRAP scoring via fallow health --coverage
This commit is contained in:
@@ -0,0 +1,292 @@
|
||||
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { spawnSync } from "node:child_process";
|
||||
import { afterEach, describe, expect, it } from "vitest";
|
||||
|
||||
/**
|
||||
* CLI-level tests for arg parsing (`parseArgs`), the `--after`/`--before`
|
||||
* time window (`withinWindow`), and report rendering (`printReport`).
|
||||
* Invalid-arg cases assert the exit contract (stderr + exit 1) that the
|
||||
* analyze-shadow CLI documents.
|
||||
*/
|
||||
|
||||
const dirs: string[] = [];
|
||||
|
||||
function tmp(): string {
|
||||
const dir = mkdtempSync(join(tmpdir(), "ai-judge-cli-args-"));
|
||||
dirs.push(dir);
|
||||
return dir;
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
for (const dir of dirs.splice(0)) {
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
interface RunResult {
|
||||
readonly stdout: string;
|
||||
readonly stderr: string;
|
||||
readonly status: number;
|
||||
}
|
||||
|
||||
function runExpect(args: readonly string[]): RunResult {
|
||||
const cli = join(import.meta.dirname, "..", "..", "src", "analyzer", "cli.ts");
|
||||
const r = spawnSync("npx", ["tsx", cli, ...args], {
|
||||
encoding: "utf-8",
|
||||
stdio: ["ignore", "pipe", "pipe"],
|
||||
});
|
||||
return {
|
||||
stdout: r.stdout ?? "",
|
||||
stderr: r.stderr ?? "",
|
||||
status: r.status ?? -1,
|
||||
};
|
||||
}
|
||||
|
||||
function reviewLog(dir: string, lines: readonly object[]): string {
|
||||
const path = join(dir, "review.jsonl");
|
||||
writeFileSync(
|
||||
path,
|
||||
lines.map((l) => JSON.stringify(l)).join("\n") + "\n",
|
||||
);
|
||||
return path;
|
||||
}
|
||||
|
||||
describe("analyze-shadow CLI — argument parsing (parseArgs)", () => {
|
||||
it("prints usage and exits 0 for --help and -h", () => {
|
||||
for (const flag of ["--help", "-h"]) {
|
||||
const r = runExpect([flag]);
|
||||
expect(r.status).toBe(0);
|
||||
expect(r.stdout.length + r.stderr.length).toBeGreaterThan(0);
|
||||
}
|
||||
});
|
||||
|
||||
it("rejects missing input path with exit 1", () => {
|
||||
const r = runExpect([]);
|
||||
expect(r.status).toBe(1);
|
||||
expect(r.stderr).toContain("missing input path");
|
||||
});
|
||||
|
||||
it("rejects multiple input paths with exit 1", () => {
|
||||
const dir = tmp();
|
||||
const a = reviewLog(dir, []);
|
||||
const b = reviewLog(dir, []).replace("review", "review2");
|
||||
const r = runExpect([a, b]);
|
||||
expect(r.status).toBe(1);
|
||||
expect(r.stderr).toContain("multiple input paths");
|
||||
});
|
||||
|
||||
it("rejects unknown options with exit 1", () => {
|
||||
const dir = tmp();
|
||||
const path = reviewLog(dir, []);
|
||||
const r = runExpect([path, "--bogus"]);
|
||||
expect(r.status).toBe(1);
|
||||
expect(r.stderr).toContain("unknown option: --bogus");
|
||||
});
|
||||
|
||||
it("rejects --after/--before/--audit without a value", () => {
|
||||
const dir = tmp();
|
||||
const path = reviewLog(dir, []);
|
||||
for (const flag of ["--after", "--before", "--audit"]) {
|
||||
const r = runExpect([path, flag]);
|
||||
expect(r.status).toBe(1);
|
||||
expect(r.stderr).toContain(`${flag} requires a value`);
|
||||
}
|
||||
});
|
||||
|
||||
it("rejects invalid --after/--before timestamps", () => {
|
||||
const dir = tmp();
|
||||
const path = reviewLog(dir, []);
|
||||
for (const flag of ["--after", "--before"]) {
|
||||
const r = runExpect([path, flag, "not-a-timestamp"]);
|
||||
expect(r.status).toBe(1);
|
||||
expect(r.stderr).toContain(`invalid ${flag} timestamp`);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe("analyze-shadow CLI — time window (withinWindow)", () => {
|
||||
const events = (ts: string) => [
|
||||
{
|
||||
timestamp: ts,
|
||||
event: "authorizer_chain_resolved",
|
||||
requestId: "req-1",
|
||||
links: ["ai-bash-judge"],
|
||||
},
|
||||
{
|
||||
timestamp: ts,
|
||||
event: "ai_bash_judge.result",
|
||||
requestId: "req-1",
|
||||
resultKind: "judgment",
|
||||
verdict: "allow",
|
||||
},
|
||||
{
|
||||
timestamp: ts,
|
||||
event: "permission_request.approved",
|
||||
requestId: "req-1",
|
||||
resolution: "approved",
|
||||
},
|
||||
];
|
||||
|
||||
it("keeps events inside --after..--before and drops events outside", () => {
|
||||
const dir = tmp();
|
||||
const path = reviewLog(dir, [
|
||||
...events("2026-08-18T00:00:05Z"), // inside
|
||||
...events("2026-08-18T00:00:01Z").map((e, i) => ({ ...e, requestId: `req-old-${i}` })), // before
|
||||
...events("2026-08-18T23:00:00Z").map((e, i) => ({ ...e, requestId: `req-new-${i}` })), // after
|
||||
]);
|
||||
const r = runExpect([
|
||||
path,
|
||||
"--after", "2026-08-18T00:00:02Z",
|
||||
"--before", "2026-08-18T12:00:00Z",
|
||||
]);
|
||||
expect(r.status).toBe(0);
|
||||
expect(r.stdout).toContain("enrollments (N): 1");
|
||||
expect(r.stdout).toContain("joined rows: 1");
|
||||
});
|
||||
|
||||
it("keeps events with a missing timestamp regardless of the window", () => {
|
||||
const dir = tmp();
|
||||
// First event has no timestamp: survives any window.
|
||||
const path = reviewLog(dir, [
|
||||
{
|
||||
event: "authorizer_chain_resolved",
|
||||
requestId: "req-1",
|
||||
links: ["ai-bash-judge"],
|
||||
},
|
||||
...events("2026-08-18T00:00:05Z").slice(1),
|
||||
]);
|
||||
const r = runExpect([
|
||||
path,
|
||||
"--after", "2026-08-18T00:00:02Z",
|
||||
]);
|
||||
expect(r.status).toBe(0);
|
||||
expect(r.stdout).toContain("enrollments (N): 1");
|
||||
});
|
||||
});
|
||||
|
||||
describe("analyze-shadow CLI — report rendering (printReport)", () => {
|
||||
function joinedLog(dir: string): string {
|
||||
return reviewLog(dir, [
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:01Z",
|
||||
event: "authorizer_chain_resolved",
|
||||
requestId: "req-1",
|
||||
links: ["ai-bash-judge"],
|
||||
},
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:02Z",
|
||||
event: "ai_bash_judge.result",
|
||||
requestId: "req-1",
|
||||
resultKind: "judgment",
|
||||
verdict: "allow",
|
||||
judgeLatencyMs: 5,
|
||||
modelLatencyMs: 3,
|
||||
},
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:03Z",
|
||||
event: "permission_request.approved",
|
||||
requestId: "req-1",
|
||||
resolution: "approved",
|
||||
},
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:04Z",
|
||||
event: "authorizer_chain_resolved",
|
||||
requestId: "req-2",
|
||||
links: ["ai-bash-judge"],
|
||||
},
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:05Z",
|
||||
event: "ai_bash_judge.result",
|
||||
requestId: "req-2",
|
||||
resultKind: "judgment",
|
||||
verdict: "deny",
|
||||
code: null,
|
||||
judgeLatencyMs: 7,
|
||||
modelLatencyMs: 4,
|
||||
},
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:06Z",
|
||||
event: "permission_request.approved",
|
||||
requestId: "req-2",
|
||||
resolution: "approved",
|
||||
},
|
||||
]);
|
||||
}
|
||||
|
||||
it("renders the comparison matrix and counters for mixed judgment rows", () => {
|
||||
const dir = tmp();
|
||||
const r = runExpect([joinedLog(dir)]);
|
||||
expect(r.status).toBe(0);
|
||||
// Header + source echo.
|
||||
expect(r.stdout).toContain("AI Bash Judge — Shadow diagnostic report");
|
||||
expect(r.stdout).toContain("grade: DIAGNOSTIC");
|
||||
expect(r.stdout).toContain("source:");
|
||||
// Denominator + join counts.
|
||||
expect(r.stdout).toContain("enrollments (N): 2");
|
||||
expect(r.stdout).toContain("joined rows: 2");
|
||||
expect(r.stdout).toContain("joined judgments: 2");
|
||||
// Matrix rows for both verdicts (allow|allow, deny|allow).
|
||||
expect(r.stdout).toContain("allow|allow: 1");
|
||||
expect(r.stdout).toContain("deny|allow: 1");
|
||||
// Conservative deny counter incremented by deny|allow.
|
||||
expect(r.stdout).toContain("conservative: deny 1, defer 0");
|
||||
// Latency lines rendered with p50/p95/max/missing.
|
||||
expect(r.stdout).toMatch(/judge latency: p50=\d+ms p95=\d+ms/);
|
||||
expect(r.stdout).toMatch(/model latency: p50=\d+ms p95=\d+ms/);
|
||||
// No --audit: the audit source line must be absent.
|
||||
expect(r.stdout).not.toContain("(enrollment source, ADR 0006)");
|
||||
});
|
||||
|
||||
it("renders an empty matrix placeholder and no quarantine section when clean", () => {
|
||||
const dir = tmp();
|
||||
const r = runExpect([joinedLog(dir)]);
|
||||
expect(r.status).toBe(0);
|
||||
expect(r.stdout).not.toContain("quarantined rows:");
|
||||
// Build an empty log: no events at all — matrix stays empty.
|
||||
const empty = reviewLog(dir, []);
|
||||
const r2 = runExpect([empty]);
|
||||
expect(r2.status).toBe(0);
|
||||
expect(r2.stdout).toContain("comparison matrix [verdict|human]:");
|
||||
expect(r2.stdout).toContain("(empty)");
|
||||
expect(r2.stdout).toContain("N/A");
|
||||
});
|
||||
|
||||
it("renders the quarantine section with category counts", () => {
|
||||
const dir = tmp();
|
||||
// Two results for one request → duplicate_result quarantine.
|
||||
const path = reviewLog(dir, [
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:01Z",
|
||||
event: "authorizer_chain_resolved",
|
||||
requestId: "req-1",
|
||||
links: ["ai-bash-judge"],
|
||||
},
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:02Z",
|
||||
event: "ai_bash_judge.result",
|
||||
requestId: "req-1",
|
||||
resultKind: "judgment",
|
||||
verdict: "allow",
|
||||
},
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:03Z",
|
||||
event: "ai_bash_judge.result",
|
||||
requestId: "req-1",
|
||||
resultKind: "judgment",
|
||||
verdict: "deny",
|
||||
},
|
||||
{
|
||||
timestamp: "2026-08-18T00:00:04Z",
|
||||
event: "permission_request.approved",
|
||||
requestId: "req-1",
|
||||
resolution: "approved",
|
||||
},
|
||||
]);
|
||||
const r = runExpect([path]);
|
||||
expect(r.status).toBe(0);
|
||||
expect(r.stdout).toContain("quarantined rows:");
|
||||
expect(r.stdout).toContain("duplicate_result: 1");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,214 @@
|
||||
import { describe, expect, it, vi } from "vitest";
|
||||
import type { ReviewEvent } from "../../src/analyzer/analyze";
|
||||
import { analyzeShadowReviewLog } from "../../src/analyzer/analyze";
|
||||
import {
|
||||
parseArgs,
|
||||
parseTimestampOption,
|
||||
printReport,
|
||||
withinWindow,
|
||||
} from "../../src/analyzer/cli";
|
||||
|
||||
/**
|
||||
* In-process unit tests for the analyze-shadow CLI's pure helpers:
|
||||
* argument parsing, the time-window filter, and report rendering. The
|
||||
* end-to-end contract (file IO, dual-log mode, exit codes) lives in
|
||||
* cli-args.test.ts / cli-audit.test.ts; these tests cover branch-level
|
||||
* behavior without spawning tsx subprocesses.
|
||||
*/
|
||||
|
||||
describe("parseArgs", () => {
|
||||
it("parses the positional path with no options", () => {
|
||||
expect(parseArgs(["node", "cli.ts", "review.jsonl"])).toEqual({
|
||||
path: "review.jsonl",
|
||||
after: null,
|
||||
before: null,
|
||||
audit: null,
|
||||
});
|
||||
});
|
||||
|
||||
it("parses --after, --before, and --audit together", () => {
|
||||
const r = parseArgs([
|
||||
"node",
|
||||
"cli.ts",
|
||||
"review.jsonl",
|
||||
"--after", "2026-08-18T00:00:00Z",
|
||||
"--before", "2026-08-19T00:00:00Z",
|
||||
"--audit", "audit.jsonl",
|
||||
]);
|
||||
expect(r).toEqual({
|
||||
path: "review.jsonl",
|
||||
after: new Date("2026-08-18T00:00:00Z"),
|
||||
before: new Date("2026-08-19T00:00:00Z"),
|
||||
audit: "audit.jsonl",
|
||||
});
|
||||
});
|
||||
|
||||
it("returns the usage error for --help and -h", () => {
|
||||
expect(parseArgs(["node", "cli.ts", "--help"])).toEqual({
|
||||
error: expect.stringContaining("usage: analyze-shadow") as unknown,
|
||||
});
|
||||
expect(parseArgs(["node", "cli.ts", "-h"])).toHaveProperty("error");
|
||||
});
|
||||
|
||||
it.each([
|
||||
["missing input path", [] as const],
|
||||
["multiple input paths", ["a.jsonl", "b.jsonl"] as const],
|
||||
["unknown option", ["a.jsonl", "--bogus"] as const],
|
||||
["--audit without value", ["a.jsonl", "--audit"] as const],
|
||||
["--after without value", ["a.jsonl", "--after"] as const],
|
||||
["invalid --after timestamp", ["a.jsonl", "--after", "nope"] as const],
|
||||
["invalid --before timestamp", ["a.jsonl", "--before", "nope"] as const],
|
||||
])("errors on %s", (_name, argv) => {
|
||||
const r = parseArgs(["node", "cli.ts", ...argv]);
|
||||
expect("error" in r && typeof r.error === "string").toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseTimestampOption", () => {
|
||||
it("requires a value", () => {
|
||||
expect(parseTimestampOption("--after", undefined)).toEqual({
|
||||
error: "--after requires a value",
|
||||
});
|
||||
});
|
||||
|
||||
it("rejects unparseable timestamps", () => {
|
||||
expect(parseTimestampOption("--before", "yesterday")).toEqual({
|
||||
error: "invalid --before timestamp: yesterday",
|
||||
});
|
||||
});
|
||||
|
||||
it("accepts an ISO instant", () => {
|
||||
expect(parseTimestampOption("--after", "2026-08-18T00:00:00Z")).toEqual({
|
||||
date: new Date("2026-08-18T00:00:00Z"),
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("withinWindow", () => {
|
||||
const evt = (timestamp?: string): ReviewEvent =>
|
||||
({ event: "e", ...(timestamp === undefined ? {} : { timestamp }) }) as ReviewEvent;
|
||||
|
||||
it("keeps events without a timestamp in any window", () => {
|
||||
const window = {
|
||||
after: new Date("2026-08-18T12:00:00Z"),
|
||||
before: new Date("2026-08-18T13:00:00Z"),
|
||||
};
|
||||
expect(withinWindow(evt(), window)).toBe(true);
|
||||
});
|
||||
|
||||
it("keeps events inside the window and drops those outside", () => {
|
||||
const window = {
|
||||
after: new Date("2026-08-18T12:00:00Z"),
|
||||
before: new Date("2026-08-18T13:00:00Z"),
|
||||
};
|
||||
expect(withinWindow(evt("2026-08-18T12:30:00Z"), window)).toBe(true);
|
||||
expect(withinWindow(evt("2026-08-18T11:59:00Z"), window)).toBe(false);
|
||||
expect(withinWindow(evt("2026-08-18T13:01:00Z"), window)).toBe(false);
|
||||
});
|
||||
|
||||
it("applies each bound independently", () => {
|
||||
expect(
|
||||
withinWindow(evt("2026-08-18T10:00:00Z"), { after: new Date("2026-08-18T09:00:00Z"), before: null }),
|
||||
).toBe(true);
|
||||
expect(
|
||||
withinWindow(evt("2026-08-18T10:00:00Z"), { after: null, before: new Date("2026-08-18T09:00:00Z") }),
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
it("keeps events with an unparseable timestamp regardless of bounds", () => {
|
||||
const window = {
|
||||
after: new Date("2026-08-18T12:00:00Z"),
|
||||
before: null,
|
||||
};
|
||||
// Unparseable -> NaN time -> neither bound applies.
|
||||
expect(withinWindow(evt("not-a-date"), window)).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("printReport", () => {
|
||||
function capture(fn: () => void): string {
|
||||
const chunks: string[] = [];
|
||||
const spy = vi
|
||||
.spyOn(process.stdout, "write")
|
||||
.mockImplementation(((s: unknown) => {
|
||||
chunks.push(String(s));
|
||||
return true;
|
||||
}) as never);
|
||||
try {
|
||||
fn();
|
||||
} finally {
|
||||
spy.mockRestore();
|
||||
}
|
||||
return chunks.join("");
|
||||
}
|
||||
|
||||
const parsed = {
|
||||
path: "review.jsonl",
|
||||
after: null,
|
||||
before: null,
|
||||
audit: null,
|
||||
};
|
||||
|
||||
function analyze(rows: readonly ReviewEvent[]): ReturnType<
|
||||
typeof import("../../src/analyzer/analyze").analyzeShadowReviewLog
|
||||
> {
|
||||
return analyzeShadowReviewLog(rows);
|
||||
}
|
||||
|
||||
const joinedEvents: ReviewEvent[] = [
|
||||
{ event: "authorizer_chain_resolved", requestId: "r1", links: ["ai-bash-judge"] },
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
requestId: "r1",
|
||||
resultKind: "judgment",
|
||||
verdict: "allow",
|
||||
judgeLatencyMs: 5,
|
||||
modelLatencyMs: 3,
|
||||
},
|
||||
{ event: "permission_request.approved", requestId: "r1", resolution: "approved" },
|
||||
];
|
||||
|
||||
it("renders header, counts, matrix, and latency lines", () => {
|
||||
const { enrollments, metrics } = analyze(joinedEvents);
|
||||
const out = capture(() => printReport(parsed, enrollments, metrics));
|
||||
expect(out).toContain("AI Bash Judge — Shadow diagnostic report");
|
||||
expect(out).toContain("grade: DIAGNOSTIC");
|
||||
expect(out).toContain("enrollments (N): 1");
|
||||
expect(out).toContain("allow|allow: 1");
|
||||
expect(out).toMatch(/judge latency: p50=5ms/);
|
||||
expect(out).not.toContain("audit:");
|
||||
expect(out).not.toContain("quarantined rows:");
|
||||
});
|
||||
|
||||
it("renders the audit source line in dual-log mode", () => {
|
||||
const { enrollments, metrics } = analyze([]);
|
||||
const out = capture(() =>
|
||||
printReport(
|
||||
{ ...parsed, audit: "audit.jsonl" },
|
||||
enrollments,
|
||||
metrics,
|
||||
),
|
||||
);
|
||||
expect(out).toContain("audit: audit.jsonl (enrollment source, ADR 0006)");
|
||||
});
|
||||
|
||||
it("renders the empty-matrix placeholder for an empty log", () => {
|
||||
const { enrollments, metrics } = analyze([]);
|
||||
const out = capture(() => printReport(parsed, enrollments, metrics));
|
||||
expect(out).toContain("(empty)");
|
||||
expect(out).toContain("N/A");
|
||||
});
|
||||
|
||||
it("renders quarantine categories when integrity faults exist", () => {
|
||||
const duplicate: ReviewEvent[] = [
|
||||
{ event: "authorizer_chain_resolved", requestId: "r1", links: ["ai-bash-judge"] },
|
||||
{ event: "ai_bash_judge.result", requestId: "r1", resultKind: "judgment", verdict: "allow" },
|
||||
{ event: "ai_bash_judge.result", requestId: "r1", resultKind: "judgment", verdict: "deny" },
|
||||
{ event: "permission_request.approved", requestId: "r1", resolution: "approved" },
|
||||
];
|
||||
const { enrollments, metrics } = analyze(duplicate);
|
||||
const out = capture(() => printReport(parsed, enrollments, metrics));
|
||||
expect(out).toContain("quarantined rows:");
|
||||
expect(out).toContain("duplicate_result: 1");
|
||||
});
|
||||
});
|
||||
@@ -83,6 +83,61 @@ describe("loadModelCatalog", () => {
|
||||
expect(diagnostics[0]?.key).toBe("file");
|
||||
expect(String(diagnostics[0]?.problem)).toMatch(/whole catalog degraded/i);
|
||||
});
|
||||
|
||||
it("rejects an entry per invalid field (every parseEntry branch)", () => {
|
||||
const cases: readonly [name: string, entry: unknown][] = [
|
||||
["non-object entry", "not-an-object"],
|
||||
["null entry", null],
|
||||
["array entry", [VALID_ENTRY]],
|
||||
["empty provider", { ...VALID_ENTRY, provider: " " }],
|
||||
["empty model", { ...VALID_ENTRY, model: "" }],
|
||||
["empty api", { ...VALID_ENTRY, api: "" }],
|
||||
["unknown status", { ...VALID_ENTRY, status: "experimental" }],
|
||||
["empty promptVersion", { ...VALID_ENTRY, promptVersion: "" }],
|
||||
["empty corpusVersion", { ...VALID_ENTRY, corpusVersion: "" }],
|
||||
["empty testedAt", { ...VALID_ENTRY, testedAt: "" }],
|
||||
["non-integer corpusCases", { ...VALID_ENTRY, corpusCases: 2.5 }],
|
||||
["zero corpusCases", { ...VALID_ENTRY, corpusCases: 0 }],
|
||||
["non-integer matched", { ...VALID_ENTRY, matched: 1.5 }],
|
||||
["non-integer infrastructureFailures", { ...VALID_ENTRY, infrastructureFailures: 0.5 }],
|
||||
["latencyMs not an object", { ...VALID_ENTRY, latencyMs: 3000 }],
|
||||
["latencyMs wrong type", { ...VALID_ENTRY, latencyMs: { p50: "3000", p95: null, max: null } }],
|
||||
["latencyMs missing key", { ...VALID_ENTRY, latencyMs: { p50: 1, p95: null } }],
|
||||
["empty reportPath", { ...VALID_ENTRY, reportPath: "" }],
|
||||
["blank notes", { ...VALID_ENTRY, notes: " " }],
|
||||
];
|
||||
for (const [name, entry] of cases) {
|
||||
const { catalog, diagnostics } = run(depsWith(JSON.stringify({
|
||||
version: 1,
|
||||
entries: [entry],
|
||||
})));
|
||||
expect(catalog?.entries, name).toEqual([]);
|
||||
expect(diagnostics.map((d) => d.key), name).toContain("file");
|
||||
}
|
||||
});
|
||||
|
||||
it("accepts latency nulls and omits notes only when absent", () => {
|
||||
const withNulls = {
|
||||
...VALID_ENTRY,
|
||||
latencyMs: { p50: null, p95: null, max: null },
|
||||
};
|
||||
const { catalog } = run(depsWith(JSON.stringify({
|
||||
version: 1,
|
||||
entries: [withNulls],
|
||||
})));
|
||||
expect(catalog?.entries).toHaveLength(1);
|
||||
expect(catalog?.entries[0]?.latencyMs).toEqual({ p50: null, p95: null, max: null });
|
||||
expect("notes" in (catalog?.entries[0] ?? {})).toBe(false);
|
||||
|
||||
const withNotes = { ...VALID_ENTRY, notes: "qualifying corpus replay report" };
|
||||
const r2 = run(depsWith(JSON.stringify({
|
||||
version: 1,
|
||||
entries: [withNotes],
|
||||
})));
|
||||
expect(r2.catalog?.entries[0]).toMatchObject({
|
||||
notes: "qualifying corpus replay report",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("classifyModel", () => {
|
||||
|
||||
@@ -0,0 +1,205 @@
|
||||
import { describe, expect, it, vi } from "vitest";
|
||||
import type { AssistantMessage, Context, Model } from "@earendil-works/pi-ai";
|
||||
import type { ModelRegistry } from "@earendil-works/pi-coding-agent";
|
||||
import {
|
||||
DEFAULT_TIMEOUT_MS,
|
||||
MIN_TIMEOUT_MS,
|
||||
MAX_TIMEOUT_MS,
|
||||
} from "../src/model";
|
||||
import { createModelAvailability } from "../src/model";
|
||||
import {
|
||||
applyTimeoutOption,
|
||||
replayCorpus,
|
||||
selectCorpusCases,
|
||||
validateCliState,
|
||||
type CliOptions,
|
||||
} from "../tools/corpus-replay";
|
||||
|
||||
/**
|
||||
* In-process tests for the corpus-replay harness helpers: option
|
||||
* application/validation, case selection, and the replay loop driven by a
|
||||
* stub model availability (no registry, no network). The CLI exit contract
|
||||
* lives in corpus-replay-cli.test.ts.
|
||||
*/
|
||||
|
||||
function state(overrides: Partial<CliOptions> = {}): { -readonly [K in keyof CliOptions]: CliOptions[K] } {
|
||||
return {
|
||||
provider: "p",
|
||||
model: "m",
|
||||
timeoutMs: DEFAULT_TIMEOUT_MS,
|
||||
cases: null,
|
||||
out: null,
|
||||
strict: false,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe("applyTimeoutOption", () => {
|
||||
it.each([
|
||||
[MIN_TIMEOUT_MS, null],
|
||||
[MAX_TIMEOUT_MS, null],
|
||||
[DEFAULT_TIMEOUT_MS, null],
|
||||
])("accepts the boundary value %i", (value, error) => {
|
||||
const s = state();
|
||||
expect(applyTimeoutOption(s, String(value))).toBe(error);
|
||||
expect(s.timeoutMs).toBe(value);
|
||||
});
|
||||
|
||||
it.each([
|
||||
[String(MIN_TIMEOUT_MS - 1)],
|
||||
[String(MAX_TIMEOUT_MS + 1)],
|
||||
["12.5"],
|
||||
["abc"],
|
||||
])("rejects %s with the documented range error", (value) => {
|
||||
const s = state();
|
||||
const r = applyTimeoutOption(s, value);
|
||||
expect(r).toMatch(/--timeout-ms must be an integer in \[\d+, \d+\]/);
|
||||
expect(s.timeoutMs).toBe(DEFAULT_TIMEOUT_MS);
|
||||
});
|
||||
|
||||
it("rejects a missing value with the range error (the usage check lives in parseArgs)", () => {
|
||||
expect(applyTimeoutOption(state(), undefined)).toMatch(/--timeout-ms must be an integer/);
|
||||
});
|
||||
});
|
||||
|
||||
describe("validateCliState", () => {
|
||||
it("requires both provider and model", () => {
|
||||
expect(validateCliState(state({ provider: "" }))).toMatch(/^usage:/);
|
||||
expect(validateCliState(state({ model: "" }))).toMatch(/^usage:/);
|
||||
});
|
||||
|
||||
it("rejects --strict combined with a --case subset", () => {
|
||||
expect(
|
||||
validateCliState(state({ strict: true, cases: new Set(["a"]) })),
|
||||
).toContain("--strict requires the full corpus");
|
||||
});
|
||||
|
||||
it("accepts a complete valid state", () => {
|
||||
expect(validateCliState(state())).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("selectCorpusCases", () => {
|
||||
it("selects the full corpus when no case filter is given", () => {
|
||||
const r = selectCorpusCases(null);
|
||||
expect("selected" in r && r.selected.length).toBeGreaterThan(10);
|
||||
});
|
||||
|
||||
it("selects exactly the requested known ids", () => {
|
||||
const r = selectCorpusCases(new Set(["requested-clean", "extra-push"]));
|
||||
expect("selected" in r ? r.selected.map((c) => c.id) : []).toEqual([
|
||||
"requested-clean",
|
||||
"extra-push",
|
||||
]);
|
||||
});
|
||||
|
||||
it("errors on unknown ids without selecting anything", () => {
|
||||
const r = selectCorpusCases(new Set(["requested-clean", "no-such-case"]));
|
||||
expect("error" in r && r.error).toContain("unknown case ids: no-such-case");
|
||||
});
|
||||
});
|
||||
|
||||
describe("replayCorpus", () => {
|
||||
const stderrSpy = () => vi.spyOn(process.stderr, "write").mockReturnValue(true);
|
||||
|
||||
function availabilityWith(
|
||||
respond: () => AssistantMessage,
|
||||
): ReturnType<typeof createModelAvailability> {
|
||||
const model = {
|
||||
id: "m",
|
||||
provider: "p",
|
||||
api: "openai-codex-responses",
|
||||
} as Model<any>;
|
||||
const registry = {
|
||||
complete: async (
|
||||
_m: Model<any>,
|
||||
_c: Context,
|
||||
_o?: Record<string, unknown>,
|
||||
) => respond(),
|
||||
} as unknown as ModelRegistry;
|
||||
return createModelAvailability(model, registry);
|
||||
}
|
||||
|
||||
function judgment(verdict: "allow" | "deny" | "defer"): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call-1",
|
||||
name: "report_verdict",
|
||||
arguments: { verdict, reason: "stub reason" },
|
||||
},
|
||||
],
|
||||
api: "openai-codex-responses",
|
||||
provider: "p",
|
||||
model: "m",
|
||||
stopReason: "toolUse",
|
||||
usage: {
|
||||
input: 10,
|
||||
output: 8,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 18,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
timestamp: Date.now(),
|
||||
} as AssistantMessage;
|
||||
}
|
||||
|
||||
it("counts matches over the selected cases", async () => {
|
||||
const spy = stderrSpy();
|
||||
try {
|
||||
// requested-clean expects allow; extra-push expects defer — one
|
||||
// canned judgment cannot satisfy both, so assert row semantics
|
||||
// rather than a perfect score.
|
||||
const availability = availabilityWith(() => judgment("allow"));
|
||||
const selected = selectCorpusCases(
|
||||
new Set(["requested-clean", "extra-push"]),
|
||||
);
|
||||
if (!("selected" in selected)) throw new Error("unreachable");
|
||||
const { rows, matched } = await replayCorpus(
|
||||
selected.selected,
|
||||
availability,
|
||||
DEFAULT_TIMEOUT_MS,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(rows).toHaveLength(2);
|
||||
expect(matched).toBe(1);
|
||||
const clean = rows.find((r) => r.case === "requested-clean");
|
||||
expect(clean).toMatchObject({ verdict: "allow", match: true });
|
||||
const push = rows.find((r) => r.case === "extra-push");
|
||||
expect(push).toMatchObject({ verdict: "allow", match: false });
|
||||
} finally {
|
||||
spy.mockRestore();
|
||||
}
|
||||
});
|
||||
|
||||
it("records infrastructure failures as unmatched rows", async () => {
|
||||
const spy = stderrSpy();
|
||||
try {
|
||||
const availability = availabilityWith(() => {
|
||||
const m = judgment("allow");
|
||||
m.stopReason = "error";
|
||||
return m;
|
||||
});
|
||||
const selected = selectCorpusCases(new Set(["requested-clean"]));
|
||||
if (!("selected" in selected)) throw new Error("unreachable");
|
||||
const { rows, matched } = await replayCorpus(
|
||||
selected.selected,
|
||||
availability,
|
||||
DEFAULT_TIMEOUT_MS,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(rows).toHaveLength(1);
|
||||
expect(rows[0]).toMatchObject({
|
||||
verdict: null,
|
||||
resultKind: "infrastructure_failure",
|
||||
match: false,
|
||||
});
|
||||
expect(matched).toBe(0);
|
||||
} finally {
|
||||
spy.mockRestore();
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -28,9 +28,14 @@ describe("classifyHighRisk — data_loss", () => {
|
||||
it.each([
|
||||
"git clean -nxd",
|
||||
"git clean -nd",
|
||||
"git clean -nxfd",
|
||||
"git clean -f -x -n",
|
||||
"git clean -fd",
|
||||
"git clean -xn",
|
||||
"git reset --soft HEAD~1",
|
||||
"git checkout main",
|
||||
"git checkout -- file.txt",
|
||||
"git checkout . file.txt",
|
||||
"git restore file.txt",
|
||||
"rm -rf build/",
|
||||
"rm -rf ./dist",
|
||||
|
||||
@@ -542,6 +542,82 @@ describe("AI judge lifecycle", () => {
|
||||
harness.shutdown();
|
||||
});
|
||||
|
||||
it("records a provider failure as an infrastructure_failure row and defers", async () => {
|
||||
let authorize: Authorizer["authorize"] | undefined;
|
||||
const service = {
|
||||
registerAuthorizer: vi.fn((_name, callback) => {
|
||||
authorize = callback;
|
||||
return vi.fn();
|
||||
}),
|
||||
checkPermission: vi.fn(),
|
||||
getToolPermission: vi.fn(),
|
||||
} as unknown as PermissionsService;
|
||||
publishPermissionsService(service);
|
||||
publishedService = service;
|
||||
|
||||
const complete = vi.fn(
|
||||
async () => {
|
||||
throw new Error("provider 500");
|
||||
},
|
||||
);
|
||||
const ctx = {
|
||||
hasUI: true,
|
||||
sessionManager: fakeSessionManager(),
|
||||
model: {
|
||||
id: "test-model",
|
||||
provider: "test-provider",
|
||||
api: "openai-codex-responses",
|
||||
} as Model<any>,
|
||||
modelRegistry: { complete },
|
||||
ui: { notify: vi.fn() },
|
||||
} as unknown as ExtensionContext;
|
||||
|
||||
const harness = createFakePi();
|
||||
extension(harness.pi);
|
||||
harness.start(ctx);
|
||||
harness.ready();
|
||||
expect(authorize).toBeDefined();
|
||||
|
||||
const reviews: Array<{
|
||||
event: string;
|
||||
details?: Record<string, unknown>;
|
||||
}> = [];
|
||||
const verdict = await authorize!(
|
||||
ask(),
|
||||
{
|
||||
checkPermission: vi.fn(),
|
||||
getToolPermission: vi.fn(),
|
||||
},
|
||||
{
|
||||
review: (event, details) => reviews.push({ event, details }),
|
||||
debug: vi.fn(),
|
||||
},
|
||||
);
|
||||
|
||||
expect(verdict).toEqual({ kind: "defer" });
|
||||
expect(complete).toHaveBeenCalledTimes(1);
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
details: expect.objectContaining({
|
||||
resultKind: "infrastructure_failure",
|
||||
verdict: null,
|
||||
effectiveVerdict: "defer",
|
||||
modelCalled: true,
|
||||
code: "model_error",
|
||||
modelSource: "session",
|
||||
riskOverride: null,
|
||||
inputUsage: null,
|
||||
outputUsage: null,
|
||||
}),
|
||||
},
|
||||
]);
|
||||
// The failure row must not leak the provider error text.
|
||||
expect(JSON.stringify(reviews)).not.toContain("provider 500");
|
||||
|
||||
harness.shutdown();
|
||||
});
|
||||
|
||||
it("does not register from a headless child", () => {
|
||||
const service = {
|
||||
registerAuthorizer: vi.fn(),
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
import { describe, expect, it, vi } from "vitest";
|
||||
import type { AssistantMessage, Context } from "@earendil-works/pi-ai";
|
||||
import type { AssistantMessage, Context, Model } from "@earendil-works/pi-ai";
|
||||
import type { ModelRegistry } from "@earendil-works/pi-coding-agent";
|
||||
import {
|
||||
createModelAvailability,
|
||||
requestStructuredVerdict,
|
||||
type ModelAvailability,
|
||||
} from "../src/model";
|
||||
@@ -50,6 +52,18 @@ function ready(
|
||||
return { kind: "ready", metadata, complete };
|
||||
}
|
||||
|
||||
/** A minimal valid report_verdict tool-call content block. */
|
||||
function verdictContent(): AssistantMessage["content"] {
|
||||
return [
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call-1",
|
||||
name: "report_verdict",
|
||||
arguments: { verdict: "allow", reason: "bounded" },
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
const evidence = {
|
||||
fullCommand: "pnpm test && git push",
|
||||
triggeringUnit: "git push",
|
||||
@@ -231,6 +245,137 @@ describe("requestStructuredVerdict", () => {
|
||||
});
|
||||
});
|
||||
|
||||
it("rejects zero and non-finite output usage", async () => {
|
||||
const zeroUsage = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
response(verdictContent(), 0),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(zeroUsage).toMatchObject({ code: "model_error" });
|
||||
|
||||
const nonFiniteUsage = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
response(verdictContent(), Number.NaN),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(nonFiniteUsage).toMatchObject({ code: "model_error" });
|
||||
});
|
||||
|
||||
it("rejects responses without exactly one report_verdict tool call", async () => {
|
||||
const noCall = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
response([{ type: "text", text: "I would allow it." } as never]),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(noCall).toMatchObject({ code: "missing_tool_call" });
|
||||
|
||||
const wrongName = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
response([
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call-1",
|
||||
name: "other_tool",
|
||||
arguments: { verdict: "allow", reason: "x" },
|
||||
},
|
||||
]),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(wrongName).toMatchObject({ code: "missing_tool_call" });
|
||||
});
|
||||
|
||||
it("rejects null and array tool-call arguments", async () => {
|
||||
const nullArgs = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
response([
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call-1",
|
||||
name: "report_verdict",
|
||||
arguments: null as unknown as Record<string, unknown>,
|
||||
},
|
||||
]),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(nullArgs).toMatchObject({ code: "invalid_arguments" });
|
||||
|
||||
const arrayArgs = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
response([
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call-1",
|
||||
name: "report_verdict",
|
||||
arguments: ["allow", "x"] as unknown as Record<string, unknown>,
|
||||
},
|
||||
]),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(arrayArgs).toMatchObject({ code: "invalid_arguments" });
|
||||
|
||||
const missingKey = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
response([
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call-1",
|
||||
name: "report_verdict",
|
||||
arguments: { verdict: "allow" },
|
||||
},
|
||||
]),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(missingKey).toMatchObject({ code: "invalid_arguments" });
|
||||
});
|
||||
|
||||
it("rejects non-string and blank-trimmed reasons", async () => {
|
||||
const nonStringReason = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
response([
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call-1",
|
||||
name: "report_verdict",
|
||||
arguments: { verdict: "defer", reason: 42 as unknown as string },
|
||||
},
|
||||
]),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(nonStringReason).toMatchObject({ code: "invalid_reason" });
|
||||
|
||||
const blankReason = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
response([
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call-1",
|
||||
name: "report_verdict",
|
||||
arguments: { verdict: "defer", reason: " " },
|
||||
},
|
||||
]),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(blankReason).toMatchObject({ code: "invalid_reason" });
|
||||
});
|
||||
|
||||
it("maps the bounded deadline to timeout", async () => {
|
||||
const waiting = ready(
|
||||
async (_context, signal) =>
|
||||
@@ -319,3 +464,92 @@ describe("requestStructuredVerdict", () => {
|
||||
expect(timedOut).toMatchObject({ code: "timeout" });
|
||||
});
|
||||
});
|
||||
|
||||
describe("createModelAvailability — per-API forced tool choice and output cap", () => {
|
||||
it("returns no_model for an undefined model", () => {
|
||||
expect(createModelAvailability(undefined, registryStub())).toEqual({
|
||||
kind: "no_model",
|
||||
});
|
||||
});
|
||||
|
||||
it("returns unsupported_api for an unknown api", () => {
|
||||
const availability = createModelAvailability(
|
||||
modelWithApi("mystery-api"),
|
||||
registryStub(),
|
||||
);
|
||||
expect(availability).toMatchObject({
|
||||
kind: "unsupported_api",
|
||||
metadata: { api: "mystery-api" },
|
||||
});
|
||||
});
|
||||
|
||||
it.each([
|
||||
// [api, expected toolChoice, enforces output cap]
|
||||
["anthropic-messages", { type: "tool", name: "report_verdict" }, true],
|
||||
["bedrock-converse-stream", { type: "tool", name: "report_verdict" }, true],
|
||||
["google-generative-ai", "any", true],
|
||||
["google-vertex", "any", true],
|
||||
["openai-completions", { type: "function", function: { name: "report_verdict" } }, true],
|
||||
["mistral-conversations", { type: "function", function: { name: "report_verdict" } }, true],
|
||||
["pi-messages", { type: "function", function: { name: "report_verdict" } }, true],
|
||||
["openai-responses", { type: "function", name: "report_verdict" }, true],
|
||||
["azure-openai-responses", { type: "function", name: "report_verdict" }, true],
|
||||
// Codex supports required but not named choice, and no output cap.
|
||||
["openai-codex-responses", "required", false],
|
||||
])("maps %s to its forced tool choice and cap policy", async (api, toolChoice, cap) => {
|
||||
const calls: Array<Record<string, unknown>> = [];
|
||||
const registry = registryStub({
|
||||
complete: (_model, _context, options) => {
|
||||
calls.push(options as Record<string, unknown>);
|
||||
return Promise.resolve(response(verdictContent()));
|
||||
},
|
||||
});
|
||||
const availability = createModelAvailability(modelWithApi(api), registry);
|
||||
expect(availability).toMatchObject({ kind: "ready", metadata: { api } });
|
||||
|
||||
if (availability.kind === "ready") {
|
||||
await availability.complete({} as never, new AbortController().signal);
|
||||
}
|
||||
expect(calls[0]?.toolChoice).toEqual(toolChoice);
|
||||
if (cap) {
|
||||
expect(calls[0]?.maxTokens).toBeTypeOf("number");
|
||||
} else {
|
||||
expect(calls[0]?.maxTokens).toBeUndefined();
|
||||
}
|
||||
});
|
||||
|
||||
it("forwards retry and cache policy to the registry", async () => {
|
||||
let seen: Record<string, unknown> | undefined;
|
||||
const registry = registryStub({
|
||||
complete: (_m, _c, options) => {
|
||||
seen = options as Record<string, unknown>;
|
||||
return Promise.resolve(response(verdictContent()));
|
||||
},
|
||||
});
|
||||
const availability = createModelAvailability(
|
||||
modelWithApi("anthropic-messages"),
|
||||
registry,
|
||||
);
|
||||
if (availability.kind === "ready") {
|
||||
await availability.complete({} as never, new AbortController().signal);
|
||||
}
|
||||
expect(seen).toMatchObject({ maxRetries: 0, cacheRetention: "none" });
|
||||
});
|
||||
});
|
||||
|
||||
function modelWithApi(api: string): Model<any> {
|
||||
return {
|
||||
provider: "test-provider",
|
||||
id: "test-model",
|
||||
api,
|
||||
} as unknown as Model<any>;
|
||||
}
|
||||
|
||||
function registryStub(overrides: Partial<ModelRegistry> = {}): ModelRegistry {
|
||||
return {
|
||||
complete: () => {
|
||||
throw new Error("not called");
|
||||
},
|
||||
...overrides,
|
||||
} as unknown as ModelRegistry;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user