feat(ai-judge): add advisory model catalog and strict corpus replay

- add versioned advisory model catalog shipped with the package and a fail-closed loader
- annotate the enforce session notice for untested, deprecated, and revoked models
- add --strict to corpus-replay with 0/1/2 exit codes and reject strict subset runs
- extract replay qualification into a pure module that recomputes matches and validates latencies
- remove the documented-but-unimplemented --thinking flag and stamp reports with a corpus version
- qualify gpt-5.6-sol as the first recommended entry and archive three real replay reports
- revise the corpus to 2026-08-21.2 changing unclear-forward expected defer to deny
This commit is contained in:
2026-08-21 23:11:40 +08:00
parent 061c60c344
commit 7987fc7a3c
13 changed files with 1544 additions and 39 deletions
@@ -0,0 +1,130 @@
import { describe, expect, it } from "vitest";
import {
classifyModel,
loadModelCatalog,
type CatalogDeps,
} from "../src/catalog";
/**
* PIEXTENSIO-24: the advisory model catalog shipped with the package.
* Advisory only — catalog state affects notifications and docs, never
* Enforce authority (ADR 0008). All expectations are worked literals
* from the schema, independent of any real entry.
*/
const VALID_ENTRY = {
provider: "openai-codex",
model: "gpt-5.6-sol",
api: "openai-codex-responses",
status: "recommended",
promptVersion: "bash-shadow-v4",
corpusVersion: "2026-08-21.1",
testedAt: "2026-08-21T10:00:00Z",
corpusCases: 21,
matched: 21,
infrastructureFailures: 0,
latencyMs: { p50: 3000, p95: 9000, max: 12000 },
reportPath: "reports/corpus-replay-x.json",
};
function depsWith(raw: string | null): CatalogDeps {
return {
catalogPath: "/nonexistent-ai-judge-catalog-test/models-catalog.json",
readFile: (_path: string) => {
if (raw === null) throw new Error("ENOENT");
return raw;
},
};
}
describe("loadModelCatalog", () => {
it("loads a valid versioned catalog with frozen entries", () => {
const { catalog } = run(depsWith(JSON.stringify({
version: 1,
entries: [VALID_ENTRY],
})));
expect(catalog?.version).toBe(1);
expect(catalog?.entries).toHaveLength(1);
expect(catalog?.entries[0]).toMatchObject({
provider: "openai-codex",
model: "gpt-5.6-sol",
status: "recommended",
});
expect(Object.isFrozen(catalog?.entries)).toBe(true);
});
it("degrades to an empty catalog with a diagnostic on an unknown version", () => {
const { catalog, diagnostics } = run(depsWith(JSON.stringify({
version: 99,
entries: [VALID_ENTRY],
})));
expect(catalog?.entries).toEqual([]);
expect(diagnostics.map((d) => d.key)).toContain("version");
});
it("degrades to an empty catalog with a diagnostic on unreadable or malformed files", () => {
const missing = run(depsWith(null));
expect(missing.catalog?.entries).toEqual([]);
expect(missing.diagnostics.map((d) => d.key)).toContain("file");
const malformed = run(depsWith("{not json"));
expect(malformed.catalog?.entries).toEqual([]);
expect(malformed.diagnostics.map((d) => d.key)).toContain("file");
});
it("degrades the whole catalog to empty when any entry is invalid", () => {
const broken = { ...VALID_ENTRY, model: 42 };
const { catalog, diagnostics } = run(depsWith(JSON.stringify({
version: 1,
entries: [broken, VALID_ENTRY],
})));
expect(catalog?.entries).toEqual([]);
expect(diagnostics).toHaveLength(1);
expect(diagnostics[0]?.key).toBe("file");
expect(String(diagnostics[0]?.problem)).toMatch(/whole catalog degraded/i);
});
});
describe("classifyModel", () => {
it("classifies catalog entries by status and unknown models as unlisted", () => {
const { catalog } = run(depsWith(JSON.stringify({
version: 1,
entries: [
VALID_ENTRY,
{
...VALID_ENTRY,
provider: "prov",
model: "old",
status: "deprecated",
},
{
...VALID_ENTRY,
provider: "prov",
model: "bad",
status: "revoked",
},
],
})));
expect(classifyModel(catalog!, "openai-codex", "gpt-5.6-sol")).toBe("recommended");
expect(classifyModel(catalog!, "prov", "old")).toBe("deprecated");
expect(classifyModel(catalog!, "prov", "bad")).toBe("revoked");
expect(classifyModel(catalog!, "openai-codex", "other-model")).toBe("unlisted");
expect(classifyModel(catalog!, "nope", "gpt-5.6-sol")).toBe("unlisted");
});
it("never blocks: classification is pure lookup with no gating semantics", () => {
const { catalog } = run(depsWith(null));
expect(catalog?.entries).toEqual([]);
expect(classifyModel(catalog!, "anything", "anything")).toBe("unlisted");
});
});
// -- helpers -------------------------------------------------------------
function run(deps: CatalogDeps): {
catalog: ReturnType<typeof loadModelCatalog>["catalog"];
diagnostics: ReturnType<typeof loadModelCatalog>["diagnostics"];
} {
const result = loadModelCatalog(deps);
return { catalog: result.catalog, diagnostics: result.diagnostics };
}
@@ -0,0 +1,41 @@
import { describe, expect, it } from "vitest";
import { parseArgs } from "../tools/corpus-replay";
/**
* PIEXTENSIO-24 CLI contract: qualification (--strict) is defined over the
* full corpus only — a --case subset is observation-only and must never be
* able to report qualified. Exit-code mapping itself lives in main()
* (0 qualified / 1 harness failure / 2 qualification failure) and needs a
* live model; the parse contract is what keeps the standard honest.
*/
const BASE = ["node", "corpus-replay.ts", "--provider", "p", "--model", "m"];
describe("corpus-replay parseArgs", () => {
it("accepts --strict with the full corpus", () => {
const parsed = parseArgs([...BASE, "--strict"]);
expect("error" in parsed && parsed.error).toBeFalsy();
expect("strict" in parsed && parsed.strict).toBe(true);
expect("cases" in parsed && parsed.cases).toBeNull();
});
it("rejects --strict together with --case", () => {
const parsed = parseArgs([...BASE, "--strict", "--case", "a,b"]);
expect("error" in parsed && /full corpus/.test(parsed.error)).toBe(true);
});
it("keeps --case usable without --strict", () => {
const parsed = parseArgs([...BASE, "--case", "a,b"]);
expect("error" in parsed && parsed.error).toBeFalsy();
expect("cases" in parsed && parsed.cases).toEqual(new Set(["a", "b"]));
expect("strict" in parsed && parsed.strict).toBe(false);
});
it("rejects unknown options and missing values", () => {
expect(
"error" in parseArgs([...BASE, "--thinking", "high"]),
).toBe(true);
expect("error" in parseArgs([...BASE, "--out"])).toBe(true);
expect("error" in parseArgs(["node", "corpus-replay.ts"])).toBe(true);
});
});
@@ -32,6 +32,25 @@ vi.mock("@earendil-works/pi-coding-agent", async (importOriginal) => {
getAgentDir: () => mockAgentDir.dir || "/nonexistent-ai-judge-test",
};
});
// Catalog seam (PIEXTENSIO-24): index.ts reads the advisory catalog once
// per session; tests inject entries through this hoisted holder while
// keeping the real classifyModel (pure lookup).
const { mockCatalog } = vi.hoisted(() => ({
mockCatalog: {
entries: [] as Array<Record<string, unknown>>,
diagnostics: [] as Array<Record<string, unknown>>,
},
}));
vi.mock("../src/catalog", async (importOriginal) => {
const actual = await importOriginal<typeof import("../src/catalog")>();
return {
...actual,
loadModelCatalog: () => ({
catalog: { version: 1, entries: mockCatalog.entries },
diagnostics: mockCatalog.diagnostics,
}),
};
});
import type { AssistantMessage, Context, Model } from "@earendil-works/pi-ai";
import type {
ExtensionAPI,
@@ -50,6 +69,7 @@ import {
unpublishPermissionsService,
} from "@gotgenes/pi-permission-system";
import extension from "../src/index";
import type { ModelCatalogEntry } from "../src/catalog";
import { PROMPT_VERSION, TOOL_SCHEMA_VERSION } from "../src/prompt";
function createFakePi(): {
@@ -191,6 +211,8 @@ afterEach(() => {
publishedService = undefined;
}
cleanupMockAgentDir();
mockCatalog.entries = [];
mockCatalog.diagnostics = [];
});
describe("AI judge lifecycle", () => {
@@ -875,4 +897,72 @@ describe("AI judge Enforce authority seam (PIEXTENSIO-23, ADR 0008)", () => {
notify.mock.calls.filter((call) => /Enforce/i.test(String(call[0]))),
).toHaveLength(0);
});
it("appends an untested-model note to the enforce notice for an out-of-catalog model", async () => {
const { notify } = await runAsk({
config: { version: 2, mode: "enforce" },
});
const enforceNotice = notify.mock.calls
.map((call) => String(call[0]))
.find((message) => /Enforce/i.test(message));
expect(enforceNotice).toBeDefined();
expect(enforceNotice).toMatch(/untested/i);
expect(enforceNotice).toMatch(/advisory catalog/i);
});
it("adds catalog-status notes for deprecated and revoked enforce models", async () => {
for (const status of ["deprecated", "revoked"] as const) {
mockCatalog.entries = [
{
provider: "session-provider",
model: "session-model",
api: "openai-codex-responses",
status,
promptVersion: "v",
corpusVersion: "v",
testedAt: "2026-01-01T00:00:00Z",
corpusCases: 21,
matched: 21,
infrastructureFailures: 0,
latencyMs: { p50: 1, p95: 2, max: 3 },
reportPath: "reports/x.json",
} satisfies ModelCatalogEntry,
];
const { notify } = await runAsk({
config: { version: 2, mode: "enforce" },
});
const enforceNotice = notify.mock.calls
.map((call) => String(call[0]))
.find((message) => /Enforce/i.test(message));
expect(enforceNotice).toMatch(new RegExp(`catalog status: ${status}`, "i"));
}
});
it("omits the untested note when the enforce model is catalog-recommended", async () => {
mockCatalog.entries = [
{
provider: "session-provider",
model: "session-model",
api: "openai-codex-responses",
status: "recommended",
promptVersion: "v",
corpusVersion: "v",
testedAt: "2026-01-01T00:00:00Z",
corpusCases: 21,
matched: 21,
infrastructureFailures: 0,
latencyMs: { p50: 1, p95: 2, max: 3 },
reportPath: "reports/x.json",
} satisfies ModelCatalogEntry,
];
const { notify } = await runAsk({
config: { version: 2, mode: "enforce" },
});
const enforceNotice = notify.mock.calls
.map((call) => String(call[0]))
.find((message) => /Enforce/i.test(message));
expect(enforceNotice).toBeDefined();
expect(enforceNotice).not.toMatch(/untested/i);
expect(enforceNotice).not.toMatch(/catalog status/i);
});
});
@@ -0,0 +1,124 @@
import { describe, expect, it } from "vitest";
import { qualifyReplay, type ReplayRow } from "../tools/replay-qualify";
/**
* PIEXTENSIO-24: the light qualification standard behind `--strict`.
* Every expectation below is an independently worked literal — the
* corpus-replay harness must be able to fail hard on quality runs while
* keeping its historical "exit 0 keeps unfavorable rows" behavior for
* observation-only runs.
*/
function judgmentRow(
id: string,
verdict: "allow" | "deny" | "defer",
expected: "allow" | "deny" | "defer",
latencyMs: number,
): ReplayRow {
return {
case: id,
expected,
verdict,
match: verdict === expected,
latencyMs,
};
}
function infraRow(id: string): ReplayRow {
return {
case: id,
expected: "allow",
verdict: null,
resultKind: "timeout",
match: false,
latencyMs: null,
};
}
describe("qualifyReplay", () => {
it("qualifies a fully matched replay with latency within budget", () => {
const rows = [
judgmentRow("a", "allow", "allow", 100),
judgmentRow("b", "deny", "deny", 200),
judgmentRow("c", "defer", "defer", 300),
judgmentRow("d", "allow", "allow", 400),
];
const result = qualifyReplay(rows, { budgetMs: 30_000 });
expect(result.qualified).toBe(true);
expect(result.reasons).toEqual([]);
expect(result.matched).toBe(4);
expect(result.mismatches).toEqual([]);
expect(result.infrastructureFailures).toEqual([]);
// Sorted latencies [100, 200, 300, 400]: lower-median p50 = 200,
// nearest-rank p95 = 300, max = 400.
expect(result.latencyMs).toEqual({ p50: 200, p95: 300, max: 400 });
});
it("rejects a mismatched case and names it", () => {
const rows = [
judgmentRow("a", "allow", "allow", 100),
judgmentRow("bad-case", "defer", "deny", 150),
];
const result = qualifyReplay(rows, { budgetMs: 30_000 });
expect(result.qualified).toBe(false);
expect(result.mismatches).toEqual(["bad-case"]);
expect(result.reasons.join(" ")).toContain("bad-case");
expect(result.reasons.join(" ")).toMatch(/mismatch/i);
});
it("rejects any infrastructure failure row", () => {
const rows = [
judgmentRow("a", "allow", "allow", 100),
infraRow("b"),
];
const result = qualifyReplay(rows, { budgetMs: 30_000 });
expect(result.qualified).toBe(false);
expect(result.infrastructureFailures).toEqual(["b"]);
expect(result.reasons.join(" ")).toMatch(/infrastructure/i);
});
it("rejects a judgment latency beyond budget", () => {
const rows = [judgmentRow("slow", "allow", "allow", 30_001)];
const result = qualifyReplay(rows, { budgetMs: 30_000 });
expect(result.qualified).toBe(false);
expect(result.reasons.join(" ")).toMatch(/budget/i);
});
it("recomputes agreement from expected/verdict: a contradictory match flag cannot qualify", () => {
// Harness-reported match=true contradicts verdict!==expected, and
// the judgment carries no usable latency — both must disqualify.
const lying = {
case: "lying-row",
expected: "deny",
verdict: "allow",
match: true,
latencyMs: null,
};
const result = qualifyReplay([lying], { budgetMs: 30_000 });
expect(result.qualified).toBe(false);
expect(result.mismatches).toEqual(["lying-row"]);
expect(result.matched).toBe(0);
expect(result.reasons.join(" ")).toMatch(/usable latency/);
});
it("disqualifies a matching judgment whose latency is missing", () => {
const row = {
case: "no-latency",
expected: "defer",
verdict: "defer",
match: true,
latencyMs: null,
};
const result = qualifyReplay([row], { budgetMs: 30_000 });
expect(result.qualified).toBe(false);
expect(result.mismatches).toEqual([]);
expect(result.reasons.join(" ")).toMatch(/usable latency: no-latency/);
});
it("rejects an empty replay", () => {
const result = qualifyReplay([], { budgetMs: 30_000 });
expect(result.qualified).toBe(false);
expect(result.reasons.join(" ")).toMatch(/no rows/i);
expect(result.latencyMs).toEqual({ p50: null, p95: null, max: null });
});
});