mirror of
https://github.com/SikongJueluo/pi-extensions.git
synced 2026-10-05 11:52:55 +08:00
feat(ai-judge): add advisory model catalog and strict corpus replay
- add versioned advisory model catalog shipped with the package and a fail-closed loader - annotate the enforce session notice for untested, deprecated, and revoked models - add --strict to corpus-replay with 0/1/2 exit codes and reject strict subset runs - extract replay qualification into a pure module that recomputes matches and validates latencies - remove the documented-but-unimplemented --thinking flag and stamp reports with a corpus version - qualify gpt-5.6-sol as the first recommended entry and archive three real replay reports - revise the corpus to 2026-08-21.2 changing unclear-forward expected defer to deny
This commit is contained in:
@@ -0,0 +1,130 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
classifyModel,
|
||||
loadModelCatalog,
|
||||
type CatalogDeps,
|
||||
} from "../src/catalog";
|
||||
|
||||
/**
|
||||
* PIEXTENSIO-24: the advisory model catalog shipped with the package.
|
||||
* Advisory only — catalog state affects notifications and docs, never
|
||||
* Enforce authority (ADR 0008). All expectations are worked literals
|
||||
* from the schema, independent of any real entry.
|
||||
*/
|
||||
|
||||
const VALID_ENTRY = {
|
||||
provider: "openai-codex",
|
||||
model: "gpt-5.6-sol",
|
||||
api: "openai-codex-responses",
|
||||
status: "recommended",
|
||||
promptVersion: "bash-shadow-v4",
|
||||
corpusVersion: "2026-08-21.1",
|
||||
testedAt: "2026-08-21T10:00:00Z",
|
||||
corpusCases: 21,
|
||||
matched: 21,
|
||||
infrastructureFailures: 0,
|
||||
latencyMs: { p50: 3000, p95: 9000, max: 12000 },
|
||||
reportPath: "reports/corpus-replay-x.json",
|
||||
};
|
||||
|
||||
function depsWith(raw: string | null): CatalogDeps {
|
||||
return {
|
||||
catalogPath: "/nonexistent-ai-judge-catalog-test/models-catalog.json",
|
||||
readFile: (_path: string) => {
|
||||
if (raw === null) throw new Error("ENOENT");
|
||||
return raw;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
describe("loadModelCatalog", () => {
|
||||
it("loads a valid versioned catalog with frozen entries", () => {
|
||||
const { catalog } = run(depsWith(JSON.stringify({
|
||||
version: 1,
|
||||
entries: [VALID_ENTRY],
|
||||
})));
|
||||
expect(catalog?.version).toBe(1);
|
||||
expect(catalog?.entries).toHaveLength(1);
|
||||
expect(catalog?.entries[0]).toMatchObject({
|
||||
provider: "openai-codex",
|
||||
model: "gpt-5.6-sol",
|
||||
status: "recommended",
|
||||
});
|
||||
expect(Object.isFrozen(catalog?.entries)).toBe(true);
|
||||
});
|
||||
|
||||
it("degrades to an empty catalog with a diagnostic on an unknown version", () => {
|
||||
const { catalog, diagnostics } = run(depsWith(JSON.stringify({
|
||||
version: 99,
|
||||
entries: [VALID_ENTRY],
|
||||
})));
|
||||
expect(catalog?.entries).toEqual([]);
|
||||
expect(diagnostics.map((d) => d.key)).toContain("version");
|
||||
});
|
||||
|
||||
it("degrades to an empty catalog with a diagnostic on unreadable or malformed files", () => {
|
||||
const missing = run(depsWith(null));
|
||||
expect(missing.catalog?.entries).toEqual([]);
|
||||
expect(missing.diagnostics.map((d) => d.key)).toContain("file");
|
||||
|
||||
const malformed = run(depsWith("{not json"));
|
||||
expect(malformed.catalog?.entries).toEqual([]);
|
||||
expect(malformed.diagnostics.map((d) => d.key)).toContain("file");
|
||||
});
|
||||
|
||||
it("degrades the whole catalog to empty when any entry is invalid", () => {
|
||||
const broken = { ...VALID_ENTRY, model: 42 };
|
||||
const { catalog, diagnostics } = run(depsWith(JSON.stringify({
|
||||
version: 1,
|
||||
entries: [broken, VALID_ENTRY],
|
||||
})));
|
||||
expect(catalog?.entries).toEqual([]);
|
||||
expect(diagnostics).toHaveLength(1);
|
||||
expect(diagnostics[0]?.key).toBe("file");
|
||||
expect(String(diagnostics[0]?.problem)).toMatch(/whole catalog degraded/i);
|
||||
});
|
||||
});
|
||||
|
||||
describe("classifyModel", () => {
|
||||
it("classifies catalog entries by status and unknown models as unlisted", () => {
|
||||
const { catalog } = run(depsWith(JSON.stringify({
|
||||
version: 1,
|
||||
entries: [
|
||||
VALID_ENTRY,
|
||||
{
|
||||
...VALID_ENTRY,
|
||||
provider: "prov",
|
||||
model: "old",
|
||||
status: "deprecated",
|
||||
},
|
||||
{
|
||||
...VALID_ENTRY,
|
||||
provider: "prov",
|
||||
model: "bad",
|
||||
status: "revoked",
|
||||
},
|
||||
],
|
||||
})));
|
||||
expect(classifyModel(catalog!, "openai-codex", "gpt-5.6-sol")).toBe("recommended");
|
||||
expect(classifyModel(catalog!, "prov", "old")).toBe("deprecated");
|
||||
expect(classifyModel(catalog!, "prov", "bad")).toBe("revoked");
|
||||
expect(classifyModel(catalog!, "openai-codex", "other-model")).toBe("unlisted");
|
||||
expect(classifyModel(catalog!, "nope", "gpt-5.6-sol")).toBe("unlisted");
|
||||
});
|
||||
|
||||
it("never blocks: classification is pure lookup with no gating semantics", () => {
|
||||
const { catalog } = run(depsWith(null));
|
||||
expect(catalog?.entries).toEqual([]);
|
||||
expect(classifyModel(catalog!, "anything", "anything")).toBe("unlisted");
|
||||
});
|
||||
});
|
||||
|
||||
// -- helpers -------------------------------------------------------------
|
||||
|
||||
function run(deps: CatalogDeps): {
|
||||
catalog: ReturnType<typeof loadModelCatalog>["catalog"];
|
||||
diagnostics: ReturnType<typeof loadModelCatalog>["diagnostics"];
|
||||
} {
|
||||
const result = loadModelCatalog(deps);
|
||||
return { catalog: result.catalog, diagnostics: result.diagnostics };
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { parseArgs } from "../tools/corpus-replay";
|
||||
|
||||
/**
|
||||
* PIEXTENSIO-24 CLI contract: qualification (--strict) is defined over the
|
||||
* full corpus only — a --case subset is observation-only and must never be
|
||||
* able to report qualified. Exit-code mapping itself lives in main()
|
||||
* (0 qualified / 1 harness failure / 2 qualification failure) and needs a
|
||||
* live model; the parse contract is what keeps the standard honest.
|
||||
*/
|
||||
|
||||
const BASE = ["node", "corpus-replay.ts", "--provider", "p", "--model", "m"];
|
||||
|
||||
describe("corpus-replay parseArgs", () => {
|
||||
it("accepts --strict with the full corpus", () => {
|
||||
const parsed = parseArgs([...BASE, "--strict"]);
|
||||
expect("error" in parsed && parsed.error).toBeFalsy();
|
||||
expect("strict" in parsed && parsed.strict).toBe(true);
|
||||
expect("cases" in parsed && parsed.cases).toBeNull();
|
||||
});
|
||||
|
||||
it("rejects --strict together with --case", () => {
|
||||
const parsed = parseArgs([...BASE, "--strict", "--case", "a,b"]);
|
||||
expect("error" in parsed && /full corpus/.test(parsed.error)).toBe(true);
|
||||
});
|
||||
|
||||
it("keeps --case usable without --strict", () => {
|
||||
const parsed = parseArgs([...BASE, "--case", "a,b"]);
|
||||
expect("error" in parsed && parsed.error).toBeFalsy();
|
||||
expect("cases" in parsed && parsed.cases).toEqual(new Set(["a", "b"]));
|
||||
expect("strict" in parsed && parsed.strict).toBe(false);
|
||||
});
|
||||
|
||||
it("rejects unknown options and missing values", () => {
|
||||
expect(
|
||||
"error" in parseArgs([...BASE, "--thinking", "high"]),
|
||||
).toBe(true);
|
||||
expect("error" in parseArgs([...BASE, "--out"])).toBe(true);
|
||||
expect("error" in parseArgs(["node", "corpus-replay.ts"])).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -32,6 +32,25 @@ vi.mock("@earendil-works/pi-coding-agent", async (importOriginal) => {
|
||||
getAgentDir: () => mockAgentDir.dir || "/nonexistent-ai-judge-test",
|
||||
};
|
||||
});
|
||||
// Catalog seam (PIEXTENSIO-24): index.ts reads the advisory catalog once
|
||||
// per session; tests inject entries through this hoisted holder while
|
||||
// keeping the real classifyModel (pure lookup).
|
||||
const { mockCatalog } = vi.hoisted(() => ({
|
||||
mockCatalog: {
|
||||
entries: [] as Array<Record<string, unknown>>,
|
||||
diagnostics: [] as Array<Record<string, unknown>>,
|
||||
},
|
||||
}));
|
||||
vi.mock("../src/catalog", async (importOriginal) => {
|
||||
const actual = await importOriginal<typeof import("../src/catalog")>();
|
||||
return {
|
||||
...actual,
|
||||
loadModelCatalog: () => ({
|
||||
catalog: { version: 1, entries: mockCatalog.entries },
|
||||
diagnostics: mockCatalog.diagnostics,
|
||||
}),
|
||||
};
|
||||
});
|
||||
import type { AssistantMessage, Context, Model } from "@earendil-works/pi-ai";
|
||||
import type {
|
||||
ExtensionAPI,
|
||||
@@ -50,6 +69,7 @@ import {
|
||||
unpublishPermissionsService,
|
||||
} from "@gotgenes/pi-permission-system";
|
||||
import extension from "../src/index";
|
||||
import type { ModelCatalogEntry } from "../src/catalog";
|
||||
import { PROMPT_VERSION, TOOL_SCHEMA_VERSION } from "../src/prompt";
|
||||
|
||||
function createFakePi(): {
|
||||
@@ -191,6 +211,8 @@ afterEach(() => {
|
||||
publishedService = undefined;
|
||||
}
|
||||
cleanupMockAgentDir();
|
||||
mockCatalog.entries = [];
|
||||
mockCatalog.diagnostics = [];
|
||||
});
|
||||
|
||||
describe("AI judge lifecycle", () => {
|
||||
@@ -875,4 +897,72 @@ describe("AI judge Enforce authority seam (PIEXTENSIO-23, ADR 0008)", () => {
|
||||
notify.mock.calls.filter((call) => /Enforce/i.test(String(call[0]))),
|
||||
).toHaveLength(0);
|
||||
});
|
||||
|
||||
it("appends an untested-model note to the enforce notice for an out-of-catalog model", async () => {
|
||||
const { notify } = await runAsk({
|
||||
config: { version: 2, mode: "enforce" },
|
||||
});
|
||||
const enforceNotice = notify.mock.calls
|
||||
.map((call) => String(call[0]))
|
||||
.find((message) => /Enforce/i.test(message));
|
||||
expect(enforceNotice).toBeDefined();
|
||||
expect(enforceNotice).toMatch(/untested/i);
|
||||
expect(enforceNotice).toMatch(/advisory catalog/i);
|
||||
});
|
||||
|
||||
it("adds catalog-status notes for deprecated and revoked enforce models", async () => {
|
||||
for (const status of ["deprecated", "revoked"] as const) {
|
||||
mockCatalog.entries = [
|
||||
{
|
||||
provider: "session-provider",
|
||||
model: "session-model",
|
||||
api: "openai-codex-responses",
|
||||
status,
|
||||
promptVersion: "v",
|
||||
corpusVersion: "v",
|
||||
testedAt: "2026-01-01T00:00:00Z",
|
||||
corpusCases: 21,
|
||||
matched: 21,
|
||||
infrastructureFailures: 0,
|
||||
latencyMs: { p50: 1, p95: 2, max: 3 },
|
||||
reportPath: "reports/x.json",
|
||||
} satisfies ModelCatalogEntry,
|
||||
];
|
||||
const { notify } = await runAsk({
|
||||
config: { version: 2, mode: "enforce" },
|
||||
});
|
||||
const enforceNotice = notify.mock.calls
|
||||
.map((call) => String(call[0]))
|
||||
.find((message) => /Enforce/i.test(message));
|
||||
expect(enforceNotice).toMatch(new RegExp(`catalog status: ${status}`, "i"));
|
||||
}
|
||||
});
|
||||
|
||||
it("omits the untested note when the enforce model is catalog-recommended", async () => {
|
||||
mockCatalog.entries = [
|
||||
{
|
||||
provider: "session-provider",
|
||||
model: "session-model",
|
||||
api: "openai-codex-responses",
|
||||
status: "recommended",
|
||||
promptVersion: "v",
|
||||
corpusVersion: "v",
|
||||
testedAt: "2026-01-01T00:00:00Z",
|
||||
corpusCases: 21,
|
||||
matched: 21,
|
||||
infrastructureFailures: 0,
|
||||
latencyMs: { p50: 1, p95: 2, max: 3 },
|
||||
reportPath: "reports/x.json",
|
||||
} satisfies ModelCatalogEntry,
|
||||
];
|
||||
const { notify } = await runAsk({
|
||||
config: { version: 2, mode: "enforce" },
|
||||
});
|
||||
const enforceNotice = notify.mock.calls
|
||||
.map((call) => String(call[0]))
|
||||
.find((message) => /Enforce/i.test(message));
|
||||
expect(enforceNotice).toBeDefined();
|
||||
expect(enforceNotice).not.toMatch(/untested/i);
|
||||
expect(enforceNotice).not.toMatch(/catalog status/i);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { qualifyReplay, type ReplayRow } from "../tools/replay-qualify";
|
||||
|
||||
/**
|
||||
* PIEXTENSIO-24: the light qualification standard behind `--strict`.
|
||||
* Every expectation below is an independently worked literal — the
|
||||
* corpus-replay harness must be able to fail hard on quality runs while
|
||||
* keeping its historical "exit 0 keeps unfavorable rows" behavior for
|
||||
* observation-only runs.
|
||||
*/
|
||||
|
||||
function judgmentRow(
|
||||
id: string,
|
||||
verdict: "allow" | "deny" | "defer",
|
||||
expected: "allow" | "deny" | "defer",
|
||||
latencyMs: number,
|
||||
): ReplayRow {
|
||||
return {
|
||||
case: id,
|
||||
expected,
|
||||
verdict,
|
||||
match: verdict === expected,
|
||||
latencyMs,
|
||||
};
|
||||
}
|
||||
|
||||
function infraRow(id: string): ReplayRow {
|
||||
return {
|
||||
case: id,
|
||||
expected: "allow",
|
||||
verdict: null,
|
||||
resultKind: "timeout",
|
||||
match: false,
|
||||
latencyMs: null,
|
||||
};
|
||||
}
|
||||
|
||||
describe("qualifyReplay", () => {
|
||||
it("qualifies a fully matched replay with latency within budget", () => {
|
||||
const rows = [
|
||||
judgmentRow("a", "allow", "allow", 100),
|
||||
judgmentRow("b", "deny", "deny", 200),
|
||||
judgmentRow("c", "defer", "defer", 300),
|
||||
judgmentRow("d", "allow", "allow", 400),
|
||||
];
|
||||
const result = qualifyReplay(rows, { budgetMs: 30_000 });
|
||||
expect(result.qualified).toBe(true);
|
||||
expect(result.reasons).toEqual([]);
|
||||
expect(result.matched).toBe(4);
|
||||
expect(result.mismatches).toEqual([]);
|
||||
expect(result.infrastructureFailures).toEqual([]);
|
||||
// Sorted latencies [100, 200, 300, 400]: lower-median p50 = 200,
|
||||
// nearest-rank p95 = 300, max = 400.
|
||||
expect(result.latencyMs).toEqual({ p50: 200, p95: 300, max: 400 });
|
||||
});
|
||||
|
||||
it("rejects a mismatched case and names it", () => {
|
||||
const rows = [
|
||||
judgmentRow("a", "allow", "allow", 100),
|
||||
judgmentRow("bad-case", "defer", "deny", 150),
|
||||
];
|
||||
const result = qualifyReplay(rows, { budgetMs: 30_000 });
|
||||
expect(result.qualified).toBe(false);
|
||||
expect(result.mismatches).toEqual(["bad-case"]);
|
||||
expect(result.reasons.join(" ")).toContain("bad-case");
|
||||
expect(result.reasons.join(" ")).toMatch(/mismatch/i);
|
||||
});
|
||||
|
||||
it("rejects any infrastructure failure row", () => {
|
||||
const rows = [
|
||||
judgmentRow("a", "allow", "allow", 100),
|
||||
infraRow("b"),
|
||||
];
|
||||
const result = qualifyReplay(rows, { budgetMs: 30_000 });
|
||||
expect(result.qualified).toBe(false);
|
||||
expect(result.infrastructureFailures).toEqual(["b"]);
|
||||
expect(result.reasons.join(" ")).toMatch(/infrastructure/i);
|
||||
});
|
||||
|
||||
it("rejects a judgment latency beyond budget", () => {
|
||||
const rows = [judgmentRow("slow", "allow", "allow", 30_001)];
|
||||
const result = qualifyReplay(rows, { budgetMs: 30_000 });
|
||||
expect(result.qualified).toBe(false);
|
||||
expect(result.reasons.join(" ")).toMatch(/budget/i);
|
||||
});
|
||||
|
||||
it("recomputes agreement from expected/verdict: a contradictory match flag cannot qualify", () => {
|
||||
// Harness-reported match=true contradicts verdict!==expected, and
|
||||
// the judgment carries no usable latency — both must disqualify.
|
||||
const lying = {
|
||||
case: "lying-row",
|
||||
expected: "deny",
|
||||
verdict: "allow",
|
||||
match: true,
|
||||
latencyMs: null,
|
||||
};
|
||||
const result = qualifyReplay([lying], { budgetMs: 30_000 });
|
||||
expect(result.qualified).toBe(false);
|
||||
expect(result.mismatches).toEqual(["lying-row"]);
|
||||
expect(result.matched).toBe(0);
|
||||
expect(result.reasons.join(" ")).toMatch(/usable latency/);
|
||||
});
|
||||
|
||||
it("disqualifies a matching judgment whose latency is missing", () => {
|
||||
const row = {
|
||||
case: "no-latency",
|
||||
expected: "defer",
|
||||
verdict: "defer",
|
||||
match: true,
|
||||
latencyMs: null,
|
||||
};
|
||||
const result = qualifyReplay([row], { budgetMs: 30_000 });
|
||||
expect(result.qualified).toBe(false);
|
||||
expect(result.mismatches).toEqual([]);
|
||||
expect(result.reasons.join(" ")).toMatch(/usable latency: no-latency/);
|
||||
});
|
||||
|
||||
it("rejects an empty replay", () => {
|
||||
const result = qualifyReplay([], { budgetMs: 30_000 });
|
||||
expect(result.qualified).toBe(false);
|
||||
expect(result.reasons.join(" ")).toMatch(/no rows/i);
|
||||
expect(result.latencyMs).toEqual({ p50: null, p95: null, max: null });
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user