feat(ai-judge): add advisory model catalog and strict corpus replay

- add versioned advisory model catalog shipped with the package and a fail-closed loader
- annotate the enforce session notice for untested, deprecated, and revoked models
- add --strict to corpus-replay with 0/1/2 exit codes and reject strict subset runs
- extract replay qualification into a pure module that recomputes matches and validates latencies
- remove the documented-but-unimplemented --thinking flag and stamp reports with a corpus version
- qualify gpt-5.6-sol as the first recommended entry and archive three real replay reports
- revise the corpus to 2026-08-21.2 changing unclear-forward expected defer to deny
This commit is contained in:
2026-08-21 23:11:40 +08:00
parent 061c60c344
commit 7987fc7a3c
13 changed files with 1544 additions and 39 deletions
@@ -0,0 +1,130 @@
import { describe, expect, it } from "vitest";
import {
classifyModel,
loadModelCatalog,
type CatalogDeps,
} from "../src/catalog";
/**
* PIEXTENSIO-24: the advisory model catalog shipped with the package.
* Advisory only — catalog state affects notifications and docs, never
* Enforce authority (ADR 0008). All expectations are worked literals
* from the schema, independent of any real entry.
*/
const VALID_ENTRY = {
provider: "openai-codex",
model: "gpt-5.6-sol",
api: "openai-codex-responses",
status: "recommended",
promptVersion: "bash-shadow-v4",
corpusVersion: "2026-08-21.1",
testedAt: "2026-08-21T10:00:00Z",
corpusCases: 21,
matched: 21,
infrastructureFailures: 0,
latencyMs: { p50: 3000, p95: 9000, max: 12000 },
reportPath: "reports/corpus-replay-x.json",
};
function depsWith(raw: string | null): CatalogDeps {
return {
catalogPath: "/nonexistent-ai-judge-catalog-test/models-catalog.json",
readFile: (_path: string) => {
if (raw === null) throw new Error("ENOENT");
return raw;
},
};
}
describe("loadModelCatalog", () => {
it("loads a valid versioned catalog with frozen entries", () => {
const { catalog } = run(depsWith(JSON.stringify({
version: 1,
entries: [VALID_ENTRY],
})));
expect(catalog?.version).toBe(1);
expect(catalog?.entries).toHaveLength(1);
expect(catalog?.entries[0]).toMatchObject({
provider: "openai-codex",
model: "gpt-5.6-sol",
status: "recommended",
});
expect(Object.isFrozen(catalog?.entries)).toBe(true);
});
it("degrades to an empty catalog with a diagnostic on an unknown version", () => {
const { catalog, diagnostics } = run(depsWith(JSON.stringify({
version: 99,
entries: [VALID_ENTRY],
})));
expect(catalog?.entries).toEqual([]);
expect(diagnostics.map((d) => d.key)).toContain("version");
});
it("degrades to an empty catalog with a diagnostic on unreadable or malformed files", () => {
const missing = run(depsWith(null));
expect(missing.catalog?.entries).toEqual([]);
expect(missing.diagnostics.map((d) => d.key)).toContain("file");
const malformed = run(depsWith("{not json"));
expect(malformed.catalog?.entries).toEqual([]);
expect(malformed.diagnostics.map((d) => d.key)).toContain("file");
});
it("degrades the whole catalog to empty when any entry is invalid", () => {
const broken = { ...VALID_ENTRY, model: 42 };
const { catalog, diagnostics } = run(depsWith(JSON.stringify({
version: 1,
entries: [broken, VALID_ENTRY],
})));
expect(catalog?.entries).toEqual([]);
expect(diagnostics).toHaveLength(1);
expect(diagnostics[0]?.key).toBe("file");
expect(String(diagnostics[0]?.problem)).toMatch(/whole catalog degraded/i);
});
});
describe("classifyModel", () => {
it("classifies catalog entries by status and unknown models as unlisted", () => {
const { catalog } = run(depsWith(JSON.stringify({
version: 1,
entries: [
VALID_ENTRY,
{
...VALID_ENTRY,
provider: "prov",
model: "old",
status: "deprecated",
},
{
...VALID_ENTRY,
provider: "prov",
model: "bad",
status: "revoked",
},
],
})));
expect(classifyModel(catalog!, "openai-codex", "gpt-5.6-sol")).toBe("recommended");
expect(classifyModel(catalog!, "prov", "old")).toBe("deprecated");
expect(classifyModel(catalog!, "prov", "bad")).toBe("revoked");
expect(classifyModel(catalog!, "openai-codex", "other-model")).toBe("unlisted");
expect(classifyModel(catalog!, "nope", "gpt-5.6-sol")).toBe("unlisted");
});
it("never blocks: classification is pure lookup with no gating semantics", () => {
const { catalog } = run(depsWith(null));
expect(catalog?.entries).toEqual([]);
expect(classifyModel(catalog!, "anything", "anything")).toBe("unlisted");
});
});
// -- helpers -------------------------------------------------------------
function run(deps: CatalogDeps): {
catalog: ReturnType<typeof loadModelCatalog>["catalog"];
diagnostics: ReturnType<typeof loadModelCatalog>["diagnostics"];
} {
const result = loadModelCatalog(deps);
return { catalog: result.catalog, diagnostics: result.diagnostics };
}