feat(ai-judge): capture model per permission request

This commit is contained in:
2026-08-17 20:00:18 +08:00
parent 0546a80497
commit 3f3bbb4c28
2 changed files with 71 additions and 5 deletions
+16 -5
View File
@@ -1,5 +1,7 @@
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
import { getAgentDir } from "@earendil-works/pi-coding-agent";
import type { Model } from "@earendil-works/pi-ai";
import type { ModelRegistry } from "@earendil-works/pi-coding-agent";
import {
getPermissionsService,
PERMISSIONS_READY_CHANNEL,
@@ -20,7 +22,9 @@ const REVIEW_SCHEMA_VERSION = 1;
interface RootSession {
readonly getSessionId: () => string;
readonly expectedSessionId: string;
readonly model: ModelAvailability;
/** Current-model probe: reads the live model at each authorize call. */
readonly getModel: () => Model<any> | undefined;
readonly modelRegistry: ModelRegistry;
readonly shutdown: AbortController;
/** Opaque per-runtime identity for cohort segmentation. */
readonly judgeRuntimeId: string;
@@ -181,10 +185,16 @@ export default function permissionAiJudge(pi: ExtensionAPI): void {
return { kind: "defer" };
}
// `captured.model` is the session-start snapshot. Config and
// model-select support are deliberately outside this slice.
// Per-request model capture (PIEXTENSIO-3 cat.3): an
// in-request switch does not change an in-flight attempt;
// a between-request switch affects the next attempt. The
// probe reads the live current model here, not at start.
const availability = createModelAvailability(
captured.getModel(),
captured.modelRegistry,
);
const result = await requestStructuredVerdict(
captured.model,
availability,
evidence,
captured.shutdown.signal,
captured.config.timeoutMs,
@@ -266,7 +276,8 @@ export default function permissionAiJudge(pi: ExtensionAPI): void {
root = {
getSessionId: () => ctx.sessionManager.getSessionId(),
expectedSessionId: sessionId,
model: createModelAvailability(ctx.model, ctx.modelRegistry),
getModel: () => ctx.model,
modelRegistry: ctx.modelRegistry,
shutdown: new AbortController(),
judgeRuntimeId: crypto.randomUUID(),
config: loadJudgeConfig({ agentDir: getAgentDir() }),
@@ -145,6 +145,61 @@ afterEach(() => {
});
describe("AI judge lifecycle", () => {
it("captures the model per request: a between-request switch changes the next call", async () => {
let authorize: Authorizer["authorize"] | undefined;
const service = {
registerAuthorizer: vi.fn((_name, callback) => {
authorize = callback;
return vi.fn();
}),
checkPermission: vi.fn(),
getToolPermission: vi.fn(),
} as unknown as PermissionsService;
publishPermissionsService(service);
publishedService = service;
// A mutable "current model" the session switches mid-run.
let currentModel = {
id: "model-a",
provider: "test-provider",
api: "openai-codex-responses",
} as Model<any>;
const seen: string[] = [];
const complete = vi.fn(async (model: Model<any>) => {
seen.push(model.id);
return modelResponse();
});
const ctx = {
hasUI: true,
sessionManager: { getSessionId: () => "session-root" },
get model() {
return currentModel;
},
modelRegistry: { complete },
ui: { notify: vi.fn() },
} as unknown as ExtensionContext;
const harness = createFakePi();
extension(harness.pi);
harness.start(ctx);
harness.ready();
const log = {
review: vi.fn(),
debug: vi.fn(),
};
await authorize!(ask(), { checkPermission: vi.fn(), getToolPermission: vi.fn() }, log);
currentModel = {
id: "model-b",
provider: "test-provider",
api: "openai-codex-responses",
} as Model<any>;
await authorize!(ask(), { checkPermission: vi.fn(), getToolPermission: vi.fn() }, log);
expect(seen).toEqual(["model-a", "model-b"]);
harness.shutdown();
});
it("calls the current model once, records metadata, and still defers in Shadow", async () => {
let authorize: Authorizer["authorize"] | undefined;
const dispose = vi.fn();