feat(ai-judge): rework enforce mode as user-assumed-risk contract

- add config v2 with fixed judge model selection and fail-closed v1 enforce migration to shadow
- add built-in high-risk override for irreversible, publish, system, and credential shapes that always defers to the human
- remove the promotion gate, promotion records tool, and their tests
- fail closed with judge_model_unavailable when a configured judge model cannot be resolved
- notify once per session in enforce mode with the judge model and risk contract
- add ADR 0008 and update README and CONTEXT
This commit is contained in:
2026-08-21 23:11:40 +08:00
parent f4ba739878
commit 061c60c344
17 changed files with 1372 additions and 1137 deletions
@@ -20,9 +20,11 @@ describe("loadJudgeConfig — missing and malformed", () => {
it("resolves a missing file to all defaults with one diagnostic", () => {
const config = loadJudgeConfig(deps());
expect(config).toEqual({
configVersion: 1,
mode: "shadow",
timeoutMs: 15_000,
timeoutCohort: "default",
judgeModel: undefined,
diagnostics: [
expect.objectContaining({ key: "file", fallback: "all defaults" }),
],
@@ -45,19 +47,76 @@ describe("loadJudgeConfig — missing and malformed", () => {
});
});
describe("loadJudgeConfig — version selection", () => {
it("treats a missing version as v1", () => {
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: '{"mode":"shadow"}' }));
expect(config.configVersion).toBe(1);
});
it("accepts version 1 explicitly", () => {
const config = loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"version":1,"mode":"shadow"}' }),
);
expect(config.configVersion).toBe(1);
expect(config.diagnostics).toEqual([]);
});
it("accepts version 2", () => {
const config = loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"version":2,"mode":"shadow"}' }),
);
expect(config.configVersion).toBe(2);
expect(config.diagnostics).toEqual([]);
});
it("rejects an unknown version to all defaults with a diagnostic", () => {
const config = loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"version":3,"mode":"enforce"}' }),
);
expect(config).toMatchObject({ configVersion: 1, mode: "shadow" });
expect(config.diagnostics[0]?.key).toBe("version");
});
});
describe("loadJudgeConfig — v1 enforce migration (fail-closed)", () => {
it("downgrades v1 enforce to shadow with a migration diagnostic", () => {
for (const raw of ['{"mode":"enforce"}', '{"version":1,"mode":"enforce"}']) {
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: raw }));
expect(config.mode).toBe("shadow");
expect(config.diagnostics).toEqual([
{
key: "mode",
problem: expect.stringMatching(/"version": 2/),
fallback: "shadow (v1 enforce requires explicit migration)",
},
]);
}
});
it("keeps v1 shadow without diagnostics", () => {
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: '{"mode":"shadow"}' }));
expect(config.mode).toBe("shadow");
expect(config.diagnostics).toEqual([]);
});
});
describe("loadJudgeConfig — mode", () => {
it("accepts shadow and enforce", () => {
it("accepts shadow and enforce in v2", () => {
expect(
loadJudgeConfig(deps({ [CONFIG_PATH]: '{"mode":"shadow"}' })).mode,
loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"version":2,"mode":"shadow"}' }),
).mode,
).toBe("shadow");
expect(
loadJudgeConfig(deps({ [CONFIG_PATH]: '{"mode":"enforce"}' })).mode,
loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"version":2,"mode":"enforce"}' }),
).mode,
).toBe("enforce");
});
it("resolves an unknown mode to shadow with a diagnostic", () => {
const config = loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"mode":"yolo"}' }),
deps({ [CONFIG_PATH]: '{"version":2,"mode":"yolo"}' }),
);
expect(config.mode).toBe("shadow");
expect(config.diagnostics).toEqual([
@@ -70,18 +129,72 @@ describe("loadJudgeConfig — mode", () => {
});
it("resolves a missing mode to shadow without diagnostics", () => {
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: "{}" }));
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: '{"version":2}' }));
expect(config.mode).toBe("shadow");
expect(config.diagnostics).toEqual([]);
});
});
describe("loadJudgeConfig — v2 judge model", () => {
it("parses an explicit fixed judge model", () => {
const config = loadJudgeConfig(
deps({
[CONFIG_PATH]:
'{"version":2,"mode":"enforce","model":{"provider":"openai-codex","id":"gpt-5.6-sol"}}',
}),
);
expect(config.mode).toBe("enforce");
expect(config.judgeModel).toEqual({
provider: "openai-codex",
id: "gpt-5.6-sol",
});
expect(config.diagnostics).toEqual([]);
});
it("allows a judge model in shadow mode too", () => {
const config = loadJudgeConfig(
deps({
[CONFIG_PATH]:
'{"version":2,"mode":"shadow","model":{"provider":"p","id":"m"}}',
}),
);
expect(config.mode).toBe("shadow");
expect(config.judgeModel).toEqual({ provider: "p", id: "m" });
});
it.each([
'{"version":2,"model":"openai-codex"}',
'{"version":2,"model":{"provider":"p"}}',
'{"version":2,"model":{"id":"m"}}',
'{"version":2,"model":{"provider":"","id":"m"}}',
'{"version":2,"model":{"provider":"p","id":""}}',
'{"version":2,"model":{"provider":1,"id":"m"}}',
'{"version":2,"model":[]}',
])("fails a malformed model %j closed to shadow with a diagnostic", (raw) => {
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: raw }));
expect(config.mode).toBe("shadow");
expect(config.judgeModel).toBeUndefined();
expect(config.diagnostics[0]?.key).toBe("model");
});
it("ignores a model field in v1 with a diagnostic", () => {
const config = loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"model":{"provider":"p","id":"m"}}' }),
);
expect(config.configVersion).toBe(1);
expect(config.judgeModel).toBeUndefined();
expect(config.diagnostics[0]?.key).toBe("model");
});
});
describe("loadJudgeConfig — timeout boundaries", () => {
it.each([4_999, 30_001, 0, -5_000, 15.5, NaN, Infinity, "20000"])(
"rejects invalid timeoutMs %p with fallback to 15,000",
(value) => {
const config = loadJudgeConfig(
deps({ [CONFIG_PATH]: JSON.stringify({ timeoutMs: value }) }),
deps({
[CONFIG_PATH]: JSON.stringify({ version: 2, timeoutMs: value }),
}),
);
expect(config.timeoutMs).toBe(15_000);
expect(config.timeoutCohort).toBe("default");
@@ -91,16 +204,20 @@ describe("loadJudgeConfig — timeout boundaries", () => {
it("accepts the inclusive boundaries 5,000 and 30,000", () => {
expect(
loadJudgeConfig(deps({ [CONFIG_PATH]: '{"timeoutMs":5000}' })),
loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"version":2,"timeoutMs":5000}' }),
),
).toMatchObject({ timeoutMs: 5_000, timeoutCohort: 5_000 });
expect(
loadJudgeConfig(deps({ [CONFIG_PATH]: '{"timeoutMs":30000}' })),
loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"version":2,"timeoutMs":30000}' }),
),
).toMatchObject({ timeoutMs: 30_000, timeoutCohort: 30_000 });
});
it("marks an explicit default timeout as the default cohort", () => {
const config = loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"timeoutMs":15000}' }),
deps({ [CONFIG_PATH]: '{"version":2,"timeoutMs":15000}' }),
);
expect(config.timeoutCohort).toBe("default");
expect(config.diagnostics).toEqual([]);
@@ -108,7 +225,7 @@ describe("loadJudgeConfig — timeout boundaries", () => {
it("marks a non-default timeout as a distinct cohort", () => {
const config = loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"timeoutMs":30000}' }),
deps({ [CONFIG_PATH]: '{"version":2,"timeoutMs":30000}' }),
);
expect(config.timeoutCohort).toBe(30_000);
});
@@ -117,13 +234,18 @@ describe("loadJudgeConfig — timeout boundaries", () => {
describe("loadJudgeConfig — snapshot immutability", () => {
it("returns an immutable effective-config snapshot", () => {
const config = loadJudgeConfig(
deps({ [CONFIG_PATH]: '{"mode":"enforce","timeoutMs":20000}' }),
deps({
[CONFIG_PATH]:
'{"version":2,"mode":"enforce","timeoutMs":20000,"model":{"provider":"p","id":"m"}}',
}),
);
expect(Object.isFrozen(config)).toBe(true);
expect(config).toEqual({
configVersion: 2,
mode: "enforce",
timeoutMs: 20_000,
timeoutCohort: 20_000,
judgeModel: { provider: "p", id: "m" },
diagnostics: [],
});
});
@@ -0,0 +1,208 @@
import { describe, expect, it } from "vitest";
import { classifyHighRisk } from "../src/highrisk";
describe("classifyHighRisk — data_loss", () => {
it.each([
"git clean -xfd",
"git clean -fdx",
"git clean -fxd",
"git clean -f -x -d",
"git clean --force -xd",
"git reset --hard",
"git reset --hard HEAD~3",
"git reset --hard origin/main",
"git checkout -- .",
"git checkout .",
"git restore .",
"git restore -- .",
"rm -rf ~",
"rm -rf /",
"rm -rf ~/*",
"rm -rf $HOME",
"rm -rf $HOME/projects",
"rm -fr ~",
])("flags %j", (command) => {
expect(classifyHighRisk(command)?.category).toBe("data_loss");
});
it.each([
"git clean -nxd",
"git clean -nd",
"git reset --soft HEAD~1",
"git checkout main",
"git checkout -- file.txt",
"git restore file.txt",
"rm -rf build/",
"rm -rf ./dist",
"rm -r build",
"rm file.txt",
"echo git clean -xfd",
"git status",
])("does not flag %j", (command) => {
expect(classifyHighRisk(command)).toBeUndefined();
});
});
describe("classifyHighRisk — history_rewrite", () => {
it.each([
"git push --force",
"git push -f",
"git push --force origin main",
"git push --force-with-lease origin main",
"git push origin +main",
])("flags %j", (command) => {
expect(classifyHighRisk(command)?.category).toBe("history_rewrite");
});
it.each(["git push origin main", "git push", "git push --tags"])(
"does not flag %j",
(command) => {
expect(classifyHighRisk(command)).toBeUndefined();
},
);
});
describe("classifyHighRisk — publish_deploy", () => {
it.each([
"npm publish",
"npm publish --access public",
"pnpm publish",
"yarn publish",
"cargo publish",
"terraform destroy",
"terraform destroy -auto-approve",
])("flags %j", (command) => {
expect(classifyHighRisk(command)?.category).toBe("publish_deploy");
});
it.each(["npm run publish", "npm install", "pnpm test", "terraform plan", "terraform apply"])(
"does not flag %j",
(command) => {
expect(classifyHighRisk(command)).toBeUndefined();
},
);
});
describe("classifyHighRisk — system_modify", () => {
it.each([
"sudo rm /etc/hosts",
"sudo -i",
"sudo apt-get install ripgrep",
"mkfs.ext4 /dev/sda1",
"mkfs /dev/sdb",
"dd if=/dev/zero of=/dev/sda bs=1M",
"shutdown now",
"shutdown -h now",
"reboot",
"halt",
"poweroff",
])("flags %j", (command) => {
expect(classifyHighRisk(command)?.category).toBe("system_modify");
});
it.each([
"dd if=/dev/zero of=/tmp/img bs=1M count=10",
"echo sudo",
"cat /etc/hosts",
])("does not flag %j", (command) => {
expect(classifyHighRisk(command)).toBeUndefined();
});
});
describe("classifyHighRisk — credential_access", () => {
it.each([
"cat ~/.ssh/id_rsa",
"cat ~/.ssh/id_ed25519",
"less ~/.ssh/id_ed25519",
"cat ~/.aws/credentials",
"head ~/.netrc",
"cat ~/.config/gcloud/application_default_credentials.json",
"rm ~/.ssh/id_ed25519",
"mv ~/.ssh/authorized_keys /tmp",
"cp ~/.gnupg/secring.gpg .",
"cat $HOME/.ssh/id_rsa",
"cat $HOME/.aws/credentials",
])("flags %j", (command) => {
expect(classifyHighRisk(command)?.category).toBe("credential_access");
});
it.each([
"echo x > ~/.aws/credentials",
"echo x >> ~/.ssh/authorized_keys",
"ssh-keygen foo 2> ~/.aws/credentials",
"printf '%s' key > ~/.ssh/id_ed25519",
"curl -s url | tee ~/.netrc",
])("flags credential replacement %j", (command) => {
expect(classifyHighRisk(command)).toMatchObject({
category: "credential_access",
});
});
it.each([
"cat ~/.ssh/config",
"cat ~/.ssh/known_hosts",
"cat package.json",
"ls ~/.ssh",
"cat ~/.bashrc",
"cat id_rsa",
"echo x > ~/.bashrc",
"curl -s url | tee notes.txt",
])("does not flag %j", (command) => {
expect(classifyHighRisk(command)).toBeUndefined();
});
});
describe("classifyHighRisk — compound inputs", () => {
it("flags when any unit matches", () => {
expect(classifyHighRisk("pnpm test && git clean -xfd")?.category).toBe(
"data_loss",
);
expect(classifyHighRisk("git push --force; echo done")?.category).toBe(
"history_rewrite",
);
expect(classifyHighRisk("npm publish | tee log")?.category).toBe(
"publish_deploy",
);
});
it("flags single-`&` background compound units", () => {
expect(classifyHighRisk("echo ready & npm publish")?.category).toBe(
"publish_deploy",
);
expect(classifyHighRisk("sleep 5 & rm -rf ~")?.category).toBe(
"data_loss",
);
});
it("does not flag separators inside quotes or escapes", () => {
expect(classifyHighRisk("printf 'x; npm publish;'")).toBeUndefined();
expect(classifyHighRisk("echo \"git push --force\"")).toBeUndefined();
expect(classifyHighRisk("echo git\\;npm\\;publish")).toBeUndefined();
expect(classifyHighRisk("echo 'rm -rf ~'")).toBeUndefined();
});
it("treats a quoted command argument as one token, not a command", () => {
// git commit -m "junk; sudo rm /" quotes an argument, not a unit.
expect(classifyHighRisk("git commit -m 'do not npm publish'")).toBeUndefined();
});
it("flags high-risk units even when another unit quotes text", () => {
expect(classifyHighRisk("echo 'a;b' && git push --force")).toMatchObject({
category: "history_rewrite",
});
});
it("returns undefined for all-benign compounds", () => {
expect(classifyHighRisk("pnpm test && pnpm check")).toBeUndefined();
expect(classifyHighRisk("echo a; echo b | wc -l")).toBeUndefined();
expect(classifyHighRisk("sleep 5 & echo done")).toBeUndefined();
});
it("reports the matching rule for audit observability", () => {
const match = classifyHighRisk("git clean -xfd");
expect(match).toMatchObject({
category: "data_loss",
rule: expect.stringMatching(/^git_clean/),
});
});
});
@@ -3,18 +3,10 @@ import {
evaluateEnforceAuthority,
type EnforceGateState,
} from "../src/judge";
import {
loadPromotionRecords,
resolvePromotionGates,
type CandidateIdentity,
} from "../src/promotion";
const ALL_OPEN: EnforceGateState = {
auditHealthy: true,
telemetryHealth: "healthy",
cohortQualified: true,
ownerApprovalRecorded: true,
activationRecorded: true,
resultKind: "judgment",
verdict: "allow",
reviewAcknowledged: true,
@@ -23,7 +15,10 @@ const ALL_OPEN: EnforceGateState = {
};
describe("evaluateEnforceAuthority — every gate independently forces defer", () => {
it("allows only when every gate holds", () => {
it("allows when every gate holds — no promotion records required", () => {
// ADR 0008: the promotion gates (cohort qualification, owner
// approval, activation) are no longer authority inputs; a session
// with zero promotion records can hold Enforce authority.
expect(evaluateEnforceAuthority(ALL_OPEN)).toEqual({ kind: "allow" });
});
@@ -37,9 +32,6 @@ describe("evaluateEnforceAuthority — every gate independently forces defer", (
{ name: "telemetry disabled", patch: { telemetryHealth: "disabled" }, expectedReason: "telemetry_disabled" },
{ name: "telemetry write failed", patch: { telemetryHealth: "write_failed" }, expectedReason: "telemetry_write_failed" },
{ name: "telemetry integrity anomaly", patch: { telemetryHealth: "integrity_anomaly" }, expectedReason: "telemetry_integrity_anomaly" },
{ name: "cohort not qualified", patch: { cohortQualified: false }, expectedReason: "cohort_not_qualified" },
{ name: "owner approval absent", patch: { ownerApprovalRecorded: false }, expectedReason: "owner_approval_absent" },
{ name: "activation absent", patch: { activationRecorded: false }, expectedReason: "activation_absent" },
{ name: "preflight result", patch: { resultKind: "preflight_defer" }, expectedReason: "result_preflight_defer" },
{ name: "infrastructure result", patch: { resultKind: "infrastructure_failure" }, expectedReason: "result_infrastructure_failure" },
{ name: "semantic deny verdict", patch: { verdict: "deny" }, expectedReason: "verdict_deny" },
@@ -58,86 +50,3 @@ describe("evaluateEnforceAuthority — every gate independently forces defer", (
});
}
});
describe("evaluateEnforceAuthority — production state with no promotion records", () => {
// The real post-PIEXTENSIO-21 seam: an empty records file (the normal
// pre-promotion state) closes all promotion gates, so every mode and
// telemetry state defers — mechanically identical to v0.1's hardcoded
// closure, now derived from actual storage.
const emptySnapshot = loadPromotionRecords({
agentDir: "/nonexistent-agent-dir",
});
const identity: CandidateIdentity = {
judge: "@sikongjueluo/pi-permission-ai-judge@0.0.1",
permissionSystem: "25.4.0",
provider: "openai-codex",
model: "gpt-5.6-sol",
api: "openai-codex-responses",
promptVersion: "bash-shadow-v4",
toolSchemaVersion: "report-verdict-v1",
reviewSchemaVersion: "1",
timeoutCohort: 30000,
};
it("never grants authority for any mode or telemetry state without records", () => {
const modes = ["shadow", "enforce"] as const;
const healths = ["healthy", "disabled", "write_failed", "integrity_anomaly"] as const;
for (const mode of modes) {
for (const health of healths) {
const gates = resolvePromotionGates(emptySnapshot, identity);
const outcome = evaluateEnforceAuthority({
...ALL_OPEN,
mode,
telemetryHealth: health,
...gates,
});
expect(outcome.kind).toBe("defer");
}
}
});
it("blocks enforce on the cohort gate first even with healthy audit", () => {
const gates = resolvePromotionGates(emptySnapshot, identity);
const outcome = evaluateEnforceAuthority({
...ALL_OPEN,
...gates,
});
expect(outcome).toEqual({
kind: "defer",
blockedBy: "cohort_not_qualified",
});
});
it("grants authority only when every record kind exists for the exact identity", () => {
const records = [
{
kind: "cohort_qualified",
candidateIdentity: identity,
recordedAt: "2026-08-20T12:00:00Z",
basis: "cohort test",
},
{
kind: "owner_approval",
candidateIdentity: identity,
recordedAt: "2026-08-20T12:01:00Z",
basis: "approved",
},
{
kind: "activation",
candidateIdentity: identity,
recordedAt: "2026-08-20T12:02:00Z",
basis: "activated",
},
] as const;
const snapshot = {
records,
healthy: true,
diagnostic: null,
path: "unused",
};
const gates = resolvePromotionGates(snapshot, identity);
expect(evaluateEnforceAuthority({ ...ALL_OPEN, ...gates })).toEqual({
kind: "allow",
});
});
});
@@ -50,7 +50,6 @@ import {
unpublishPermissionsService,
} from "@gotgenes/pi-permission-system";
import extension from "../src/index";
import { appendPromotionRecord, type CandidateIdentity } from "../src/promotion";
import { PROMPT_VERSION, TOOL_SCHEMA_VERSION } from "../src/prompt";
function createFakePi(): {
@@ -542,33 +541,39 @@ describe("AI judge lifecycle", () => {
});
});
describe("AI judge Enforce authority seam (PIEXTENSIO-21)", () => {
describe("AI judge Enforce authority seam (PIEXTENSIO-23, ADR 0008)", () => {
beforeEach(() => {
createMockAgentDir();
writeFileSync(
join(mockAgentDir.dir, "pi-permission-ai-judge.config.json"),
JSON.stringify({ mode: "enforce" }),
);
});
/** Records identity matching the lifecycle fake model + static fields. */
function lifecycleIdentity(): CandidateIdentity {
return {
judge: "@sikongjueluo/pi-permission-ai-judge@0.0.1",
permissionSystem: "25.4.0",
provider: "test-provider",
model: "test-model",
api: "openai-codex-responses",
promptVersion: PROMPT_VERSION,
toolSchemaVersion: TOOL_SCHEMA_VERSION,
reviewSchemaVersion: "1",
timeoutCohort: "default",
};
function writeConfig(config: Record<string, unknown>): void {
writeFileSync(
join(mockAgentDir.dir, "pi-permission-ai-judge.config.json"),
JSON.stringify(config),
);
}
async function runAsk(): Promise<{
interface RunAskOptions {
/** Config file contents (written before session start). */
config: Record<string, unknown>;
/** Command unit for the ask; full command becomes `pnpm test && <unit>`. */
command?: string;
/** modelRegistry.find result for a configured judge model (when set). */
findResult?: Model<any> | undefined;
/** Whether the found model has configured auth. */
authConfigured?: boolean;
/** Overridable model verdict. */
response?: AssistantMessage;
}
async function runAsk(
options: RunAskOptions,
): Promise<{
verdict: { kind: string };
reviews: Array<{ event: string; details?: Record<string, unknown> }>;
notify: ReturnType<typeof vi.fn>;
complete: ReturnType<typeof vi.fn>;
find: ReturnType<typeof vi.fn>;
}> {
let authorize: Authorizer["authorize"] | undefined;
const service = {
@@ -582,28 +587,45 @@ describe("AI judge Enforce authority seam (PIEXTENSIO-21)", () => {
publishPermissionsService(service);
publishedService = service;
const complete = vi.fn(async () => modelResponse());
const complete = vi.fn(async () => options.response ?? modelResponse());
const find = vi.fn(() => options.findResult);
const hasConfiguredAuth = vi.fn(() => options.authConfigured !== false);
const notify = vi.fn();
const sessionManager = fakeSessionManager();
const ctx = {
hasUI: true,
sessionManager: fakeSessionManager(),
sessionManager,
model: {
id: "test-model",
provider: "test-provider",
id: "session-model",
provider: "session-provider",
api: "openai-codex-responses",
} as Model<any>,
modelRegistry: { complete },
ui: { notify: vi.fn() },
modelRegistry: { complete, find, hasConfiguredAuth },
ui: { notify },
} as unknown as ExtensionContext;
const harness = createFakePi();
extension(harness.pi);
writeConfig(options.config);
harness.start(ctx);
harness.ready();
expect(authorize).toBeDefined();
const unit = options.command ?? "git push origin main";
const askDetails = ask();
askDetails.command = unit;
const request = askDetails.payload.request as { value: string };
request.value = unit;
const evidenceEntry = (
askDetails.payload as { evidence: ReadonlyArray<{ label: string; text: string }> }
).evidence[0];
if (evidenceEntry !== undefined) {
evidenceEntry.text = `pnpm test && ${unit}`;
}
const reviews: Array<{ event: string; details?: Record<string, unknown> }> = [];
const verdict = await authorize!(
ask(),
askDetails,
{
checkPermission: vi.fn(),
getToolPermission: vi.fn(),
@@ -614,45 +636,13 @@ describe("AI judge Enforce authority seam (PIEXTENSIO-21)", () => {
},
);
harness.shutdown();
return { verdict, reviews };
return { verdict, reviews, notify, complete, find };
}
it("defers in enforce mode when no promotion records exist", async () => {
const { verdict, reviews } = await runAsk();
expect(verdict).toEqual({ kind: "defer" });
expect(reviews).toMatchObject([
{
event: "ai_bash_judge.result",
details: expect.objectContaining({
mode: "enforce",
verdict: "allow",
effectiveVerdict: "defer",
authorityBlockedBy: "cohort_not_qualified",
}),
},
]);
});
it("grants authority in enforce mode only with all three exact-identity records", async () => {
const identity = lifecycleIdentity();
for (const [kind, basis] of [
["cohort_qualified", "cohort piextensio-test"],
["owner_approval", "approved for test"],
["activation", "activated for test"],
] as const) {
expect(
appendPromotionRecord({
agentDir: mockAgentDir.dir,
record: {
kind,
candidateIdentity: identity,
recordedAt: "2026-08-21T10:00:00Z",
basis,
},
}),
).toBeNull();
}
const { verdict, reviews } = await runAsk();
it("grants authority in v2 enforce mode with no promotion records", async () => {
const { verdict, reviews } = await runAsk({
config: { version: 2, mode: "enforce" },
});
expect(verdict).toEqual({ kind: "allow" });
expect(reviews).toMatchObject([
{
@@ -662,73 +652,227 @@ describe("AI judge Enforce authority seam (PIEXTENSIO-21)", () => {
verdict: "allow",
effectiveVerdict: "allow",
authorityBlockedBy: null,
modelSource: "session",
}),
},
]);
});
it("defers in enforce mode when records exist for another identity", async () => {
const identity = { ...lifecycleIdentity(), model: "other-model" };
for (const kind of [
"cohort_qualified",
"owner_approval",
"activation",
] as const) {
appendPromotionRecord({
agentDir: mockAgentDir.dir,
record: {
kind,
candidateIdentity: identity,
recordedAt: "2026-08-21T10:00:00Z",
basis: "other identity",
},
});
}
const { verdict, reviews } = await runAsk();
expect(verdict).toEqual({ kind: "defer" });
expect(reviews).toMatchObject([
{
event: "ai_bash_judge.result",
details: expect.objectContaining({
effectiveVerdict: "defer",
authorityBlockedBy: "cohort_not_qualified",
}),
},
]);
});
it("never grants authority in shadow mode regardless of records", async () => {
writeFileSync(
join(mockAgentDir.dir, "pi-permission-ai-judge.config.json"),
JSON.stringify({ mode: "shadow" }),
);
const identity = lifecycleIdentity();
for (const kind of [
"cohort_qualified",
"owner_approval",
"activation",
] as const) {
appendPromotionRecord({
agentDir: mockAgentDir.dir,
record: {
kind,
candidateIdentity: identity,
recordedAt: "2026-08-21T10:00:00Z",
basis: "shadow still defers",
},
});
}
const { verdict, reviews } = await runAsk();
it("downgrades v1 enforce to shadow with a migration diagnostic notification", async () => {
const { verdict, reviews, notify, complete } = await runAsk({
config: { mode: "enforce" },
});
expect(verdict).toEqual({ kind: "defer" });
expect(reviews).toMatchObject([
{
event: "ai_bash_judge.result",
details: expect.objectContaining({
mode: "shadow",
verdict: "allow",
effectiveVerdict: "defer",
authorityBlockedBy: "mode_shadow",
}),
},
]);
const notified = notify.mock.calls.map((call) => String(call[0]));
expect(
notified.some((message) =>
/v1 enforce requires explicit migration|version 2/.test(message),
),
).toBe(true);
expect(complete).toHaveBeenCalledTimes(1);
});
it("defers immediately on a high-risk command in enforce mode without calling the model", async () => {
const { verdict, reviews, complete } = await runAsk({
config: { version: 2, mode: "enforce" },
command: "git clean -xfd",
});
expect(verdict).toEqual({ kind: "defer" });
expect(complete).not.toHaveBeenCalled();
expect(reviews).toMatchObject([
{
event: "ai_bash_judge.result",
details: expect.objectContaining({
resultKind: "preflight_defer",
verdict: null,
effectiveVerdict: "defer",
modelCalled: false,
code: "high_risk_override",
riskCategory: "data_loss",
riskRule: expect.any(String),
}),
},
]);
});
it("still calls the model on a high-risk command in shadow mode, records the override, and defers", async () => {
const { verdict, reviews, complete } = await runAsk({
config: { version: 2, mode: "shadow" },
command: "git clean -xfd",
});
expect(complete).toHaveBeenCalledTimes(1);
expect(verdict).toEqual({ kind: "defer" });
expect(reviews).toMatchObject([
{
event: "ai_bash_judge.result",
details: expect.objectContaining({
resultKind: "judgment",
verdict: "allow",
effectiveVerdict: "defer",
riskOverride: { category: "data_loss", rule: expect.any(String) },
}),
},
]);
});
it("resolves a configured v2 judge model instead of the session model", async () => {
const { verdict, reviews, complete, find } = await runAsk({
config: {
version: 2,
mode: "enforce",
model: { provider: "fixed-provider", id: "fixed-model" },
},
findResult: {
id: "fixed-model",
provider: "fixed-provider",
api: "openai-codex-responses",
} as Model<any>,
});
expect(find).toHaveBeenCalledWith("fixed-provider", "fixed-model");
expect(complete).toHaveBeenCalledTimes(1);
expect((complete.mock.calls[0] as unknown[])[0]).toMatchObject({
id: "fixed-model",
provider: "fixed-provider",
});
expect(verdict).toEqual({ kind: "allow" });
expect(reviews[0]?.details).toMatchObject({
modelSource: "configured",
provider: "fixed-provider",
model: "fixed-model",
});
});
it("defers with judge_model_unavailable when a configured model cannot be resolved, without fallback", async () => {
const { verdict, reviews, complete } = await runAsk({
config: {
version: 2,
mode: "enforce",
model: { provider: "ghost-provider", id: "ghost-model" },
},
findResult: undefined,
});
expect(verdict).toEqual({ kind: "defer" });
expect(complete).not.toHaveBeenCalled();
expect(reviews).toMatchObject([
{
event: "ai_bash_judge.result",
details: expect.objectContaining({
resultKind: "infrastructure_failure",
code: "judge_model_unavailable",
modelCalled: false,
provider: "ghost-provider",
model: "ghost-model",
}),
},
]);
});
it("keeps riskOverride on the early judge_model_unavailable row for a high-risk shadow ask", async () => {
const { reviews } = await runAsk({
config: {
version: 2,
mode: "shadow",
model: { provider: "ghost-provider", id: "ghost-model" },
},
command: "git clean -xfd",
findResult: undefined,
});
expect(reviews[0]?.details).toMatchObject({
code: "judge_model_unavailable",
riskOverride: { category: "data_loss", rule: expect.any(String) },
});
});
it("defers with judge_model_unavailable when a configured model lacks auth", async () => {
const { verdict, reviews, complete } = await runAsk({
config: {
version: 2,
mode: "enforce",
model: { provider: "p", id: "m" },
},
findResult: { id: "m", provider: "p", api: "openai-codex-responses" } as Model<any>,
authConfigured: false,
});
expect(verdict).toEqual({ kind: "defer" });
expect(complete).not.toHaveBeenCalled();
expect(reviews[0]?.details).toMatchObject({
resultKind: "infrastructure_failure",
code: "judge_model_unavailable",
});
});
it("notifies once per session in v2 enforce mode with the judge model and risk contract", async () => {
let authorize: Authorizer["authorize"] | undefined;
const service = {
registerAuthorizer: vi.fn((_name, callback) => {
authorize = callback;
return vi.fn();
}),
checkPermission: vi.fn(),
getToolPermission: vi.fn(),
} as unknown as PermissionsService;
publishPermissionsService(service);
publishedService = service;
const complete = vi.fn(async () => modelResponse());
const notify = vi.fn();
const ctx = {
hasUI: true,
sessionManager: fakeSessionManager(),
model: {
id: "session-model",
provider: "session-provider",
api: "openai-codex-responses",
} as Model<any>,
modelRegistry: { complete, find: vi.fn(), hasConfiguredAuth: vi.fn(() => true) },
ui: { notify },
} as unknown as ExtensionContext;
const harness = createFakePi();
extension(harness.pi);
writeConfig({ version: 2, mode: "enforce" });
harness.start(ctx);
harness.ready();
const enforceNotice = notify.mock.calls.filter((call) =>
/Enforce/i.test(String(call[0])),
);
expect(enforceNotice).toHaveLength(1);
expect(String(enforceNotice[0]?.[0])).toContain(
"session-provider/session-model",
);
expect(String(enforceNotice[0]?.[0])).toMatch(/risk/i);
// Two more asks must not repeat the notification.
for (let i = 0; i < 2; i += 1) {
await authorize!(ask(), { checkPermission: vi.fn(), getToolPermission: vi.fn() }, {
review: vi.fn(),
debug: vi.fn(),
});
}
expect(
notify.mock.calls.filter((call) => /Enforce/i.test(String(call[0]))),
).toHaveLength(1);
harness.shutdown();
});
it("does not show the enforce notification in shadow mode", async () => {
const { notify } = await runAsk({
config: { version: 2, mode: "shadow" },
});
expect(
notify.mock.calls.filter((call) => /Enforce/i.test(String(call[0]))),
).toHaveLength(0);
});
});
@@ -1,260 +0,0 @@
import { mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { dirname, join } from "node:path";
import { afterEach, beforeEach, describe, expect, it } from "vitest";
import {
appendPromotionRecord,
loadPromotionRecords,
promotionRecordsPath,
resolvePromotionGates,
type CandidateIdentity,
type PromotionRecord,
} from "../src/promotion";
const IDENTITY: CandidateIdentity = {
judge: "@sikongjueluo/pi-permission-ai-judge@0.0.1",
permissionSystem: "25.4.0",
provider: "openai-codex",
model: "gpt-5.6-sol",
api: "openai-codex-responses",
promptVersion: "bash-shadow-v4",
toolSchemaVersion: "report-verdict-v1",
reviewSchemaVersion: "1",
timeoutCohort: 30000,
};
function record(
kind: PromotionRecord["kind"],
identity: CandidateIdentity = IDENTITY,
basis = "test basis",
): PromotionRecord {
return {
kind,
candidateIdentity: identity,
recordedAt: "2026-08-20T12:00:00Z",
basis,
};
}
function line(value: unknown): string {
return `${JSON.stringify(value)}\n`;
}
/** writeFileSync, but creating the records directory first. */
function writeRecords(dir: string, content: string): void {
const path = promotionRecordsPath(dir);
mkdirSync(dirname(path), { recursive: true });
writeFileSync(path, content);
}
describe("promotion records — loadPromotionRecords", () => {
let dir: string;
beforeEach(() => {
dir = mkdtempSync(join(tmpdir(), "ai-judge-promotion-"));
});
afterEach(() => {
rmSync(dir, { recursive: true, force: true });
});
it("treats a missing file as the healthy pre-promotion state", () => {
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(snapshot).toEqual({
records: [],
healthy: true,
diagnostic: null,
path: promotionRecordsPath(dir),
});
});
it("parses well-formed records of every kind", () => {
writeRecords(
dir,
[record("cohort_qualified"), record("owner_approval"), record("activation")]
.map(line)
.join(""),
);
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(snapshot.healthy).toBe(true);
expect(snapshot.records).toHaveLength(3);
expect(snapshot.records.map((r) => r.kind)).toEqual([
"cohort_qualified",
"owner_approval",
"activation",
]);
});
it("fails closed on malformed JSON lines", () => {
writeRecords(
dir,
line(record("cohort_qualified")) + "{not json\n",
);
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(snapshot.healthy).toBe(false);
expect(snapshot.records).toEqual([]);
expect(snapshot.diagnostic).toContain("malformed");
});
it("fails closed on shape-invalid records", () => {
writeRecords(
dir,
line({ kind: "activation" }), // missing identity/basis/recordedAt
);
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(snapshot.healthy).toBe(false);
expect(snapshot.diagnostic).toContain("malformed");
});
it("skips blank lines without failing", () => {
writeRecords(
dir,
"\n" + line(record("activation")) + "\n\n",
);
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(snapshot.healthy).toBe(true);
expect(snapshot.records).toHaveLength(1);
});
});
describe("promotion records — resolvePromotionGates", () => {
let dir: string;
beforeEach(() => {
dir = mkdtempSync(join(tmpdir(), "ai-judge-promotion-"));
});
afterEach(() => {
rmSync(dir, { recursive: true, force: true });
});
it("flips only its own gate per record kind", () => {
writeRecords(dir, line(record("owner_approval")));
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(resolvePromotionGates(snapshot, IDENTITY)).toEqual({
cohortQualified: false,
ownerApprovalRecorded: true,
activationRecorded: false,
});
});
it("all three gates open with all three records for the exact identity", () => {
writeRecords(
dir,
[
record("cohort_qualified", IDENTITY, "cohort id x"),
record("owner_approval", IDENTITY, "approved"),
record("activation", IDENTITY, "activated"),
]
.map(line)
.join(""),
);
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(resolvePromotionGates(snapshot, IDENTITY)).toEqual({
cohortQualified: true,
ownerApprovalRecorded: true,
activationRecorded: true,
});
});
const identityDrifts: ReadonlyArray<[string, Partial<CandidateIdentity>]> = [
["provider", { provider: "other-provider" }],
["model", { model: "gpt-5.7" }],
["api", { api: "openai-responses" }],
["promptVersion", { promptVersion: "bash-shadow-v5" }],
["toolSchemaVersion", { toolSchemaVersion: "report-verdict-v2" }],
["reviewSchemaVersion", { reviewSchemaVersion: "2" }],
["timeoutCohort", { timeoutCohort: "default" }],
["permissionSystem", { permissionSystem: "25.5.0" }],
["judge package", { judge: "@sikongjueluo/pi-permission-ai-judge@0.0.2" }],
];
for (const [name, patch] of identityDrifts) {
it(`keeps every gate closed when the live identity drifts on ${name}`, () => {
writeRecords(
dir,
[
record("cohort_qualified"),
record("owner_approval"),
record("activation"),
]
.map(line)
.join(""),
);
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(
resolvePromotionGates(snapshot, { ...IDENTITY, ...patch }),
).toEqual({
cohortQualified: false,
ownerApprovalRecorded: false,
activationRecorded: false,
});
});
}
it("closes every gate when the snapshot is unhealthy", () => {
writeRecords(dir, "garbage\n");
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(snapshot.healthy).toBe(false);
expect(resolvePromotionGates(snapshot, IDENTITY)).toEqual({
cohortQualified: false,
ownerApprovalRecorded: false,
activationRecorded: false,
});
});
it("leaves other-identity records inert, not malformed", () => {
const other: CandidateIdentity = {
...IDENTITY,
model: "glm-5.2",
};
writeRecords(
dir,
[
record("cohort_qualified", other),
record("activation", other),
]
.map(line)
.join(""),
);
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(snapshot.healthy).toBe(true);
expect(resolvePromotionGates(snapshot, IDENTITY)).toEqual({
cohortQualified: false,
ownerApprovalRecorded: false,
activationRecorded: false,
});
});
});
describe("promotion records — appendPromotionRecord", () => {
let dir: string;
beforeEach(() => {
dir = mkdtempSync(join(tmpdir(), "ai-judge-promotion-"));
});
afterEach(() => {
rmSync(dir, { recursive: true, force: true });
});
it("appends a shape-valid record that round-trips through the loader", () => {
const error = appendPromotionRecord({
agentDir: dir,
record: record("owner_approval", IDENTITY, "approved v4"),
now: () => "2026-08-21T09:00:00Z",
});
expect(error).toBeNull();
const raw = readFileSync(promotionRecordsPath(dir), "utf-8");
expect(raw).toContain('"recordedAt":"2026-08-21T09:00:00Z"');
expect(raw).toContain('"basis":"approved v4"');
const snapshot = loadPromotionRecords({ agentDir: dir });
expect(snapshot.healthy).toBe(true);
expect(resolvePromotionGates(snapshot, IDENTITY).ownerApprovalRecorded).toBe(true);
});
it("rejects a shape-invalid record without touching the file", () => {
const bad = {
kind: "activation",
candidateIdentity: { judge: "x" },
recordedAt: "",
basis: "",
} as unknown as PromotionRecord;
const error = appendPromotionRecord({ agentDir: dir, record: bad });
expect(error).toBe("record is not shape-valid");
expect(loadPromotionRecords({ agentDir: dir }).records).toEqual([]);
});
});