mirror of
https://github.com/SikongJueluo/pi-extensions.git
synced 2026-10-05 11:52:55 +08:00
feat(ai-judge): rework enforce mode as user-assumed-risk contract
- add config v2 with fixed judge model selection and fail-closed v1 enforce migration to shadow - add built-in high-risk override for irreversible, publish, system, and credential shapes that always defers to the human - remove the promotion gate, promotion records tool, and their tests - fail closed with judge_model_unavailable when a configured judge model cannot be resolved - notify once per session in enforce mode with the judge model and risk contract - add ADR 0008 and update README and CONTEXT
This commit is contained in:
@@ -20,9 +20,11 @@ describe("loadJudgeConfig — missing and malformed", () => {
|
||||
it("resolves a missing file to all defaults with one diagnostic", () => {
|
||||
const config = loadJudgeConfig(deps());
|
||||
expect(config).toEqual({
|
||||
configVersion: 1,
|
||||
mode: "shadow",
|
||||
timeoutMs: 15_000,
|
||||
timeoutCohort: "default",
|
||||
judgeModel: undefined,
|
||||
diagnostics: [
|
||||
expect.objectContaining({ key: "file", fallback: "all defaults" }),
|
||||
],
|
||||
@@ -45,19 +47,76 @@ describe("loadJudgeConfig — missing and malformed", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("loadJudgeConfig — version selection", () => {
|
||||
it("treats a missing version as v1", () => {
|
||||
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: '{"mode":"shadow"}' }));
|
||||
expect(config.configVersion).toBe(1);
|
||||
});
|
||||
|
||||
it("accepts version 1 explicitly", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"version":1,"mode":"shadow"}' }),
|
||||
);
|
||||
expect(config.configVersion).toBe(1);
|
||||
expect(config.diagnostics).toEqual([]);
|
||||
});
|
||||
|
||||
it("accepts version 2", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"version":2,"mode":"shadow"}' }),
|
||||
);
|
||||
expect(config.configVersion).toBe(2);
|
||||
expect(config.diagnostics).toEqual([]);
|
||||
});
|
||||
|
||||
it("rejects an unknown version to all defaults with a diagnostic", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"version":3,"mode":"enforce"}' }),
|
||||
);
|
||||
expect(config).toMatchObject({ configVersion: 1, mode: "shadow" });
|
||||
expect(config.diagnostics[0]?.key).toBe("version");
|
||||
});
|
||||
});
|
||||
|
||||
describe("loadJudgeConfig — v1 enforce migration (fail-closed)", () => {
|
||||
it("downgrades v1 enforce to shadow with a migration diagnostic", () => {
|
||||
for (const raw of ['{"mode":"enforce"}', '{"version":1,"mode":"enforce"}']) {
|
||||
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: raw }));
|
||||
expect(config.mode).toBe("shadow");
|
||||
expect(config.diagnostics).toEqual([
|
||||
{
|
||||
key: "mode",
|
||||
problem: expect.stringMatching(/"version": 2/),
|
||||
fallback: "shadow (v1 enforce requires explicit migration)",
|
||||
},
|
||||
]);
|
||||
}
|
||||
});
|
||||
|
||||
it("keeps v1 shadow without diagnostics", () => {
|
||||
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: '{"mode":"shadow"}' }));
|
||||
expect(config.mode).toBe("shadow");
|
||||
expect(config.diagnostics).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("loadJudgeConfig — mode", () => {
|
||||
it("accepts shadow and enforce", () => {
|
||||
it("accepts shadow and enforce in v2", () => {
|
||||
expect(
|
||||
loadJudgeConfig(deps({ [CONFIG_PATH]: '{"mode":"shadow"}' })).mode,
|
||||
loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"version":2,"mode":"shadow"}' }),
|
||||
).mode,
|
||||
).toBe("shadow");
|
||||
expect(
|
||||
loadJudgeConfig(deps({ [CONFIG_PATH]: '{"mode":"enforce"}' })).mode,
|
||||
loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"version":2,"mode":"enforce"}' }),
|
||||
).mode,
|
||||
).toBe("enforce");
|
||||
});
|
||||
|
||||
it("resolves an unknown mode to shadow with a diagnostic", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"mode":"yolo"}' }),
|
||||
deps({ [CONFIG_PATH]: '{"version":2,"mode":"yolo"}' }),
|
||||
);
|
||||
expect(config.mode).toBe("shadow");
|
||||
expect(config.diagnostics).toEqual([
|
||||
@@ -70,18 +129,72 @@ describe("loadJudgeConfig — mode", () => {
|
||||
});
|
||||
|
||||
it("resolves a missing mode to shadow without diagnostics", () => {
|
||||
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: "{}" }));
|
||||
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: '{"version":2}' }));
|
||||
expect(config.mode).toBe("shadow");
|
||||
expect(config.diagnostics).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("loadJudgeConfig — v2 judge model", () => {
|
||||
it("parses an explicit fixed judge model", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({
|
||||
[CONFIG_PATH]:
|
||||
'{"version":2,"mode":"enforce","model":{"provider":"openai-codex","id":"gpt-5.6-sol"}}',
|
||||
}),
|
||||
);
|
||||
expect(config.mode).toBe("enforce");
|
||||
expect(config.judgeModel).toEqual({
|
||||
provider: "openai-codex",
|
||||
id: "gpt-5.6-sol",
|
||||
});
|
||||
expect(config.diagnostics).toEqual([]);
|
||||
});
|
||||
|
||||
it("allows a judge model in shadow mode too", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({
|
||||
[CONFIG_PATH]:
|
||||
'{"version":2,"mode":"shadow","model":{"provider":"p","id":"m"}}',
|
||||
}),
|
||||
);
|
||||
expect(config.mode).toBe("shadow");
|
||||
expect(config.judgeModel).toEqual({ provider: "p", id: "m" });
|
||||
});
|
||||
|
||||
it.each([
|
||||
'{"version":2,"model":"openai-codex"}',
|
||||
'{"version":2,"model":{"provider":"p"}}',
|
||||
'{"version":2,"model":{"id":"m"}}',
|
||||
'{"version":2,"model":{"provider":"","id":"m"}}',
|
||||
'{"version":2,"model":{"provider":"p","id":""}}',
|
||||
'{"version":2,"model":{"provider":1,"id":"m"}}',
|
||||
'{"version":2,"model":[]}',
|
||||
])("fails a malformed model %j closed to shadow with a diagnostic", (raw) => {
|
||||
const config = loadJudgeConfig(deps({ [CONFIG_PATH]: raw }));
|
||||
expect(config.mode).toBe("shadow");
|
||||
expect(config.judgeModel).toBeUndefined();
|
||||
expect(config.diagnostics[0]?.key).toBe("model");
|
||||
});
|
||||
|
||||
it("ignores a model field in v1 with a diagnostic", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"model":{"provider":"p","id":"m"}}' }),
|
||||
);
|
||||
expect(config.configVersion).toBe(1);
|
||||
expect(config.judgeModel).toBeUndefined();
|
||||
expect(config.diagnostics[0]?.key).toBe("model");
|
||||
});
|
||||
});
|
||||
|
||||
describe("loadJudgeConfig — timeout boundaries", () => {
|
||||
it.each([4_999, 30_001, 0, -5_000, 15.5, NaN, Infinity, "20000"])(
|
||||
"rejects invalid timeoutMs %p with fallback to 15,000",
|
||||
(value) => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: JSON.stringify({ timeoutMs: value }) }),
|
||||
deps({
|
||||
[CONFIG_PATH]: JSON.stringify({ version: 2, timeoutMs: value }),
|
||||
}),
|
||||
);
|
||||
expect(config.timeoutMs).toBe(15_000);
|
||||
expect(config.timeoutCohort).toBe("default");
|
||||
@@ -91,16 +204,20 @@ describe("loadJudgeConfig — timeout boundaries", () => {
|
||||
|
||||
it("accepts the inclusive boundaries 5,000 and 30,000", () => {
|
||||
expect(
|
||||
loadJudgeConfig(deps({ [CONFIG_PATH]: '{"timeoutMs":5000}' })),
|
||||
loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"version":2,"timeoutMs":5000}' }),
|
||||
),
|
||||
).toMatchObject({ timeoutMs: 5_000, timeoutCohort: 5_000 });
|
||||
expect(
|
||||
loadJudgeConfig(deps({ [CONFIG_PATH]: '{"timeoutMs":30000}' })),
|
||||
loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"version":2,"timeoutMs":30000}' }),
|
||||
),
|
||||
).toMatchObject({ timeoutMs: 30_000, timeoutCohort: 30_000 });
|
||||
});
|
||||
|
||||
it("marks an explicit default timeout as the default cohort", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"timeoutMs":15000}' }),
|
||||
deps({ [CONFIG_PATH]: '{"version":2,"timeoutMs":15000}' }),
|
||||
);
|
||||
expect(config.timeoutCohort).toBe("default");
|
||||
expect(config.diagnostics).toEqual([]);
|
||||
@@ -108,7 +225,7 @@ describe("loadJudgeConfig — timeout boundaries", () => {
|
||||
|
||||
it("marks a non-default timeout as a distinct cohort", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"timeoutMs":30000}' }),
|
||||
deps({ [CONFIG_PATH]: '{"version":2,"timeoutMs":30000}' }),
|
||||
);
|
||||
expect(config.timeoutCohort).toBe(30_000);
|
||||
});
|
||||
@@ -117,13 +234,18 @@ describe("loadJudgeConfig — timeout boundaries", () => {
|
||||
describe("loadJudgeConfig — snapshot immutability", () => {
|
||||
it("returns an immutable effective-config snapshot", () => {
|
||||
const config = loadJudgeConfig(
|
||||
deps({ [CONFIG_PATH]: '{"mode":"enforce","timeoutMs":20000}' }),
|
||||
deps({
|
||||
[CONFIG_PATH]:
|
||||
'{"version":2,"mode":"enforce","timeoutMs":20000,"model":{"provider":"p","id":"m"}}',
|
||||
}),
|
||||
);
|
||||
expect(Object.isFrozen(config)).toBe(true);
|
||||
expect(config).toEqual({
|
||||
configVersion: 2,
|
||||
mode: "enforce",
|
||||
timeoutMs: 20_000,
|
||||
timeoutCohort: 20_000,
|
||||
judgeModel: { provider: "p", id: "m" },
|
||||
diagnostics: [],
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { classifyHighRisk } from "../src/highrisk";
|
||||
|
||||
describe("classifyHighRisk — data_loss", () => {
|
||||
it.each([
|
||||
"git clean -xfd",
|
||||
"git clean -fdx",
|
||||
"git clean -fxd",
|
||||
"git clean -f -x -d",
|
||||
"git clean --force -xd",
|
||||
"git reset --hard",
|
||||
"git reset --hard HEAD~3",
|
||||
"git reset --hard origin/main",
|
||||
"git checkout -- .",
|
||||
"git checkout .",
|
||||
"git restore .",
|
||||
"git restore -- .",
|
||||
"rm -rf ~",
|
||||
"rm -rf /",
|
||||
"rm -rf ~/*",
|
||||
"rm -rf $HOME",
|
||||
"rm -rf $HOME/projects",
|
||||
"rm -fr ~",
|
||||
])("flags %j", (command) => {
|
||||
expect(classifyHighRisk(command)?.category).toBe("data_loss");
|
||||
});
|
||||
|
||||
it.each([
|
||||
"git clean -nxd",
|
||||
"git clean -nd",
|
||||
"git reset --soft HEAD~1",
|
||||
"git checkout main",
|
||||
"git checkout -- file.txt",
|
||||
"git restore file.txt",
|
||||
"rm -rf build/",
|
||||
"rm -rf ./dist",
|
||||
"rm -r build",
|
||||
"rm file.txt",
|
||||
"echo git clean -xfd",
|
||||
"git status",
|
||||
])("does not flag %j", (command) => {
|
||||
expect(classifyHighRisk(command)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("classifyHighRisk — history_rewrite", () => {
|
||||
it.each([
|
||||
"git push --force",
|
||||
"git push -f",
|
||||
"git push --force origin main",
|
||||
"git push --force-with-lease origin main",
|
||||
"git push origin +main",
|
||||
])("flags %j", (command) => {
|
||||
expect(classifyHighRisk(command)?.category).toBe("history_rewrite");
|
||||
});
|
||||
|
||||
it.each(["git push origin main", "git push", "git push --tags"])(
|
||||
"does not flag %j",
|
||||
(command) => {
|
||||
expect(classifyHighRisk(command)).toBeUndefined();
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
describe("classifyHighRisk — publish_deploy", () => {
|
||||
it.each([
|
||||
"npm publish",
|
||||
"npm publish --access public",
|
||||
"pnpm publish",
|
||||
"yarn publish",
|
||||
"cargo publish",
|
||||
"terraform destroy",
|
||||
"terraform destroy -auto-approve",
|
||||
])("flags %j", (command) => {
|
||||
expect(classifyHighRisk(command)?.category).toBe("publish_deploy");
|
||||
});
|
||||
|
||||
it.each(["npm run publish", "npm install", "pnpm test", "terraform plan", "terraform apply"])(
|
||||
"does not flag %j",
|
||||
(command) => {
|
||||
expect(classifyHighRisk(command)).toBeUndefined();
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
describe("classifyHighRisk — system_modify", () => {
|
||||
it.each([
|
||||
"sudo rm /etc/hosts",
|
||||
"sudo -i",
|
||||
"sudo apt-get install ripgrep",
|
||||
"mkfs.ext4 /dev/sda1",
|
||||
"mkfs /dev/sdb",
|
||||
"dd if=/dev/zero of=/dev/sda bs=1M",
|
||||
"shutdown now",
|
||||
"shutdown -h now",
|
||||
"reboot",
|
||||
"halt",
|
||||
"poweroff",
|
||||
])("flags %j", (command) => {
|
||||
expect(classifyHighRisk(command)?.category).toBe("system_modify");
|
||||
});
|
||||
|
||||
it.each([
|
||||
"dd if=/dev/zero of=/tmp/img bs=1M count=10",
|
||||
"echo sudo",
|
||||
"cat /etc/hosts",
|
||||
])("does not flag %j", (command) => {
|
||||
expect(classifyHighRisk(command)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("classifyHighRisk — credential_access", () => {
|
||||
it.each([
|
||||
"cat ~/.ssh/id_rsa",
|
||||
"cat ~/.ssh/id_ed25519",
|
||||
"less ~/.ssh/id_ed25519",
|
||||
"cat ~/.aws/credentials",
|
||||
"head ~/.netrc",
|
||||
"cat ~/.config/gcloud/application_default_credentials.json",
|
||||
"rm ~/.ssh/id_ed25519",
|
||||
"mv ~/.ssh/authorized_keys /tmp",
|
||||
"cp ~/.gnupg/secring.gpg .",
|
||||
"cat $HOME/.ssh/id_rsa",
|
||||
"cat $HOME/.aws/credentials",
|
||||
])("flags %j", (command) => {
|
||||
expect(classifyHighRisk(command)?.category).toBe("credential_access");
|
||||
});
|
||||
|
||||
it.each([
|
||||
"echo x > ~/.aws/credentials",
|
||||
"echo x >> ~/.ssh/authorized_keys",
|
||||
"ssh-keygen foo 2> ~/.aws/credentials",
|
||||
"printf '%s' key > ~/.ssh/id_ed25519",
|
||||
"curl -s url | tee ~/.netrc",
|
||||
])("flags credential replacement %j", (command) => {
|
||||
expect(classifyHighRisk(command)).toMatchObject({
|
||||
category: "credential_access",
|
||||
});
|
||||
});
|
||||
|
||||
it.each([
|
||||
"cat ~/.ssh/config",
|
||||
"cat ~/.ssh/known_hosts",
|
||||
"cat package.json",
|
||||
"ls ~/.ssh",
|
||||
"cat ~/.bashrc",
|
||||
"cat id_rsa",
|
||||
"echo x > ~/.bashrc",
|
||||
"curl -s url | tee notes.txt",
|
||||
])("does not flag %j", (command) => {
|
||||
expect(classifyHighRisk(command)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("classifyHighRisk — compound inputs", () => {
|
||||
it("flags when any unit matches", () => {
|
||||
expect(classifyHighRisk("pnpm test && git clean -xfd")?.category).toBe(
|
||||
"data_loss",
|
||||
);
|
||||
expect(classifyHighRisk("git push --force; echo done")?.category).toBe(
|
||||
"history_rewrite",
|
||||
);
|
||||
expect(classifyHighRisk("npm publish | tee log")?.category).toBe(
|
||||
"publish_deploy",
|
||||
);
|
||||
});
|
||||
|
||||
it("flags single-`&` background compound units", () => {
|
||||
expect(classifyHighRisk("echo ready & npm publish")?.category).toBe(
|
||||
"publish_deploy",
|
||||
);
|
||||
expect(classifyHighRisk("sleep 5 & rm -rf ~")?.category).toBe(
|
||||
"data_loss",
|
||||
);
|
||||
});
|
||||
|
||||
it("does not flag separators inside quotes or escapes", () => {
|
||||
expect(classifyHighRisk("printf 'x; npm publish;'")).toBeUndefined();
|
||||
expect(classifyHighRisk("echo \"git push --force\"")).toBeUndefined();
|
||||
expect(classifyHighRisk("echo git\\;npm\\;publish")).toBeUndefined();
|
||||
expect(classifyHighRisk("echo 'rm -rf ~'")).toBeUndefined();
|
||||
});
|
||||
|
||||
it("treats a quoted command argument as one token, not a command", () => {
|
||||
// git commit -m "junk; sudo rm /" quotes an argument, not a unit.
|
||||
expect(classifyHighRisk("git commit -m 'do not npm publish'")).toBeUndefined();
|
||||
});
|
||||
|
||||
it("flags high-risk units even when another unit quotes text", () => {
|
||||
expect(classifyHighRisk("echo 'a;b' && git push --force")).toMatchObject({
|
||||
category: "history_rewrite",
|
||||
});
|
||||
});
|
||||
|
||||
it("returns undefined for all-benign compounds", () => {
|
||||
expect(classifyHighRisk("pnpm test && pnpm check")).toBeUndefined();
|
||||
expect(classifyHighRisk("echo a; echo b | wc -l")).toBeUndefined();
|
||||
expect(classifyHighRisk("sleep 5 & echo done")).toBeUndefined();
|
||||
});
|
||||
|
||||
it("reports the matching rule for audit observability", () => {
|
||||
const match = classifyHighRisk("git clean -xfd");
|
||||
expect(match).toMatchObject({
|
||||
category: "data_loss",
|
||||
rule: expect.stringMatching(/^git_clean/),
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -3,18 +3,10 @@ import {
|
||||
evaluateEnforceAuthority,
|
||||
type EnforceGateState,
|
||||
} from "../src/judge";
|
||||
import {
|
||||
loadPromotionRecords,
|
||||
resolvePromotionGates,
|
||||
type CandidateIdentity,
|
||||
} from "../src/promotion";
|
||||
|
||||
const ALL_OPEN: EnforceGateState = {
|
||||
auditHealthy: true,
|
||||
telemetryHealth: "healthy",
|
||||
cohortQualified: true,
|
||||
ownerApprovalRecorded: true,
|
||||
activationRecorded: true,
|
||||
resultKind: "judgment",
|
||||
verdict: "allow",
|
||||
reviewAcknowledged: true,
|
||||
@@ -23,7 +15,10 @@ const ALL_OPEN: EnforceGateState = {
|
||||
};
|
||||
|
||||
describe("evaluateEnforceAuthority — every gate independently forces defer", () => {
|
||||
it("allows only when every gate holds", () => {
|
||||
it("allows when every gate holds — no promotion records required", () => {
|
||||
// ADR 0008: the promotion gates (cohort qualification, owner
|
||||
// approval, activation) are no longer authority inputs; a session
|
||||
// with zero promotion records can hold Enforce authority.
|
||||
expect(evaluateEnforceAuthority(ALL_OPEN)).toEqual({ kind: "allow" });
|
||||
});
|
||||
|
||||
@@ -37,9 +32,6 @@ describe("evaluateEnforceAuthority — every gate independently forces defer", (
|
||||
{ name: "telemetry disabled", patch: { telemetryHealth: "disabled" }, expectedReason: "telemetry_disabled" },
|
||||
{ name: "telemetry write failed", patch: { telemetryHealth: "write_failed" }, expectedReason: "telemetry_write_failed" },
|
||||
{ name: "telemetry integrity anomaly", patch: { telemetryHealth: "integrity_anomaly" }, expectedReason: "telemetry_integrity_anomaly" },
|
||||
{ name: "cohort not qualified", patch: { cohortQualified: false }, expectedReason: "cohort_not_qualified" },
|
||||
{ name: "owner approval absent", patch: { ownerApprovalRecorded: false }, expectedReason: "owner_approval_absent" },
|
||||
{ name: "activation absent", patch: { activationRecorded: false }, expectedReason: "activation_absent" },
|
||||
{ name: "preflight result", patch: { resultKind: "preflight_defer" }, expectedReason: "result_preflight_defer" },
|
||||
{ name: "infrastructure result", patch: { resultKind: "infrastructure_failure" }, expectedReason: "result_infrastructure_failure" },
|
||||
{ name: "semantic deny verdict", patch: { verdict: "deny" }, expectedReason: "verdict_deny" },
|
||||
@@ -58,86 +50,3 @@ describe("evaluateEnforceAuthority — every gate independently forces defer", (
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
describe("evaluateEnforceAuthority — production state with no promotion records", () => {
|
||||
// The real post-PIEXTENSIO-21 seam: an empty records file (the normal
|
||||
// pre-promotion state) closes all promotion gates, so every mode and
|
||||
// telemetry state defers — mechanically identical to v0.1's hardcoded
|
||||
// closure, now derived from actual storage.
|
||||
const emptySnapshot = loadPromotionRecords({
|
||||
agentDir: "/nonexistent-agent-dir",
|
||||
});
|
||||
const identity: CandidateIdentity = {
|
||||
judge: "@sikongjueluo/pi-permission-ai-judge@0.0.1",
|
||||
permissionSystem: "25.4.0",
|
||||
provider: "openai-codex",
|
||||
model: "gpt-5.6-sol",
|
||||
api: "openai-codex-responses",
|
||||
promptVersion: "bash-shadow-v4",
|
||||
toolSchemaVersion: "report-verdict-v1",
|
||||
reviewSchemaVersion: "1",
|
||||
timeoutCohort: 30000,
|
||||
};
|
||||
|
||||
it("never grants authority for any mode or telemetry state without records", () => {
|
||||
const modes = ["shadow", "enforce"] as const;
|
||||
const healths = ["healthy", "disabled", "write_failed", "integrity_anomaly"] as const;
|
||||
for (const mode of modes) {
|
||||
for (const health of healths) {
|
||||
const gates = resolvePromotionGates(emptySnapshot, identity);
|
||||
const outcome = evaluateEnforceAuthority({
|
||||
...ALL_OPEN,
|
||||
mode,
|
||||
telemetryHealth: health,
|
||||
...gates,
|
||||
});
|
||||
expect(outcome.kind).toBe("defer");
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
it("blocks enforce on the cohort gate first even with healthy audit", () => {
|
||||
const gates = resolvePromotionGates(emptySnapshot, identity);
|
||||
const outcome = evaluateEnforceAuthority({
|
||||
...ALL_OPEN,
|
||||
...gates,
|
||||
});
|
||||
expect(outcome).toEqual({
|
||||
kind: "defer",
|
||||
blockedBy: "cohort_not_qualified",
|
||||
});
|
||||
});
|
||||
|
||||
it("grants authority only when every record kind exists for the exact identity", () => {
|
||||
const records = [
|
||||
{
|
||||
kind: "cohort_qualified",
|
||||
candidateIdentity: identity,
|
||||
recordedAt: "2026-08-20T12:00:00Z",
|
||||
basis: "cohort test",
|
||||
},
|
||||
{
|
||||
kind: "owner_approval",
|
||||
candidateIdentity: identity,
|
||||
recordedAt: "2026-08-20T12:01:00Z",
|
||||
basis: "approved",
|
||||
},
|
||||
{
|
||||
kind: "activation",
|
||||
candidateIdentity: identity,
|
||||
recordedAt: "2026-08-20T12:02:00Z",
|
||||
basis: "activated",
|
||||
},
|
||||
] as const;
|
||||
const snapshot = {
|
||||
records,
|
||||
healthy: true,
|
||||
diagnostic: null,
|
||||
path: "unused",
|
||||
};
|
||||
const gates = resolvePromotionGates(snapshot, identity);
|
||||
expect(evaluateEnforceAuthority({ ...ALL_OPEN, ...gates })).toEqual({
|
||||
kind: "allow",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -50,7 +50,6 @@ import {
|
||||
unpublishPermissionsService,
|
||||
} from "@gotgenes/pi-permission-system";
|
||||
import extension from "../src/index";
|
||||
import { appendPromotionRecord, type CandidateIdentity } from "../src/promotion";
|
||||
import { PROMPT_VERSION, TOOL_SCHEMA_VERSION } from "../src/prompt";
|
||||
|
||||
function createFakePi(): {
|
||||
@@ -542,33 +541,39 @@ describe("AI judge lifecycle", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("AI judge Enforce authority seam (PIEXTENSIO-21)", () => {
|
||||
describe("AI judge Enforce authority seam (PIEXTENSIO-23, ADR 0008)", () => {
|
||||
beforeEach(() => {
|
||||
createMockAgentDir();
|
||||
writeFileSync(
|
||||
join(mockAgentDir.dir, "pi-permission-ai-judge.config.json"),
|
||||
JSON.stringify({ mode: "enforce" }),
|
||||
);
|
||||
});
|
||||
|
||||
/** Records identity matching the lifecycle fake model + static fields. */
|
||||
function lifecycleIdentity(): CandidateIdentity {
|
||||
return {
|
||||
judge: "@sikongjueluo/pi-permission-ai-judge@0.0.1",
|
||||
permissionSystem: "25.4.0",
|
||||
provider: "test-provider",
|
||||
model: "test-model",
|
||||
api: "openai-codex-responses",
|
||||
promptVersion: PROMPT_VERSION,
|
||||
toolSchemaVersion: TOOL_SCHEMA_VERSION,
|
||||
reviewSchemaVersion: "1",
|
||||
timeoutCohort: "default",
|
||||
};
|
||||
function writeConfig(config: Record<string, unknown>): void {
|
||||
writeFileSync(
|
||||
join(mockAgentDir.dir, "pi-permission-ai-judge.config.json"),
|
||||
JSON.stringify(config),
|
||||
);
|
||||
}
|
||||
|
||||
async function runAsk(): Promise<{
|
||||
interface RunAskOptions {
|
||||
/** Config file contents (written before session start). */
|
||||
config: Record<string, unknown>;
|
||||
/** Command unit for the ask; full command becomes `pnpm test && <unit>`. */
|
||||
command?: string;
|
||||
/** modelRegistry.find result for a configured judge model (when set). */
|
||||
findResult?: Model<any> | undefined;
|
||||
/** Whether the found model has configured auth. */
|
||||
authConfigured?: boolean;
|
||||
/** Overridable model verdict. */
|
||||
response?: AssistantMessage;
|
||||
}
|
||||
|
||||
async function runAsk(
|
||||
options: RunAskOptions,
|
||||
): Promise<{
|
||||
verdict: { kind: string };
|
||||
reviews: Array<{ event: string; details?: Record<string, unknown> }>;
|
||||
notify: ReturnType<typeof vi.fn>;
|
||||
complete: ReturnType<typeof vi.fn>;
|
||||
find: ReturnType<typeof vi.fn>;
|
||||
}> {
|
||||
let authorize: Authorizer["authorize"] | undefined;
|
||||
const service = {
|
||||
@@ -582,28 +587,45 @@ describe("AI judge Enforce authority seam (PIEXTENSIO-21)", () => {
|
||||
publishPermissionsService(service);
|
||||
publishedService = service;
|
||||
|
||||
const complete = vi.fn(async () => modelResponse());
|
||||
const complete = vi.fn(async () => options.response ?? modelResponse());
|
||||
const find = vi.fn(() => options.findResult);
|
||||
const hasConfiguredAuth = vi.fn(() => options.authConfigured !== false);
|
||||
const notify = vi.fn();
|
||||
const sessionManager = fakeSessionManager();
|
||||
const ctx = {
|
||||
hasUI: true,
|
||||
sessionManager: fakeSessionManager(),
|
||||
sessionManager,
|
||||
model: {
|
||||
id: "test-model",
|
||||
provider: "test-provider",
|
||||
id: "session-model",
|
||||
provider: "session-provider",
|
||||
api: "openai-codex-responses",
|
||||
} as Model<any>,
|
||||
modelRegistry: { complete },
|
||||
ui: { notify: vi.fn() },
|
||||
modelRegistry: { complete, find, hasConfiguredAuth },
|
||||
ui: { notify },
|
||||
} as unknown as ExtensionContext;
|
||||
|
||||
const harness = createFakePi();
|
||||
extension(harness.pi);
|
||||
writeConfig(options.config);
|
||||
harness.start(ctx);
|
||||
harness.ready();
|
||||
expect(authorize).toBeDefined();
|
||||
|
||||
const unit = options.command ?? "git push origin main";
|
||||
const askDetails = ask();
|
||||
askDetails.command = unit;
|
||||
const request = askDetails.payload.request as { value: string };
|
||||
request.value = unit;
|
||||
const evidenceEntry = (
|
||||
askDetails.payload as { evidence: ReadonlyArray<{ label: string; text: string }> }
|
||||
).evidence[0];
|
||||
if (evidenceEntry !== undefined) {
|
||||
evidenceEntry.text = `pnpm test && ${unit}`;
|
||||
}
|
||||
|
||||
const reviews: Array<{ event: string; details?: Record<string, unknown> }> = [];
|
||||
const verdict = await authorize!(
|
||||
ask(),
|
||||
askDetails,
|
||||
{
|
||||
checkPermission: vi.fn(),
|
||||
getToolPermission: vi.fn(),
|
||||
@@ -614,45 +636,13 @@ describe("AI judge Enforce authority seam (PIEXTENSIO-21)", () => {
|
||||
},
|
||||
);
|
||||
harness.shutdown();
|
||||
return { verdict, reviews };
|
||||
return { verdict, reviews, notify, complete, find };
|
||||
}
|
||||
|
||||
it("defers in enforce mode when no promotion records exist", async () => {
|
||||
const { verdict, reviews } = await runAsk();
|
||||
expect(verdict).toEqual({ kind: "defer" });
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
details: expect.objectContaining({
|
||||
mode: "enforce",
|
||||
verdict: "allow",
|
||||
effectiveVerdict: "defer",
|
||||
authorityBlockedBy: "cohort_not_qualified",
|
||||
}),
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("grants authority in enforce mode only with all three exact-identity records", async () => {
|
||||
const identity = lifecycleIdentity();
|
||||
for (const [kind, basis] of [
|
||||
["cohort_qualified", "cohort piextensio-test"],
|
||||
["owner_approval", "approved for test"],
|
||||
["activation", "activated for test"],
|
||||
] as const) {
|
||||
expect(
|
||||
appendPromotionRecord({
|
||||
agentDir: mockAgentDir.dir,
|
||||
record: {
|
||||
kind,
|
||||
candidateIdentity: identity,
|
||||
recordedAt: "2026-08-21T10:00:00Z",
|
||||
basis,
|
||||
},
|
||||
}),
|
||||
).toBeNull();
|
||||
}
|
||||
const { verdict, reviews } = await runAsk();
|
||||
it("grants authority in v2 enforce mode with no promotion records", async () => {
|
||||
const { verdict, reviews } = await runAsk({
|
||||
config: { version: 2, mode: "enforce" },
|
||||
});
|
||||
expect(verdict).toEqual({ kind: "allow" });
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
@@ -662,73 +652,227 @@ describe("AI judge Enforce authority seam (PIEXTENSIO-21)", () => {
|
||||
verdict: "allow",
|
||||
effectiveVerdict: "allow",
|
||||
authorityBlockedBy: null,
|
||||
modelSource: "session",
|
||||
}),
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("defers in enforce mode when records exist for another identity", async () => {
|
||||
const identity = { ...lifecycleIdentity(), model: "other-model" };
|
||||
for (const kind of [
|
||||
"cohort_qualified",
|
||||
"owner_approval",
|
||||
"activation",
|
||||
] as const) {
|
||||
appendPromotionRecord({
|
||||
agentDir: mockAgentDir.dir,
|
||||
record: {
|
||||
kind,
|
||||
candidateIdentity: identity,
|
||||
recordedAt: "2026-08-21T10:00:00Z",
|
||||
basis: "other identity",
|
||||
},
|
||||
});
|
||||
}
|
||||
const { verdict, reviews } = await runAsk();
|
||||
expect(verdict).toEqual({ kind: "defer" });
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
details: expect.objectContaining({
|
||||
effectiveVerdict: "defer",
|
||||
authorityBlockedBy: "cohort_not_qualified",
|
||||
}),
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("never grants authority in shadow mode regardless of records", async () => {
|
||||
writeFileSync(
|
||||
join(mockAgentDir.dir, "pi-permission-ai-judge.config.json"),
|
||||
JSON.stringify({ mode: "shadow" }),
|
||||
);
|
||||
const identity = lifecycleIdentity();
|
||||
for (const kind of [
|
||||
"cohort_qualified",
|
||||
"owner_approval",
|
||||
"activation",
|
||||
] as const) {
|
||||
appendPromotionRecord({
|
||||
agentDir: mockAgentDir.dir,
|
||||
record: {
|
||||
kind,
|
||||
candidateIdentity: identity,
|
||||
recordedAt: "2026-08-21T10:00:00Z",
|
||||
basis: "shadow still defers",
|
||||
},
|
||||
});
|
||||
}
|
||||
const { verdict, reviews } = await runAsk();
|
||||
it("downgrades v1 enforce to shadow with a migration diagnostic notification", async () => {
|
||||
const { verdict, reviews, notify, complete } = await runAsk({
|
||||
config: { mode: "enforce" },
|
||||
});
|
||||
expect(verdict).toEqual({ kind: "defer" });
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
details: expect.objectContaining({
|
||||
mode: "shadow",
|
||||
verdict: "allow",
|
||||
effectiveVerdict: "defer",
|
||||
authorityBlockedBy: "mode_shadow",
|
||||
}),
|
||||
},
|
||||
]);
|
||||
const notified = notify.mock.calls.map((call) => String(call[0]));
|
||||
expect(
|
||||
notified.some((message) =>
|
||||
/v1 enforce requires explicit migration|version 2/.test(message),
|
||||
),
|
||||
).toBe(true);
|
||||
expect(complete).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("defers immediately on a high-risk command in enforce mode without calling the model", async () => {
|
||||
const { verdict, reviews, complete } = await runAsk({
|
||||
config: { version: 2, mode: "enforce" },
|
||||
command: "git clean -xfd",
|
||||
});
|
||||
expect(verdict).toEqual({ kind: "defer" });
|
||||
expect(complete).not.toHaveBeenCalled();
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
details: expect.objectContaining({
|
||||
resultKind: "preflight_defer",
|
||||
verdict: null,
|
||||
effectiveVerdict: "defer",
|
||||
modelCalled: false,
|
||||
code: "high_risk_override",
|
||||
riskCategory: "data_loss",
|
||||
riskRule: expect.any(String),
|
||||
}),
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("still calls the model on a high-risk command in shadow mode, records the override, and defers", async () => {
|
||||
const { verdict, reviews, complete } = await runAsk({
|
||||
config: { version: 2, mode: "shadow" },
|
||||
command: "git clean -xfd",
|
||||
});
|
||||
expect(complete).toHaveBeenCalledTimes(1);
|
||||
expect(verdict).toEqual({ kind: "defer" });
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
details: expect.objectContaining({
|
||||
resultKind: "judgment",
|
||||
verdict: "allow",
|
||||
effectiveVerdict: "defer",
|
||||
riskOverride: { category: "data_loss", rule: expect.any(String) },
|
||||
}),
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("resolves a configured v2 judge model instead of the session model", async () => {
|
||||
const { verdict, reviews, complete, find } = await runAsk({
|
||||
config: {
|
||||
version: 2,
|
||||
mode: "enforce",
|
||||
model: { provider: "fixed-provider", id: "fixed-model" },
|
||||
},
|
||||
findResult: {
|
||||
id: "fixed-model",
|
||||
provider: "fixed-provider",
|
||||
api: "openai-codex-responses",
|
||||
} as Model<any>,
|
||||
});
|
||||
expect(find).toHaveBeenCalledWith("fixed-provider", "fixed-model");
|
||||
expect(complete).toHaveBeenCalledTimes(1);
|
||||
expect((complete.mock.calls[0] as unknown[])[0]).toMatchObject({
|
||||
id: "fixed-model",
|
||||
provider: "fixed-provider",
|
||||
});
|
||||
expect(verdict).toEqual({ kind: "allow" });
|
||||
expect(reviews[0]?.details).toMatchObject({
|
||||
modelSource: "configured",
|
||||
provider: "fixed-provider",
|
||||
model: "fixed-model",
|
||||
});
|
||||
});
|
||||
|
||||
it("defers with judge_model_unavailable when a configured model cannot be resolved, without fallback", async () => {
|
||||
const { verdict, reviews, complete } = await runAsk({
|
||||
config: {
|
||||
version: 2,
|
||||
mode: "enforce",
|
||||
model: { provider: "ghost-provider", id: "ghost-model" },
|
||||
},
|
||||
findResult: undefined,
|
||||
});
|
||||
expect(verdict).toEqual({ kind: "defer" });
|
||||
expect(complete).not.toHaveBeenCalled();
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
details: expect.objectContaining({
|
||||
resultKind: "infrastructure_failure",
|
||||
code: "judge_model_unavailable",
|
||||
modelCalled: false,
|
||||
provider: "ghost-provider",
|
||||
model: "ghost-model",
|
||||
}),
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("keeps riskOverride on the early judge_model_unavailable row for a high-risk shadow ask", async () => {
|
||||
const { reviews } = await runAsk({
|
||||
config: {
|
||||
version: 2,
|
||||
mode: "shadow",
|
||||
model: { provider: "ghost-provider", id: "ghost-model" },
|
||||
},
|
||||
command: "git clean -xfd",
|
||||
findResult: undefined,
|
||||
});
|
||||
expect(reviews[0]?.details).toMatchObject({
|
||||
code: "judge_model_unavailable",
|
||||
riskOverride: { category: "data_loss", rule: expect.any(String) },
|
||||
});
|
||||
});
|
||||
|
||||
it("defers with judge_model_unavailable when a configured model lacks auth", async () => {
|
||||
const { verdict, reviews, complete } = await runAsk({
|
||||
config: {
|
||||
version: 2,
|
||||
mode: "enforce",
|
||||
model: { provider: "p", id: "m" },
|
||||
},
|
||||
findResult: { id: "m", provider: "p", api: "openai-codex-responses" } as Model<any>,
|
||||
authConfigured: false,
|
||||
});
|
||||
expect(verdict).toEqual({ kind: "defer" });
|
||||
expect(complete).not.toHaveBeenCalled();
|
||||
expect(reviews[0]?.details).toMatchObject({
|
||||
resultKind: "infrastructure_failure",
|
||||
code: "judge_model_unavailable",
|
||||
});
|
||||
});
|
||||
|
||||
it("notifies once per session in v2 enforce mode with the judge model and risk contract", async () => {
|
||||
let authorize: Authorizer["authorize"] | undefined;
|
||||
const service = {
|
||||
registerAuthorizer: vi.fn((_name, callback) => {
|
||||
authorize = callback;
|
||||
return vi.fn();
|
||||
}),
|
||||
checkPermission: vi.fn(),
|
||||
getToolPermission: vi.fn(),
|
||||
} as unknown as PermissionsService;
|
||||
publishPermissionsService(service);
|
||||
publishedService = service;
|
||||
|
||||
const complete = vi.fn(async () => modelResponse());
|
||||
const notify = vi.fn();
|
||||
const ctx = {
|
||||
hasUI: true,
|
||||
sessionManager: fakeSessionManager(),
|
||||
model: {
|
||||
id: "session-model",
|
||||
provider: "session-provider",
|
||||
api: "openai-codex-responses",
|
||||
} as Model<any>,
|
||||
modelRegistry: { complete, find: vi.fn(), hasConfiguredAuth: vi.fn(() => true) },
|
||||
ui: { notify },
|
||||
} as unknown as ExtensionContext;
|
||||
|
||||
const harness = createFakePi();
|
||||
extension(harness.pi);
|
||||
writeConfig({ version: 2, mode: "enforce" });
|
||||
harness.start(ctx);
|
||||
harness.ready();
|
||||
|
||||
const enforceNotice = notify.mock.calls.filter((call) =>
|
||||
/Enforce/i.test(String(call[0])),
|
||||
);
|
||||
expect(enforceNotice).toHaveLength(1);
|
||||
expect(String(enforceNotice[0]?.[0])).toContain(
|
||||
"session-provider/session-model",
|
||||
);
|
||||
expect(String(enforceNotice[0]?.[0])).toMatch(/risk/i);
|
||||
|
||||
// Two more asks must not repeat the notification.
|
||||
for (let i = 0; i < 2; i += 1) {
|
||||
await authorize!(ask(), { checkPermission: vi.fn(), getToolPermission: vi.fn() }, {
|
||||
review: vi.fn(),
|
||||
debug: vi.fn(),
|
||||
});
|
||||
}
|
||||
expect(
|
||||
notify.mock.calls.filter((call) => /Enforce/i.test(String(call[0]))),
|
||||
).toHaveLength(1);
|
||||
harness.shutdown();
|
||||
});
|
||||
|
||||
it("does not show the enforce notification in shadow mode", async () => {
|
||||
const { notify } = await runAsk({
|
||||
config: { version: 2, mode: "shadow" },
|
||||
});
|
||||
expect(
|
||||
notify.mock.calls.filter((call) => /Enforce/i.test(String(call[0]))),
|
||||
).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,260 +0,0 @@
|
||||
import { mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { dirname, join } from "node:path";
|
||||
import { afterEach, beforeEach, describe, expect, it } from "vitest";
|
||||
import {
|
||||
appendPromotionRecord,
|
||||
loadPromotionRecords,
|
||||
promotionRecordsPath,
|
||||
resolvePromotionGates,
|
||||
type CandidateIdentity,
|
||||
type PromotionRecord,
|
||||
} from "../src/promotion";
|
||||
|
||||
const IDENTITY: CandidateIdentity = {
|
||||
judge: "@sikongjueluo/pi-permission-ai-judge@0.0.1",
|
||||
permissionSystem: "25.4.0",
|
||||
provider: "openai-codex",
|
||||
model: "gpt-5.6-sol",
|
||||
api: "openai-codex-responses",
|
||||
promptVersion: "bash-shadow-v4",
|
||||
toolSchemaVersion: "report-verdict-v1",
|
||||
reviewSchemaVersion: "1",
|
||||
timeoutCohort: 30000,
|
||||
};
|
||||
|
||||
function record(
|
||||
kind: PromotionRecord["kind"],
|
||||
identity: CandidateIdentity = IDENTITY,
|
||||
basis = "test basis",
|
||||
): PromotionRecord {
|
||||
return {
|
||||
kind,
|
||||
candidateIdentity: identity,
|
||||
recordedAt: "2026-08-20T12:00:00Z",
|
||||
basis,
|
||||
};
|
||||
}
|
||||
|
||||
function line(value: unknown): string {
|
||||
return `${JSON.stringify(value)}\n`;
|
||||
}
|
||||
|
||||
/** writeFileSync, but creating the records directory first. */
|
||||
function writeRecords(dir: string, content: string): void {
|
||||
const path = promotionRecordsPath(dir);
|
||||
mkdirSync(dirname(path), { recursive: true });
|
||||
writeFileSync(path, content);
|
||||
}
|
||||
|
||||
describe("promotion records — loadPromotionRecords", () => {
|
||||
let dir: string;
|
||||
beforeEach(() => {
|
||||
dir = mkdtempSync(join(tmpdir(), "ai-judge-promotion-"));
|
||||
});
|
||||
afterEach(() => {
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it("treats a missing file as the healthy pre-promotion state", () => {
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(snapshot).toEqual({
|
||||
records: [],
|
||||
healthy: true,
|
||||
diagnostic: null,
|
||||
path: promotionRecordsPath(dir),
|
||||
});
|
||||
});
|
||||
|
||||
it("parses well-formed records of every kind", () => {
|
||||
writeRecords(
|
||||
dir,
|
||||
[record("cohort_qualified"), record("owner_approval"), record("activation")]
|
||||
.map(line)
|
||||
.join(""),
|
||||
);
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(snapshot.healthy).toBe(true);
|
||||
expect(snapshot.records).toHaveLength(3);
|
||||
expect(snapshot.records.map((r) => r.kind)).toEqual([
|
||||
"cohort_qualified",
|
||||
"owner_approval",
|
||||
"activation",
|
||||
]);
|
||||
});
|
||||
|
||||
it("fails closed on malformed JSON lines", () => {
|
||||
writeRecords(
|
||||
dir,
|
||||
line(record("cohort_qualified")) + "{not json\n",
|
||||
);
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(snapshot.healthy).toBe(false);
|
||||
expect(snapshot.records).toEqual([]);
|
||||
expect(snapshot.diagnostic).toContain("malformed");
|
||||
});
|
||||
|
||||
it("fails closed on shape-invalid records", () => {
|
||||
writeRecords(
|
||||
dir,
|
||||
line({ kind: "activation" }), // missing identity/basis/recordedAt
|
||||
);
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(snapshot.healthy).toBe(false);
|
||||
expect(snapshot.diagnostic).toContain("malformed");
|
||||
});
|
||||
|
||||
it("skips blank lines without failing", () => {
|
||||
writeRecords(
|
||||
dir,
|
||||
"\n" + line(record("activation")) + "\n\n",
|
||||
);
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(snapshot.healthy).toBe(true);
|
||||
expect(snapshot.records).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("promotion records — resolvePromotionGates", () => {
|
||||
let dir: string;
|
||||
beforeEach(() => {
|
||||
dir = mkdtempSync(join(tmpdir(), "ai-judge-promotion-"));
|
||||
});
|
||||
afterEach(() => {
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it("flips only its own gate per record kind", () => {
|
||||
writeRecords(dir, line(record("owner_approval")));
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(resolvePromotionGates(snapshot, IDENTITY)).toEqual({
|
||||
cohortQualified: false,
|
||||
ownerApprovalRecorded: true,
|
||||
activationRecorded: false,
|
||||
});
|
||||
});
|
||||
|
||||
it("all three gates open with all three records for the exact identity", () => {
|
||||
writeRecords(
|
||||
dir,
|
||||
[
|
||||
record("cohort_qualified", IDENTITY, "cohort id x"),
|
||||
record("owner_approval", IDENTITY, "approved"),
|
||||
record("activation", IDENTITY, "activated"),
|
||||
]
|
||||
.map(line)
|
||||
.join(""),
|
||||
);
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(resolvePromotionGates(snapshot, IDENTITY)).toEqual({
|
||||
cohortQualified: true,
|
||||
ownerApprovalRecorded: true,
|
||||
activationRecorded: true,
|
||||
});
|
||||
});
|
||||
|
||||
const identityDrifts: ReadonlyArray<[string, Partial<CandidateIdentity>]> = [
|
||||
["provider", { provider: "other-provider" }],
|
||||
["model", { model: "gpt-5.7" }],
|
||||
["api", { api: "openai-responses" }],
|
||||
["promptVersion", { promptVersion: "bash-shadow-v5" }],
|
||||
["toolSchemaVersion", { toolSchemaVersion: "report-verdict-v2" }],
|
||||
["reviewSchemaVersion", { reviewSchemaVersion: "2" }],
|
||||
["timeoutCohort", { timeoutCohort: "default" }],
|
||||
["permissionSystem", { permissionSystem: "25.5.0" }],
|
||||
["judge package", { judge: "@sikongjueluo/pi-permission-ai-judge@0.0.2" }],
|
||||
];
|
||||
for (const [name, patch] of identityDrifts) {
|
||||
it(`keeps every gate closed when the live identity drifts on ${name}`, () => {
|
||||
writeRecords(
|
||||
dir,
|
||||
[
|
||||
record("cohort_qualified"),
|
||||
record("owner_approval"),
|
||||
record("activation"),
|
||||
]
|
||||
.map(line)
|
||||
.join(""),
|
||||
);
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(
|
||||
resolvePromotionGates(snapshot, { ...IDENTITY, ...patch }),
|
||||
).toEqual({
|
||||
cohortQualified: false,
|
||||
ownerApprovalRecorded: false,
|
||||
activationRecorded: false,
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
it("closes every gate when the snapshot is unhealthy", () => {
|
||||
writeRecords(dir, "garbage\n");
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(snapshot.healthy).toBe(false);
|
||||
expect(resolvePromotionGates(snapshot, IDENTITY)).toEqual({
|
||||
cohortQualified: false,
|
||||
ownerApprovalRecorded: false,
|
||||
activationRecorded: false,
|
||||
});
|
||||
});
|
||||
|
||||
it("leaves other-identity records inert, not malformed", () => {
|
||||
const other: CandidateIdentity = {
|
||||
...IDENTITY,
|
||||
model: "glm-5.2",
|
||||
};
|
||||
writeRecords(
|
||||
dir,
|
||||
[
|
||||
record("cohort_qualified", other),
|
||||
record("activation", other),
|
||||
]
|
||||
.map(line)
|
||||
.join(""),
|
||||
);
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(snapshot.healthy).toBe(true);
|
||||
expect(resolvePromotionGates(snapshot, IDENTITY)).toEqual({
|
||||
cohortQualified: false,
|
||||
ownerApprovalRecorded: false,
|
||||
activationRecorded: false,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("promotion records — appendPromotionRecord", () => {
|
||||
let dir: string;
|
||||
beforeEach(() => {
|
||||
dir = mkdtempSync(join(tmpdir(), "ai-judge-promotion-"));
|
||||
});
|
||||
afterEach(() => {
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it("appends a shape-valid record that round-trips through the loader", () => {
|
||||
const error = appendPromotionRecord({
|
||||
agentDir: dir,
|
||||
record: record("owner_approval", IDENTITY, "approved v4"),
|
||||
now: () => "2026-08-21T09:00:00Z",
|
||||
});
|
||||
expect(error).toBeNull();
|
||||
const raw = readFileSync(promotionRecordsPath(dir), "utf-8");
|
||||
expect(raw).toContain('"recordedAt":"2026-08-21T09:00:00Z"');
|
||||
expect(raw).toContain('"basis":"approved v4"');
|
||||
const snapshot = loadPromotionRecords({ agentDir: dir });
|
||||
expect(snapshot.healthy).toBe(true);
|
||||
expect(resolvePromotionGates(snapshot, IDENTITY).ownerApprovalRecorded).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects a shape-invalid record without touching the file", () => {
|
||||
const bad = {
|
||||
kind: "activation",
|
||||
candidateIdentity: { judge: "x" },
|
||||
recordedAt: "",
|
||||
basis: "",
|
||||
} as unknown as PromotionRecord;
|
||||
const error = appendPromotionRecord({ agentDir: dir, record: bad });
|
||||
expect(error).toBe("record is not shape-valid");
|
||||
expect(loadPromotionRecords({ agentDir: dir }).records).toEqual([]);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user