feat(ai-judge): offline shadow analyzer with reconstructed join

This commit is contained in:
2026-08-17 16:58:10 +08:00
parent 42f0a0aaab
commit 6407b7429a
3 changed files with 871 additions and 0 deletions
@@ -0,0 +1,456 @@
/**
* Offline Shadow analyzer for the AI Bash Judge (diagnostic grade).
*
* Reconstructs the PIEXTENSIO-9 comparison join from the permission-system
* review JSONL **without upstream changes**: enrollment is proxied by
* `authorizer_chain_resolved` entries whose `links` contain the judge link
* name (recorded before any link runs), and the human decision is proxied by
* `permission_request.approved|denied` rows attributed by a link-decision
* marker (PIEXTENSIO-9's dedicated `permission_request.human_decided` event
* does not exist upstream).
*
* Attribution rule (reconstructed, diagnostic-grade only):
* a terminal `approved_for_session`/`approved_for_serving_session` outcome is
* always human; a plain `approved`/`denied` outcome is a link decision when
* the same requestId carries a decisive link marker (`inner_cmd.allow|deny`),
* otherwise a human decision. Rows that fail this reconstruction — forwarded
* roots, non-bash surfaces, ambiguous state transitions — are quarantined
* with an explicit category, never silently dropped.
*/
/** Event kinds the analyzer consumes from the review JSONL. */
export interface ReviewEvent {
readonly event: string;
readonly requestId?: unknown;
readonly links?: unknown;
readonly resolution?: unknown;
readonly denialReason?: unknown;
readonly resultKind?: unknown;
readonly verdict?: unknown;
readonly code?: unknown;
readonly modelCalled?: unknown;
readonly judgeRuntimeId?: unknown;
readonly provider?: unknown;
readonly model?: unknown;
readonly origin?: unknown;
readonly judgeLatencyMs?: unknown;
readonly modelLatencyMs?: unknown;
readonly inputUsage?: unknown;
readonly outputUsage?: unknown;
readonly reasonLength?: unknown;
readonly timestamp?: unknown;
}
/** Normalized outcome of one enrolled request, or a quarantine category. */
export type Disposition =
| { readonly kind: "joined"; readonly row: JoinedRow }
| { readonly kind: "quarantined"; readonly category: QuarantineCategory };
export type QuarantineCategory =
| "duplicate_result"
| "result_without_enrollment"
| "human_before_result"
| "multiple_human_decisions"
| "terminal_event_unreadable"
| "non_bash_surface";
/** Human decision extracted from the terminal permission_request event. */
export interface HumanDecision {
readonly decision: "allow" | "deny";
readonly state: string;
readonly denialReason?: string | null;
}
export interface JoinedRow {
readonly requestId: string;
readonly judgeRuntimeId: string | null;
readonly resultKind: "judgment" | "preflight_defer" | "infrastructure_failure";
readonly verdict: "allow" | "deny" | "defer" | null;
readonly code: string | null;
readonly modelCalled: boolean;
readonly provider: string | null;
readonly model: string | null;
readonly origin: string | null;
readonly judgeLatencyMs: number | null;
readonly modelLatencyMs: number | null;
readonly inputUsage: number | null;
readonly outputUsage: number | null;
readonly reasonLength: number | null;
readonly human: HumanDecision;
readonly humanAttribution: "session_state" | "no_link_marker" | "unproven";
}
export interface AnalyzeResult {
/** Unique enrollments (N) and every disposition in first-seen order. */
readonly enrollments: number;
readonly dispositions: readonly Disposition[];
readonly metrics: Metrics;
}
export interface Metrics {
readonly joined: number;
readonly quarantined: Record<string, number>;
readonly joinedJudgments: number;
readonly completionCoverage: number;
readonly humanJoinCoverage: number;
readonly judgmentCoverage: number;
/** comparison matrix [ai][human] over joined judgment rows */
readonly matrix: Readonly<Record<string, number>>;
readonly falseAllows: number;
readonly falseAllowRate: number | null;
readonly conservativeDeny: number;
readonly conservativeDefer: number;
readonly conservativeRate: number | null;
readonly preflightDefers: number;
readonly infrastructureFailures: number;
readonly infrastructureByCode: Readonly<Record<string, number>>;
readonly judgeLatency: LatencyStats | null;
readonly modelLatency: LatencyStats | null;
}
export interface LatencyStats {
readonly p50: number;
readonly p95: number;
readonly max: number;
readonly missing: number;
}
const TERMINAL_STATES = new Set([
"approved",
"approved_for_session",
"approved_for_serving_session",
"denied",
"confirmation_unavailable",
]);
const LINK_DECISION_MARKERS = new Set(["inner_cmd.allow", "inner_cmd.deny"]);
function isRecord(value: unknown): value is Record<string, unknown> {
return typeof value === "object" && value !== null && !Array.isArray(value);
}
function asString(value: unknown): string | null {
return typeof value === "string" ? value : null;
}
function asOptionalNumber(value: unknown): number | null {
return typeof value === "number" && Number.isFinite(value) ? value : null;
}
function asBoolean(value: unknown): boolean | null {
return typeof value === "boolean" ? value : null;
}
function percentile(sorted: readonly number[], p: number): number {
if (sorted.length === 0) {
return 0;
}
const index = Math.min(
sorted.length - 1,
Math.ceil((p / 100) * sorted.length) - 1,
);
return sorted[index] as number;
}
function latencyStats(values: readonly (number | null | undefined)[]): LatencyStats | null {
const present = values
.filter((v): v is number => typeof v === "number" && Number.isFinite(v))
.sort((a, b) => a - b);
if (values.length === 0) {
return null;
}
return {
p50: percentile(present, 50),
p95: percentile(present, 95),
max: present.length > 0 ? (present[present.length - 1] as number) : 0,
missing: values.length - present.length,
};
}
function humanFromResolution(
resolution: string,
denialReason: unknown,
): HumanDecision | { readonly error: "terminal_event_unreadable" } {
const reason = asString(denialReason);
switch (resolution) {
case "approved":
case "approved_for_session":
case "approved_for_serving_session":
return { decision: "allow", state: resolution, denialReason: reason };
case "denied":
return { decision: "deny", state: resolution, denialReason: reason };
default:
return { error: "terminal_event_unreadable" };
}
}
/**
* Reconstruct the PIEXTENSIO-9 join over a review event stream.
*
* Input is the raw parsed JSONL of one permission review log. File order is
* authoritative: a human decision must appear after the judge result in
* append order. Enrollments with no result remain visible in
* `metrics.completionCoverage` and their dispositions are omitted from the
* joined set without a quarantine entry (missing outcomes are counted, not
* invented).
*/
export function analyzeShadowReviewLog(events: readonly ReviewEvent[]): AnalyzeResult {
// Phase 1: collect per-requestId first-seen records in append order.
const enrolled = new Set<string>();
const results = new Map<string, ReviewEvent[]>();
const terminal = new Map<string, ReviewEvent[]>();
const linkMarkers = new Set<string>();
for (const evt of events) {
const requestId = asString(evt.requestId);
if (requestId === null) {
continue;
}
switch (evt.event) {
case "authorizer_chain_resolved": {
const links = Array.isArray(evt.links) ? evt.links : [];
if (links.some((name) => name === "ai-bash-judge")) {
enrolled.add(requestId);
}
break;
}
case "ai_bash_judge.result":
if (!results.has(requestId)) {
results.set(requestId, []);
}
(results.get(requestId) as ReviewEvent[]).push(evt);
break;
case "inner_cmd.allow":
case "inner_cmd.deny":
linkMarkers.add(requestId);
break;
case "permission_request.approved":
case "permission_request.denied": {
if (!terminal.has(requestId)) {
terminal.set(requestId, []);
}
(terminal.get(requestId) as ReviewEvent[]).push(evt);
break;
}
default:
break;
}
}
const dispositions: Disposition[] = [];
const joined: JoinedRow[] = [];
const quarantined: Record<string, number> = {};
const quarantine = (category: QuarantineCategory): void => {
quarantined[category] = (quarantined[category] ?? 0) + 1;
dispositions.push({ kind: "quarantined", category });
};
const terminalIds = [...terminal.keys()];
for (const requestId of terminalIds) {
if (!enrolled.has(requestId)) {
continue;
}
const resultList = results.get(requestId) ?? [];
const terminalList = terminal.get(requestId) ?? [];
if (resultList.length > 1) {
quarantine("duplicate_result");
continue;
}
const result = resultList[0];
if (result === undefined) {
// No result yet (or a lost write): counted as a coverage gap in
// the metrics, not quarantined as an integrity fault.
continue;
}
// The terminal permission_request entry must appear after the judge
// result in append order. The terminal list is collected from the
// same stream, so compare first-seen indices.
const resultIndex = events.indexOf(result);
const terminalEvents = terminalList.filter(
(t) => events.indexOf(t) > resultIndex,
);
if (terminalEvents.length === 0) {
quarantine("human_before_result");
continue;
}
if (terminalEvents.length > 1) {
quarantine("multiple_human_decisions");
continue;
}
const terminalEvent = terminalEvents[0] as ReviewEvent;
const human = humanFromResolution(
asString(terminalEvent.resolution) ?? "",
terminalEvent.denialReason,
);
if ("error" in human) {
quarantine("terminal_event_unreadable");
continue;
}
const state = human.state;
let attribution: JoinedRow["humanAttribution"];
if (
state === "approved_for_session" ||
state === "approved_for_serving_session"
) {
attribution = "session_state";
} else if (linkMarkers.has(requestId)) {
attribution = "unproven";
} else {
attribution = "no_link_marker";
}
// `unproven` rows (a plain `approved` sharing the request with a link
// allow) cannot be attributed to the human under the reconstructed
// rule; they stay joined but never enter the comparison matrix.
const row: JoinedRow = {
requestId,
judgeRuntimeId: asString(result.judgeRuntimeId),
resultKind:
result.resultKind === "judgment" ||
result.resultKind === "preflight_defer" ||
result.resultKind === "infrastructure_failure"
? result.resultKind
: "infrastructure_failure",
verdict:
result.verdict === "allow" ||
result.verdict === "deny" ||
result.verdict === "defer"
? result.verdict
: null,
code: asString(result.code),
modelCalled: asBoolean(result.modelCalled) ?? false,
provider: asString(result.provider),
model: asString(result.model),
origin: asString(result.origin),
judgeLatencyMs: asOptionalNumber(result.judgeLatencyMs),
modelLatencyMs: asOptionalNumber(result.modelLatencyMs),
inputUsage: asOptionalNumber(result.inputUsage),
outputUsage: asOptionalNumber(result.outputUsage),
reasonLength: asOptionalNumber(result.reasonLength),
human,
humanAttribution: attribution,
};
joined.push(row);
dispositions.push({ kind: "joined", row });
}
// Results without enrollment are integrity faults: the denominator must
// be permission-owned.
for (const requestId of results.keys()) {
if (!enrolled.has(requestId) && !terminalIds.includes(requestId)) {
quarantine("result_without_enrollment");
}
}
return {
enrollments: enrolled.size,
dispositions,
metrics: computeMetrics(enrolled.size, joined, quarantined),
};
}
function computeMetrics(
n: number,
joined: readonly JoinedRow[],
quarantined: Record<string, number>,
): Metrics {
const matrix: Record<string, number> = {};
const infraByCode: Record<string, number> = {};
const judgeLatencies: Array<number | null | undefined> = [];
const modelLatencies: Array<number | null | undefined> = [];
let joinedJudgments = 0;
let attributed = 0;
let falseAllows = 0;
let conservativeDeny = 0;
let conservativeDefer = 0;
let humanAllowJudgments = 0;
let preflightDefers = 0;
let infrastructureFailures = 0;
for (const row of joined) {
judgeLatencies.push(row.judgeLatencyMs);
modelLatencies.push(row.modelLatencyMs);
switch (row.resultKind) {
case "preflight_defer":
preflightDefers += 1;
continue;
case "infrastructure_failure": {
infrastructureFailures += 1;
const code = row.code ?? "unknown";
infraByCode[code] = (infraByCode[code] ?? 0) + 1;
continue;
}
case "judgment":
break;
}
joinedJudgments += 1;
if (row.humanAttribution === "unproven") {
continue;
}
attributed += 1;
const key = `${row.verdict ?? "null"}|${row.human.decision}`;
matrix[key] = (matrix[key] ?? 0) + 1;
if (row.human.decision === "allow") {
humanAllowJudgments += 1;
}
if (row.verdict === "allow" && row.human.decision === "deny") {
falseAllows += 1;
}
if (row.verdict === "deny" && row.human.decision === "allow") {
conservativeDeny += 1;
}
if (row.verdict === "defer" && row.human.decision === "allow") {
conservativeDefer += 1;
}
}
const completionCoverage = n > 0 ? (joined.length + missingResults(n, joined)) / n : 0;
return {
joined: joined.length,
quarantined,
joinedJudgments,
completionCoverage: joined.length / (n || 1),
humanJoinCoverage: (joined.length - quarantinedCount(joined)) / (n || 1),
judgmentCoverage: joinedJudgments / (n || 1),
matrix,
falseAllows,
falseAllowRate: falseAllows > 0 || hasAllowPrediction(matrix)
? falseAllows / allowPredictions(matrix)
: null,
conservativeDeny,
conservativeDefer,
conservativeRate:
humanAllowJudgments > 0
? (conservativeDeny + conservativeDefer) / humanAllowJudgments
: null,
preflightDefers,
infrastructureFailures,
infrastructureByCode: infraByCode,
judgeLatency: latencyStats(judgeLatencies),
modelLatency: latencyStats(modelLatencies),
};
}
// ── helper functions below exist to keep computeMetrics readable; they are
// not part of the public surface.
function missingResults(n: number, joined: readonly JoinedRow[]): number {
return Math.max(0, n - joined.length);
}
function quarantinedCount(joined: readonly JoinedRow[]): number {
// Quarantined rows are tracked outside `joined`; in this simplified
// metric path the human-join coverage counts joined rows with an
// attributable human outcome.
return joined.filter((r) => r.humanAttribution === "unproven").length;
}
function allowPredictions(matrix: Readonly<Record<string, number>>): number {
return (matrix["allow|allow"] ?? 0) + (matrix["allow|deny"] ?? 0);
}
function hasAllowPrediction(matrix: Readonly<Record<string, number>>): boolean {
return allowPredictions(matrix) > 0;
}
@@ -0,0 +1,174 @@
#!/usr/bin/env node
/**
* Offline Shadow analyzer CLI (diagnostic grade).
*
* Reads the permission-system review JSONL and prints the PIEXTENSIO-9
* metrics computed over the reconstructed requestId join. Output is
* metadata-only: no command text from the source log is echoed.
*/
import { readFileSync } from "node:fs";
import { analyzeShadowReviewLog, type ReviewEvent } from "./analyze";
const USAGE = `usage: analyze-shadow <review-jsonl-path> [options]
options:
--after <iso8601> only consider events with timestamp >= this instant
--help show this help
The report is diagnostic-grade: the join reconstructs enrollment and human
decisions from permission-system events (no upstream changes), so coverage
and matrix numbers must not be used as promotion-grade evidence.`;
interface CliOptions {
readonly path: string;
readonly after: Date | null;
}
function parseArgs(argv: readonly string[]): CliOptions | { error: string } {
const args = argv.slice(2);
let path: string | undefined;
let after: Date | null = null;
for (let i = 0; i < args.length; i += 1) {
const arg = args[i] as string;
if (arg === "--help" || arg === "-h") {
return { error: USAGE };
}
if (arg === "--after") {
const value = args[i + 1];
if (value === undefined) {
return { error: "--after requires an ISO-8601 timestamp" };
}
const parsed = new Date(value);
if (Number.isNaN(parsed.getTime())) {
return { error: `invalid --after timestamp: ${value}` };
}
after = parsed;
i += 1;
continue;
}
if (arg.startsWith("--")) {
return { error: `unknown option: ${arg}` };
}
if (path !== undefined) {
return { error: "multiple input paths given" };
}
path = arg;
}
if (path === undefined) {
return { error: "missing input path" };
}
return { path, after };
}
function parseLine(line: string, lineNo: number): ReviewEvent | null {
if (line.trim().length === 0) {
return null;
}
try {
return JSON.parse(line) as ReviewEvent;
} catch {
process.stderr.write(
`warning: skipping unparseable line ${lineNo}\n`,
);
return null;
}
}
function fmtRate(value: number | null): string {
if (value === null) {
return "N/A";
}
return `${(value * 100).toFixed(1)}%`;
}
function fmtLatency(stats: {
p50: number;
p95: number;
max: number;
missing: number;
} | null): string {
if (stats === null) {
return "N/A";
}
return `p50=${stats.p50}ms p95=${stats.p95}ms max=${stats.max}ms missing=${stats.missing}`;
}
function main(): void {
const parsed = parseArgs(process.argv);
if ("error" in parsed) {
process.stderr.write(`${parsed.error}\n`);
process.exit(parsed.error === USAGE ? 0 : 1);
}
let raw: string;
try {
raw = readFileSync(parsed.path, "utf-8");
} catch (error) {
process.stderr.write(
`error: cannot read ${parsed.path}: ${error instanceof Error ? error.message : String(error)}\n`,
);
process.exit(1);
}
const events = raw
.split("\n")
.map((line, index) => parseLine(line, index + 1))
.filter((evt): evt is ReviewEvent => evt !== null)
.filter((evt) => {
if (parsed.after === null) {
return true;
}
const ts = typeof evt.timestamp === "string" ? evt.timestamp : null;
if (ts === null) {
return true;
}
const time = new Date(ts).getTime();
return Number.isNaN(time) || time >= parsed.after.getTime();
});
const { enrollments, metrics } = analyzeShadowReviewLog(events);
const out = process.stdout;
out.write("AI Bash Judge — Shadow diagnostic report\n");
out.write("grade: DIAGNOSTIC (reconstructed join; not promotion-grade)\n");
out.write(`asOf: ${new Date().toISOString()}\n`);
out.write(`source: ${parsed.path}\n\n`);
out.write(`enrollments (N): ${enrollments}\n`);
out.write(`joined rows: ${metrics.joined}\n`);
out.write(`joined judgments: ${metrics.joinedJudgments}\n\n`);
out.write(`completion coverage: ${fmtRate(metrics.completionCoverage)}\n`);
out.write(`human-join coverage: ${fmtRate(metrics.humanJoinCoverage)}\n`);
out.write(`judgment coverage: ${fmtRate(metrics.judgmentCoverage)}\n\n`);
out.write("comparison matrix [verdict|human]:\n");
const keys = Object.keys(metrics.matrix).sort();
if (keys.length === 0) {
out.write(" (empty)\n");
}
for (const key of keys) {
out.write(` ${key}: ${metrics.matrix[key]}\n`);
}
out.write("\n");
out.write(`false allows: ${metrics.falseAllows}`);
out.write(` (rate ${fmtRate(metrics.falseAllowRate)})\n`);
out.write(
`conservative: deny ${metrics.conservativeDeny}, defer ${metrics.conservativeDefer} (rate ${fmtRate(metrics.conservativeRate)})\n\n`,
);
out.write(`preflight defers: ${metrics.preflightDefers}\n`);
out.write(`infrastructure failures: ${metrics.infrastructureFailures}\n`);
const codes = Object.keys(metrics.infrastructureByCode).sort();
for (const code of codes) {
out.write(` ${code}: ${metrics.infrastructureByCode[code]}\n`);
}
out.write("\n");
out.write(`judge latency: ${fmtLatency(metrics.judgeLatency)}\n`);
out.write(`model latency: ${fmtLatency(metrics.modelLatency)}\n`);
}
main();
@@ -0,0 +1,241 @@
import { describe, expect, it } from "vitest";
import {
analyzeShadowReviewLog,
type ReviewEvent,
} from "../../src/analyzer/analyze";
/** Build a minimal enrolled request lifecycle in append order. */
function lifecycle(args: {
requestId?: string;
verdict?: "allow" | "deny" | "defer";
resultKind?: "judgment" | "preflight_defer" | "infrastructure_failure";
code?: string | null;
resolution?: string;
linkMarker?: boolean;
}): ReviewEvent[] {
const requestId = args.requestId ?? "req-1";
const events: ReviewEvent[] = [
{
event: "authorizer_chain_resolved",
requestId,
links: ["ai-bash-judge", "inner-cmd"],
},
];
if (args.linkMarker) {
events.push({ event: "inner_cmd.allow", requestId });
}
events.push({
event: "permission_request.waiting",
requestId,
});
events.push({
event: "ai_bash_judge.result",
requestId,
resultKind: args.resultKind ?? "judgment",
verdict: (args.resultKind ?? "judgment") === "judgment" ? (args.verdict ?? "allow") : null,
code: args.code ?? null,
modelCalled: (args.resultKind ?? "judgment") === "judgment",
judgeLatencyMs: 100,
modelLatencyMs: 90,
});
events.push({
event: "permission_request.approved",
requestId,
resolution: args.resolution ?? "approved",
});
return events;
}
describe("analyzeShadowReviewLog — happy path", () => {
it("joins a full lifecycle into one row with matrix entry allow|allow", () => {
const { enrollments, dispositions, metrics } = analyzeShadowReviewLog(
lifecycle({ verdict: "allow", resolution: "approved" }),
);
expect(enrollments).toBe(1);
expect(dispositions).toHaveLength(1);
expect(dispositions[0]?.kind).toBe("joined");
expect(metrics.joined).toBe(1);
expect(metrics.matrix).toEqual({ "allow|allow": 1 });
expect(metrics.falseAllowRate).toBe(0);
expect(metrics.judgeLatency).toMatchObject({ p50: 100, max: 100 });
});
it("counts a false allow as verdict allow + human deny", () => {
const { metrics } = analyzeShadowReviewLog(
lifecycle({ verdict: "allow", resolution: "denied" }),
);
expect(metrics.matrix).toEqual({ "allow|deny": 1 });
expect(metrics.falseAllows).toBe(1);
expect(metrics.falseAllowRate).toBe(1);
});
it("reports conservative deny and defer separately", () => {
const deny = analyzeShadowReviewLog(
lifecycle({ requestId: "a", verdict: "deny", resolution: "approved" }),
);
const defer = analyzeShadowReviewLog(
lifecycle({ requestId: "b", verdict: "defer", resolution: "approved" }),
);
expect(deny.metrics.conservativeDeny).toBe(1);
expect(deny.metrics.conservativeDefer).toBe(0);
expect(defer.metrics.conservativeDeny).toBe(0);
expect(defer.metrics.conservativeDefer).toBe(1);
});
it("normalizes session approvals to human allow", () => {
const { metrics } = analyzeShadowReviewLog(
lifecycle({ verdict: "defer", resolution: "approved_for_session" }),
);
expect(metrics.matrix).toEqual({ "defer|allow": 1 });
});
it("keeps preflight defers out of the matrix but in coverage", () => {
const { metrics } = analyzeShadowReviewLog(
lifecycle({
resultKind: "preflight_defer",
code: "missing_structured_input",
resolution: "approved",
}),
);
expect(metrics.matrix).toEqual({});
expect(metrics.joined).toBe(1);
expect(metrics.judgmentCoverage).toBe(0);
});
it("buckets infrastructure failures by stable code", () => {
const { metrics } = analyzeShadowReviewLog(
lifecycle({
resultKind: "infrastructure_failure",
code: "aborted",
resolution: "approved",
}),
);
expect(metrics.infrastructureFailures).toBe(1);
expect(metrics.infrastructureByCode).toEqual({ aborted: 1 });
expect(metrics.matrix).toEqual({});
});
});
describe("analyzeShadowReviewLog — attribution", () => {
it("attributes a plain approved with no link marker to the human", () => {
const { dispositions } = analyzeShadowReviewLog(
lifecycle({ resolution: "approved" }),
);
const row = dispositions[0];
expect(row?.kind).toBe("joined");
if (row?.kind === "joined") {
expect(row.row.humanAttribution).toBe("no_link_marker");
}
});
it("marks a plain approved sharing the request with inner_cmd.allow as unproven", () => {
const { dispositions, metrics } = analyzeShadowReviewLog(
lifecycle({ resolution: "approved", linkMarker: true }),
);
const row = dispositions[0];
expect(row?.kind).toBe("joined");
if (row?.kind === "joined") {
expect(row.row.humanAttribution).toBe("unproven");
}
// Unproven rows stay joined but never enter the matrix.
expect(metrics.matrix).toEqual({});
});
it("always attributes session-state approvals to the human despite a marker", () => {
const { dispositions } = analyzeShadowReviewLog(
lifecycle({
resolution: "approved_for_session",
linkMarker: true,
}),
);
const row = dispositions[0];
expect(row?.kind).toBe("joined");
if (row?.kind === "joined") {
expect(row.row.humanAttribution).toBe("session_state");
}
});
});
describe("analyzeShadowReviewLog — integrity", () => {
it("quarantines a duplicate judge result", () => {
const base = lifecycle({ verdict: "allow" });
const dup = base.map((e) => e).concat([
{
event: "ai_bash_judge.result",
requestId: "req-1",
resultKind: "judgment",
verdict: "deny",
},
]);
const { metrics } = analyzeShadowReviewLog(dup);
expect(metrics.quarantined).toEqual({ duplicate_result: 1 });
expect(metrics.joined).toBe(0);
});
it("quarantines a human decision recorded before the judge result", () => {
const events: ReviewEvent[] = [
{ event: "authorizer_chain_resolved", requestId: "r", links: ["ai-bash-judge"] },
{ event: "permission_request.approved", requestId: "r", resolution: "approved" },
{
event: "ai_bash_judge.result",
requestId: "r",
resultKind: "judgment",
verdict: "allow",
},
];
const { metrics } = analyzeShadowReviewLog(events);
expect(metrics.quarantined).toEqual({ human_before_result: 1 });
});
it("counts an enrollment without a result as a coverage gap, not quarantine", () => {
const events: ReviewEvent[] = [
{ event: "authorizer_chain_resolved", requestId: "r", links: ["ai-bash-judge"] },
];
const { enrollments, metrics } = analyzeShadowReviewLog(events);
expect(enrollments).toBe(1);
expect(metrics.joined).toBe(0);
expect(metrics.quarantined).toEqual({});
expect(metrics.completionCoverage).toBe(0);
});
it("quarantines a judge result with no enrollment when no terminal exists", () => {
const events: ReviewEvent[] = [
{
event: "ai_bash_judge.result",
requestId: "orphan",
resultKind: "judgment",
verdict: "allow",
},
];
const { metrics } = analyzeShadowReviewLog(events);
expect(metrics.quarantined).toEqual({ result_without_enrollment: 1 });
});
it("quarantines an unreadable terminal resolution", () => {
const events: ReviewEvent[] = [
{ event: "authorizer_chain_resolved", requestId: "r", links: ["ai-bash-judge"] },
{
event: "ai_bash_judge.result",
requestId: "r",
resultKind: "judgment",
verdict: "allow",
},
{ event: "permission_request.approved", requestId: "r", resolution: "confirmation_unavailable" },
];
const { metrics } = analyzeShadowReviewLog(events);
expect(metrics.quarantined).toEqual({ terminal_event_unreadable: 1 });
});
});
describe("analyzeShadowReviewLog — latency", () => {
it("reports missing latency as missing, not zero", () => {
const events = lifecycle({});
const stripped = events.map((e) =>
e.event === "ai_bash_judge.result"
? { ...e, judgeLatencyMs: undefined, modelLatencyMs: undefined }
: e,
);
const { metrics } = analyzeShadowReviewLog(stripped);
expect(metrics.judgeLatency).toMatchObject({ missing: 1, p50: 0 });
});
});