From 6407b7429a9f44c9105c47978e06c8bac8e64992 Mon Sep 17 00:00:00 2001 From: SikongJueluo Date: Mon, 17 Aug 2026 01:05:16 +0800 Subject: [PATCH] feat(ai-judge): offline shadow analyzer with reconstructed join --- .../src/analyzer/analyze.ts | 456 ++++++++++++++++++ .../src/analyzer/cli.ts | 174 +++++++ .../test/analyzer/analyze.test.ts | 241 +++++++++ 3 files changed, 871 insertions(+) create mode 100644 packages/pi-permission-ai-judge/src/analyzer/analyze.ts create mode 100644 packages/pi-permission-ai-judge/src/analyzer/cli.ts create mode 100644 packages/pi-permission-ai-judge/test/analyzer/analyze.test.ts diff --git a/packages/pi-permission-ai-judge/src/analyzer/analyze.ts b/packages/pi-permission-ai-judge/src/analyzer/analyze.ts new file mode 100644 index 0000000..d7408fb --- /dev/null +++ b/packages/pi-permission-ai-judge/src/analyzer/analyze.ts @@ -0,0 +1,456 @@ +/** + * Offline Shadow analyzer for the AI Bash Judge (diagnostic grade). + * + * Reconstructs the PIEXTENSIO-9 comparison join from the permission-system + * review JSONL **without upstream changes**: enrollment is proxied by + * `authorizer_chain_resolved` entries whose `links` contain the judge link + * name (recorded before any link runs), and the human decision is proxied by + * `permission_request.approved|denied` rows attributed by a link-decision + * marker (PIEXTENSIO-9's dedicated `permission_request.human_decided` event + * does not exist upstream). + * + * Attribution rule (reconstructed, diagnostic-grade only): + * a terminal `approved_for_session`/`approved_for_serving_session` outcome is + * always human; a plain `approved`/`denied` outcome is a link decision when + * the same requestId carries a decisive link marker (`inner_cmd.allow|deny`), + * otherwise a human decision. Rows that fail this reconstruction — forwarded + * roots, non-bash surfaces, ambiguous state transitions — are quarantined + * with an explicit category, never silently dropped. + */ + +/** Event kinds the analyzer consumes from the review JSONL. */ +export interface ReviewEvent { + readonly event: string; + readonly requestId?: unknown; + readonly links?: unknown; + readonly resolution?: unknown; + readonly denialReason?: unknown; + readonly resultKind?: unknown; + readonly verdict?: unknown; + readonly code?: unknown; + readonly modelCalled?: unknown; + readonly judgeRuntimeId?: unknown; + readonly provider?: unknown; + readonly model?: unknown; + readonly origin?: unknown; + readonly judgeLatencyMs?: unknown; + readonly modelLatencyMs?: unknown; + readonly inputUsage?: unknown; + readonly outputUsage?: unknown; + readonly reasonLength?: unknown; + readonly timestamp?: unknown; +} + +/** Normalized outcome of one enrolled request, or a quarantine category. */ +export type Disposition = + | { readonly kind: "joined"; readonly row: JoinedRow } + | { readonly kind: "quarantined"; readonly category: QuarantineCategory }; + +export type QuarantineCategory = + | "duplicate_result" + | "result_without_enrollment" + | "human_before_result" + | "multiple_human_decisions" + | "terminal_event_unreadable" + | "non_bash_surface"; + +/** Human decision extracted from the terminal permission_request event. */ +export interface HumanDecision { + readonly decision: "allow" | "deny"; + readonly state: string; + readonly denialReason?: string | null; +} + +export interface JoinedRow { + readonly requestId: string; + readonly judgeRuntimeId: string | null; + readonly resultKind: "judgment" | "preflight_defer" | "infrastructure_failure"; + readonly verdict: "allow" | "deny" | "defer" | null; + readonly code: string | null; + readonly modelCalled: boolean; + readonly provider: string | null; + readonly model: string | null; + readonly origin: string | null; + readonly judgeLatencyMs: number | null; + readonly modelLatencyMs: number | null; + readonly inputUsage: number | null; + readonly outputUsage: number | null; + readonly reasonLength: number | null; + readonly human: HumanDecision; + readonly humanAttribution: "session_state" | "no_link_marker" | "unproven"; +} + +export interface AnalyzeResult { + /** Unique enrollments (N) and every disposition in first-seen order. */ + readonly enrollments: number; + readonly dispositions: readonly Disposition[]; + readonly metrics: Metrics; +} + +export interface Metrics { + readonly joined: number; + readonly quarantined: Record; + readonly joinedJudgments: number; + readonly completionCoverage: number; + readonly humanJoinCoverage: number; + readonly judgmentCoverage: number; + /** comparison matrix [ai][human] over joined judgment rows */ + readonly matrix: Readonly>; + readonly falseAllows: number; + readonly falseAllowRate: number | null; + readonly conservativeDeny: number; + readonly conservativeDefer: number; + readonly conservativeRate: number | null; + readonly preflightDefers: number; + readonly infrastructureFailures: number; + readonly infrastructureByCode: Readonly>; + readonly judgeLatency: LatencyStats | null; + readonly modelLatency: LatencyStats | null; +} + +export interface LatencyStats { + readonly p50: number; + readonly p95: number; + readonly max: number; + readonly missing: number; +} + +const TERMINAL_STATES = new Set([ + "approved", + "approved_for_session", + "approved_for_serving_session", + "denied", + "confirmation_unavailable", +]); + +const LINK_DECISION_MARKERS = new Set(["inner_cmd.allow", "inner_cmd.deny"]); + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function asString(value: unknown): string | null { + return typeof value === "string" ? value : null; +} + +function asOptionalNumber(value: unknown): number | null { + return typeof value === "number" && Number.isFinite(value) ? value : null; +} + +function asBoolean(value: unknown): boolean | null { + return typeof value === "boolean" ? value : null; +} + +function percentile(sorted: readonly number[], p: number): number { + if (sorted.length === 0) { + return 0; + } + const index = Math.min( + sorted.length - 1, + Math.ceil((p / 100) * sorted.length) - 1, + ); + return sorted[index] as number; +} + +function latencyStats(values: readonly (number | null | undefined)[]): LatencyStats | null { + const present = values + .filter((v): v is number => typeof v === "number" && Number.isFinite(v)) + .sort((a, b) => a - b); + if (values.length === 0) { + return null; + } + return { + p50: percentile(present, 50), + p95: percentile(present, 95), + max: present.length > 0 ? (present[present.length - 1] as number) : 0, + missing: values.length - present.length, + }; +} + +function humanFromResolution( + resolution: string, + denialReason: unknown, +): HumanDecision | { readonly error: "terminal_event_unreadable" } { + const reason = asString(denialReason); + switch (resolution) { + case "approved": + case "approved_for_session": + case "approved_for_serving_session": + return { decision: "allow", state: resolution, denialReason: reason }; + case "denied": + return { decision: "deny", state: resolution, denialReason: reason }; + default: + return { error: "terminal_event_unreadable" }; + } +} + +/** + * Reconstruct the PIEXTENSIO-9 join over a review event stream. + * + * Input is the raw parsed JSONL of one permission review log. File order is + * authoritative: a human decision must appear after the judge result in + * append order. Enrollments with no result remain visible in + * `metrics.completionCoverage` and their dispositions are omitted from the + * joined set without a quarantine entry (missing outcomes are counted, not + * invented). + */ +export function analyzeShadowReviewLog(events: readonly ReviewEvent[]): AnalyzeResult { + // Phase 1: collect per-requestId first-seen records in append order. + const enrolled = new Set(); + const results = new Map(); + const terminal = new Map(); + const linkMarkers = new Set(); + + for (const evt of events) { + const requestId = asString(evt.requestId); + if (requestId === null) { + continue; + } + switch (evt.event) { + case "authorizer_chain_resolved": { + const links = Array.isArray(evt.links) ? evt.links : []; + if (links.some((name) => name === "ai-bash-judge")) { + enrolled.add(requestId); + } + break; + } + case "ai_bash_judge.result": + if (!results.has(requestId)) { + results.set(requestId, []); + } + (results.get(requestId) as ReviewEvent[]).push(evt); + break; + case "inner_cmd.allow": + case "inner_cmd.deny": + linkMarkers.add(requestId); + break; + case "permission_request.approved": + case "permission_request.denied": { + if (!terminal.has(requestId)) { + terminal.set(requestId, []); + } + (terminal.get(requestId) as ReviewEvent[]).push(evt); + break; + } + default: + break; + } + } + + const dispositions: Disposition[] = []; + const joined: JoinedRow[] = []; + const quarantined: Record = {}; + const quarantine = (category: QuarantineCategory): void => { + quarantined[category] = (quarantined[category] ?? 0) + 1; + dispositions.push({ kind: "quarantined", category }); + }; + + const terminalIds = [...terminal.keys()]; + for (const requestId of terminalIds) { + if (!enrolled.has(requestId)) { + continue; + } + const resultList = results.get(requestId) ?? []; + const terminalList = terminal.get(requestId) ?? []; + + if (resultList.length > 1) { + quarantine("duplicate_result"); + continue; + } + const result = resultList[0]; + if (result === undefined) { + // No result yet (or a lost write): counted as a coverage gap in + // the metrics, not quarantined as an integrity fault. + continue; + } + // The terminal permission_request entry must appear after the judge + // result in append order. The terminal list is collected from the + // same stream, so compare first-seen indices. + const resultIndex = events.indexOf(result); + const terminalEvents = terminalList.filter( + (t) => events.indexOf(t) > resultIndex, + ); + if (terminalEvents.length === 0) { + quarantine("human_before_result"); + continue; + } + if (terminalEvents.length > 1) { + quarantine("multiple_human_decisions"); + continue; + } + const terminalEvent = terminalEvents[0] as ReviewEvent; + const human = humanFromResolution( + asString(terminalEvent.resolution) ?? "", + terminalEvent.denialReason, + ); + if ("error" in human) { + quarantine("terminal_event_unreadable"); + continue; + } + + const state = human.state; + let attribution: JoinedRow["humanAttribution"]; + if ( + state === "approved_for_session" || + state === "approved_for_serving_session" + ) { + attribution = "session_state"; + } else if (linkMarkers.has(requestId)) { + attribution = "unproven"; + } else { + attribution = "no_link_marker"; + } + // `unproven` rows (a plain `approved` sharing the request with a link + // allow) cannot be attributed to the human under the reconstructed + // rule; they stay joined but never enter the comparison matrix. + + const row: JoinedRow = { + requestId, + judgeRuntimeId: asString(result.judgeRuntimeId), + resultKind: + result.resultKind === "judgment" || + result.resultKind === "preflight_defer" || + result.resultKind === "infrastructure_failure" + ? result.resultKind + : "infrastructure_failure", + verdict: + result.verdict === "allow" || + result.verdict === "deny" || + result.verdict === "defer" + ? result.verdict + : null, + code: asString(result.code), + modelCalled: asBoolean(result.modelCalled) ?? false, + provider: asString(result.provider), + model: asString(result.model), + origin: asString(result.origin), + judgeLatencyMs: asOptionalNumber(result.judgeLatencyMs), + modelLatencyMs: asOptionalNumber(result.modelLatencyMs), + inputUsage: asOptionalNumber(result.inputUsage), + outputUsage: asOptionalNumber(result.outputUsage), + reasonLength: asOptionalNumber(result.reasonLength), + human, + humanAttribution: attribution, + }; + joined.push(row); + dispositions.push({ kind: "joined", row }); + } + + // Results without enrollment are integrity faults: the denominator must + // be permission-owned. + for (const requestId of results.keys()) { + if (!enrolled.has(requestId) && !terminalIds.includes(requestId)) { + quarantine("result_without_enrollment"); + } + } + + return { + enrollments: enrolled.size, + dispositions, + metrics: computeMetrics(enrolled.size, joined, quarantined), + }; +} + +function computeMetrics( + n: number, + joined: readonly JoinedRow[], + quarantined: Record, +): Metrics { + const matrix: Record = {}; + const infraByCode: Record = {}; + const judgeLatencies: Array = []; + const modelLatencies: Array = []; + + let joinedJudgments = 0; + let attributed = 0; + let falseAllows = 0; + let conservativeDeny = 0; + let conservativeDefer = 0; + let humanAllowJudgments = 0; + let preflightDefers = 0; + let infrastructureFailures = 0; + + for (const row of joined) { + judgeLatencies.push(row.judgeLatencyMs); + modelLatencies.push(row.modelLatencyMs); + switch (row.resultKind) { + case "preflight_defer": + preflightDefers += 1; + continue; + case "infrastructure_failure": { + infrastructureFailures += 1; + const code = row.code ?? "unknown"; + infraByCode[code] = (infraByCode[code] ?? 0) + 1; + continue; + } + case "judgment": + break; + } + joinedJudgments += 1; + if (row.humanAttribution === "unproven") { + continue; + } + attributed += 1; + const key = `${row.verdict ?? "null"}|${row.human.decision}`; + matrix[key] = (matrix[key] ?? 0) + 1; + if (row.human.decision === "allow") { + humanAllowJudgments += 1; + } + if (row.verdict === "allow" && row.human.decision === "deny") { + falseAllows += 1; + } + if (row.verdict === "deny" && row.human.decision === "allow") { + conservativeDeny += 1; + } + if (row.verdict === "defer" && row.human.decision === "allow") { + conservativeDefer += 1; + } + } + + const completionCoverage = n > 0 ? (joined.length + missingResults(n, joined)) / n : 0; + return { + joined: joined.length, + quarantined, + joinedJudgments, + completionCoverage: joined.length / (n || 1), + humanJoinCoverage: (joined.length - quarantinedCount(joined)) / (n || 1), + judgmentCoverage: joinedJudgments / (n || 1), + matrix, + falseAllows, + falseAllowRate: falseAllows > 0 || hasAllowPrediction(matrix) + ? falseAllows / allowPredictions(matrix) + : null, + conservativeDeny, + conservativeDefer, + conservativeRate: + humanAllowJudgments > 0 + ? (conservativeDeny + conservativeDefer) / humanAllowJudgments + : null, + preflightDefers, + infrastructureFailures, + infrastructureByCode: infraByCode, + judgeLatency: latencyStats(judgeLatencies), + modelLatency: latencyStats(modelLatencies), + }; +} + +// ── helper functions below exist to keep computeMetrics readable; they are +// not part of the public surface. + +function missingResults(n: number, joined: readonly JoinedRow[]): number { + return Math.max(0, n - joined.length); +} +function quarantinedCount(joined: readonly JoinedRow[]): number { + // Quarantined rows are tracked outside `joined`; in this simplified + // metric path the human-join coverage counts joined rows with an + // attributable human outcome. + return joined.filter((r) => r.humanAttribution === "unproven").length; +} + +function allowPredictions(matrix: Readonly>): number { + return (matrix["allow|allow"] ?? 0) + (matrix["allow|deny"] ?? 0); +} + +function hasAllowPrediction(matrix: Readonly>): boolean { + return allowPredictions(matrix) > 0; +} diff --git a/packages/pi-permission-ai-judge/src/analyzer/cli.ts b/packages/pi-permission-ai-judge/src/analyzer/cli.ts new file mode 100644 index 0000000..07fa92b --- /dev/null +++ b/packages/pi-permission-ai-judge/src/analyzer/cli.ts @@ -0,0 +1,174 @@ +#!/usr/bin/env node +/** + * Offline Shadow analyzer CLI (diagnostic grade). + * + * Reads the permission-system review JSONL and prints the PIEXTENSIO-9 + * metrics computed over the reconstructed requestId join. Output is + * metadata-only: no command text from the source log is echoed. + */ + +import { readFileSync } from "node:fs"; +import { analyzeShadowReviewLog, type ReviewEvent } from "./analyze"; + +const USAGE = `usage: analyze-shadow [options] + +options: + --after only consider events with timestamp >= this instant + --help show this help + +The report is diagnostic-grade: the join reconstructs enrollment and human +decisions from permission-system events (no upstream changes), so coverage +and matrix numbers must not be used as promotion-grade evidence.`; + +interface CliOptions { + readonly path: string; + readonly after: Date | null; +} + +function parseArgs(argv: readonly string[]): CliOptions | { error: string } { + const args = argv.slice(2); + let path: string | undefined; + let after: Date | null = null; + for (let i = 0; i < args.length; i += 1) { + const arg = args[i] as string; + if (arg === "--help" || arg === "-h") { + return { error: USAGE }; + } + if (arg === "--after") { + const value = args[i + 1]; + if (value === undefined) { + return { error: "--after requires an ISO-8601 timestamp" }; + } + const parsed = new Date(value); + if (Number.isNaN(parsed.getTime())) { + return { error: `invalid --after timestamp: ${value}` }; + } + after = parsed; + i += 1; + continue; + } + if (arg.startsWith("--")) { + return { error: `unknown option: ${arg}` }; + } + if (path !== undefined) { + return { error: "multiple input paths given" }; + } + path = arg; + } + if (path === undefined) { + return { error: "missing input path" }; + } + return { path, after }; +} + +function parseLine(line: string, lineNo: number): ReviewEvent | null { + if (line.trim().length === 0) { + return null; + } + try { + return JSON.parse(line) as ReviewEvent; + } catch { + process.stderr.write( + `warning: skipping unparseable line ${lineNo}\n`, + ); + return null; + } +} + +function fmtRate(value: number | null): string { + if (value === null) { + return "N/A"; + } + return `${(value * 100).toFixed(1)}%`; +} + +function fmtLatency(stats: { + p50: number; + p95: number; + max: number; + missing: number; +} | null): string { + if (stats === null) { + return "N/A"; + } + return `p50=${stats.p50}ms p95=${stats.p95}ms max=${stats.max}ms missing=${stats.missing}`; +} + +function main(): void { + const parsed = parseArgs(process.argv); + if ("error" in parsed) { + process.stderr.write(`${parsed.error}\n`); + process.exit(parsed.error === USAGE ? 0 : 1); + } + + let raw: string; + try { + raw = readFileSync(parsed.path, "utf-8"); + } catch (error) { + process.stderr.write( + `error: cannot read ${parsed.path}: ${error instanceof Error ? error.message : String(error)}\n`, + ); + process.exit(1); + } + + const events = raw + .split("\n") + .map((line, index) => parseLine(line, index + 1)) + .filter((evt): evt is ReviewEvent => evt !== null) + .filter((evt) => { + if (parsed.after === null) { + return true; + } + const ts = typeof evt.timestamp === "string" ? evt.timestamp : null; + if (ts === null) { + return true; + } + const time = new Date(ts).getTime(); + return Number.isNaN(time) || time >= parsed.after.getTime(); + }); + + const { enrollments, metrics } = analyzeShadowReviewLog(events); + const out = process.stdout; + + out.write("AI Bash Judge — Shadow diagnostic report\n"); + out.write("grade: DIAGNOSTIC (reconstructed join; not promotion-grade)\n"); + out.write(`asOf: ${new Date().toISOString()}\n`); + out.write(`source: ${parsed.path}\n\n`); + + out.write(`enrollments (N): ${enrollments}\n`); + out.write(`joined rows: ${metrics.joined}\n`); + out.write(`joined judgments: ${metrics.joinedJudgments}\n\n`); + + out.write(`completion coverage: ${fmtRate(metrics.completionCoverage)}\n`); + out.write(`human-join coverage: ${fmtRate(metrics.humanJoinCoverage)}\n`); + out.write(`judgment coverage: ${fmtRate(metrics.judgmentCoverage)}\n\n`); + + out.write("comparison matrix [verdict|human]:\n"); + const keys = Object.keys(metrics.matrix).sort(); + if (keys.length === 0) { + out.write(" (empty)\n"); + } + for (const key of keys) { + out.write(` ${key}: ${metrics.matrix[key]}\n`); + } + out.write("\n"); + + out.write(`false allows: ${metrics.falseAllows}`); + out.write(` (rate ${fmtRate(metrics.falseAllowRate)})\n`); + out.write( + `conservative: deny ${metrics.conservativeDeny}, defer ${metrics.conservativeDefer} (rate ${fmtRate(metrics.conservativeRate)})\n\n`, + ); + + out.write(`preflight defers: ${metrics.preflightDefers}\n`); + out.write(`infrastructure failures: ${metrics.infrastructureFailures}\n`); + const codes = Object.keys(metrics.infrastructureByCode).sort(); + for (const code of codes) { + out.write(` ${code}: ${metrics.infrastructureByCode[code]}\n`); + } + out.write("\n"); + + out.write(`judge latency: ${fmtLatency(metrics.judgeLatency)}\n`); + out.write(`model latency: ${fmtLatency(metrics.modelLatency)}\n`); +} + +main(); diff --git a/packages/pi-permission-ai-judge/test/analyzer/analyze.test.ts b/packages/pi-permission-ai-judge/test/analyzer/analyze.test.ts new file mode 100644 index 0000000..50add7f --- /dev/null +++ b/packages/pi-permission-ai-judge/test/analyzer/analyze.test.ts @@ -0,0 +1,241 @@ +import { describe, expect, it } from "vitest"; +import { + analyzeShadowReviewLog, + type ReviewEvent, +} from "../../src/analyzer/analyze"; + +/** Build a minimal enrolled request lifecycle in append order. */ +function lifecycle(args: { + requestId?: string; + verdict?: "allow" | "deny" | "defer"; + resultKind?: "judgment" | "preflight_defer" | "infrastructure_failure"; + code?: string | null; + resolution?: string; + linkMarker?: boolean; +}): ReviewEvent[] { + const requestId = args.requestId ?? "req-1"; + const events: ReviewEvent[] = [ + { + event: "authorizer_chain_resolved", + requestId, + links: ["ai-bash-judge", "inner-cmd"], + }, + ]; + if (args.linkMarker) { + events.push({ event: "inner_cmd.allow", requestId }); + } + events.push({ + event: "permission_request.waiting", + requestId, + }); + events.push({ + event: "ai_bash_judge.result", + requestId, + resultKind: args.resultKind ?? "judgment", + verdict: (args.resultKind ?? "judgment") === "judgment" ? (args.verdict ?? "allow") : null, + code: args.code ?? null, + modelCalled: (args.resultKind ?? "judgment") === "judgment", + judgeLatencyMs: 100, + modelLatencyMs: 90, + }); + events.push({ + event: "permission_request.approved", + requestId, + resolution: args.resolution ?? "approved", + }); + return events; +} + +describe("analyzeShadowReviewLog — happy path", () => { + it("joins a full lifecycle into one row with matrix entry allow|allow", () => { + const { enrollments, dispositions, metrics } = analyzeShadowReviewLog( + lifecycle({ verdict: "allow", resolution: "approved" }), + ); + expect(enrollments).toBe(1); + expect(dispositions).toHaveLength(1); + expect(dispositions[0]?.kind).toBe("joined"); + expect(metrics.joined).toBe(1); + expect(metrics.matrix).toEqual({ "allow|allow": 1 }); + expect(metrics.falseAllowRate).toBe(0); + expect(metrics.judgeLatency).toMatchObject({ p50: 100, max: 100 }); + }); + + it("counts a false allow as verdict allow + human deny", () => { + const { metrics } = analyzeShadowReviewLog( + lifecycle({ verdict: "allow", resolution: "denied" }), + ); + expect(metrics.matrix).toEqual({ "allow|deny": 1 }); + expect(metrics.falseAllows).toBe(1); + expect(metrics.falseAllowRate).toBe(1); + }); + + it("reports conservative deny and defer separately", () => { + const deny = analyzeShadowReviewLog( + lifecycle({ requestId: "a", verdict: "deny", resolution: "approved" }), + ); + const defer = analyzeShadowReviewLog( + lifecycle({ requestId: "b", verdict: "defer", resolution: "approved" }), + ); + expect(deny.metrics.conservativeDeny).toBe(1); + expect(deny.metrics.conservativeDefer).toBe(0); + expect(defer.metrics.conservativeDeny).toBe(0); + expect(defer.metrics.conservativeDefer).toBe(1); + }); + + it("normalizes session approvals to human allow", () => { + const { metrics } = analyzeShadowReviewLog( + lifecycle({ verdict: "defer", resolution: "approved_for_session" }), + ); + expect(metrics.matrix).toEqual({ "defer|allow": 1 }); + }); + + it("keeps preflight defers out of the matrix but in coverage", () => { + const { metrics } = analyzeShadowReviewLog( + lifecycle({ + resultKind: "preflight_defer", + code: "missing_structured_input", + resolution: "approved", + }), + ); + expect(metrics.matrix).toEqual({}); + expect(metrics.joined).toBe(1); + expect(metrics.judgmentCoverage).toBe(0); + }); + + it("buckets infrastructure failures by stable code", () => { + const { metrics } = analyzeShadowReviewLog( + lifecycle({ + resultKind: "infrastructure_failure", + code: "aborted", + resolution: "approved", + }), + ); + expect(metrics.infrastructureFailures).toBe(1); + expect(metrics.infrastructureByCode).toEqual({ aborted: 1 }); + expect(metrics.matrix).toEqual({}); + }); +}); + +describe("analyzeShadowReviewLog — attribution", () => { + it("attributes a plain approved with no link marker to the human", () => { + const { dispositions } = analyzeShadowReviewLog( + lifecycle({ resolution: "approved" }), + ); + const row = dispositions[0]; + expect(row?.kind).toBe("joined"); + if (row?.kind === "joined") { + expect(row.row.humanAttribution).toBe("no_link_marker"); + } + }); + + it("marks a plain approved sharing the request with inner_cmd.allow as unproven", () => { + const { dispositions, metrics } = analyzeShadowReviewLog( + lifecycle({ resolution: "approved", linkMarker: true }), + ); + const row = dispositions[0]; + expect(row?.kind).toBe("joined"); + if (row?.kind === "joined") { + expect(row.row.humanAttribution).toBe("unproven"); + } + // Unproven rows stay joined but never enter the matrix. + expect(metrics.matrix).toEqual({}); + }); + + it("always attributes session-state approvals to the human despite a marker", () => { + const { dispositions } = analyzeShadowReviewLog( + lifecycle({ + resolution: "approved_for_session", + linkMarker: true, + }), + ); + const row = dispositions[0]; + expect(row?.kind).toBe("joined"); + if (row?.kind === "joined") { + expect(row.row.humanAttribution).toBe("session_state"); + } + }); +}); + +describe("analyzeShadowReviewLog — integrity", () => { + it("quarantines a duplicate judge result", () => { + const base = lifecycle({ verdict: "allow" }); + const dup = base.map((e) => e).concat([ + { + event: "ai_bash_judge.result", + requestId: "req-1", + resultKind: "judgment", + verdict: "deny", + }, + ]); + const { metrics } = analyzeShadowReviewLog(dup); + expect(metrics.quarantined).toEqual({ duplicate_result: 1 }); + expect(metrics.joined).toBe(0); + }); + + it("quarantines a human decision recorded before the judge result", () => { + const events: ReviewEvent[] = [ + { event: "authorizer_chain_resolved", requestId: "r", links: ["ai-bash-judge"] }, + { event: "permission_request.approved", requestId: "r", resolution: "approved" }, + { + event: "ai_bash_judge.result", + requestId: "r", + resultKind: "judgment", + verdict: "allow", + }, + ]; + const { metrics } = analyzeShadowReviewLog(events); + expect(metrics.quarantined).toEqual({ human_before_result: 1 }); + }); + + it("counts an enrollment without a result as a coverage gap, not quarantine", () => { + const events: ReviewEvent[] = [ + { event: "authorizer_chain_resolved", requestId: "r", links: ["ai-bash-judge"] }, + ]; + const { enrollments, metrics } = analyzeShadowReviewLog(events); + expect(enrollments).toBe(1); + expect(metrics.joined).toBe(0); + expect(metrics.quarantined).toEqual({}); + expect(metrics.completionCoverage).toBe(0); + }); + + it("quarantines a judge result with no enrollment when no terminal exists", () => { + const events: ReviewEvent[] = [ + { + event: "ai_bash_judge.result", + requestId: "orphan", + resultKind: "judgment", + verdict: "allow", + }, + ]; + const { metrics } = analyzeShadowReviewLog(events); + expect(metrics.quarantined).toEqual({ result_without_enrollment: 1 }); + }); + + it("quarantines an unreadable terminal resolution", () => { + const events: ReviewEvent[] = [ + { event: "authorizer_chain_resolved", requestId: "r", links: ["ai-bash-judge"] }, + { + event: "ai_bash_judge.result", + requestId: "r", + resultKind: "judgment", + verdict: "allow", + }, + { event: "permission_request.approved", requestId: "r", resolution: "confirmation_unavailable" }, + ]; + const { metrics } = analyzeShadowReviewLog(events); + expect(metrics.quarantined).toEqual({ terminal_event_unreadable: 1 }); + }); +}); + +describe("analyzeShadowReviewLog — latency", () => { + it("reports missing latency as missing, not zero", () => { + const events = lifecycle({}); + const stripped = events.map((e) => + e.event === "ai_bash_judge.result" + ? { ...e, judgeLatencyMs: undefined, modelLatencyMs: undefined } + : e, + ); + const { metrics } = analyzeShadowReviewLog(stripped); + expect(metrics.judgeLatency).toMatchObject({ missing: 1, p50: 0 }); + }); +});