diff --git a/packages/pi-permission-ai-judge/src/analyzer/analyze.ts b/packages/pi-permission-ai-judge/src/analyzer/analyze.ts index d91b12c..38553b5 100644 --- a/packages/pi-permission-ai-judge/src/analyzer/analyze.ts +++ b/packages/pi-permission-ai-judge/src/analyzer/analyze.ts @@ -187,17 +187,16 @@ function humanFromResolution( } /** - * Reconstruct the PIEXTENSIO-9 join over a review event stream. - * - * Input is the raw parsed JSONL of one permission review log. File order is - * authoritative: a human decision must appear after the judge result in - * append order. Enrollments with no result remain visible in - * `metrics.completionCoverage` and their dispositions are omitted from the - * joined set without a quarantine entry (missing outcomes are counted, not - * invented). + * Phase 1: collect per-requestId first-seen records in append order. */ -export function analyzeShadowReviewLog(events: readonly ReviewEvent[]): AnalyzeResult { - // Phase 1: collect per-requestId first-seen records in append order. +interface CollectedEvents { + readonly enrolled: ReadonlySet; + readonly results: ReadonlyMap; + readonly terminal: ReadonlyMap; + readonly linkMarkers: ReadonlySet; +} + +function collectReviewEvents(events: readonly ReviewEvent[]): CollectedEvents { const enrolled = new Set(); const results = new Map(); const terminal = new Map(); @@ -238,6 +237,142 @@ export function analyzeShadowReviewLog(events: readonly ReviewEvent[]): AnalyzeR break; } } + return { enrolled, results, terminal, linkMarkers }; +} + +/** Outcome of joining one terminal request against its judge result. */ +type JoinOutcome = + | { readonly kind: "joined"; readonly row: JoinedRow } + | { readonly kind: "quarantined"; readonly category: QuarantineCategory } + | { readonly kind: "skipped" }; + +/** + * Phase 2: join one enrolled request's terminal permission event against + * its judge result. Guards run in append-order integrity order; the first + * failure quarantines with an explicit category. + */ +function joinTerminalRequest( + requestId: string, + events: readonly ReviewEvent[], + collected: CollectedEvents, +): JoinOutcome { + const resultList = collected.results.get(requestId) ?? []; + const terminalList = collected.terminal.get(requestId) ?? []; + + if (resultList.length > 1) { + return { kind: "quarantined", category: "duplicate_result" }; + } + const result = resultList[0]; + if (result === undefined) { + // No result yet (or a lost write): counted as a coverage gap in + // the metrics, not quarantined as an integrity fault. + return { kind: "skipped" }; + } + // The terminal permission_request entry must appear after the judge + // result in append order. The terminal list is collected from the + // same stream, so compare first-seen indices. + const resultIndex = events.indexOf(result); + const terminalEvents = terminalList.filter( + (t) => events.indexOf(t) > resultIndex, + ); + if (terminalEvents.length === 0) { + return { kind: "quarantined", category: "human_before_result" }; + } + if (terminalEvents.length > 1) { + // Upstream's forwarded decision path double-writes the terminal + // event with the same requestId and resolution in adjacent file + // order (round 1: two `approved` or two `denied_with_reason` + // rows in the same second). Identical-resolution duplicates are + // that pattern, not an integrity fault: collapse to the first + // row. Conflicting resolutions stay quarantined — the analyzer + // must never pick a convenient outcome among alternatives + // (PIEXTENSIO-9). + const distinct = new Set( + terminalEvents.map((t) => asString(t.resolution) ?? ""), + ); + if (distinct.size > 1) { + return { kind: "quarantined", category: "multiple_human_decisions" }; + } + } + const terminalEvent = terminalEvents[0] as ReviewEvent; + const human = humanFromResolution( + asString(terminalEvent.resolution) ?? "", + terminalEvent.denialReason, + ); + if ("error" in human) { + return { kind: "quarantined", category: "terminal_event_unreadable" }; + } + + return { + kind: "joined", + row: buildJoinedRow(requestId, result, human, collected.linkMarkers), + }; +} + +function buildJoinedRow( + requestId: string, + result: ReviewEvent, + human: HumanDecision, + linkMarkers: ReadonlySet, +): JoinedRow { + const state = human.state; + let attribution: JoinedRow["humanAttribution"]; + if ( + state === "approved_for_session" || + state === "approved_for_serving_session" + ) { + attribution = "session_state"; + } else if (linkMarkers.has(requestId)) { + attribution = "unproven"; + } else { + attribution = "no_link_marker"; + } + // `unproven` rows (a plain `approved` sharing the request with a link + // allow) cannot be attributed to the human under the reconstructed + // rule; they stay joined but never enter the comparison matrix. + + return { + requestId, + judgeRuntimeId: asString(result.judgeRuntimeId), + resultKind: + result.resultKind === "judgment" || + result.resultKind === "preflight_defer" || + result.resultKind === "infrastructure_failure" + ? result.resultKind + : "infrastructure_failure", + verdict: + result.verdict === "allow" || + result.verdict === "deny" || + result.verdict === "defer" + ? result.verdict + : null, + code: asString(result.code), + modelCalled: asBoolean(result.modelCalled) ?? false, + provider: asString(result.provider), + model: asString(result.model), + origin: asString(result.origin), + judgeLatencyMs: asOptionalNumber(result.judgeLatencyMs), + modelLatencyMs: asOptionalNumber(result.modelLatencyMs), + inputUsage: asOptionalNumber(result.inputUsage), + outputUsage: asOptionalNumber(result.outputUsage), + reasonLength: asOptionalNumber(result.reasonLength), + human, + humanAttribution: attribution, + }; +} + +/** + * Reconstruct the PIEXTENSIO-9 join over a review event stream. + * + * Input is the raw parsed JSONL of one permission review log. File order is + * authoritative: a human decision must appear after the judge result in + * append order. Enrollments with no result remain visible in + * `metrics.completionCoverage` and their dispositions are omitted from the + * joined set without a quarantine entry (missing outcomes are counted, not + * invented). + */ +export function analyzeShadowReviewLog(events: readonly ReviewEvent[]): AnalyzeResult { + const collected = collectReviewEvents(events); const dispositions: Disposition[] = []; const joined: JoinedRow[] = []; @@ -247,122 +382,32 @@ export function analyzeShadowReviewLog(events: readonly ReviewEvent[]): AnalyzeR dispositions.push({ kind: "quarantined", category }); }; - const terminalIds = [...terminal.keys()]; - for (const requestId of terminalIds) { - if (!enrolled.has(requestId)) { + for (const requestId of collected.terminal.keys()) { + if (!collected.enrolled.has(requestId)) { continue; } - const resultList = results.get(requestId) ?? []; - const terminalList = terminal.get(requestId) ?? []; - - if (resultList.length > 1) { - quarantine("duplicate_result"); - continue; + const outcome = joinTerminalRequest(requestId, events, collected); + if (outcome.kind === "quarantined") { + quarantine(outcome.category); + } else if (outcome.kind === "joined") { + joined.push(outcome.row); + dispositions.push({ kind: "joined", row: outcome.row }); } - const result = resultList[0]; - if (result === undefined) { - // No result yet (or a lost write): counted as a coverage gap in - // the metrics, not quarantined as an integrity fault. - continue; - } - // The terminal permission_request entry must appear after the judge - // result in append order. The terminal list is collected from the - // same stream, so compare first-seen indices. - const resultIndex = events.indexOf(result); - const terminalEvents = terminalList.filter( - (t) => events.indexOf(t) > resultIndex, - ); - if (terminalEvents.length === 0) { - quarantine("human_before_result"); - continue; - } - if (terminalEvents.length > 1) { - // Upstream's forwarded decision path double-writes the terminal - // event with the same requestId and resolution in adjacent file - // order (round 1: two `approved` or two `denied_with_reason` - // rows in the same second). Identical-resolution duplicates are - // that pattern, not an integrity fault: collapse to the first - // row. Conflicting resolutions stay quarantined — the analyzer - // must never pick a convenient outcome among alternatives - // (PIEXTENSIO-9). - const distinct = new Set( - terminalEvents.map((t) => asString(t.resolution) ?? ""), - ); - if (distinct.size > 1) { - quarantine("multiple_human_decisions"); - continue; - } - } - const terminalEvent = terminalEvents[0] as ReviewEvent; - const human = humanFromResolution( - asString(terminalEvent.resolution) ?? "", - terminalEvent.denialReason, - ); - if ("error" in human) { - quarantine("terminal_event_unreadable"); - continue; - } - - const state = human.state; - let attribution: JoinedRow["humanAttribution"]; - if ( - state === "approved_for_session" || - state === "approved_for_serving_session" - ) { - attribution = "session_state"; - } else if (linkMarkers.has(requestId)) { - attribution = "unproven"; - } else { - attribution = "no_link_marker"; - } - // `unproven` rows (a plain `approved` sharing the request with a link - // allow) cannot be attributed to the human under the reconstructed - // rule; they stay joined but never enter the comparison matrix. - - const row: JoinedRow = { - requestId, - judgeRuntimeId: asString(result.judgeRuntimeId), - resultKind: - result.resultKind === "judgment" || - result.resultKind === "preflight_defer" || - result.resultKind === "infrastructure_failure" - ? result.resultKind - : "infrastructure_failure", - verdict: - result.verdict === "allow" || - result.verdict === "deny" || - result.verdict === "defer" - ? result.verdict - : null, - code: asString(result.code), - modelCalled: asBoolean(result.modelCalled) ?? false, - provider: asString(result.provider), - model: asString(result.model), - origin: asString(result.origin), - judgeLatencyMs: asOptionalNumber(result.judgeLatencyMs), - modelLatencyMs: asOptionalNumber(result.modelLatencyMs), - inputUsage: asOptionalNumber(result.inputUsage), - outputUsage: asOptionalNumber(result.outputUsage), - reasonLength: asOptionalNumber(result.reasonLength), - human, - humanAttribution: attribution, - }; - joined.push(row); - dispositions.push({ kind: "joined", row }); } // Results without enrollment are integrity faults: the denominator must // be permission-owned. - for (const requestId of results.keys()) { - if (!enrolled.has(requestId) && !terminalIds.includes(requestId)) { + const terminalIds = [...collected.terminal.keys()]; + for (const requestId of collected.results.keys()) { + if (!collected.enrolled.has(requestId) && !terminalIds.includes(requestId)) { quarantine("result_without_enrollment"); } } return { - enrollments: enrolled.size, + enrollments: collected.enrolled.size, dispositions, - metrics: computeMetrics(enrolled.size, joined, quarantined), + metrics: computeMetrics(collected.enrolled.size, joined, quarantined), }; } diff --git a/packages/pi-permission-ai-judge/src/config.ts b/packages/pi-permission-ai-judge/src/config.ts index a401531..2574a9d 100644 --- a/packages/pi-permission-ai-judge/src/config.ts +++ b/packages/pi-permission-ai-judge/src/config.ts @@ -92,6 +92,132 @@ function parseJudgeModel(value: unknown): JudgeModelSelection | undefined { }; } +/** Parse the `version` field; unknown versions are a hard failure. */ +function parseConfigVersion( + record: Record, +): { version: 1 | 2 } | { problem: string } { + // Version: unversioned files are legacy v1; anything but 1 or 2 fails + // closed to all defaults (the consent expression is uninterpretable). + if (record.version === undefined) { + return { version: 1 }; + } + if (record.version === 1 || record.version === 2) { + return { version: record.version }; + } + return { + problem: `unknown version ${JSON.stringify(record.version)}`, + }; +} + +/** Parse the `mode` field: unknown or missing resolves to shadow (fail-closed). */ +function parseMode( + record: Record, +): { mode: JudgeMode; diagnostics: ConfigDiagnostic[] } { + const diagnostics: ConfigDiagnostic[] = []; + let mode: JudgeMode = "shadow"; + if (record.mode !== undefined) { + if (record.mode === "shadow" || record.mode === "enforce") { + mode = record.mode; + } else { + diagnostics.push({ + key: "mode", + problem: `unknown mode ${JSON.stringify(record.mode)}`, + fallback: "shadow", + }); + } + } + return { mode, diagnostics }; +} + +/** + * Parse the v2-only `model` field. `demoteToShadow` marks that a malformed + * model poisons the consent file — the caller fails the whole config closed + * to shadow rather than silently substituting the session model. + */ +function parseJudgeModelField( + record: Record, + configVersion: 1 | 2, +): { + judgeModel: JudgeModelSelection | undefined; + demoteToShadow: boolean; + diagnostics: ConfigDiagnostic[]; +} { + if (record.model === undefined) { + return { judgeModel: undefined, demoteToShadow: false, diagnostics: [] }; + } + if (configVersion === 2) { + const judgeModel = parseJudgeModel(record.model); + if (judgeModel === undefined) { + return { + judgeModel: undefined, + demoteToShadow: true, + diagnostics: [ + { + key: "model", + problem: `invalid model ${JSON.stringify(record.model)} (expected {provider, id} with non-empty strings)`, + fallback: "shadow (no judge model)", + }, + ], + }; + } + return { judgeModel, demoteToShadow: false, diagnostics: [] }; + } + return { + judgeModel: undefined, + demoteToShadow: false, + diagnostics: [ + { + key: "model", + problem: "model selection requires \"version\": 2", + fallback: "session model", + }, + ], + }; +} + +/** + * Parse `timeoutMs`: integers in [5_000, 30_000]; anything else falls back + * to the documented 15,000 ms default. Boundary semantics: 4,999 and 30,001 + * are invalid, 5,000 and 30,000 are valid (PIEXTENSIO-3 boundaries). + */ +function parseTimeout(record: Record): { + timeoutMs: number; + timeoutCohort: EffectiveJudgeConfig["timeoutCohort"]; + diagnostics: ConfigDiagnostic[]; +} { + if (record.timeoutMs === undefined) { + return { + timeoutMs: DEFAULT_TIMEOUT_MS, + timeoutCohort: "default", + diagnostics: [], + }; + } + const value = record.timeoutMs; + if ( + typeof value === "number" && + Number.isInteger(value) && + value >= MIN_TIMEOUT_MS && + value <= MAX_TIMEOUT_MS + ) { + return { + timeoutMs: value, + timeoutCohort: value === DEFAULT_TIMEOUT_MS ? "default" : value, + diagnostics: [], + }; + } + return { + timeoutMs: DEFAULT_TIMEOUT_MS, + timeoutCohort: "default", + diagnostics: [ + { + key: "timeoutMs", + problem: `invalid timeoutMs ${JSON.stringify(value)}`, + fallback: `${DEFAULT_TIMEOUT_MS} (default)`, + }, + ], + }; +} + /** * Load and validate the global config. Missing file, malformed JSON, or * an unknown version resolve to the documented defaults with one @@ -128,40 +254,22 @@ export function loadJudgeConfig( const record = parsed as Record; const diagnostics: ConfigDiagnostic[] = []; - // Version: unversioned files are legacy v1; anything but 1 or 2 fails - // closed to all defaults (the consent expression is uninterpretable). - let configVersion: 1 | 2 = 1; - if (record.version !== undefined) { - if (record.version === 1 || record.version === 2) { - configVersion = record.version; - } else { - return { - ...DEFAULT_CONFIG, - diagnostics: [ - { - key: "version", - problem: `unknown version ${JSON.stringify(record.version)}`, - fallback: "all defaults", - }, - ], - }; - } + const versionResult = parseConfigVersion(record); + if ("problem" in versionResult) { + return { + ...DEFAULT_CONFIG, + diagnostics: [ + { key: "version", problem: versionResult.problem, fallback: "all defaults" }, + ], + }; } + const configVersion = versionResult.version; - // Mode: unknown or missing resolves to shadow (fail-closed). A v1 - // enforce cannot silently inherit the v2 risk contract (ADR 0008). - let mode: JudgeMode = "shadow"; - if (record.mode !== undefined) { - if (record.mode === "shadow" || record.mode === "enforce") { - mode = record.mode; - } else { - diagnostics.push({ - key: "mode", - problem: `unknown mode ${JSON.stringify(record.mode)}`, - fallback: "shadow", - }); - } - } + const modeResult = parseMode(record); + diagnostics.push(...modeResult.diagnostics); + let mode = modeResult.mode; + + // A v1 enforce cannot silently inherit the v2 risk contract (ADR 0008). if (configVersion === 1 && mode === "enforce") { mode = "shadow"; diagnostics.push({ @@ -172,60 +280,21 @@ export function loadJudgeConfig( }); } - // Judge model: v2-only. A malformed model poisons the consent file — - // fail the whole config closed to shadow rather than silently - // substituting the session model. - let judgeModel: JudgeModelSelection | undefined; - if (record.model !== undefined) { - if (configVersion === 2) { - judgeModel = parseJudgeModel(record.model); - if (judgeModel === undefined) { - mode = "shadow"; - diagnostics.push({ - key: "model", - problem: `invalid model ${JSON.stringify(record.model)} (expected {provider, id} with non-empty strings)`, - fallback: "shadow (no judge model)", - }); - } - } else { - diagnostics.push({ - key: "model", - problem: "model selection requires \"version\": 2", - fallback: "session model", - }); - } + const modelResult = parseJudgeModelField(record, configVersion); + diagnostics.push(...modelResult.diagnostics); + if (modelResult.demoteToShadow) { + mode = "shadow"; } - // Timeout: integers in [5_000, 30_000]; anything else falls back to the - // documented 15,000 ms default. Boundary semantics: 4,999 and 30,001 - // are invalid, 5,000 and 30,000 are valid (PIEXTENSIO-3 boundaries). - let timeoutMs = DEFAULT_TIMEOUT_MS; - let timeoutCohort: EffectiveJudgeConfig["timeoutCohort"] = "default"; - if (record.timeoutMs !== undefined) { - const value = record.timeoutMs; - if ( - typeof value === "number" && - Number.isInteger(value) && - value >= MIN_TIMEOUT_MS && - value <= MAX_TIMEOUT_MS - ) { - timeoutMs = value; - timeoutCohort = value === DEFAULT_TIMEOUT_MS ? "default" : value; - } else { - diagnostics.push({ - key: "timeoutMs", - problem: `invalid timeoutMs ${JSON.stringify(value)}`, - fallback: `${DEFAULT_TIMEOUT_MS} (default)`, - }); - } - } + const timeoutResult = parseTimeout(record); + diagnostics.push(...timeoutResult.diagnostics); return Object.freeze({ configVersion, mode, - timeoutMs, - timeoutCohort, - judgeModel, + timeoutMs: timeoutResult.timeoutMs, + timeoutCohort: timeoutResult.timeoutCohort, + judgeModel: modelResult.judgeModel, diagnostics, }); } diff --git a/packages/pi-permission-ai-judge/src/index.ts b/packages/pi-permission-ai-judge/src/index.ts index e42ddc2..52af57f 100644 --- a/packages/pi-permission-ai-judge/src/index.ts +++ b/packages/pi-permission-ai-judge/src/index.ts @@ -8,12 +8,15 @@ import { getPermissionsService, PERMISSIONS_READY_CHANNEL, type PromptPermissionDetails, + type AuthorizerLog, + type AuthorizerVerdict, } from "@gotgenes/pi-permission-system"; import { buildBashJudgmentEvidence } from "./evidence"; import { createModelAvailability, requestStructuredVerdict, type ModelAvailability, + type ModelAttempt, } from "./model"; import { PROMPT_VERSION, TOOL_SCHEMA_VERSION } from "./prompt"; import { loadJudgeConfig, type EffectiveJudgeConfig } from "./config"; @@ -31,6 +34,7 @@ import { loadModelCatalog, DEFAULT_CATALOG_PATH, type ModelCatalogClassification, + type LoadedModelCatalog, } from "./catalog"; const LINK_NAME = "ai-bash-judge"; @@ -166,6 +170,384 @@ function resolveJudgeModel( return { kind: "model", model: found, source: "configured" }; } +/** Per-call emission context shared by the result emitters. */ +interface EmitContext { + readonly captured: RootSession; + readonly details: PromptPermissionDetails; + readonly startedAt: number; + readonly emitResult: (record: Record) => void; +} + +/** Shared identity/cohort fields; latency is measured at emit time. */ +function emitBase(ctx: EmitContext): Record { + return resultBase( + ctx.captured.judgeRuntimeId, + ctx.details, + ctx.startedAt, + ctx.captured.config, + ); +} + +/** + * Emit a `preflight_defer` result (fail-closed before the model runs) and + * return the matching defer verdict. + */ +function preflightDefer( + ctx: EmitContext, + code: string, + eq: Record, + extra: Record = {}, +): AuthorizerVerdict { + ctx.emitResult({ + ...emitBase(ctx), + resultKind: "preflight_defer", + verdict: null, + effectiveVerdict: "defer", + modelCalled: false, + code, + ...extra, + evidenceQuality: eq, + }); + return { kind: "defer" }; +} + +/** + * Emit an `infrastructure_failure` result with a defer verdict (e.g. the + * judge model cannot be resolved). + */ +function infrastructureDefer( + ctx: EmitContext, + code: string, + eq: Record, + extra: Record = {}, +): AuthorizerVerdict { + ctx.emitResult({ + ...emitBase(ctx), + resultKind: "infrastructure_failure", + verdict: null, + effectiveVerdict: "defer", + modelCalled: false, + code, + ...extra, + evidenceQuality: eq, + }); + return { kind: "defer" }; +} + +/** Emit the `judgment` result for a model call that returned a verdict. */ +function emitJudgmentResult( + ctx: EmitContext, + result: Extract, + conversation: ConversationEvidence, + effectiveVerdict: "allow" | "defer", + authorityBlockedBy: string | null, + modelSource: "configured" | "session", + risk: HighRiskMatch | undefined, +): void { + ctx.emitResult({ + ...emitBase(ctx), + resultKind: "judgment", + verdict: result.verdict, + effectiveVerdict, + authorityBlockedBy, + modelCalled: true, + code: null, + modelSource, + provider: result.metadata.provider, + model: result.metadata.model, + api: result.metadata.api, + riskOverride: risk ?? null, + // Log keys deliberately avoid the substring + // "token": permission-system masks any key matching + // /token/i (structural key-name redaction), which + // would erase usage telemetry from the review log. + inputUsage: result.inputTokens, + outputUsage: result.outputTokens, + modelLatencyMs: result.modelLatencyMs, + reasonLength: reasonLength(result.reason), + evidenceQuality: evidenceQuality(true, conversation, ctx.captured.getCwd()), + }); +} + +/** Emit the post-model `infrastructure_failure` result under authority. */ +function emitInfrastructureResult( + ctx: EmitContext, + result: Extract, + conversation: ConversationEvidence, + effectiveVerdict: "allow" | "defer", + authorityBlockedBy: string | null, + modelSource: "configured" | "session", + risk: HighRiskMatch | undefined, +): void { + ctx.emitResult({ + ...emitBase(ctx), + resultKind: "infrastructure_failure", + verdict: null, + effectiveVerdict, + authorityBlockedBy, + modelCalled: result.modelCalled, + code: result.code, + modelSource, + provider: result.metadata?.provider ?? null, + model: result.metadata?.model ?? null, + api: result.metadata?.api ?? null, + riskOverride: risk ?? null, + inputUsage: result.inputTokens ?? null, + outputUsage: result.outputTokens ?? null, + modelLatencyMs: result.modelLatencyMs, + evidenceQuality: evidenceQuality(true, conversation, ctx.captured.getCwd()), + }); +} + +/** + * One authorize call: enroll, preflight-gate, judge, and enforce the truth + * table. Extracted from the registerAuthorizer callback so each stage reads + * linearly; any exception fails closed (defer) without logging raw errors. + */ +async function judgeAuthorize( + captured: RootSession, + details: PromptPermissionDetails, + log: AuthorizerLog, +): Promise { + const startedAt = Date.now(); + const sink: ReviewSink = createReviewSink({ + log, + reviewLogEnabled: captured.reviewLogEnabled, + }); + try { + const ctx: EmitContext = { + captured, + details, + startedAt, + emitResult: (record) => { + sink.review("ai_bash_judge.result", record); + captured.auditLog.audit("ai_bash_judge.result", record); + }, + }; + + // Judge-owned enrollment record (ADR 0006 denominator: + // asks the Judge received, per its own audit log). + // Non-bash surfaces are ignored below without a shadow + // row; they also do not enroll (v0.1 cohort is bash-only). + if ( + details.forwarding !== undefined || + details.payload.kind === "forwarded" || + details.payload.kind === "bash" + ) { + captured.auditLog.audit("ai_bash_judge.enrolled", { + requestId: details.requestId, + origin: + details.forwarding !== undefined || + details.payload.kind === "forwarded" + ? "forwarded" + : "local", + surface: "bash", + command: + details.payload.kind === "bash" && + details.payload.request?.value + ? details.payload.request.value + : (details.command ?? null), + }); + } + // Forwarded asks do not carry a structured child full + // command in permission-system 25.3/25.4. Never parse the + // legacy prose. The deferral is recorded so the request + // stays visible in the offline denominator instead of + // silently vanishing. + if ( + details.forwarding !== undefined || + details.payload.kind === "forwarded" + ) { + return preflightDefer( + ctx, + "missing_structured_input", + evidenceQuality(false, EMPTY_CONVERSATION, "", false), + ); + } + + // Ignore unrelated permission surfaces without producing a + // Shadow row or invoking the model: the v0.1 cohort selects + // accessSurface = bash only. + if (details.payload.kind !== "bash") { + return { kind: "defer" }; + } + + if (captured.getSessionId() !== captured.expectedSessionId) { + return preflightDefer( + ctx, + "session_ownership_unproven", + evidenceQuality(false, EMPTY_CONVERSATION, ""), + ); + } + + const evidence = buildBashJudgmentEvidence(details); + if (evidence === undefined) { + return preflightDefer( + ctx, + "invalid_evidence", + evidenceQuality(false, EMPTY_CONVERSATION, ""), + ); + } + + // Built-in high-risk override (ADR 0008): clear-cut + // irreversible/system shapes always defer. In Enforce the + // model is skipped entirely; in Shadow it still runs for + // quality observation and the override is recorded. + const risk: HighRiskMatch | undefined = classifyHighRisk( + evidence.fullCommand, + ); + if (risk !== undefined && captured.config.mode === "enforce") { + return preflightDefer( + ctx, + "high_risk_override", + evidenceQuality(true, EMPTY_CONVERSATION, captured.getCwd()), + { riskCategory: risk.category, riskRule: risk.rule }, + ); + } + + // Per-request judge-model resolution (PIEXTENSIO-3 cat.3 + // for the session model; ADR 0008 for a configured fixed + // model). A configured model that cannot be resolved is + // an observable infrastructure failure — never a silent + // fallback to the session model. + const resolved = resolveJudgeModel( + captured.config, + captured.getModel(), + captured.modelRegistry, + ); + if (resolved.kind === "unavailable") { + return infrastructureDefer( + ctx, + "judge_model_unavailable", + evidenceQuality(true, EMPTY_CONVERSATION, captured.getCwd()), + { + provider: captured.config.judgeModel?.provider ?? null, + model: captured.config.judgeModel?.id ?? null, + api: null, + riskOverride: risk ?? null, + }, + ); + } + const modelSource = resolved.source; + const availability: ModelAvailability = createModelAvailability( + resolved.model, + captured.modelRegistry, + ); + // Conversation evidence is captured at ask time from the + // live serving branch, not at session start: the newest + // user intent is the ask's intent. + const conversation: ConversationEvidence = + buildConversationEvidence(captured.conversation); + const result = await requestStructuredVerdict( + availability, + evidence, + captured.shutdown.signal, + captured.config.timeoutMs, + conversation, + ); + + // Enforce truth table (PIEXTENSIO-3 cat.4 / M5; ADR 0008): + // the fail-closed runtime health gates — audit health, + // telemetry, result kind, verdict, review + // acknowledgement, generation currency — are the single + // authority seam; the retired promotion gates are no + // longer inputs. reviewAcknowledged is true in the ADR + // 0006 sense: the Judge-owned audit write for this result + // happens before the authority return, and a failed write + // flips auditHealthy sticky-unhealthy, closing authority + // for every later ask. + const gateState: EnforceGateState = { + auditHealthy: captured.auditLog.healthy(), + telemetryHealth: sink.health(), + resultKind: + result.kind === "judgment" + ? "judgment" + : result.kind, + verdict: + result.kind === "judgment" ? result.verdict : null, + reviewAcknowledged: true, + generationCurrent: !captured.shutdown.signal.aborted, + mode: captured.config.mode, + }; + const authority = evaluateEnforceAuthority(gateState); + const effectiveVerdict = + authority.kind === "allow" ? "allow" : "defer"; + const authorityBlockedBy = + authority.kind === "allow" ? null : authority.blockedBy; + + if (result.kind === "judgment") { + emitJudgmentResult( + ctx, result, conversation, + effectiveVerdict, authorityBlockedBy, modelSource, risk, + ); + } else { + emitInfrastructureResult( + ctx, result, conversation, + effectiveVerdict, authorityBlockedBy, modelSource, risk, + ); + } + + return authority.kind === "allow" + ? { kind: "allow" } + : { kind: "defer" }; + } catch { + // A link exception would abort the whole authority chain. + // Keep provider/payload/session failures fail-closed and do + // not include raw errors or authorization evidence in logs. + sink.debug("ai_bash_judge.exception"); + return { kind: "defer" }; + } +} + +/** + * One non-blocking session notice in Enforce mode: the risk contract and + * the effective judge model (ADR 0008). Not repeated per ask. The advisory + * model catalog (PIEXTENSIO-24) only annotates this notice — it never gates + * authority. + */ +function notifyEnforceActive( + notify: (message: string, kind: "info" | "warning") => void, + config: EffectiveJudgeConfig, + sessionModel: Model | undefined, + catalog: LoadedModelCatalog, +): void { + const configured = config.judgeModel; + let judgeModelDescription: string; + let classification: ModelCatalogClassification | null; + if (configured !== undefined) { + judgeModelDescription = `${configured.provider}/${configured.id} (configured)`; + classification = classifyModel( + catalog, + configured.provider, + configured.id, + ); + } else { + if (sessionModel === undefined) { + judgeModelDescription = + "the current session model (none resolved yet)"; + classification = null; + } else { + judgeModelDescription = `${sessionModel.provider}/${sessionModel.id} (current session model)`; + classification = classifyModel( + catalog, + sessionModel.provider, + sessionModel.id, + ); + } + } + const catalogNote = + classification === "unlisted" + ? " This model is untested in the advisory catalog — used at your own risk." + : classification === "deprecated" || classification === "revoked" + ? ` Advisory catalog status: ${classification}.` + : ""; + notify( + `ai-bash-judge Enforce active: ${judgeModelDescription} judges Bash asks; allow skips the dialog — you accept the risk of model misjudgment (ADR 0008). High-risk shapes (irreversible, publish, system, credentials) always ask.${catalogNote}`, + "info", + ); +} + + /** Register a Shadow-only structured-output judge for local native Bash asks. */ export default function permissionAiJudge(pi: ExtensionAPI): void { let root: RootSession | undefined; @@ -184,284 +566,8 @@ export default function permissionAiJudge(pi: ExtensionAPI): void { const captured = root; disposeAuthorizer = service.registerAuthorizer( LINK_NAME, - async (details, _query, log) => { - const startedAt = Date.now(); - const sink: ReviewSink = createReviewSink({ - log, - reviewLogEnabled: captured.reviewLogEnabled, - }); - try { - // Judge-owned enrollment record (ADR 0006 denominator: - // asks the Judge received, per its own audit log). - // Non-bash surfaces are ignored below without a shadow - // row; they also do not enroll (v0.1 cohort is bash-only). - if ( - details.forwarding !== undefined || - details.payload.kind === "forwarded" || - details.payload.kind === "bash" - ) { - captured.auditLog.audit("ai_bash_judge.enrolled", { - requestId: details.requestId, - origin: - details.forwarding !== undefined || - details.payload.kind === "forwarded" - ? "forwarded" - : "local", - surface: "bash", - command: - details.payload.kind === "bash" && - details.payload.request?.value - ? details.payload.request.value - : (details.command ?? null), - }); - } - const emitResult = (record: Record): void => { - sink.review("ai_bash_judge.result", record); - captured.auditLog.audit("ai_bash_judge.result", record); - }; - // Forwarded asks do not carry a structured child full - // command in permission-system 25.3/25.4. Never parse the - // legacy prose. The deferral is recorded so the request - // stays visible in the offline denominator instead of - // silently vanishing. - if ( - details.forwarding !== undefined || - details.payload.kind === "forwarded" - ) { - emitResult({ - ...resultBase( - captured.judgeRuntimeId, - details, - startedAt, - captured.config, - ), - resultKind: "preflight_defer", - verdict: null, - effectiveVerdict: "defer", - modelCalled: false, - code: "missing_structured_input", - evidenceQuality: evidenceQuality(false, EMPTY_CONVERSATION, "", false), - }); - return { kind: "defer" }; - } - - // Ignore unrelated permission surfaces without producing a - // Shadow row or invoking the model: the v0.1 cohort selects - // accessSurface = bash only. - if (details.payload.kind !== "bash") { - return { kind: "defer" }; - } - - if ( - captured.getSessionId() !== captured.expectedSessionId - ) { - emitResult({ - ...resultBase( - captured.judgeRuntimeId, - details, - startedAt, - captured.config, - ), - resultKind: "preflight_defer", - verdict: null, - effectiveVerdict: "defer", - modelCalled: false, - code: "session_ownership_unproven", - evidenceQuality: evidenceQuality(false, EMPTY_CONVERSATION, ""), - }); - return { kind: "defer" }; - } - - const evidence = buildBashJudgmentEvidence(details); - if (evidence === undefined) { - emitResult({ - ...resultBase( - captured.judgeRuntimeId, - details, - startedAt, - captured.config, - ), - resultKind: "preflight_defer", - verdict: null, - effectiveVerdict: "defer", - modelCalled: false, - code: "invalid_evidence", - evidenceQuality: evidenceQuality(false, EMPTY_CONVERSATION, ""), - }); - return { kind: "defer" }; - } - - // Built-in high-risk override (ADR 0008): clear-cut - // irreversible/system shapes always defer. In Enforce the - // model is skipped entirely; in Shadow it still runs for - // quality observation and the override is recorded. - const risk: HighRiskMatch | undefined = classifyHighRisk( - evidence.fullCommand, - ); - if (risk !== undefined && captured.config.mode === "enforce") { - emitResult({ - ...resultBase( - captured.judgeRuntimeId, - details, - startedAt, - captured.config, - ), - resultKind: "preflight_defer", - verdict: null, - effectiveVerdict: "defer", - modelCalled: false, - code: "high_risk_override", - riskCategory: risk.category, - riskRule: risk.rule, - evidenceQuality: evidenceQuality(true, EMPTY_CONVERSATION, captured.getCwd()), - }); - return { kind: "defer" }; - } - - // Per-request judge-model resolution (PIEXTENSIO-3 cat.3 - // for the session model; ADR 0008 for a configured fixed - // model). A configured model that cannot be resolved is - // an observable infrastructure failure — never a silent - // fallback to the session model. - const resolved = resolveJudgeModel( - captured.config, - captured.getModel(), - captured.modelRegistry, - ); - if (resolved.kind === "unavailable") { - emitResult({ - ...resultBase( - captured.judgeRuntimeId, - details, - startedAt, - captured.config, - ), - resultKind: "infrastructure_failure", - verdict: null, - effectiveVerdict: "defer", - modelCalled: false, - code: "judge_model_unavailable", - provider: captured.config.judgeModel?.provider ?? null, - model: captured.config.judgeModel?.id ?? null, - api: null, - riskOverride: risk ?? null, - evidenceQuality: evidenceQuality(true, EMPTY_CONVERSATION, captured.getCwd()), - }); - return { kind: "defer" }; - } - const modelSource = resolved.source; - const availability: ModelAvailability = createModelAvailability( - resolved.model, - captured.modelRegistry, - ); - // Conversation evidence is captured at ask time from the - // live serving branch, not at session start: the newest - // user intent is the ask's intent. - const conversation: ConversationEvidence = - buildConversationEvidence(captured.conversation); - const result = await requestStructuredVerdict( - availability, - evidence, - captured.shutdown.signal, - captured.config.timeoutMs, - conversation, - ); - - // Enforce truth table (PIEXTENSIO-3 cat.4 / M5; ADR 0008): - // the fail-closed runtime health gates — audit health, - // telemetry, result kind, verdict, review - // acknowledgement, generation currency — are the single - // authority seam; the retired promotion gates are no - // longer inputs. reviewAcknowledged is true in the ADR - // 0006 sense: the Judge-owned audit write for this result - // happens before the authority return, and a failed write - // flips auditHealthy sticky-unhealthy, closing authority - // for every later ask. - const gateState: EnforceGateState = { - auditHealthy: captured.auditLog.healthy(), - telemetryHealth: sink.health(), - resultKind: - result.kind === "judgment" - ? "judgment" - : result.kind, - verdict: - result.kind === "judgment" ? result.verdict : null, - reviewAcknowledged: true, - generationCurrent: !captured.shutdown.signal.aborted, - mode: captured.config.mode, - }; - const authority = evaluateEnforceAuthority(gateState); - const effectiveVerdict = - authority.kind === "allow" ? "allow" : "defer"; - const authorityBlockedBy = - authority.kind === "allow" ? null : authority.blockedBy; - - if (result.kind === "judgment") { - emitResult({ - ...resultBase( - captured.judgeRuntimeId, - details, - startedAt, - captured.config, - ), - resultKind: "judgment", - verdict: result.verdict, - effectiveVerdict, - authorityBlockedBy, - modelCalled: true, - code: null, - modelSource, - provider: result.metadata.provider, - model: result.metadata.model, - api: result.metadata.api, - riskOverride: risk ?? null, - // Log keys deliberately avoid the substring - // "token": permission-system masks any key matching - // /token/i (structural key-name redaction), which - // would erase usage telemetry from the review log. - inputUsage: result.inputTokens, - outputUsage: result.outputTokens, - modelLatencyMs: result.modelLatencyMs, - reasonLength: reasonLength(result.reason), - evidenceQuality: evidenceQuality(true, conversation, captured.getCwd()), - }); - } else { - emitResult({ - ...resultBase( - captured.judgeRuntimeId, - details, - startedAt, - captured.config, - ), - resultKind: "infrastructure_failure", - verdict: null, - effectiveVerdict, - authorityBlockedBy, - modelCalled: result.modelCalled, - code: result.code, - modelSource, - provider: result.metadata?.provider ?? null, - model: result.metadata?.model ?? null, - api: result.metadata?.api ?? null, - riskOverride: risk ?? null, - inputUsage: result.inputTokens ?? null, - outputUsage: result.outputTokens ?? null, - modelLatencyMs: result.modelLatencyMs, - evidenceQuality: evidenceQuality(true, conversation, captured.getCwd()), - }); - } - - return authority.kind === "allow" - ? { kind: "allow" } - : { kind: "defer" }; - } catch { - // A link exception would abort the whole authority chain. - // Keep provider/payload/session failures fail-closed and do - // not include raw errors or authorization evidence in logs. - sink.debug("ai_bash_judge.exception"); - return { kind: "defer" }; - } - }, + async (details, _query, log) => + judgeAuthorize(captured, details, log), ); } @@ -513,40 +619,11 @@ export default function permissionAiJudge(pi: ExtensionAPI): void { ); } if (root.config.mode === "enforce") { - const configured = root.config.judgeModel; - let judgeModelDescription: string; - let classification: ModelCatalogClassification | null; - if (configured !== undefined) { - judgeModelDescription = `${configured.provider}/${configured.id} (configured)`; - classification = classifyModel( - catalogResult.catalog, - configured.provider, - configured.id, - ); - } else { - const sessionModel = ctx.model; - if (sessionModel === undefined) { - judgeModelDescription = - "the current session model (none resolved yet)"; - classification = null; - } else { - judgeModelDescription = `${sessionModel.provider}/${sessionModel.id} (current session model)`; - classification = classifyModel( - catalogResult.catalog, - sessionModel.provider, - sessionModel.id, - ); - } - } - const catalogNote = - classification === "unlisted" - ? " This model is untested in the advisory catalog — used at your own risk." - : classification === "deprecated" || classification === "revoked" - ? ` Advisory catalog status: ${classification}.` - : ""; - ctx.ui.notify( - `ai-bash-judge Enforce active: ${judgeModelDescription} judges Bash asks; allow skips the dialog — you accept the risk of model misjudgment (ADR 0008). High-risk shapes (irreversible, publish, system, credentials) always ask.${catalogNote}`, - "info", + notifyEnforceActive( + (message, kind) => ctx.ui.notify(message, kind), + root.config, + ctx.model, + catalogResult.catalog, ); } tryRegister();