mirror of
https://github.com/SikongJueluo/pi-extensions.git
synced 2026-10-05 11:52:55 +08:00
refactor(ai-judge): split remaining hotspots and cover CRAP branches
- stage-ify judgeAuthorize into auditEnrollment, runPreflightGates, prepareModelCall, and enforceAndEmit - extract tallyJoinedRows and tallyAttributable from computeMetrics, removing three dead locals - extract validateVerdictResponse from requestStructuredVerdict - extract collectUserTexts and charBudgetStart from buildConversationEvidence - table-drive corpus-replay parseArgs and split main into resolveReplayModel, selectCorpusCases, and replayCorpus - split analyzer cli main into loadReviewEvents, withinWindow, loadAuditEnrolled, and printReport with a run-as-script guard - add 81 tests covering parseEntry, extractBashCommandEvidence, validateVerdictResponse, forcedToolChoice, classifyGit dry-run paths, CLI arg/window/report rendering, and the infra-failure result path - add @vitest/coverage-istanbul for exact per-function CRAP scoring via fallow health --coverage
This commit is contained in:
File diff suppressed because one or more lines are too long
@@ -21,6 +21,8 @@
|
|||||||
"@earendil-works/pi-coding-agent": "*",
|
"@earendil-works/pi-coding-agent": "*",
|
||||||
"@gotgenes/pi-permission-system": ">=25.4.0",
|
"@gotgenes/pi-permission-system": ">=25.4.0",
|
||||||
"@types/node": "^26.0.0",
|
"@types/node": "^26.0.0",
|
||||||
|
"@vitest/coverage-istanbul": "3.2.7",
|
||||||
|
"@vitest/coverage-v8": "^3.2.7",
|
||||||
"typescript": "^5",
|
"typescript": "^5",
|
||||||
"vitest": "^3"
|
"vitest": "^3"
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -411,24 +411,73 @@ export function analyzeShadowReviewLog(events: readonly ReviewEvent[]): AnalyzeR
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
function computeMetrics(
|
/** Per-kind tallies collected in one pass over the joined rows. */
|
||||||
n: number,
|
interface RowTallies {
|
||||||
joined: readonly JoinedRow[],
|
readonly joinedJudgments: number;
|
||||||
quarantined: Record<string, number>,
|
readonly falseAllows: number;
|
||||||
): Metrics {
|
readonly conservativeDeny: number;
|
||||||
const matrix: Record<string, number> = {};
|
readonly conservativeDefer: number;
|
||||||
|
readonly humanAllowJudgments: number;
|
||||||
|
readonly preflightDefers: number;
|
||||||
|
readonly infrastructureFailures: number;
|
||||||
|
readonly infraByCode: Record<string, number>;
|
||||||
|
/** Comparison matrix [ai|human] over attributable judgment rows. */
|
||||||
|
readonly matrix: Record<string, number>;
|
||||||
|
readonly judgeLatencies: Array<number | null | undefined>;
|
||||||
|
readonly modelLatencies: Array<number | null | undefined>;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Counters accumulated over attributable judgment rows. */
|
||||||
|
interface MatrixCounters {
|
||||||
|
matrix: Record<string, number>;
|
||||||
|
falseAllows: number;
|
||||||
|
conservativeDeny: number;
|
||||||
|
conservativeDefer: number;
|
||||||
|
humanAllowJudgments: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Tally one attributable judgment row into the comparison matrix and the
|
||||||
|
* conservative-direction counters. Unattributable rows never reach here.
|
||||||
|
*/
|
||||||
|
function tallyAttributable(row: JoinedRow, c: MatrixCounters): void {
|
||||||
|
const key = `${row.verdict ?? "null"}|${row.human.decision}`;
|
||||||
|
c.matrix[key] = (c.matrix[key] ?? 0) + 1;
|
||||||
|
if (row.human.decision === "allow") {
|
||||||
|
c.humanAllowJudgments += 1;
|
||||||
|
}
|
||||||
|
if (row.verdict === "allow" && row.human.decision === "deny") {
|
||||||
|
c.falseAllows += 1;
|
||||||
|
}
|
||||||
|
if (row.verdict === "deny" && row.human.decision === "allow") {
|
||||||
|
c.conservativeDeny += 1;
|
||||||
|
}
|
||||||
|
if (row.verdict === "defer" && row.human.decision === "allow") {
|
||||||
|
c.conservativeDefer += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* One pass over joined rows: latencies are collected for every row;
|
||||||
|
* non-judgment rows (preflight/infra) are tallied and excluded; judgment
|
||||||
|
* rows enter the comparison matrix unless their human attribution is
|
||||||
|
* `unproven` (those stay joined but never enter the matrix).
|
||||||
|
*/
|
||||||
|
function tallyJoinedRows(joined: readonly JoinedRow[]): RowTallies {
|
||||||
const infraByCode: Record<string, number> = {};
|
const infraByCode: Record<string, number> = {};
|
||||||
const judgeLatencies: Array<number | null | undefined> = [];
|
const judgeLatencies: Array<number | null | undefined> = [];
|
||||||
const modelLatencies: Array<number | null | undefined> = [];
|
const modelLatencies: Array<number | null | undefined> = [];
|
||||||
|
|
||||||
let joinedJudgments = 0;
|
let joinedJudgments = 0;
|
||||||
let attributed = 0;
|
|
||||||
let falseAllows = 0;
|
|
||||||
let conservativeDeny = 0;
|
|
||||||
let conservativeDefer = 0;
|
|
||||||
let humanAllowJudgments = 0;
|
|
||||||
let preflightDefers = 0;
|
let preflightDefers = 0;
|
||||||
let infrastructureFailures = 0;
|
let infrastructureFailures = 0;
|
||||||
|
const counters: MatrixCounters = {
|
||||||
|
matrix: {},
|
||||||
|
falseAllows: 0,
|
||||||
|
conservativeDeny: 0,
|
||||||
|
conservativeDefer: 0,
|
||||||
|
humanAllowJudgments: 0,
|
||||||
|
};
|
||||||
|
|
||||||
for (const row of joined) {
|
for (const row of joined) {
|
||||||
judgeLatencies.push(row.judgeLatencyMs);
|
judgeLatencies.push(row.judgeLatencyMs);
|
||||||
@@ -447,63 +496,65 @@ function computeMetrics(
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
joinedJudgments += 1;
|
joinedJudgments += 1;
|
||||||
if (row.humanAttribution === "unproven") {
|
if (row.humanAttribution !== "unproven") {
|
||||||
continue;
|
tallyAttributable(row, counters);
|
||||||
}
|
|
||||||
attributed += 1;
|
|
||||||
const key = `${row.verdict ?? "null"}|${row.human.decision}`;
|
|
||||||
matrix[key] = (matrix[key] ?? 0) + 1;
|
|
||||||
if (row.human.decision === "allow") {
|
|
||||||
humanAllowJudgments += 1;
|
|
||||||
}
|
|
||||||
if (row.verdict === "allow" && row.human.decision === "deny") {
|
|
||||||
falseAllows += 1;
|
|
||||||
}
|
|
||||||
if (row.verdict === "deny" && row.human.decision === "allow") {
|
|
||||||
conservativeDeny += 1;
|
|
||||||
}
|
|
||||||
if (row.verdict === "defer" && row.human.decision === "allow") {
|
|
||||||
conservativeDefer += 1;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
const completionCoverage = n > 0 ? (joined.length + missingResults(n, joined)) / n : 0;
|
return {
|
||||||
|
joinedJudgments,
|
||||||
|
falseAllows: counters.falseAllows,
|
||||||
|
conservativeDeny: counters.conservativeDeny,
|
||||||
|
conservativeDefer: counters.conservativeDefer,
|
||||||
|
humanAllowJudgments: counters.humanAllowJudgments,
|
||||||
|
preflightDefers,
|
||||||
|
infrastructureFailures,
|
||||||
|
infraByCode,
|
||||||
|
matrix: counters.matrix,
|
||||||
|
judgeLatencies,
|
||||||
|
modelLatencies,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function computeMetrics(
|
||||||
|
n: number,
|
||||||
|
joined: readonly JoinedRow[],
|
||||||
|
quarantined: Record<string, number>,
|
||||||
|
): Metrics {
|
||||||
|
const t = tallyJoinedRows(joined);
|
||||||
return {
|
return {
|
||||||
joined: joined.length,
|
joined: joined.length,
|
||||||
quarantined,
|
quarantined,
|
||||||
joinedJudgments,
|
joinedJudgments: t.joinedJudgments,
|
||||||
completionCoverage: joined.length / (n || 1),
|
completionCoverage: joined.length / (n || 1),
|
||||||
humanJoinCoverage: (joined.length - quarantinedCount(joined)) / (n || 1),
|
humanJoinCoverage: (joined.length - unprovenCount(joined)) / (n || 1),
|
||||||
judgmentCoverage: joinedJudgments / (n || 1),
|
judgmentCoverage: t.joinedJudgments / (n || 1),
|
||||||
matrix,
|
matrix: t.matrix,
|
||||||
falseAllows,
|
falseAllows: t.falseAllows,
|
||||||
falseAllowRate: falseAllows > 0 || hasAllowPrediction(matrix)
|
falseAllowRate: t.falseAllows > 0 || hasAllowPrediction(t.matrix)
|
||||||
? falseAllows / allowPredictions(matrix)
|
? t.falseAllows / allowPredictions(t.matrix)
|
||||||
: null,
|
: null,
|
||||||
conservativeDeny,
|
conservativeDeny: t.conservativeDeny,
|
||||||
conservativeDefer,
|
conservativeDefer: t.conservativeDefer,
|
||||||
conservativeRate:
|
conservativeRate:
|
||||||
humanAllowJudgments > 0
|
t.humanAllowJudgments > 0
|
||||||
? (conservativeDeny + conservativeDefer) / humanAllowJudgments
|
? (t.conservativeDeny + t.conservativeDefer) / t.humanAllowJudgments
|
||||||
: null,
|
: null,
|
||||||
preflightDefers,
|
preflightDefers: t.preflightDefers,
|
||||||
infrastructureFailures,
|
infrastructureFailures: t.infrastructureFailures,
|
||||||
infrastructureByCode: infraByCode,
|
infrastructureByCode: t.infraByCode,
|
||||||
judgeLatency: latencyStats(judgeLatencies),
|
judgeLatency: latencyStats(t.judgeLatencies),
|
||||||
modelLatency: latencyStats(modelLatencies),
|
modelLatency: latencyStats(t.modelLatencies),
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
// ── helper functions below exist to keep computeMetrics readable; they are
|
// ── helper functions below exist to keep computeMetrics readable; they are
|
||||||
// not part of the public surface.
|
// not part of the public surface.
|
||||||
|
|
||||||
function missingResults(n: number, joined: readonly JoinedRow[]): number {
|
function unprovenCount(joined: readonly JoinedRow[]): number {
|
||||||
return Math.max(0, n - joined.length);
|
// The human-join coverage counts joined rows with an
|
||||||
}
|
// attributable human outcome; `unproven` rows are joined but
|
||||||
function quarantinedCount(joined: readonly JoinedRow[]): number {
|
// never attributable.
|
||||||
// Quarantined rows are tracked outside `joined`; in this simplified
|
|
||||||
// metric path the human-join coverage counts joined rows with an
|
|
||||||
// attributable human outcome.
|
|
||||||
return joined.filter((r) => r.humanAttribution === "unproven").length;
|
return joined.filter((r) => r.humanAttribution === "unproven").length;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -8,6 +8,7 @@
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
import { readFileSync } from "node:fs";
|
import { readFileSync } from "node:fs";
|
||||||
|
import { pathToFileURL } from "node:url";
|
||||||
import { analyzeShadowReviewLog, type ReviewEvent } from "./analyze";
|
import { analyzeShadowReviewLog, type ReviewEvent } from "./analyze";
|
||||||
|
|
||||||
const USAGE = `usage: analyze-shadow <review-jsonl-path> [options]
|
const USAGE = `usage: analyze-shadow <review-jsonl-path> [options]
|
||||||
@@ -32,7 +33,23 @@ interface CliOptions {
|
|||||||
readonly audit: string | null;
|
readonly audit: string | null;
|
||||||
}
|
}
|
||||||
|
|
||||||
function parseArgs(argv: readonly string[]): CliOptions | { error: string } {
|
/** Parse one `--after`/`--before` timestamp value; error string on failure. */
|
||||||
|
export function parseTimestampOption(
|
||||||
|
arg: string,
|
||||||
|
value: string | undefined,
|
||||||
|
): { date: Date } | { error: string } {
|
||||||
|
if (value === undefined) {
|
||||||
|
return { error: `${arg} requires a value` };
|
||||||
|
}
|
||||||
|
const date = new Date(value);
|
||||||
|
if (Number.isNaN(date.getTime())) {
|
||||||
|
return { error: `invalid ${arg} timestamp: ${value}` };
|
||||||
|
}
|
||||||
|
return { date };
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Parse CLI options; exported for in-process tests. */
|
||||||
|
export function parseArgs(argv: readonly string[]): CliOptions | { error: string } {
|
||||||
const args = argv.slice(2);
|
const args = argv.slice(2);
|
||||||
let path: string | undefined;
|
let path: string | undefined;
|
||||||
let after: Date | null = null;
|
let after: Date | null = null;
|
||||||
@@ -43,24 +60,24 @@ function parseArgs(argv: readonly string[]): CliOptions | { error: string } {
|
|||||||
if (arg === "--help" || arg === "-h") {
|
if (arg === "--help" || arg === "-h") {
|
||||||
return { error: USAGE };
|
return { error: USAGE };
|
||||||
}
|
}
|
||||||
if (arg === "--after" || arg === "--before" || arg === "--audit") {
|
if (arg === "--audit") {
|
||||||
const value = args[i + 1];
|
const value = args[i + 1];
|
||||||
if (value === undefined) {
|
if (value === undefined) {
|
||||||
return { error: `${arg} requires a value` };
|
return { error: `${arg} requires a value` };
|
||||||
}
|
}
|
||||||
if (arg === "--audit") {
|
|
||||||
audit = value;
|
audit = value;
|
||||||
i += 1;
|
i += 1;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
const parsed = new Date(value);
|
if (arg === "--after" || arg === "--before") {
|
||||||
if (Number.isNaN(parsed.getTime())) {
|
const parsed = parseTimestampOption(arg, args[i + 1]);
|
||||||
return { error: `invalid ${arg} timestamp: ${value}` };
|
if ("error" in parsed) {
|
||||||
|
return parsed;
|
||||||
}
|
}
|
||||||
if (arg === "--after") {
|
if (arg === "--after") {
|
||||||
after = parsed;
|
after = parsed.date;
|
||||||
} else {
|
} else {
|
||||||
before = parsed;
|
before = parsed.date;
|
||||||
}
|
}
|
||||||
i += 1;
|
i += 1;
|
||||||
continue;
|
continue;
|
||||||
@@ -112,6 +129,63 @@ function fmtLatency(stats: {
|
|||||||
return `p50=${stats.p50}ms p95=${stats.p95}ms max=${stats.max}ms missing=${stats.missing}`;
|
return `p50=${stats.p50}ms p95=${stats.p95}ms max=${stats.max}ms missing=${stats.missing}`;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Parse one review-log JSONL file, skipping blank/unparseable lines. */
|
||||||
|
function loadReviewEvents(path: string): ReviewEvent[] {
|
||||||
|
let raw: string;
|
||||||
|
try {
|
||||||
|
raw = readFileSync(path, "utf-8");
|
||||||
|
} catch (error) {
|
||||||
|
process.stderr.write(
|
||||||
|
`error: cannot read ${path}: ${error instanceof Error ? error.message : String(error)}\n`,
|
||||||
|
);
|
||||||
|
process.exit(1);
|
||||||
|
}
|
||||||
|
return raw
|
||||||
|
.split("\n")
|
||||||
|
.map((line, index) => parseLine(line, index + 1))
|
||||||
|
.filter((evt): evt is ReviewEvent => evt !== null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The `--after`/`--before` selection window; exported for tests. */
|
||||||
|
export interface TimeWindow {
|
||||||
|
readonly after: Date | null;
|
||||||
|
readonly before: Date | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** True when an event's timestamp falls inside the window (or is absent). */
|
||||||
|
export function withinWindow(evt: ReviewEvent, window: TimeWindow): boolean {
|
||||||
|
const ts = typeof evt.timestamp === "string" ? evt.timestamp : null;
|
||||||
|
if (ts === null) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
const time = new Date(ts).getTime();
|
||||||
|
if (window.after !== null && !Number.isNaN(time) && time < window.after.getTime()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (window.before !== null && !Number.isNaN(time) && time > window.before.getTime()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* ADR 0006 dual-log mode: with --audit, the denominator comes from the
|
||||||
|
* Judge's own enrolled rows (asks it received), rewritten as synthetic
|
||||||
|
* enrollment events so the reconstructed permission-system enrollment
|
||||||
|
* proxy can be dropped to avoid a second denominator source. Human
|
||||||
|
* decisions still come from the permission log (attribution join).
|
||||||
|
*/
|
||||||
|
function loadAuditEnrolled(auditPath: string, window: TimeWindow): ReviewEvent[] {
|
||||||
|
return loadReviewEvents(auditPath)
|
||||||
|
.filter((evt) => evt.event === "ai_bash_judge.enrolled")
|
||||||
|
.filter((evt) => withinWindow(evt, window))
|
||||||
|
.map((evt) => ({
|
||||||
|
...evt,
|
||||||
|
event: "authorizer_chain_resolved",
|
||||||
|
links: ["ai-bash-judge"],
|
||||||
|
}) as ReviewEvent);
|
||||||
|
}
|
||||||
|
|
||||||
function main(): void {
|
function main(): void {
|
||||||
const parsed = parseArgs(process.argv);
|
const parsed = parseArgs(process.argv);
|
||||||
if ("error" in parsed) {
|
if ("error" in parsed) {
|
||||||
@@ -119,75 +193,13 @@ function main(): void {
|
|||||||
process.exit(parsed.error === USAGE ? 0 : 1);
|
process.exit(parsed.error === USAGE ? 0 : 1);
|
||||||
}
|
}
|
||||||
|
|
||||||
let raw: string;
|
const window: TimeWindow = { after: parsed.after, before: parsed.before };
|
||||||
try {
|
const events = loadReviewEvents(parsed.path).filter((evt) =>
|
||||||
raw = readFileSync(parsed.path, "utf-8");
|
withinWindow(evt, window),
|
||||||
} catch (error) {
|
|
||||||
process.stderr.write(
|
|
||||||
`error: cannot read ${parsed.path}: ${error instanceof Error ? error.message : String(error)}\n`,
|
|
||||||
);
|
);
|
||||||
process.exit(1);
|
|
||||||
}
|
|
||||||
|
|
||||||
const events = raw
|
|
||||||
.split("\n")
|
|
||||||
.map((line, index) => parseLine(line, index + 1))
|
|
||||||
.filter((evt): evt is ReviewEvent => evt !== null)
|
|
||||||
.filter((evt) => {
|
|
||||||
const ts = typeof evt.timestamp === "string" ? evt.timestamp : null;
|
|
||||||
if (ts === null) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
const time = new Date(ts).getTime();
|
|
||||||
if (parsed.after !== null && !Number.isNaN(time) && time < parsed.after.getTime()) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
if (parsed.before !== null && !Number.isNaN(time) && time > parsed.before.getTime()) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
});
|
|
||||||
|
|
||||||
// ADR 0006 dual-log mode: with --audit, the denominator comes from the
|
|
||||||
// Judge's own enrolled rows (asks it received), and the reconstructed
|
|
||||||
// permission-system enrollment proxy is dropped to avoid a second
|
|
||||||
// denominator source. Human decisions still come from the permission
|
|
||||||
// log (attribution join).
|
|
||||||
let joined = events;
|
let joined = events;
|
||||||
if (parsed.audit !== null) {
|
if (parsed.audit !== null) {
|
||||||
let auditRaw: string;
|
const auditEnrolled = loadAuditEnrolled(parsed.audit, window);
|
||||||
try {
|
|
||||||
auditRaw = readFileSync(parsed.audit, "utf-8");
|
|
||||||
} catch (error) {
|
|
||||||
process.stderr.write(
|
|
||||||
`error: cannot read audit log ${parsed.audit}: ${error instanceof Error ? error.message : String(error)}\n`,
|
|
||||||
);
|
|
||||||
process.exit(1);
|
|
||||||
}
|
|
||||||
const auditEnrolled = auditRaw
|
|
||||||
.split("\n")
|
|
||||||
.map((line, index) => parseLine(line, index + 1))
|
|
||||||
.filter((evt): evt is ReviewEvent => evt !== null)
|
|
||||||
.filter((evt) => evt.event === "ai_bash_judge.enrolled")
|
|
||||||
.filter((evt) => {
|
|
||||||
const ts = typeof evt.timestamp === "string" ? evt.timestamp : null;
|
|
||||||
if (ts === null) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
const time = new Date(ts).getTime();
|
|
||||||
if (parsed.after !== null && !Number.isNaN(time) && time < parsed.after.getTime()) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
if (parsed.before !== null && !Number.isNaN(time) && time > parsed.before.getTime()) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
})
|
|
||||||
.map((evt) => ({
|
|
||||||
...evt,
|
|
||||||
event: "authorizer_chain_resolved",
|
|
||||||
links: ["ai-bash-judge"],
|
|
||||||
}) as ReviewEvent);
|
|
||||||
joined = [
|
joined = [
|
||||||
...events.filter((evt) => evt.event !== "authorizer_chain_resolved"),
|
...events.filter((evt) => evt.event !== "authorizer_chain_resolved"),
|
||||||
...auditEnrolled,
|
...auditEnrolled,
|
||||||
@@ -195,6 +207,14 @@ function main(): void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
const { enrollments, metrics } = analyzeShadowReviewLog(joined);
|
const { enrollments, metrics } = analyzeShadowReviewLog(joined);
|
||||||
|
printReport(parsed, enrollments, metrics);
|
||||||
|
}
|
||||||
|
|
||||||
|
export function printReport(
|
||||||
|
parsed: CliOptions,
|
||||||
|
enrollments: number,
|
||||||
|
metrics: ReturnType<typeof analyzeShadowReviewLog>["metrics"],
|
||||||
|
): void {
|
||||||
const out = process.stdout;
|
const out = process.stdout;
|
||||||
|
|
||||||
out.write("AI Bash Judge — Shadow diagnostic report\n");
|
out.write("AI Bash Judge — Shadow diagnostic report\n");
|
||||||
@@ -253,4 +273,11 @@ function main(): void {
|
|||||||
out.write(`model latency: ${fmtLatency(metrics.modelLatency)}\n`);
|
out.write(`model latency: ${fmtLatency(metrics.modelLatency)}\n`);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Run as a script only (not under vitest imports): same guard as
|
||||||
|
// tools/corpus-replay.ts.
|
||||||
|
if (
|
||||||
|
process.argv[1] !== undefined &&
|
||||||
|
import.meta.url === pathToFileURL(process.argv[1]).href
|
||||||
|
) {
|
||||||
main();
|
main();
|
||||||
|
}
|
||||||
|
|||||||
@@ -92,22 +92,20 @@ function userTextFrom(message: unknown): string | null {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Build bounded conversation evidence from the active branch.
|
* Newest-first user texts from the active branch, bounded by
|
||||||
*
|
* MAX_CONVERSATION_ITEMS. Compaction boundaries are flagged, not kept.
|
||||||
* Iterates entries newest-first to guarantee latest-user preservation,
|
|
||||||
* then reverses for newest-last output. Head/middle/tail truncation: with
|
|
||||||
* more items than fit, the newest `MAX_CONVERSATION_ITEMS` are kept and
|
|
||||||
* `truncated` is set — the head is the part dropped, which keeps the most
|
|
||||||
* recent intent window intact and matches "latest-user preservation".
|
|
||||||
*/
|
*/
|
||||||
export function buildConversationEvidence(
|
function collectUserTexts(
|
||||||
probe: ConversationProbe,
|
entries: readonly unknown[],
|
||||||
): ConversationEvidence {
|
): { texts: string[]; hasCompaction: boolean } {
|
||||||
const entries = probe.getActiveEntries();
|
|
||||||
const collected: string[] = [];
|
const collected: string[] = [];
|
||||||
let hasCompaction = false;
|
let hasCompaction = false;
|
||||||
|
|
||||||
for (let i = entries.length - 1; i >= 0 && collected.length < MAX_CONVERSATION_ITEMS; i -= 1) {
|
for (
|
||||||
|
let i = entries.length - 1;
|
||||||
|
i >= 0 && collected.length < MAX_CONVERSATION_ITEMS;
|
||||||
|
i -= 1
|
||||||
|
) {
|
||||||
const entry = entries[i];
|
const entry = entries[i];
|
||||||
if (!isRecord(entry)) {
|
if (!isRecord(entry)) {
|
||||||
continue;
|
continue;
|
||||||
@@ -124,6 +122,39 @@ export function buildConversationEvidence(
|
|||||||
collected.push(text);
|
collected.push(text);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
return { texts: collected, hasCompaction };
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Character budget: the head (oldest) drop index so the remainder fits
|
||||||
|
* within MAX_CONVERSATION_CHARS; 0 when everything fits. Always keeps at
|
||||||
|
* least the newest item.
|
||||||
|
*/
|
||||||
|
function charBudgetStart(kept: readonly string[]): number {
|
||||||
|
let rendered = 0;
|
||||||
|
for (let i = 0; i < kept.length; i += 1) {
|
||||||
|
rendered += kept[i]?.length ?? 0;
|
||||||
|
if (rendered > MAX_CONVERSATION_CHARS) {
|
||||||
|
return Math.max(1, i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Build bounded conversation evidence from the active branch.
|
||||||
|
*
|
||||||
|
* Iterates entries newest-first to guarantee latest-user preservation,
|
||||||
|
* then reverses for newest-last output. Head/middle/tail truncation: with
|
||||||
|
* more items than fit, the newest `MAX_CONVERSATION_ITEMS` are kept and
|
||||||
|
* `truncated` is set — the head is the part dropped, which keeps the most
|
||||||
|
* recent intent window intact and matches "latest-user preservation".
|
||||||
|
*/
|
||||||
|
export function buildConversationEvidence(
|
||||||
|
probe: ConversationProbe,
|
||||||
|
): ConversationEvidence {
|
||||||
|
const entries = probe.getActiveEntries();
|
||||||
|
const { texts: collected, hasCompaction } = collectUserTexts(entries);
|
||||||
|
|
||||||
const kept = collected.reverse();
|
const kept = collected.reverse();
|
||||||
const totalItems = countUserEntries(entries);
|
const totalItems = countUserEntries(entries);
|
||||||
@@ -131,22 +162,14 @@ export function buildConversationEvidence(
|
|||||||
|
|
||||||
// Character budget: drop from the head (oldest) until it fits; the
|
// Character budget: drop from the head (oldest) until it fits; the
|
||||||
// newest items are preserved. A dropped head is recorded by `truncated`.
|
// newest items are preserved. A dropped head is recorded by `truncated`.
|
||||||
let rendered = 0;
|
const start = charBudgetStart(kept);
|
||||||
let start = 0;
|
|
||||||
for (let i = 0; i < kept.length; i += 1) {
|
|
||||||
rendered += kept[i]?.length ?? 0;
|
|
||||||
if (rendered > MAX_CONVERSATION_CHARS) {
|
|
||||||
start = Math.max(1, i); // keep at least the newest item
|
|
||||||
rendered = 0;
|
|
||||||
for (let j = start; j < kept.length; j += 1) {
|
|
||||||
rendered += kept[j]?.length ?? 0;
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const finalItems = kept
|
const finalItems = kept
|
||||||
.slice(start)
|
.slice(start)
|
||||||
.map((text, index) => ({ position: index + 1, role: "user" as const, text }));
|
.map((text, index) => ({ position: index + 1, role: "user" as const, text }));
|
||||||
|
const rendered = finalItems.reduce(
|
||||||
|
(sum, item) => sum + item.text.length,
|
||||||
|
0,
|
||||||
|
);
|
||||||
|
|
||||||
return {
|
return {
|
||||||
items: finalItems,
|
items: finalItems,
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ import {
|
|||||||
type AuthorizerLog,
|
type AuthorizerLog,
|
||||||
type AuthorizerVerdict,
|
type AuthorizerVerdict,
|
||||||
} from "@gotgenes/pi-permission-system";
|
} from "@gotgenes/pi-permission-system";
|
||||||
import { buildBashJudgmentEvidence } from "./evidence";
|
import { buildBashJudgmentEvidence, type BashJudgmentEvidence } from "./evidence";
|
||||||
import {
|
import {
|
||||||
createModelAvailability,
|
createModelAvailability,
|
||||||
requestStructuredVerdict,
|
requestStructuredVerdict,
|
||||||
@@ -300,35 +300,14 @@ function emitInfrastructureResult(
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* One authorize call: enroll, preflight-gate, judge, and enforce the truth
|
* Judge-owned enrollment record (ADR 0006 denominator: asks the Judge
|
||||||
* table. Extracted from the registerAuthorizer callback so each stage reads
|
* received, per its own audit log). Non-bash surfaces are ignored later
|
||||||
* linearly; any exception fails closed (defer) without logging raw errors.
|
* without a shadow row; they also do not enroll (v0.1 cohort is bash-only).
|
||||||
*/
|
*/
|
||||||
async function judgeAuthorize(
|
function auditEnrollment(
|
||||||
captured: RootSession,
|
captured: RootSession,
|
||||||
details: PromptPermissionDetails,
|
details: PromptPermissionDetails,
|
||||||
log: AuthorizerLog,
|
): void {
|
||||||
): Promise<AuthorizerVerdict> {
|
|
||||||
const startedAt = Date.now();
|
|
||||||
const sink: ReviewSink = createReviewSink({
|
|
||||||
log,
|
|
||||||
reviewLogEnabled: captured.reviewLogEnabled,
|
|
||||||
});
|
|
||||||
try {
|
|
||||||
const ctx: EmitContext = {
|
|
||||||
captured,
|
|
||||||
details,
|
|
||||||
startedAt,
|
|
||||||
emitResult: (record) => {
|
|
||||||
sink.review("ai_bash_judge.result", record);
|
|
||||||
captured.auditLog.audit("ai_bash_judge.result", record);
|
|
||||||
},
|
|
||||||
};
|
|
||||||
|
|
||||||
// Judge-owned enrollment record (ADR 0006 denominator:
|
|
||||||
// asks the Judge received, per its own audit log).
|
|
||||||
// Non-bash surfaces are ignored below without a shadow
|
|
||||||
// row; they also do not enroll (v0.1 cohort is bash-only).
|
|
||||||
if (
|
if (
|
||||||
details.forwarding !== undefined ||
|
details.forwarding !== undefined ||
|
||||||
details.payload.kind === "forwarded" ||
|
details.payload.kind === "forwarded" ||
|
||||||
@@ -349,74 +328,118 @@ async function judgeAuthorize(
|
|||||||
: (details.command ?? null),
|
: (details.command ?? null),
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
// Forwarded asks do not carry a structured child full
|
}
|
||||||
// command in permission-system 25.3/25.4. Never parse the
|
|
||||||
// legacy prose. The deferral is recorded so the request
|
/** A preflight-gate outcome: proceed with usable evidence, or stop. */
|
||||||
// stays visible in the offline denominator instead of
|
type PreflightGate =
|
||||||
// silently vanishing.
|
| {
|
||||||
|
readonly kind: "proceed";
|
||||||
|
readonly evidence: BashJudgmentEvidence;
|
||||||
|
readonly risk: HighRiskMatch | undefined;
|
||||||
|
}
|
||||||
|
| { readonly kind: "stop"; readonly verdict: AuthorizerVerdict };
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fail-closed preflight gates, in order:
|
||||||
|
* 1. Forwarded asks do not carry a structured child full command in
|
||||||
|
* permission-system 25.3/25.4 — never parse the legacy prose. The
|
||||||
|
* deferral is recorded so the request stays visible in the offline
|
||||||
|
* denominator instead of silently vanishing.
|
||||||
|
* 2. Unrelated permission surfaces produce no Shadow row and no model
|
||||||
|
* call: the v0.1 cohort selects accessSurface = bash only.
|
||||||
|
* 3. Session ownership must be proven.
|
||||||
|
* 4. Evidence must be structured and valid.
|
||||||
|
* 5. Built-in high-risk override (ADR 0008): clear-cut irreversible/system
|
||||||
|
* shapes always defer. In Enforce the model is skipped entirely; in
|
||||||
|
* Shadow it still runs for quality observation and the override is
|
||||||
|
* recorded.
|
||||||
|
*/
|
||||||
|
function runPreflightGates(
|
||||||
|
ctx: EmitContext,
|
||||||
|
captured: RootSession,
|
||||||
|
details: PromptPermissionDetails,
|
||||||
|
): PreflightGate {
|
||||||
if (
|
if (
|
||||||
details.forwarding !== undefined ||
|
details.forwarding !== undefined ||
|
||||||
details.payload.kind === "forwarded"
|
details.payload.kind === "forwarded"
|
||||||
) {
|
) {
|
||||||
return preflightDefer(
|
return {
|
||||||
|
kind: "stop",
|
||||||
|
verdict: preflightDefer(
|
||||||
ctx,
|
ctx,
|
||||||
"missing_structured_input",
|
"missing_structured_input",
|
||||||
evidenceQuality(false, EMPTY_CONVERSATION, "", false),
|
evidenceQuality(false, EMPTY_CONVERSATION, "", false),
|
||||||
);
|
),
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
// Ignore unrelated permission surfaces without producing a
|
|
||||||
// Shadow row or invoking the model: the v0.1 cohort selects
|
|
||||||
// accessSurface = bash only.
|
|
||||||
if (details.payload.kind !== "bash") {
|
if (details.payload.kind !== "bash") {
|
||||||
return { kind: "defer" };
|
return { kind: "stop", verdict: { kind: "defer" } };
|
||||||
}
|
}
|
||||||
|
|
||||||
if (captured.getSessionId() !== captured.expectedSessionId) {
|
if (captured.getSessionId() !== captured.expectedSessionId) {
|
||||||
return preflightDefer(
|
return {
|
||||||
|
kind: "stop",
|
||||||
|
verdict: preflightDefer(
|
||||||
ctx,
|
ctx,
|
||||||
"session_ownership_unproven",
|
"session_ownership_unproven",
|
||||||
evidenceQuality(false, EMPTY_CONVERSATION, ""),
|
evidenceQuality(false, EMPTY_CONVERSATION, ""),
|
||||||
);
|
),
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
const evidence = buildBashJudgmentEvidence(details);
|
const evidence = buildBashJudgmentEvidence(details);
|
||||||
if (evidence === undefined) {
|
if (evidence === undefined) {
|
||||||
return preflightDefer(
|
return {
|
||||||
|
kind: "stop",
|
||||||
|
verdict: preflightDefer(
|
||||||
ctx,
|
ctx,
|
||||||
"invalid_evidence",
|
"invalid_evidence",
|
||||||
evidenceQuality(false, EMPTY_CONVERSATION, ""),
|
evidenceQuality(false, EMPTY_CONVERSATION, ""),
|
||||||
);
|
),
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
// Built-in high-risk override (ADR 0008): clear-cut
|
const risk = classifyHighRisk(evidence.fullCommand);
|
||||||
// irreversible/system shapes always defer. In Enforce the
|
|
||||||
// model is skipped entirely; in Shadow it still runs for
|
|
||||||
// quality observation and the override is recorded.
|
|
||||||
const risk: HighRiskMatch | undefined = classifyHighRisk(
|
|
||||||
evidence.fullCommand,
|
|
||||||
);
|
|
||||||
if (risk !== undefined && captured.config.mode === "enforce") {
|
if (risk !== undefined && captured.config.mode === "enforce") {
|
||||||
return preflightDefer(
|
return {
|
||||||
|
kind: "stop",
|
||||||
|
verdict: preflightDefer(
|
||||||
ctx,
|
ctx,
|
||||||
"high_risk_override",
|
"high_risk_override",
|
||||||
evidenceQuality(true, EMPTY_CONVERSATION, captured.getCwd()),
|
evidenceQuality(true, EMPTY_CONVERSATION, captured.getCwd()),
|
||||||
{ riskCategory: risk.category, riskRule: risk.rule },
|
{ riskCategory: risk.category, riskRule: risk.rule },
|
||||||
);
|
),
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
// Per-request judge-model resolution (PIEXTENSIO-3 cat.3
|
return { kind: "proceed", evidence, risk };
|
||||||
// for the session model; ADR 0008 for a configured fixed
|
}
|
||||||
// model). A configured model that cannot be resolved is
|
|
||||||
// an observable infrastructure failure — never a silent
|
/** Prepared model-call inputs; undefined when an infra defer was emitted. */
|
||||||
// fallback to the session model.
|
interface PreparedModelCall {
|
||||||
|
readonly availability: ModelAvailability;
|
||||||
|
readonly modelSource: "configured" | "session";
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Per-request judge-model resolution (PIEXTENSIO-3 cat.3 for the session
|
||||||
|
* model; ADR 0008 for a configured fixed model). A configured model that
|
||||||
|
* cannot be resolved is an observable infrastructure failure — never a
|
||||||
|
* silent fallback to the session model.
|
||||||
|
*/
|
||||||
|
function prepareModelCall(
|
||||||
|
ctx: EmitContext,
|
||||||
|
captured: RootSession,
|
||||||
|
risk: HighRiskMatch | undefined,
|
||||||
|
): PreparedModelCall | undefined {
|
||||||
const resolved = resolveJudgeModel(
|
const resolved = resolveJudgeModel(
|
||||||
captured.config,
|
captured.config,
|
||||||
captured.getModel(),
|
captured.getModel(),
|
||||||
captured.modelRegistry,
|
captured.modelRegistry,
|
||||||
);
|
);
|
||||||
if (resolved.kind === "unavailable") {
|
if (resolved.kind === "unavailable") {
|
||||||
return infrastructureDefer(
|
infrastructureDefer(
|
||||||
ctx,
|
ctx,
|
||||||
"judge_model_unavailable",
|
"judge_model_unavailable",
|
||||||
evidenceQuality(true, EMPTY_CONVERSATION, captured.getCwd()),
|
evidenceQuality(true, EMPTY_CONVERSATION, captured.getCwd()),
|
||||||
@@ -427,35 +450,36 @@ async function judgeAuthorize(
|
|||||||
riskOverride: risk ?? null,
|
riskOverride: risk ?? null,
|
||||||
},
|
},
|
||||||
);
|
);
|
||||||
|
return undefined;
|
||||||
}
|
}
|
||||||
const modelSource = resolved.source;
|
return {
|
||||||
const availability: ModelAvailability = createModelAvailability(
|
availability: createModelAvailability(
|
||||||
resolved.model,
|
resolved.model,
|
||||||
captured.modelRegistry,
|
captured.modelRegistry,
|
||||||
);
|
),
|
||||||
// Conversation evidence is captured at ask time from the
|
modelSource: resolved.source,
|
||||||
// live serving branch, not at session start: the newest
|
};
|
||||||
// user intent is the ask's intent.
|
}
|
||||||
const conversation: ConversationEvidence =
|
|
||||||
buildConversationEvidence(captured.conversation);
|
|
||||||
const result = await requestStructuredVerdict(
|
|
||||||
availability,
|
|
||||||
evidence,
|
|
||||||
captured.shutdown.signal,
|
|
||||||
captured.config.timeoutMs,
|
|
||||||
conversation,
|
|
||||||
);
|
|
||||||
|
|
||||||
// Enforce truth table (PIEXTENSIO-3 cat.4 / M5; ADR 0008):
|
/**
|
||||||
// the fail-closed runtime health gates — audit health,
|
* Enforce truth table (PIEXTENSIO-3 cat.4 / M5; ADR 0008): the fail-closed
|
||||||
// telemetry, result kind, verdict, review
|
* runtime health gates — audit health, telemetry, result kind, verdict,
|
||||||
// acknowledgement, generation currency — are the single
|
* review acknowledgement, generation currency — are the single authority
|
||||||
// authority seam; the retired promotion gates are no
|
* seam; the retired promotion gates are no longer inputs.
|
||||||
// longer inputs. reviewAcknowledged is true in the ADR
|
* reviewAcknowledged is true in the ADR 0006 sense: the Judge-owned audit
|
||||||
// 0006 sense: the Judge-owned audit write for this result
|
* write for this result happens before the authority return, and a failed
|
||||||
// happens before the authority return, and a failed write
|
* write flips auditHealthy sticky-unhealthy, closing authority for every
|
||||||
// flips auditHealthy sticky-unhealthy, closing authority
|
* later ask.
|
||||||
// for every later ask.
|
*/
|
||||||
|
function enforceAndEmit(
|
||||||
|
ctx: EmitContext,
|
||||||
|
captured: RootSession,
|
||||||
|
sink: ReviewSink,
|
||||||
|
result: ModelAttempt,
|
||||||
|
conversation: ConversationEvidence,
|
||||||
|
risk: HighRiskMatch | undefined,
|
||||||
|
modelSource: "configured" | "session",
|
||||||
|
): AuthorizerVerdict {
|
||||||
const gateState: EnforceGateState = {
|
const gateState: EnforceGateState = {
|
||||||
auditHealthy: captured.auditLog.healthy(),
|
auditHealthy: captured.auditLog.healthy(),
|
||||||
telemetryHealth: sink.health(),
|
telemetryHealth: sink.health(),
|
||||||
@@ -490,6 +514,63 @@ async function judgeAuthorize(
|
|||||||
return authority.kind === "allow"
|
return authority.kind === "allow"
|
||||||
? { kind: "allow" }
|
? { kind: "allow" }
|
||||||
: { kind: "defer" };
|
: { kind: "defer" };
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* One authorize call: enroll, preflight-gate, judge, and enforce the truth
|
||||||
|
* table. Extracted from the registerAuthorizer callback so each stage reads
|
||||||
|
* linearly; any exception fails closed (defer) without logging raw errors.
|
||||||
|
*/
|
||||||
|
async function judgeAuthorize(
|
||||||
|
captured: RootSession,
|
||||||
|
details: PromptPermissionDetails,
|
||||||
|
log: AuthorizerLog,
|
||||||
|
): Promise<AuthorizerVerdict> {
|
||||||
|
const startedAt = Date.now();
|
||||||
|
const sink: ReviewSink = createReviewSink({
|
||||||
|
log,
|
||||||
|
reviewLogEnabled: captured.reviewLogEnabled,
|
||||||
|
});
|
||||||
|
try {
|
||||||
|
const ctx: EmitContext = {
|
||||||
|
captured,
|
||||||
|
details,
|
||||||
|
startedAt,
|
||||||
|
emitResult: (record) => {
|
||||||
|
sink.review("ai_bash_judge.result", record);
|
||||||
|
captured.auditLog.audit("ai_bash_judge.result", record);
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
auditEnrollment(captured, details);
|
||||||
|
|
||||||
|
const gate = runPreflightGates(ctx, captured, details);
|
||||||
|
if (gate.kind === "stop") {
|
||||||
|
return gate.verdict;
|
||||||
|
}
|
||||||
|
|
||||||
|
const prepared = prepareModelCall(ctx, captured, gate.risk);
|
||||||
|
if (prepared === undefined) {
|
||||||
|
return { kind: "defer" };
|
||||||
|
}
|
||||||
|
|
||||||
|
// Conversation evidence is captured at ask time from the
|
||||||
|
// live serving branch, not at session start: the newest
|
||||||
|
// user intent is the ask's intent.
|
||||||
|
const conversation: ConversationEvidence =
|
||||||
|
buildConversationEvidence(captured.conversation);
|
||||||
|
const result = await requestStructuredVerdict(
|
||||||
|
prepared.availability,
|
||||||
|
gate.evidence,
|
||||||
|
captured.shutdown.signal,
|
||||||
|
captured.config.timeoutMs,
|
||||||
|
conversation,
|
||||||
|
);
|
||||||
|
|
||||||
|
return enforceAndEmit(
|
||||||
|
ctx, captured, sink, result, conversation, gate.risk,
|
||||||
|
prepared.modelSource,
|
||||||
|
);
|
||||||
} catch {
|
} catch {
|
||||||
// A link exception would abort the whole authority chain.
|
// A link exception would abort the whole authority chain.
|
||||||
// Keep provider/payload/session failures fail-closed and do
|
// Keep provider/payload/session failures fail-closed and do
|
||||||
|
|||||||
@@ -159,6 +159,72 @@ function codePointLength(value: string): number {
|
|||||||
return [...value].length;
|
return [...value].length;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** The completion type produced by a ready availability. */
|
||||||
|
type VerdictResponse = Awaited<
|
||||||
|
ReturnType<Extract<ModelAvailability, { kind: "ready" }>["complete"]>
|
||||||
|
>;
|
||||||
|
|
||||||
|
/** Validation outcome for one completion response. */
|
||||||
|
type ValidatedVerdict =
|
||||||
|
| {
|
||||||
|
readonly kind: "judgment";
|
||||||
|
readonly verdict: SemanticVerdict;
|
||||||
|
readonly reason: string;
|
||||||
|
}
|
||||||
|
| { readonly kind: "invalid"; readonly code: InfrastructureCode };
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Validate one completion response against the report_verdict contract:
|
||||||
|
* bounded output usage, exactly one report_verdict tool call, exactly the
|
||||||
|
* {verdict, reason} argument shape, and a non-empty reason within the
|
||||||
|
* code-point bound.
|
||||||
|
*/
|
||||||
|
function validateVerdictResponse(response: VerdictResponse): ValidatedVerdict {
|
||||||
|
const outputTokens = response.usage.output;
|
||||||
|
if (
|
||||||
|
!Number.isFinite(outputTokens) ||
|
||||||
|
outputTokens <= 0 ||
|
||||||
|
outputTokens > MAX_OUTPUT_TOKENS
|
||||||
|
) {
|
||||||
|
return { kind: "invalid", code: "model_error" };
|
||||||
|
}
|
||||||
|
|
||||||
|
const calls = response.content.filter(
|
||||||
|
(part) => part.type === "toolCall",
|
||||||
|
);
|
||||||
|
if (
|
||||||
|
calls.length !== 1 ||
|
||||||
|
calls[0]?.name !== REPORT_VERDICT_TOOL_NAME
|
||||||
|
) {
|
||||||
|
return { kind: "invalid", code: "missing_tool_call" };
|
||||||
|
}
|
||||||
|
|
||||||
|
const args = calls[0].arguments;
|
||||||
|
if (args === null || typeof args !== "object" || Array.isArray(args)) {
|
||||||
|
return { kind: "invalid", code: "invalid_arguments" };
|
||||||
|
}
|
||||||
|
const keys = Object.keys(args).sort();
|
||||||
|
if (keys.length !== 2 || keys[0] !== "reason" || keys[1] !== "verdict") {
|
||||||
|
return { kind: "invalid", code: "invalid_arguments" };
|
||||||
|
}
|
||||||
|
if (!isVerdict(args.verdict)) {
|
||||||
|
return { kind: "invalid", code: "invalid_verdict" };
|
||||||
|
}
|
||||||
|
if (typeof args.reason !== "string") {
|
||||||
|
return { kind: "invalid", code: "invalid_reason" };
|
||||||
|
}
|
||||||
|
|
||||||
|
const reason = args.reason.trim();
|
||||||
|
if (
|
||||||
|
reason.length === 0 ||
|
||||||
|
codePointLength(reason) > MAX_REASON_CODE_POINTS
|
||||||
|
) {
|
||||||
|
return { kind: "invalid", code: "invalid_reason" };
|
||||||
|
}
|
||||||
|
|
||||||
|
return { kind: "judgment", verdict: args.verdict, reason };
|
||||||
|
}
|
||||||
|
|
||||||
/** Make one bounded completion and accept only one `report_verdict` tool call. */
|
/** Make one bounded completion and accept only one `report_verdict` tool call. */
|
||||||
export async function requestStructuredVerdict(
|
export async function requestStructuredVerdict(
|
||||||
availability: ModelAvailability,
|
availability: ModelAvailability,
|
||||||
@@ -232,59 +298,26 @@ export async function requestStructuredVerdict(
|
|||||||
return failure("model_error");
|
return failure("model_error");
|
||||||
}
|
}
|
||||||
|
|
||||||
const outputTokens = response.usage.output;
|
// Usage is observed before validation so failure rows carry the
|
||||||
|
// telemetry of the response that failed the contract.
|
||||||
const inputTokens = response.usage.input;
|
const inputTokens = response.usage.input;
|
||||||
|
const outputTokens = response.usage.output;
|
||||||
if (Number.isFinite(inputTokens)) {
|
if (Number.isFinite(inputTokens)) {
|
||||||
observedInputTokens = inputTokens;
|
observedInputTokens = inputTokens;
|
||||||
}
|
}
|
||||||
if (Number.isFinite(outputTokens)) {
|
if (Number.isFinite(outputTokens)) {
|
||||||
observedOutputTokens = outputTokens;
|
observedOutputTokens = outputTokens;
|
||||||
}
|
}
|
||||||
if (
|
|
||||||
!Number.isFinite(outputTokens) ||
|
|
||||||
outputTokens <= 0 ||
|
|
||||||
outputTokens > MAX_OUTPUT_TOKENS
|
|
||||||
) {
|
|
||||||
return failure("model_error");
|
|
||||||
}
|
|
||||||
|
|
||||||
const calls = response.content.filter(
|
const validated = validateVerdictResponse(response);
|
||||||
(part) => part.type === "toolCall",
|
if (validated.kind === "invalid") {
|
||||||
);
|
return failure(validated.code);
|
||||||
if (
|
|
||||||
calls.length !== 1 ||
|
|
||||||
calls[0]?.name !== REPORT_VERDICT_TOOL_NAME
|
|
||||||
) {
|
|
||||||
return failure("missing_tool_call");
|
|
||||||
}
|
|
||||||
|
|
||||||
const args = calls[0].arguments;
|
|
||||||
if (args === null || typeof args !== "object" || Array.isArray(args)) {
|
|
||||||
return failure("invalid_arguments");
|
|
||||||
}
|
|
||||||
const keys = Object.keys(args).sort();
|
|
||||||
if (keys.length !== 2 || keys[0] !== "reason" || keys[1] !== "verdict") {
|
|
||||||
return failure("invalid_arguments");
|
|
||||||
}
|
|
||||||
if (!isVerdict(args.verdict)) {
|
|
||||||
return failure("invalid_verdict");
|
|
||||||
}
|
|
||||||
if (typeof args.reason !== "string") {
|
|
||||||
return failure("invalid_reason");
|
|
||||||
}
|
|
||||||
|
|
||||||
const reason = args.reason.trim();
|
|
||||||
if (
|
|
||||||
reason.length === 0 ||
|
|
||||||
codePointLength(reason) > MAX_REASON_CODE_POINTS
|
|
||||||
) {
|
|
||||||
return failure("invalid_reason");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
return {
|
return {
|
||||||
kind: "judgment",
|
kind: "judgment",
|
||||||
verdict: args.verdict,
|
verdict: validated.verdict,
|
||||||
reason,
|
reason: validated.reason,
|
||||||
metadata: availability.metadata,
|
metadata: availability.metadata,
|
||||||
inputTokens: observedInputTokens,
|
inputTokens: observedInputTokens,
|
||||||
outputTokens,
|
outputTokens,
|
||||||
|
|||||||
@@ -0,0 +1,292 @@
|
|||||||
|
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||||
|
import { tmpdir } from "node:os";
|
||||||
|
import { join } from "node:path";
|
||||||
|
import { spawnSync } from "node:child_process";
|
||||||
|
import { afterEach, describe, expect, it } from "vitest";
|
||||||
|
|
||||||
|
/**
|
||||||
|
* CLI-level tests for arg parsing (`parseArgs`), the `--after`/`--before`
|
||||||
|
* time window (`withinWindow`), and report rendering (`printReport`).
|
||||||
|
* Invalid-arg cases assert the exit contract (stderr + exit 1) that the
|
||||||
|
* analyze-shadow CLI documents.
|
||||||
|
*/
|
||||||
|
|
||||||
|
const dirs: string[] = [];
|
||||||
|
|
||||||
|
function tmp(): string {
|
||||||
|
const dir = mkdtempSync(join(tmpdir(), "ai-judge-cli-args-"));
|
||||||
|
dirs.push(dir);
|
||||||
|
return dir;
|
||||||
|
}
|
||||||
|
|
||||||
|
afterEach(() => {
|
||||||
|
for (const dir of dirs.splice(0)) {
|
||||||
|
rmSync(dir, { recursive: true, force: true });
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
interface RunResult {
|
||||||
|
readonly stdout: string;
|
||||||
|
readonly stderr: string;
|
||||||
|
readonly status: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
function runExpect(args: readonly string[]): RunResult {
|
||||||
|
const cli = join(import.meta.dirname, "..", "..", "src", "analyzer", "cli.ts");
|
||||||
|
const r = spawnSync("npx", ["tsx", cli, ...args], {
|
||||||
|
encoding: "utf-8",
|
||||||
|
stdio: ["ignore", "pipe", "pipe"],
|
||||||
|
});
|
||||||
|
return {
|
||||||
|
stdout: r.stdout ?? "",
|
||||||
|
stderr: r.stderr ?? "",
|
||||||
|
status: r.status ?? -1,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
function reviewLog(dir: string, lines: readonly object[]): string {
|
||||||
|
const path = join(dir, "review.jsonl");
|
||||||
|
writeFileSync(
|
||||||
|
path,
|
||||||
|
lines.map((l) => JSON.stringify(l)).join("\n") + "\n",
|
||||||
|
);
|
||||||
|
return path;
|
||||||
|
}
|
||||||
|
|
||||||
|
describe("analyze-shadow CLI — argument parsing (parseArgs)", () => {
|
||||||
|
it("prints usage and exits 0 for --help and -h", () => {
|
||||||
|
for (const flag of ["--help", "-h"]) {
|
||||||
|
const r = runExpect([flag]);
|
||||||
|
expect(r.status).toBe(0);
|
||||||
|
expect(r.stdout.length + r.stderr.length).toBeGreaterThan(0);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects missing input path with exit 1", () => {
|
||||||
|
const r = runExpect([]);
|
||||||
|
expect(r.status).toBe(1);
|
||||||
|
expect(r.stderr).toContain("missing input path");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects multiple input paths with exit 1", () => {
|
||||||
|
const dir = tmp();
|
||||||
|
const a = reviewLog(dir, []);
|
||||||
|
const b = reviewLog(dir, []).replace("review", "review2");
|
||||||
|
const r = runExpect([a, b]);
|
||||||
|
expect(r.status).toBe(1);
|
||||||
|
expect(r.stderr).toContain("multiple input paths");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects unknown options with exit 1", () => {
|
||||||
|
const dir = tmp();
|
||||||
|
const path = reviewLog(dir, []);
|
||||||
|
const r = runExpect([path, "--bogus"]);
|
||||||
|
expect(r.status).toBe(1);
|
||||||
|
expect(r.stderr).toContain("unknown option: --bogus");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects --after/--before/--audit without a value", () => {
|
||||||
|
const dir = tmp();
|
||||||
|
const path = reviewLog(dir, []);
|
||||||
|
for (const flag of ["--after", "--before", "--audit"]) {
|
||||||
|
const r = runExpect([path, flag]);
|
||||||
|
expect(r.status).toBe(1);
|
||||||
|
expect(r.stderr).toContain(`${flag} requires a value`);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects invalid --after/--before timestamps", () => {
|
||||||
|
const dir = tmp();
|
||||||
|
const path = reviewLog(dir, []);
|
||||||
|
for (const flag of ["--after", "--before"]) {
|
||||||
|
const r = runExpect([path, flag, "not-a-timestamp"]);
|
||||||
|
expect(r.status).toBe(1);
|
||||||
|
expect(r.stderr).toContain(`invalid ${flag} timestamp`);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("analyze-shadow CLI — time window (withinWindow)", () => {
|
||||||
|
const events = (ts: string) => [
|
||||||
|
{
|
||||||
|
timestamp: ts,
|
||||||
|
event: "authorizer_chain_resolved",
|
||||||
|
requestId: "req-1",
|
||||||
|
links: ["ai-bash-judge"],
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: ts,
|
||||||
|
event: "ai_bash_judge.result",
|
||||||
|
requestId: "req-1",
|
||||||
|
resultKind: "judgment",
|
||||||
|
verdict: "allow",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: ts,
|
||||||
|
event: "permission_request.approved",
|
||||||
|
requestId: "req-1",
|
||||||
|
resolution: "approved",
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
it("keeps events inside --after..--before and drops events outside", () => {
|
||||||
|
const dir = tmp();
|
||||||
|
const path = reviewLog(dir, [
|
||||||
|
...events("2026-08-18T00:00:05Z"), // inside
|
||||||
|
...events("2026-08-18T00:00:01Z").map((e, i) => ({ ...e, requestId: `req-old-${i}` })), // before
|
||||||
|
...events("2026-08-18T23:00:00Z").map((e, i) => ({ ...e, requestId: `req-new-${i}` })), // after
|
||||||
|
]);
|
||||||
|
const r = runExpect([
|
||||||
|
path,
|
||||||
|
"--after", "2026-08-18T00:00:02Z",
|
||||||
|
"--before", "2026-08-18T12:00:00Z",
|
||||||
|
]);
|
||||||
|
expect(r.status).toBe(0);
|
||||||
|
expect(r.stdout).toContain("enrollments (N): 1");
|
||||||
|
expect(r.stdout).toContain("joined rows: 1");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("keeps events with a missing timestamp regardless of the window", () => {
|
||||||
|
const dir = tmp();
|
||||||
|
// First event has no timestamp: survives any window.
|
||||||
|
const path = reviewLog(dir, [
|
||||||
|
{
|
||||||
|
event: "authorizer_chain_resolved",
|
||||||
|
requestId: "req-1",
|
||||||
|
links: ["ai-bash-judge"],
|
||||||
|
},
|
||||||
|
...events("2026-08-18T00:00:05Z").slice(1),
|
||||||
|
]);
|
||||||
|
const r = runExpect([
|
||||||
|
path,
|
||||||
|
"--after", "2026-08-18T00:00:02Z",
|
||||||
|
]);
|
||||||
|
expect(r.status).toBe(0);
|
||||||
|
expect(r.stdout).toContain("enrollments (N): 1");
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("analyze-shadow CLI — report rendering (printReport)", () => {
|
||||||
|
function joinedLog(dir: string): string {
|
||||||
|
return reviewLog(dir, [
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:01Z",
|
||||||
|
event: "authorizer_chain_resolved",
|
||||||
|
requestId: "req-1",
|
||||||
|
links: ["ai-bash-judge"],
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:02Z",
|
||||||
|
event: "ai_bash_judge.result",
|
||||||
|
requestId: "req-1",
|
||||||
|
resultKind: "judgment",
|
||||||
|
verdict: "allow",
|
||||||
|
judgeLatencyMs: 5,
|
||||||
|
modelLatencyMs: 3,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:03Z",
|
||||||
|
event: "permission_request.approved",
|
||||||
|
requestId: "req-1",
|
||||||
|
resolution: "approved",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:04Z",
|
||||||
|
event: "authorizer_chain_resolved",
|
||||||
|
requestId: "req-2",
|
||||||
|
links: ["ai-bash-judge"],
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:05Z",
|
||||||
|
event: "ai_bash_judge.result",
|
||||||
|
requestId: "req-2",
|
||||||
|
resultKind: "judgment",
|
||||||
|
verdict: "deny",
|
||||||
|
code: null,
|
||||||
|
judgeLatencyMs: 7,
|
||||||
|
modelLatencyMs: 4,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:06Z",
|
||||||
|
event: "permission_request.approved",
|
||||||
|
requestId: "req-2",
|
||||||
|
resolution: "approved",
|
||||||
|
},
|
||||||
|
]);
|
||||||
|
}
|
||||||
|
|
||||||
|
it("renders the comparison matrix and counters for mixed judgment rows", () => {
|
||||||
|
const dir = tmp();
|
||||||
|
const r = runExpect([joinedLog(dir)]);
|
||||||
|
expect(r.status).toBe(0);
|
||||||
|
// Header + source echo.
|
||||||
|
expect(r.stdout).toContain("AI Bash Judge — Shadow diagnostic report");
|
||||||
|
expect(r.stdout).toContain("grade: DIAGNOSTIC");
|
||||||
|
expect(r.stdout).toContain("source:");
|
||||||
|
// Denominator + join counts.
|
||||||
|
expect(r.stdout).toContain("enrollments (N): 2");
|
||||||
|
expect(r.stdout).toContain("joined rows: 2");
|
||||||
|
expect(r.stdout).toContain("joined judgments: 2");
|
||||||
|
// Matrix rows for both verdicts (allow|allow, deny|allow).
|
||||||
|
expect(r.stdout).toContain("allow|allow: 1");
|
||||||
|
expect(r.stdout).toContain("deny|allow: 1");
|
||||||
|
// Conservative deny counter incremented by deny|allow.
|
||||||
|
expect(r.stdout).toContain("conservative: deny 1, defer 0");
|
||||||
|
// Latency lines rendered with p50/p95/max/missing.
|
||||||
|
expect(r.stdout).toMatch(/judge latency: p50=\d+ms p95=\d+ms/);
|
||||||
|
expect(r.stdout).toMatch(/model latency: p50=\d+ms p95=\d+ms/);
|
||||||
|
// No --audit: the audit source line must be absent.
|
||||||
|
expect(r.stdout).not.toContain("(enrollment source, ADR 0006)");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("renders an empty matrix placeholder and no quarantine section when clean", () => {
|
||||||
|
const dir = tmp();
|
||||||
|
const r = runExpect([joinedLog(dir)]);
|
||||||
|
expect(r.status).toBe(0);
|
||||||
|
expect(r.stdout).not.toContain("quarantined rows:");
|
||||||
|
// Build an empty log: no events at all — matrix stays empty.
|
||||||
|
const empty = reviewLog(dir, []);
|
||||||
|
const r2 = runExpect([empty]);
|
||||||
|
expect(r2.status).toBe(0);
|
||||||
|
expect(r2.stdout).toContain("comparison matrix [verdict|human]:");
|
||||||
|
expect(r2.stdout).toContain("(empty)");
|
||||||
|
expect(r2.stdout).toContain("N/A");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("renders the quarantine section with category counts", () => {
|
||||||
|
const dir = tmp();
|
||||||
|
// Two results for one request → duplicate_result quarantine.
|
||||||
|
const path = reviewLog(dir, [
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:01Z",
|
||||||
|
event: "authorizer_chain_resolved",
|
||||||
|
requestId: "req-1",
|
||||||
|
links: ["ai-bash-judge"],
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:02Z",
|
||||||
|
event: "ai_bash_judge.result",
|
||||||
|
requestId: "req-1",
|
||||||
|
resultKind: "judgment",
|
||||||
|
verdict: "allow",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:03Z",
|
||||||
|
event: "ai_bash_judge.result",
|
||||||
|
requestId: "req-1",
|
||||||
|
resultKind: "judgment",
|
||||||
|
verdict: "deny",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
timestamp: "2026-08-18T00:00:04Z",
|
||||||
|
event: "permission_request.approved",
|
||||||
|
requestId: "req-1",
|
||||||
|
resolution: "approved",
|
||||||
|
},
|
||||||
|
]);
|
||||||
|
const r = runExpect([path]);
|
||||||
|
expect(r.status).toBe(0);
|
||||||
|
expect(r.stdout).toContain("quarantined rows:");
|
||||||
|
expect(r.stdout).toContain("duplicate_result: 1");
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,214 @@
|
|||||||
|
import { describe, expect, it, vi } from "vitest";
|
||||||
|
import type { ReviewEvent } from "../../src/analyzer/analyze";
|
||||||
|
import { analyzeShadowReviewLog } from "../../src/analyzer/analyze";
|
||||||
|
import {
|
||||||
|
parseArgs,
|
||||||
|
parseTimestampOption,
|
||||||
|
printReport,
|
||||||
|
withinWindow,
|
||||||
|
} from "../../src/analyzer/cli";
|
||||||
|
|
||||||
|
/**
|
||||||
|
* In-process unit tests for the analyze-shadow CLI's pure helpers:
|
||||||
|
* argument parsing, the time-window filter, and report rendering. The
|
||||||
|
* end-to-end contract (file IO, dual-log mode, exit codes) lives in
|
||||||
|
* cli-args.test.ts / cli-audit.test.ts; these tests cover branch-level
|
||||||
|
* behavior without spawning tsx subprocesses.
|
||||||
|
*/
|
||||||
|
|
||||||
|
describe("parseArgs", () => {
|
||||||
|
it("parses the positional path with no options", () => {
|
||||||
|
expect(parseArgs(["node", "cli.ts", "review.jsonl"])).toEqual({
|
||||||
|
path: "review.jsonl",
|
||||||
|
after: null,
|
||||||
|
before: null,
|
||||||
|
audit: null,
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it("parses --after, --before, and --audit together", () => {
|
||||||
|
const r = parseArgs([
|
||||||
|
"node",
|
||||||
|
"cli.ts",
|
||||||
|
"review.jsonl",
|
||||||
|
"--after", "2026-08-18T00:00:00Z",
|
||||||
|
"--before", "2026-08-19T00:00:00Z",
|
||||||
|
"--audit", "audit.jsonl",
|
||||||
|
]);
|
||||||
|
expect(r).toEqual({
|
||||||
|
path: "review.jsonl",
|
||||||
|
after: new Date("2026-08-18T00:00:00Z"),
|
||||||
|
before: new Date("2026-08-19T00:00:00Z"),
|
||||||
|
audit: "audit.jsonl",
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns the usage error for --help and -h", () => {
|
||||||
|
expect(parseArgs(["node", "cli.ts", "--help"])).toEqual({
|
||||||
|
error: expect.stringContaining("usage: analyze-shadow") as unknown,
|
||||||
|
});
|
||||||
|
expect(parseArgs(["node", "cli.ts", "-h"])).toHaveProperty("error");
|
||||||
|
});
|
||||||
|
|
||||||
|
it.each([
|
||||||
|
["missing input path", [] as const],
|
||||||
|
["multiple input paths", ["a.jsonl", "b.jsonl"] as const],
|
||||||
|
["unknown option", ["a.jsonl", "--bogus"] as const],
|
||||||
|
["--audit without value", ["a.jsonl", "--audit"] as const],
|
||||||
|
["--after without value", ["a.jsonl", "--after"] as const],
|
||||||
|
["invalid --after timestamp", ["a.jsonl", "--after", "nope"] as const],
|
||||||
|
["invalid --before timestamp", ["a.jsonl", "--before", "nope"] as const],
|
||||||
|
])("errors on %s", (_name, argv) => {
|
||||||
|
const r = parseArgs(["node", "cli.ts", ...argv]);
|
||||||
|
expect("error" in r && typeof r.error === "string").toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("parseTimestampOption", () => {
|
||||||
|
it("requires a value", () => {
|
||||||
|
expect(parseTimestampOption("--after", undefined)).toEqual({
|
||||||
|
error: "--after requires a value",
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects unparseable timestamps", () => {
|
||||||
|
expect(parseTimestampOption("--before", "yesterday")).toEqual({
|
||||||
|
error: "invalid --before timestamp: yesterday",
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it("accepts an ISO instant", () => {
|
||||||
|
expect(parseTimestampOption("--after", "2026-08-18T00:00:00Z")).toEqual({
|
||||||
|
date: new Date("2026-08-18T00:00:00Z"),
|
||||||
|
});
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("withinWindow", () => {
|
||||||
|
const evt = (timestamp?: string): ReviewEvent =>
|
||||||
|
({ event: "e", ...(timestamp === undefined ? {} : { timestamp }) }) as ReviewEvent;
|
||||||
|
|
||||||
|
it("keeps events without a timestamp in any window", () => {
|
||||||
|
const window = {
|
||||||
|
after: new Date("2026-08-18T12:00:00Z"),
|
||||||
|
before: new Date("2026-08-18T13:00:00Z"),
|
||||||
|
};
|
||||||
|
expect(withinWindow(evt(), window)).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("keeps events inside the window and drops those outside", () => {
|
||||||
|
const window = {
|
||||||
|
after: new Date("2026-08-18T12:00:00Z"),
|
||||||
|
before: new Date("2026-08-18T13:00:00Z"),
|
||||||
|
};
|
||||||
|
expect(withinWindow(evt("2026-08-18T12:30:00Z"), window)).toBe(true);
|
||||||
|
expect(withinWindow(evt("2026-08-18T11:59:00Z"), window)).toBe(false);
|
||||||
|
expect(withinWindow(evt("2026-08-18T13:01:00Z"), window)).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("applies each bound independently", () => {
|
||||||
|
expect(
|
||||||
|
withinWindow(evt("2026-08-18T10:00:00Z"), { after: new Date("2026-08-18T09:00:00Z"), before: null }),
|
||||||
|
).toBe(true);
|
||||||
|
expect(
|
||||||
|
withinWindow(evt("2026-08-18T10:00:00Z"), { after: null, before: new Date("2026-08-18T09:00:00Z") }),
|
||||||
|
).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("keeps events with an unparseable timestamp regardless of bounds", () => {
|
||||||
|
const window = {
|
||||||
|
after: new Date("2026-08-18T12:00:00Z"),
|
||||||
|
before: null,
|
||||||
|
};
|
||||||
|
// Unparseable -> NaN time -> neither bound applies.
|
||||||
|
expect(withinWindow(evt("not-a-date"), window)).toBe(true);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("printReport", () => {
|
||||||
|
function capture(fn: () => void): string {
|
||||||
|
const chunks: string[] = [];
|
||||||
|
const spy = vi
|
||||||
|
.spyOn(process.stdout, "write")
|
||||||
|
.mockImplementation(((s: unknown) => {
|
||||||
|
chunks.push(String(s));
|
||||||
|
return true;
|
||||||
|
}) as never);
|
||||||
|
try {
|
||||||
|
fn();
|
||||||
|
} finally {
|
||||||
|
spy.mockRestore();
|
||||||
|
}
|
||||||
|
return chunks.join("");
|
||||||
|
}
|
||||||
|
|
||||||
|
const parsed = {
|
||||||
|
path: "review.jsonl",
|
||||||
|
after: null,
|
||||||
|
before: null,
|
||||||
|
audit: null,
|
||||||
|
};
|
||||||
|
|
||||||
|
function analyze(rows: readonly ReviewEvent[]): ReturnType<
|
||||||
|
typeof import("../../src/analyzer/analyze").analyzeShadowReviewLog
|
||||||
|
> {
|
||||||
|
return analyzeShadowReviewLog(rows);
|
||||||
|
}
|
||||||
|
|
||||||
|
const joinedEvents: ReviewEvent[] = [
|
||||||
|
{ event: "authorizer_chain_resolved", requestId: "r1", links: ["ai-bash-judge"] },
|
||||||
|
{
|
||||||
|
event: "ai_bash_judge.result",
|
||||||
|
requestId: "r1",
|
||||||
|
resultKind: "judgment",
|
||||||
|
verdict: "allow",
|
||||||
|
judgeLatencyMs: 5,
|
||||||
|
modelLatencyMs: 3,
|
||||||
|
},
|
||||||
|
{ event: "permission_request.approved", requestId: "r1", resolution: "approved" },
|
||||||
|
];
|
||||||
|
|
||||||
|
it("renders header, counts, matrix, and latency lines", () => {
|
||||||
|
const { enrollments, metrics } = analyze(joinedEvents);
|
||||||
|
const out = capture(() => printReport(parsed, enrollments, metrics));
|
||||||
|
expect(out).toContain("AI Bash Judge — Shadow diagnostic report");
|
||||||
|
expect(out).toContain("grade: DIAGNOSTIC");
|
||||||
|
expect(out).toContain("enrollments (N): 1");
|
||||||
|
expect(out).toContain("allow|allow: 1");
|
||||||
|
expect(out).toMatch(/judge latency: p50=5ms/);
|
||||||
|
expect(out).not.toContain("audit:");
|
||||||
|
expect(out).not.toContain("quarantined rows:");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("renders the audit source line in dual-log mode", () => {
|
||||||
|
const { enrollments, metrics } = analyze([]);
|
||||||
|
const out = capture(() =>
|
||||||
|
printReport(
|
||||||
|
{ ...parsed, audit: "audit.jsonl" },
|
||||||
|
enrollments,
|
||||||
|
metrics,
|
||||||
|
),
|
||||||
|
);
|
||||||
|
expect(out).toContain("audit: audit.jsonl (enrollment source, ADR 0006)");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("renders the empty-matrix placeholder for an empty log", () => {
|
||||||
|
const { enrollments, metrics } = analyze([]);
|
||||||
|
const out = capture(() => printReport(parsed, enrollments, metrics));
|
||||||
|
expect(out).toContain("(empty)");
|
||||||
|
expect(out).toContain("N/A");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("renders quarantine categories when integrity faults exist", () => {
|
||||||
|
const duplicate: ReviewEvent[] = [
|
||||||
|
{ event: "authorizer_chain_resolved", requestId: "r1", links: ["ai-bash-judge"] },
|
||||||
|
{ event: "ai_bash_judge.result", requestId: "r1", resultKind: "judgment", verdict: "allow" },
|
||||||
|
{ event: "ai_bash_judge.result", requestId: "r1", resultKind: "judgment", verdict: "deny" },
|
||||||
|
{ event: "permission_request.approved", requestId: "r1", resolution: "approved" },
|
||||||
|
];
|
||||||
|
const { enrollments, metrics } = analyze(duplicate);
|
||||||
|
const out = capture(() => printReport(parsed, enrollments, metrics));
|
||||||
|
expect(out).toContain("quarantined rows:");
|
||||||
|
expect(out).toContain("duplicate_result: 1");
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -83,6 +83,61 @@ describe("loadModelCatalog", () => {
|
|||||||
expect(diagnostics[0]?.key).toBe("file");
|
expect(diagnostics[0]?.key).toBe("file");
|
||||||
expect(String(diagnostics[0]?.problem)).toMatch(/whole catalog degraded/i);
|
expect(String(diagnostics[0]?.problem)).toMatch(/whole catalog degraded/i);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("rejects an entry per invalid field (every parseEntry branch)", () => {
|
||||||
|
const cases: readonly [name: string, entry: unknown][] = [
|
||||||
|
["non-object entry", "not-an-object"],
|
||||||
|
["null entry", null],
|
||||||
|
["array entry", [VALID_ENTRY]],
|
||||||
|
["empty provider", { ...VALID_ENTRY, provider: " " }],
|
||||||
|
["empty model", { ...VALID_ENTRY, model: "" }],
|
||||||
|
["empty api", { ...VALID_ENTRY, api: "" }],
|
||||||
|
["unknown status", { ...VALID_ENTRY, status: "experimental" }],
|
||||||
|
["empty promptVersion", { ...VALID_ENTRY, promptVersion: "" }],
|
||||||
|
["empty corpusVersion", { ...VALID_ENTRY, corpusVersion: "" }],
|
||||||
|
["empty testedAt", { ...VALID_ENTRY, testedAt: "" }],
|
||||||
|
["non-integer corpusCases", { ...VALID_ENTRY, corpusCases: 2.5 }],
|
||||||
|
["zero corpusCases", { ...VALID_ENTRY, corpusCases: 0 }],
|
||||||
|
["non-integer matched", { ...VALID_ENTRY, matched: 1.5 }],
|
||||||
|
["non-integer infrastructureFailures", { ...VALID_ENTRY, infrastructureFailures: 0.5 }],
|
||||||
|
["latencyMs not an object", { ...VALID_ENTRY, latencyMs: 3000 }],
|
||||||
|
["latencyMs wrong type", { ...VALID_ENTRY, latencyMs: { p50: "3000", p95: null, max: null } }],
|
||||||
|
["latencyMs missing key", { ...VALID_ENTRY, latencyMs: { p50: 1, p95: null } }],
|
||||||
|
["empty reportPath", { ...VALID_ENTRY, reportPath: "" }],
|
||||||
|
["blank notes", { ...VALID_ENTRY, notes: " " }],
|
||||||
|
];
|
||||||
|
for (const [name, entry] of cases) {
|
||||||
|
const { catalog, diagnostics } = run(depsWith(JSON.stringify({
|
||||||
|
version: 1,
|
||||||
|
entries: [entry],
|
||||||
|
})));
|
||||||
|
expect(catalog?.entries, name).toEqual([]);
|
||||||
|
expect(diagnostics.map((d) => d.key), name).toContain("file");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
it("accepts latency nulls and omits notes only when absent", () => {
|
||||||
|
const withNulls = {
|
||||||
|
...VALID_ENTRY,
|
||||||
|
latencyMs: { p50: null, p95: null, max: null },
|
||||||
|
};
|
||||||
|
const { catalog } = run(depsWith(JSON.stringify({
|
||||||
|
version: 1,
|
||||||
|
entries: [withNulls],
|
||||||
|
})));
|
||||||
|
expect(catalog?.entries).toHaveLength(1);
|
||||||
|
expect(catalog?.entries[0]?.latencyMs).toEqual({ p50: null, p95: null, max: null });
|
||||||
|
expect("notes" in (catalog?.entries[0] ?? {})).toBe(false);
|
||||||
|
|
||||||
|
const withNotes = { ...VALID_ENTRY, notes: "qualifying corpus replay report" };
|
||||||
|
const r2 = run(depsWith(JSON.stringify({
|
||||||
|
version: 1,
|
||||||
|
entries: [withNotes],
|
||||||
|
})));
|
||||||
|
expect(r2.catalog?.entries[0]).toMatchObject({
|
||||||
|
notes: "qualifying corpus replay report",
|
||||||
|
});
|
||||||
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
describe("classifyModel", () => {
|
describe("classifyModel", () => {
|
||||||
|
|||||||
@@ -0,0 +1,205 @@
|
|||||||
|
import { describe, expect, it, vi } from "vitest";
|
||||||
|
import type { AssistantMessage, Context, Model } from "@earendil-works/pi-ai";
|
||||||
|
import type { ModelRegistry } from "@earendil-works/pi-coding-agent";
|
||||||
|
import {
|
||||||
|
DEFAULT_TIMEOUT_MS,
|
||||||
|
MIN_TIMEOUT_MS,
|
||||||
|
MAX_TIMEOUT_MS,
|
||||||
|
} from "../src/model";
|
||||||
|
import { createModelAvailability } from "../src/model";
|
||||||
|
import {
|
||||||
|
applyTimeoutOption,
|
||||||
|
replayCorpus,
|
||||||
|
selectCorpusCases,
|
||||||
|
validateCliState,
|
||||||
|
type CliOptions,
|
||||||
|
} from "../tools/corpus-replay";
|
||||||
|
|
||||||
|
/**
|
||||||
|
* In-process tests for the corpus-replay harness helpers: option
|
||||||
|
* application/validation, case selection, and the replay loop driven by a
|
||||||
|
* stub model availability (no registry, no network). The CLI exit contract
|
||||||
|
* lives in corpus-replay-cli.test.ts.
|
||||||
|
*/
|
||||||
|
|
||||||
|
function state(overrides: Partial<CliOptions> = {}): { -readonly [K in keyof CliOptions]: CliOptions[K] } {
|
||||||
|
return {
|
||||||
|
provider: "p",
|
||||||
|
model: "m",
|
||||||
|
timeoutMs: DEFAULT_TIMEOUT_MS,
|
||||||
|
cases: null,
|
||||||
|
out: null,
|
||||||
|
strict: false,
|
||||||
|
...overrides,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
describe("applyTimeoutOption", () => {
|
||||||
|
it.each([
|
||||||
|
[MIN_TIMEOUT_MS, null],
|
||||||
|
[MAX_TIMEOUT_MS, null],
|
||||||
|
[DEFAULT_TIMEOUT_MS, null],
|
||||||
|
])("accepts the boundary value %i", (value, error) => {
|
||||||
|
const s = state();
|
||||||
|
expect(applyTimeoutOption(s, String(value))).toBe(error);
|
||||||
|
expect(s.timeoutMs).toBe(value);
|
||||||
|
});
|
||||||
|
|
||||||
|
it.each([
|
||||||
|
[String(MIN_TIMEOUT_MS - 1)],
|
||||||
|
[String(MAX_TIMEOUT_MS + 1)],
|
||||||
|
["12.5"],
|
||||||
|
["abc"],
|
||||||
|
])("rejects %s with the documented range error", (value) => {
|
||||||
|
const s = state();
|
||||||
|
const r = applyTimeoutOption(s, value);
|
||||||
|
expect(r).toMatch(/--timeout-ms must be an integer in \[\d+, \d+\]/);
|
||||||
|
expect(s.timeoutMs).toBe(DEFAULT_TIMEOUT_MS);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects a missing value with the range error (the usage check lives in parseArgs)", () => {
|
||||||
|
expect(applyTimeoutOption(state(), undefined)).toMatch(/--timeout-ms must be an integer/);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("validateCliState", () => {
|
||||||
|
it("requires both provider and model", () => {
|
||||||
|
expect(validateCliState(state({ provider: "" }))).toMatch(/^usage:/);
|
||||||
|
expect(validateCliState(state({ model: "" }))).toMatch(/^usage:/);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects --strict combined with a --case subset", () => {
|
||||||
|
expect(
|
||||||
|
validateCliState(state({ strict: true, cases: new Set(["a"]) })),
|
||||||
|
).toContain("--strict requires the full corpus");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("accepts a complete valid state", () => {
|
||||||
|
expect(validateCliState(state())).toBeNull();
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("selectCorpusCases", () => {
|
||||||
|
it("selects the full corpus when no case filter is given", () => {
|
||||||
|
const r = selectCorpusCases(null);
|
||||||
|
expect("selected" in r && r.selected.length).toBeGreaterThan(10);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("selects exactly the requested known ids", () => {
|
||||||
|
const r = selectCorpusCases(new Set(["requested-clean", "extra-push"]));
|
||||||
|
expect("selected" in r ? r.selected.map((c) => c.id) : []).toEqual([
|
||||||
|
"requested-clean",
|
||||||
|
"extra-push",
|
||||||
|
]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("errors on unknown ids without selecting anything", () => {
|
||||||
|
const r = selectCorpusCases(new Set(["requested-clean", "no-such-case"]));
|
||||||
|
expect("error" in r && r.error).toContain("unknown case ids: no-such-case");
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe("replayCorpus", () => {
|
||||||
|
const stderrSpy = () => vi.spyOn(process.stderr, "write").mockReturnValue(true);
|
||||||
|
|
||||||
|
function availabilityWith(
|
||||||
|
respond: () => AssistantMessage,
|
||||||
|
): ReturnType<typeof createModelAvailability> {
|
||||||
|
const model = {
|
||||||
|
id: "m",
|
||||||
|
provider: "p",
|
||||||
|
api: "openai-codex-responses",
|
||||||
|
} as Model<any>;
|
||||||
|
const registry = {
|
||||||
|
complete: async (
|
||||||
|
_m: Model<any>,
|
||||||
|
_c: Context,
|
||||||
|
_o?: Record<string, unknown>,
|
||||||
|
) => respond(),
|
||||||
|
} as unknown as ModelRegistry;
|
||||||
|
return createModelAvailability(model, registry);
|
||||||
|
}
|
||||||
|
|
||||||
|
function judgment(verdict: "allow" | "deny" | "defer"): AssistantMessage {
|
||||||
|
return {
|
||||||
|
role: "assistant",
|
||||||
|
content: [
|
||||||
|
{
|
||||||
|
type: "toolCall",
|
||||||
|
id: "call-1",
|
||||||
|
name: "report_verdict",
|
||||||
|
arguments: { verdict, reason: "stub reason" },
|
||||||
|
},
|
||||||
|
],
|
||||||
|
api: "openai-codex-responses",
|
||||||
|
provider: "p",
|
||||||
|
model: "m",
|
||||||
|
stopReason: "toolUse",
|
||||||
|
usage: {
|
||||||
|
input: 10,
|
||||||
|
output: 8,
|
||||||
|
cacheRead: 0,
|
||||||
|
cacheWrite: 0,
|
||||||
|
totalTokens: 18,
|
||||||
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||||
|
},
|
||||||
|
timestamp: Date.now(),
|
||||||
|
} as AssistantMessage;
|
||||||
|
}
|
||||||
|
|
||||||
|
it("counts matches over the selected cases", async () => {
|
||||||
|
const spy = stderrSpy();
|
||||||
|
try {
|
||||||
|
// requested-clean expects allow; extra-push expects defer — one
|
||||||
|
// canned judgment cannot satisfy both, so assert row semantics
|
||||||
|
// rather than a perfect score.
|
||||||
|
const availability = availabilityWith(() => judgment("allow"));
|
||||||
|
const selected = selectCorpusCases(
|
||||||
|
new Set(["requested-clean", "extra-push"]),
|
||||||
|
);
|
||||||
|
if (!("selected" in selected)) throw new Error("unreachable");
|
||||||
|
const { rows, matched } = await replayCorpus(
|
||||||
|
selected.selected,
|
||||||
|
availability,
|
||||||
|
DEFAULT_TIMEOUT_MS,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(rows).toHaveLength(2);
|
||||||
|
expect(matched).toBe(1);
|
||||||
|
const clean = rows.find((r) => r.case === "requested-clean");
|
||||||
|
expect(clean).toMatchObject({ verdict: "allow", match: true });
|
||||||
|
const push = rows.find((r) => r.case === "extra-push");
|
||||||
|
expect(push).toMatchObject({ verdict: "allow", match: false });
|
||||||
|
} finally {
|
||||||
|
spy.mockRestore();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
it("records infrastructure failures as unmatched rows", async () => {
|
||||||
|
const spy = stderrSpy();
|
||||||
|
try {
|
||||||
|
const availability = availabilityWith(() => {
|
||||||
|
const m = judgment("allow");
|
||||||
|
m.stopReason = "error";
|
||||||
|
return m;
|
||||||
|
});
|
||||||
|
const selected = selectCorpusCases(new Set(["requested-clean"]));
|
||||||
|
if (!("selected" in selected)) throw new Error("unreachable");
|
||||||
|
const { rows, matched } = await replayCorpus(
|
||||||
|
selected.selected,
|
||||||
|
availability,
|
||||||
|
DEFAULT_TIMEOUT_MS,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(rows).toHaveLength(1);
|
||||||
|
expect(rows[0]).toMatchObject({
|
||||||
|
verdict: null,
|
||||||
|
resultKind: "infrastructure_failure",
|
||||||
|
match: false,
|
||||||
|
});
|
||||||
|
expect(matched).toBe(0);
|
||||||
|
} finally {
|
||||||
|
spy.mockRestore();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -28,9 +28,14 @@ describe("classifyHighRisk — data_loss", () => {
|
|||||||
it.each([
|
it.each([
|
||||||
"git clean -nxd",
|
"git clean -nxd",
|
||||||
"git clean -nd",
|
"git clean -nd",
|
||||||
|
"git clean -nxfd",
|
||||||
|
"git clean -f -x -n",
|
||||||
|
"git clean -fd",
|
||||||
|
"git clean -xn",
|
||||||
"git reset --soft HEAD~1",
|
"git reset --soft HEAD~1",
|
||||||
"git checkout main",
|
"git checkout main",
|
||||||
"git checkout -- file.txt",
|
"git checkout -- file.txt",
|
||||||
|
"git checkout . file.txt",
|
||||||
"git restore file.txt",
|
"git restore file.txt",
|
||||||
"rm -rf build/",
|
"rm -rf build/",
|
||||||
"rm -rf ./dist",
|
"rm -rf ./dist",
|
||||||
|
|||||||
@@ -542,6 +542,82 @@ describe("AI judge lifecycle", () => {
|
|||||||
harness.shutdown();
|
harness.shutdown();
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("records a provider failure as an infrastructure_failure row and defers", async () => {
|
||||||
|
let authorize: Authorizer["authorize"] | undefined;
|
||||||
|
const service = {
|
||||||
|
registerAuthorizer: vi.fn((_name, callback) => {
|
||||||
|
authorize = callback;
|
||||||
|
return vi.fn();
|
||||||
|
}),
|
||||||
|
checkPermission: vi.fn(),
|
||||||
|
getToolPermission: vi.fn(),
|
||||||
|
} as unknown as PermissionsService;
|
||||||
|
publishPermissionsService(service);
|
||||||
|
publishedService = service;
|
||||||
|
|
||||||
|
const complete = vi.fn(
|
||||||
|
async () => {
|
||||||
|
throw new Error("provider 500");
|
||||||
|
},
|
||||||
|
);
|
||||||
|
const ctx = {
|
||||||
|
hasUI: true,
|
||||||
|
sessionManager: fakeSessionManager(),
|
||||||
|
model: {
|
||||||
|
id: "test-model",
|
||||||
|
provider: "test-provider",
|
||||||
|
api: "openai-codex-responses",
|
||||||
|
} as Model<any>,
|
||||||
|
modelRegistry: { complete },
|
||||||
|
ui: { notify: vi.fn() },
|
||||||
|
} as unknown as ExtensionContext;
|
||||||
|
|
||||||
|
const harness = createFakePi();
|
||||||
|
extension(harness.pi);
|
||||||
|
harness.start(ctx);
|
||||||
|
harness.ready();
|
||||||
|
expect(authorize).toBeDefined();
|
||||||
|
|
||||||
|
const reviews: Array<{
|
||||||
|
event: string;
|
||||||
|
details?: Record<string, unknown>;
|
||||||
|
}> = [];
|
||||||
|
const verdict = await authorize!(
|
||||||
|
ask(),
|
||||||
|
{
|
||||||
|
checkPermission: vi.fn(),
|
||||||
|
getToolPermission: vi.fn(),
|
||||||
|
},
|
||||||
|
{
|
||||||
|
review: (event, details) => reviews.push({ event, details }),
|
||||||
|
debug: vi.fn(),
|
||||||
|
},
|
||||||
|
);
|
||||||
|
|
||||||
|
expect(verdict).toEqual({ kind: "defer" });
|
||||||
|
expect(complete).toHaveBeenCalledTimes(1);
|
||||||
|
expect(reviews).toMatchObject([
|
||||||
|
{
|
||||||
|
event: "ai_bash_judge.result",
|
||||||
|
details: expect.objectContaining({
|
||||||
|
resultKind: "infrastructure_failure",
|
||||||
|
verdict: null,
|
||||||
|
effectiveVerdict: "defer",
|
||||||
|
modelCalled: true,
|
||||||
|
code: "model_error",
|
||||||
|
modelSource: "session",
|
||||||
|
riskOverride: null,
|
||||||
|
inputUsage: null,
|
||||||
|
outputUsage: null,
|
||||||
|
}),
|
||||||
|
},
|
||||||
|
]);
|
||||||
|
// The failure row must not leak the provider error text.
|
||||||
|
expect(JSON.stringify(reviews)).not.toContain("provider 500");
|
||||||
|
|
||||||
|
harness.shutdown();
|
||||||
|
});
|
||||||
|
|
||||||
it("does not register from a headless child", () => {
|
it("does not register from a headless child", () => {
|
||||||
const service = {
|
const service = {
|
||||||
registerAuthorizer: vi.fn(),
|
registerAuthorizer: vi.fn(),
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
import { describe, expect, it, vi } from "vitest";
|
import { describe, expect, it, vi } from "vitest";
|
||||||
import type { AssistantMessage, Context } from "@earendil-works/pi-ai";
|
import type { AssistantMessage, Context, Model } from "@earendil-works/pi-ai";
|
||||||
|
import type { ModelRegistry } from "@earendil-works/pi-coding-agent";
|
||||||
import {
|
import {
|
||||||
|
createModelAvailability,
|
||||||
requestStructuredVerdict,
|
requestStructuredVerdict,
|
||||||
type ModelAvailability,
|
type ModelAvailability,
|
||||||
} from "../src/model";
|
} from "../src/model";
|
||||||
@@ -50,6 +52,18 @@ function ready(
|
|||||||
return { kind: "ready", metadata, complete };
|
return { kind: "ready", metadata, complete };
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** A minimal valid report_verdict tool-call content block. */
|
||||||
|
function verdictContent(): AssistantMessage["content"] {
|
||||||
|
return [
|
||||||
|
{
|
||||||
|
type: "toolCall",
|
||||||
|
id: "call-1",
|
||||||
|
name: "report_verdict",
|
||||||
|
arguments: { verdict: "allow", reason: "bounded" },
|
||||||
|
},
|
||||||
|
];
|
||||||
|
}
|
||||||
|
|
||||||
const evidence = {
|
const evidence = {
|
||||||
fullCommand: "pnpm test && git push",
|
fullCommand: "pnpm test && git push",
|
||||||
triggeringUnit: "git push",
|
triggeringUnit: "git push",
|
||||||
@@ -231,6 +245,137 @@ describe("requestStructuredVerdict", () => {
|
|||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("rejects zero and non-finite output usage", async () => {
|
||||||
|
const zeroUsage = await requestStructuredVerdict(
|
||||||
|
ready(async () =>
|
||||||
|
response(verdictContent(), 0),
|
||||||
|
),
|
||||||
|
evidence,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(zeroUsage).toMatchObject({ code: "model_error" });
|
||||||
|
|
||||||
|
const nonFiniteUsage = await requestStructuredVerdict(
|
||||||
|
ready(async () =>
|
||||||
|
response(verdictContent(), Number.NaN),
|
||||||
|
),
|
||||||
|
evidence,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(nonFiniteUsage).toMatchObject({ code: "model_error" });
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects responses without exactly one report_verdict tool call", async () => {
|
||||||
|
const noCall = await requestStructuredVerdict(
|
||||||
|
ready(async () =>
|
||||||
|
response([{ type: "text", text: "I would allow it." } as never]),
|
||||||
|
),
|
||||||
|
evidence,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(noCall).toMatchObject({ code: "missing_tool_call" });
|
||||||
|
|
||||||
|
const wrongName = await requestStructuredVerdict(
|
||||||
|
ready(async () =>
|
||||||
|
response([
|
||||||
|
{
|
||||||
|
type: "toolCall",
|
||||||
|
id: "call-1",
|
||||||
|
name: "other_tool",
|
||||||
|
arguments: { verdict: "allow", reason: "x" },
|
||||||
|
},
|
||||||
|
]),
|
||||||
|
),
|
||||||
|
evidence,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(wrongName).toMatchObject({ code: "missing_tool_call" });
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects null and array tool-call arguments", async () => {
|
||||||
|
const nullArgs = await requestStructuredVerdict(
|
||||||
|
ready(async () =>
|
||||||
|
response([
|
||||||
|
{
|
||||||
|
type: "toolCall",
|
||||||
|
id: "call-1",
|
||||||
|
name: "report_verdict",
|
||||||
|
arguments: null as unknown as Record<string, unknown>,
|
||||||
|
},
|
||||||
|
]),
|
||||||
|
),
|
||||||
|
evidence,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(nullArgs).toMatchObject({ code: "invalid_arguments" });
|
||||||
|
|
||||||
|
const arrayArgs = await requestStructuredVerdict(
|
||||||
|
ready(async () =>
|
||||||
|
response([
|
||||||
|
{
|
||||||
|
type: "toolCall",
|
||||||
|
id: "call-1",
|
||||||
|
name: "report_verdict",
|
||||||
|
arguments: ["allow", "x"] as unknown as Record<string, unknown>,
|
||||||
|
},
|
||||||
|
]),
|
||||||
|
),
|
||||||
|
evidence,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(arrayArgs).toMatchObject({ code: "invalid_arguments" });
|
||||||
|
|
||||||
|
const missingKey = await requestStructuredVerdict(
|
||||||
|
ready(async () =>
|
||||||
|
response([
|
||||||
|
{
|
||||||
|
type: "toolCall",
|
||||||
|
id: "call-1",
|
||||||
|
name: "report_verdict",
|
||||||
|
arguments: { verdict: "allow" },
|
||||||
|
},
|
||||||
|
]),
|
||||||
|
),
|
||||||
|
evidence,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(missingKey).toMatchObject({ code: "invalid_arguments" });
|
||||||
|
});
|
||||||
|
|
||||||
|
it("rejects non-string and blank-trimmed reasons", async () => {
|
||||||
|
const nonStringReason = await requestStructuredVerdict(
|
||||||
|
ready(async () =>
|
||||||
|
response([
|
||||||
|
{
|
||||||
|
type: "toolCall",
|
||||||
|
id: "call-1",
|
||||||
|
name: "report_verdict",
|
||||||
|
arguments: { verdict: "defer", reason: 42 as unknown as string },
|
||||||
|
},
|
||||||
|
]),
|
||||||
|
),
|
||||||
|
evidence,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(nonStringReason).toMatchObject({ code: "invalid_reason" });
|
||||||
|
|
||||||
|
const blankReason = await requestStructuredVerdict(
|
||||||
|
ready(async () =>
|
||||||
|
response([
|
||||||
|
{
|
||||||
|
type: "toolCall",
|
||||||
|
id: "call-1",
|
||||||
|
name: "report_verdict",
|
||||||
|
arguments: { verdict: "defer", reason: " " },
|
||||||
|
},
|
||||||
|
]),
|
||||||
|
),
|
||||||
|
evidence,
|
||||||
|
new AbortController().signal,
|
||||||
|
);
|
||||||
|
expect(blankReason).toMatchObject({ code: "invalid_reason" });
|
||||||
|
});
|
||||||
|
|
||||||
it("maps the bounded deadline to timeout", async () => {
|
it("maps the bounded deadline to timeout", async () => {
|
||||||
const waiting = ready(
|
const waiting = ready(
|
||||||
async (_context, signal) =>
|
async (_context, signal) =>
|
||||||
@@ -319,3 +464,92 @@ describe("requestStructuredVerdict", () => {
|
|||||||
expect(timedOut).toMatchObject({ code: "timeout" });
|
expect(timedOut).toMatchObject({ code: "timeout" });
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("createModelAvailability — per-API forced tool choice and output cap", () => {
|
||||||
|
it("returns no_model for an undefined model", () => {
|
||||||
|
expect(createModelAvailability(undefined, registryStub())).toEqual({
|
||||||
|
kind: "no_model",
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns unsupported_api for an unknown api", () => {
|
||||||
|
const availability = createModelAvailability(
|
||||||
|
modelWithApi("mystery-api"),
|
||||||
|
registryStub(),
|
||||||
|
);
|
||||||
|
expect(availability).toMatchObject({
|
||||||
|
kind: "unsupported_api",
|
||||||
|
metadata: { api: "mystery-api" },
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it.each([
|
||||||
|
// [api, expected toolChoice, enforces output cap]
|
||||||
|
["anthropic-messages", { type: "tool", name: "report_verdict" }, true],
|
||||||
|
["bedrock-converse-stream", { type: "tool", name: "report_verdict" }, true],
|
||||||
|
["google-generative-ai", "any", true],
|
||||||
|
["google-vertex", "any", true],
|
||||||
|
["openai-completions", { type: "function", function: { name: "report_verdict" } }, true],
|
||||||
|
["mistral-conversations", { type: "function", function: { name: "report_verdict" } }, true],
|
||||||
|
["pi-messages", { type: "function", function: { name: "report_verdict" } }, true],
|
||||||
|
["openai-responses", { type: "function", name: "report_verdict" }, true],
|
||||||
|
["azure-openai-responses", { type: "function", name: "report_verdict" }, true],
|
||||||
|
// Codex supports required but not named choice, and no output cap.
|
||||||
|
["openai-codex-responses", "required", false],
|
||||||
|
])("maps %s to its forced tool choice and cap policy", async (api, toolChoice, cap) => {
|
||||||
|
const calls: Array<Record<string, unknown>> = [];
|
||||||
|
const registry = registryStub({
|
||||||
|
complete: (_model, _context, options) => {
|
||||||
|
calls.push(options as Record<string, unknown>);
|
||||||
|
return Promise.resolve(response(verdictContent()));
|
||||||
|
},
|
||||||
|
});
|
||||||
|
const availability = createModelAvailability(modelWithApi(api), registry);
|
||||||
|
expect(availability).toMatchObject({ kind: "ready", metadata: { api } });
|
||||||
|
|
||||||
|
if (availability.kind === "ready") {
|
||||||
|
await availability.complete({} as never, new AbortController().signal);
|
||||||
|
}
|
||||||
|
expect(calls[0]?.toolChoice).toEqual(toolChoice);
|
||||||
|
if (cap) {
|
||||||
|
expect(calls[0]?.maxTokens).toBeTypeOf("number");
|
||||||
|
} else {
|
||||||
|
expect(calls[0]?.maxTokens).toBeUndefined();
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
it("forwards retry and cache policy to the registry", async () => {
|
||||||
|
let seen: Record<string, unknown> | undefined;
|
||||||
|
const registry = registryStub({
|
||||||
|
complete: (_m, _c, options) => {
|
||||||
|
seen = options as Record<string, unknown>;
|
||||||
|
return Promise.resolve(response(verdictContent()));
|
||||||
|
},
|
||||||
|
});
|
||||||
|
const availability = createModelAvailability(
|
||||||
|
modelWithApi("anthropic-messages"),
|
||||||
|
registry,
|
||||||
|
);
|
||||||
|
if (availability.kind === "ready") {
|
||||||
|
await availability.complete({} as never, new AbortController().signal);
|
||||||
|
}
|
||||||
|
expect(seen).toMatchObject({ maxRetries: 0, cacheRetention: "none" });
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
function modelWithApi(api: string): Model<any> {
|
||||||
|
return {
|
||||||
|
provider: "test-provider",
|
||||||
|
id: "test-model",
|
||||||
|
api,
|
||||||
|
} as unknown as Model<any>;
|
||||||
|
}
|
||||||
|
|
||||||
|
function registryStub(overrides: Partial<ModelRegistry> = {}): ModelRegistry {
|
||||||
|
return {
|
||||||
|
complete: () => {
|
||||||
|
throw new Error("not called");
|
||||||
|
},
|
||||||
|
...overrides,
|
||||||
|
} as unknown as ModelRegistry;
|
||||||
|
}
|
||||||
|
|||||||
@@ -359,7 +359,8 @@ import { qualifyReplay, type ReplayRow } from "./replay-qualify";
|
|||||||
* visibly stale. Date-based: <yyyy-mm-dd>.<n>. */
|
* visibly stale. Date-based: <yyyy-mm-dd>.<n>. */
|
||||||
export const CORPUS_VERSION = "2026-08-21.2";
|
export const CORPUS_VERSION = "2026-08-21.2";
|
||||||
|
|
||||||
interface CliOptions {
|
/** Parsed CLI options; exported for in-process tests. */
|
||||||
|
export interface CliOptions {
|
||||||
provider: string;
|
provider: string;
|
||||||
model: string;
|
model: string;
|
||||||
timeoutMs: number;
|
timeoutMs: number;
|
||||||
@@ -368,33 +369,33 @@ interface CliOptions {
|
|||||||
strict: boolean;
|
strict: boolean;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Parse CLI options; exported for exit-contract tests. Any invalid
|
/** Mutable parse state; frozen into a CliOptions at the end. */
|
||||||
* combination is an error string the caller turns into exit 1. */
|
type CliState = { -readonly [K in keyof CliOptions]: CliOptions[K] };
|
||||||
export function parseArgs(argv: readonly string[]): CliOptions | { error: string } {
|
|
||||||
const args = argv.slice(2);
|
const VALUE_OPTIONS: Readonly<
|
||||||
let provider = "";
|
Record<string, (state: CliState, value: string) => void>
|
||||||
let model = "";
|
> = {
|
||||||
let timeoutMs = DEFAULT_TIMEOUT_MS;
|
"--provider": (state, value) => {
|
||||||
let cases: ReadonlySet<string> | null = null;
|
state.provider = value;
|
||||||
let out: string | null = null;
|
},
|
||||||
let strict = false;
|
"--model": (state, value) => {
|
||||||
for (let i = 0; i < args.length; i += 1) {
|
state.model = value;
|
||||||
const arg = args[i] as string;
|
},
|
||||||
const value = args[i + 1];
|
"--out": (state, value) => {
|
||||||
if (arg === "--provider" || arg === "--model" || arg === "--out" || arg === "--case") {
|
state.out = value;
|
||||||
if (value === undefined) return { error: `${arg} requires a value` };
|
},
|
||||||
if (arg === "--provider") provider = value;
|
"--case": (state, value) => {
|
||||||
if (arg === "--model") model = value;
|
state.cases = new Set(
|
||||||
if (arg === "--out") out = value;
|
|
||||||
if (arg === "--case") {
|
|
||||||
cases = new Set(
|
|
||||||
value.split(",").map((c) => c.trim()).filter((c) => c.length > 0),
|
value.split(",").map((c) => c.trim()).filter((c) => c.length > 0),
|
||||||
);
|
);
|
||||||
}
|
},
|
||||||
i += 1;
|
};
|
||||||
continue;
|
|
||||||
}
|
/** Apply `--timeout-ms`; exported for in-process tests. */
|
||||||
if (arg === "--timeout-ms") {
|
export function applyTimeoutOption(
|
||||||
|
state: CliState,
|
||||||
|
value: string | undefined,
|
||||||
|
): string | null {
|
||||||
const n = Number(value);
|
const n = Number(value);
|
||||||
if (
|
if (
|
||||||
value === undefined ||
|
value === undefined ||
|
||||||
@@ -402,88 +403,135 @@ export function parseArgs(argv: readonly string[]): CliOptions | { error: string
|
|||||||
n < MIN_TIMEOUT_MS ||
|
n < MIN_TIMEOUT_MS ||
|
||||||
n > MAX_TIMEOUT_MS
|
n > MAX_TIMEOUT_MS
|
||||||
) {
|
) {
|
||||||
return { error: `--timeout-ms must be an integer in [${MIN_TIMEOUT_MS}, ${MAX_TIMEOUT_MS}]` };
|
return `--timeout-ms must be an integer in [${MIN_TIMEOUT_MS}, ${MAX_TIMEOUT_MS}]`;
|
||||||
|
}
|
||||||
|
state.timeoutMs = n;
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Cross-option validation; exported for in-process tests. */
|
||||||
|
export function validateCliState(state: CliState): string | null {
|
||||||
|
if (state.provider === "" || state.model === "") {
|
||||||
|
return "usage: corpus-replay --provider <p> --model <m> [--timeout-ms N] [--case a,b] [--out file.json] [--strict]";
|
||||||
|
}
|
||||||
|
// Qualification is defined over the full corpus; a --case subset is
|
||||||
|
// observation-only and must never be able to report qualified.
|
||||||
|
if (state.strict && state.cases !== null) {
|
||||||
|
return "--strict requires the full corpus; --case selects an observation-only subset";
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Parse CLI options; exported for exit-contract tests. Any invalid
|
||||||
|
* combination is an error string the caller turns into exit 1. */
|
||||||
|
export function parseArgs(argv: readonly string[]): CliOptions | { error: string } {
|
||||||
|
const args = argv.slice(2);
|
||||||
|
const state: CliState = {
|
||||||
|
provider: "",
|
||||||
|
model: "",
|
||||||
|
timeoutMs: DEFAULT_TIMEOUT_MS,
|
||||||
|
cases: null,
|
||||||
|
out: null,
|
||||||
|
strict: false,
|
||||||
|
};
|
||||||
|
for (let i = 0; i < args.length; i += 1) {
|
||||||
|
const arg = args[i] as string;
|
||||||
|
const apply = VALUE_OPTIONS[arg];
|
||||||
|
if (apply !== undefined) {
|
||||||
|
const value = args[i + 1];
|
||||||
|
if (value === undefined) {
|
||||||
|
return { error: `${arg} requires a value` };
|
||||||
|
}
|
||||||
|
apply(state, value);
|
||||||
|
i += 1;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (arg === "--timeout-ms") {
|
||||||
|
const error = applyTimeoutOption(state, args[i + 1]);
|
||||||
|
if (error !== null) {
|
||||||
|
return { error };
|
||||||
}
|
}
|
||||||
timeoutMs = n;
|
|
||||||
i += 1;
|
i += 1;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (arg === "--strict") {
|
if (arg === "--strict") {
|
||||||
strict = true;
|
state.strict = true;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
return { error: `unknown option: ${arg}` };
|
return { error: `unknown option: ${arg}` };
|
||||||
}
|
}
|
||||||
if (provider === "" || model === "") {
|
const error = validateCliState(state);
|
||||||
return { error: "usage: corpus-replay --provider <p> --model <m> [--timeout-ms N] [--case a,b] [--out file.json] [--strict]" };
|
if (error !== null) {
|
||||||
|
return { error };
|
||||||
}
|
}
|
||||||
// Qualification is defined over the full corpus; a --case subset is
|
return state;
|
||||||
// observation-only and must never be able to report qualified.
|
|
||||||
if (strict && cases !== null) {
|
|
||||||
return {
|
|
||||||
error:
|
|
||||||
"--strict requires the full corpus; --case selects an observation-only subset",
|
|
||||||
};
|
|
||||||
}
|
|
||||||
return { provider, model, timeoutMs, cases, out, strict };
|
|
||||||
}
|
|
||||||
|
|
||||||
async function main(): Promise<number> {
|
|
||||||
const parsed = parseArgs(process.argv);
|
|
||||||
if ("error" in parsed) {
|
|
||||||
process.stderr.write(`${parsed.error}\n`);
|
|
||||||
return 1;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Resolve the replay model through the production seam: ModelRegistry
|
||||||
|
* lookup, configured-auth check, then createModelAvailability — the same
|
||||||
|
* forced-tool/output-cap path the online judge uses.
|
||||||
|
*/
|
||||||
|
async function resolveReplayModel(
|
||||||
|
provider: string,
|
||||||
|
model: string,
|
||||||
|
): Promise<ModelAvailability | number> {
|
||||||
const registry = new ModelRegistry(await ModelRuntime.create());
|
const registry = new ModelRegistry(await ModelRuntime.create());
|
||||||
await registry.refresh();
|
await registry.refresh();
|
||||||
const model = registry.find(parsed.provider, parsed.model) as Model<any> | undefined;
|
const found = registry.find(provider, model) as Model<any> | undefined;
|
||||||
if (model === undefined) {
|
if (found === undefined) {
|
||||||
process.stderr.write(
|
process.stderr.write(
|
||||||
`error: model ${parsed.provider}/${parsed.model} not found in models.json\n`,
|
`error: model ${provider}/${model} not found in models.json\n`,
|
||||||
);
|
);
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
if (!registry.hasConfiguredAuth(model)) {
|
if (!registry.hasConfiguredAuth(found)) {
|
||||||
process.stderr.write(
|
process.stderr.write(
|
||||||
`error: no configured auth for ${parsed.provider}/${parsed.model}\n`,
|
`error: no configured auth for ${provider}/${model}\n`,
|
||||||
);
|
);
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Production adapter: registry.complete through the same seam the
|
|
||||||
// online judge uses (createModelAvailability), so this harness shares
|
|
||||||
// its forced-tool/output-cap behavior.
|
|
||||||
const { createModelAvailability } = await import("../src/model");
|
const { createModelAvailability } = await import("../src/model");
|
||||||
const availability: ModelAvailability = createModelAvailability(model, registry);
|
const availability: ModelAvailability = createModelAvailability(found, registry);
|
||||||
if (availability.kind !== "ready") {
|
if (availability.kind !== "ready") {
|
||||||
process.stderr.write(
|
process.stderr.write(
|
||||||
`error: model unavailable: ${availability.kind} (${JSON.stringify(availability.kind === "unsupported_api" ? availability.metadata : null)})\n`,
|
`error: model unavailable: ${availability.kind} (${JSON.stringify(availability.kind === "unsupported_api" ? availability.metadata : null)})\n`,
|
||||||
);
|
);
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
|
return availability;
|
||||||
const shutdown = new AbortController();
|
|
||||||
const selected = parsed.cases === null
|
|
||||||
? CORPUS
|
|
||||||
: CORPUS.filter((c) => (parsed.cases as ReadonlySet<string>).has(c.id));
|
|
||||||
const unknown = parsed.cases === null
|
|
||||||
? []
|
|
||||||
: [...parsed.cases].filter((id) => !CORPUS.some((c) => c.id === id));
|
|
||||||
if (unknown.length > 0) {
|
|
||||||
process.stderr.write(`error: unknown case ids: ${unknown.join(", ")}\n`);
|
|
||||||
return 1;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Select corpus cases by id; unknown ids are a hard error. Exported for tests. */
|
||||||
|
export function selectCorpusCases(
|
||||||
|
cases: ReadonlySet<string> | null,
|
||||||
|
): { selected: readonly CorpusCase[] } | { error: string } {
|
||||||
|
if (cases === null) {
|
||||||
|
return { selected: CORPUS };
|
||||||
|
}
|
||||||
|
const selected = CORPUS.filter((c) => cases.has(c.id));
|
||||||
|
const unknown = [...cases].filter((id) => !CORPUS.some((c) => c.id === id));
|
||||||
|
if (unknown.length > 0) {
|
||||||
|
return { error: `unknown case ids: ${unknown.join(", ")}` };
|
||||||
|
}
|
||||||
|
return { selected };
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Replay the selected cases, printing per-case progress to stderr. Exported for tests. */
|
||||||
|
export async function replayCorpus(
|
||||||
|
selected: readonly CorpusCase[],
|
||||||
|
availability: ModelAvailability,
|
||||||
|
timeoutMs: number,
|
||||||
|
shutdownSignal: AbortSignal,
|
||||||
|
): Promise<{ rows: ReplayRow[]; matched: number }> {
|
||||||
const rows: ReplayRow[] = [];
|
const rows: ReplayRow[] = [];
|
||||||
let matched = 0;
|
let matched = 0;
|
||||||
const startedAll = Date.now();
|
|
||||||
for (const c of selected) {
|
for (const c of selected) {
|
||||||
const attempt = await requestStructuredVerdict(
|
const attempt = await requestStructuredVerdict(
|
||||||
availability,
|
availability,
|
||||||
c.evidence,
|
c.evidence,
|
||||||
shutdown.signal,
|
shutdownSignal,
|
||||||
parsed.timeoutMs,
|
timeoutMs,
|
||||||
c.conversation,
|
c.conversation,
|
||||||
);
|
);
|
||||||
let row: ReplayRow & { boundary?: string; code?: string };
|
let row: ReplayRow & { boundary?: string; code?: string };
|
||||||
@@ -515,6 +563,36 @@ async function main(): Promise<number> {
|
|||||||
`${c.id}: ${row.verdict ?? String(row.resultKind)} (expected ${c.expected}) ${row.match ? "MATCH" : "MISS"}\n`,
|
`${c.id}: ${row.verdict ?? String(row.resultKind)} (expected ${c.expected}) ${row.match ? "MATCH" : "MISS"}\n`,
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
return { rows, matched };
|
||||||
|
}
|
||||||
|
|
||||||
|
async function main(): Promise<number> {
|
||||||
|
const parsed = parseArgs(process.argv);
|
||||||
|
if ("error" in parsed) {
|
||||||
|
process.stderr.write(`${parsed.error}\n`);
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
const resolved = await resolveReplayModel(parsed.provider, parsed.model);
|
||||||
|
if (typeof resolved === "number") {
|
||||||
|
return resolved;
|
||||||
|
}
|
||||||
|
|
||||||
|
const selection = selectCorpusCases(parsed.cases);
|
||||||
|
if ("error" in selection) {
|
||||||
|
process.stderr.write(`error: ${selection.error}\n`);
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
const shutdown = new AbortController();
|
||||||
|
const startedAll = Date.now();
|
||||||
|
const { rows, matched } = await replayCorpus(
|
||||||
|
selection.selected,
|
||||||
|
resolved,
|
||||||
|
parsed.timeoutMs,
|
||||||
|
shutdown.signal,
|
||||||
|
);
|
||||||
|
const selected = selection.selected;
|
||||||
|
|
||||||
const qualification = qualifyReplay(rows, { budgetMs: parsed.timeoutMs });
|
const qualification = qualifyReplay(rows, { budgetMs: parsed.timeoutMs });
|
||||||
const report = {
|
const report = {
|
||||||
|
|||||||
@@ -23,6 +23,8 @@
|
|||||||
"devDependencies": {
|
"devDependencies": {
|
||||||
"@earendil-works/pi-coding-agent": "*",
|
"@earendil-works/pi-coding-agent": "*",
|
||||||
"@types/node": "^26.0.0",
|
"@types/node": "^26.0.0",
|
||||||
|
"@vitest/coverage-istanbul": "3.2.7",
|
||||||
|
"@vitest/coverage-v8": "^3.2.7",
|
||||||
"typescript": "^5",
|
"typescript": "^5",
|
||||||
"vitest": "^3"
|
"vitest": "^3"
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -444,6 +444,140 @@ describe("authorizeInnerCommand — fail-closed deferrals", () => {
|
|||||||
expect(check).toEqual([]);
|
expect(check).toEqual([]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// -- extractBashCommandEvidence guard branches: every malformed shape
|
||||||
|
// of the structured payload must defer silently (fail-closed) without
|
||||||
|
// reaching the deterministic query.
|
||||||
|
|
||||||
|
it("defers silently when payload.evidence is not an array", async () => {
|
||||||
|
const details = bashDetails("call_1", null, "timeout 30s pnpm test");
|
||||||
|
const { verdict, log, check } = await run({
|
||||||
|
recoveredCommand: "timeout 30s pnpm test",
|
||||||
|
states: { "pnpm test": "allow" },
|
||||||
|
details: {
|
||||||
|
payload: {
|
||||||
|
...details.payload,
|
||||||
|
evidence: "not-an-array" as unknown as [],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
});
|
||||||
|
expect(verdict.kind).toBe("defer");
|
||||||
|
expect(log).toEqual([]);
|
||||||
|
expect(check).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("defers silently when payload.request is missing", async () => {
|
||||||
|
const details = bashDetails("call_1", null, "timeout 30s pnpm test");
|
||||||
|
const { verdict, check } = await run({
|
||||||
|
recoveredCommand: "timeout 30s pnpm test",
|
||||||
|
states: { "pnpm test": "allow" },
|
||||||
|
details: {
|
||||||
|
payload: {
|
||||||
|
...details.payload,
|
||||||
|
request: undefined as unknown as (typeof details.payload)["request"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
});
|
||||||
|
expect(verdict.kind).toBe("defer");
|
||||||
|
expect(check).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("defers silently when the requester was forwarded", async () => {
|
||||||
|
const details = bashDetails("call_1", null, "timeout 30s pnpm test");
|
||||||
|
const { verdict, check } = await run({
|
||||||
|
recoveredCommand: "timeout 30s pnpm test",
|
||||||
|
states: { "pnpm test": "allow" },
|
||||||
|
details: {
|
||||||
|
payload: {
|
||||||
|
...details.payload,
|
||||||
|
request: {
|
||||||
|
...details.payload.request,
|
||||||
|
requester: { agentName: null, forwarded: true, sessionId: "s-child" },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
});
|
||||||
|
expect(verdict.kind).toBe("defer");
|
||||||
|
expect(check).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("defers silently when the ask came through an invoking tool", async () => {
|
||||||
|
const details = bashDetails("call_1", null, "timeout 30s pnpm test");
|
||||||
|
const { verdict, check } = await run({
|
||||||
|
recoveredCommand: "timeout 30s pnpm test",
|
||||||
|
states: { "pnpm test": "allow" },
|
||||||
|
details: {
|
||||||
|
payload: {
|
||||||
|
...details.payload,
|
||||||
|
request: {
|
||||||
|
...details.payload.request,
|
||||||
|
invokedToolName: "custom_tool",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
});
|
||||||
|
expect(verdict.kind).toBe("defer");
|
||||||
|
expect(check).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("defers silently when the request surface is not Bash", async () => {
|
||||||
|
const details = bashDetails("call_1", null, "timeout 30s pnpm test");
|
||||||
|
const { verdict, check } = await run({
|
||||||
|
recoveredCommand: "timeout 30s pnpm test",
|
||||||
|
states: { "pnpm test": "allow" },
|
||||||
|
details: {
|
||||||
|
payload: {
|
||||||
|
...details.payload,
|
||||||
|
request: { ...details.payload.request, surface: "edit" },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
});
|
||||||
|
expect(verdict.kind).toBe("defer");
|
||||||
|
expect(check).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("defers silently on a blank request value", async () => {
|
||||||
|
const details = bashDetails("call_1", null, "timeout 30s pnpm test");
|
||||||
|
const { verdict, check } = await run({
|
||||||
|
recoveredCommand: "timeout 30s pnpm test",
|
||||||
|
states: { "pnpm test": "allow" },
|
||||||
|
details: {
|
||||||
|
payload: {
|
||||||
|
...details.payload,
|
||||||
|
request: { ...details.payload.request, value: " " },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
});
|
||||||
|
expect(verdict.kind).toBe("defer");
|
||||||
|
expect(check).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("defers silently when the legacy command projection disagrees with the structured value", async () => {
|
||||||
|
const { verdict, check } = await run({
|
||||||
|
recoveredCommand: "timeout 30s pnpm test",
|
||||||
|
states: { "pnpm test": "allow" },
|
||||||
|
details: { command: "echo different" },
|
||||||
|
});
|
||||||
|
expect(verdict.kind).toBe("defer");
|
||||||
|
expect(check).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("defers silently on a blank full-command evidence text", async () => {
|
||||||
|
const details = bashDetails(
|
||||||
|
"call_1",
|
||||||
|
null,
|
||||||
|
"timeout 30s pnpm test",
|
||||||
|
" ",
|
||||||
|
);
|
||||||
|
const { verdict, check } = await run({
|
||||||
|
recoveredCommand: " ",
|
||||||
|
unitCommand: "timeout 30s pnpm test",
|
||||||
|
states: { "pnpm test": "allow" },
|
||||||
|
details: { payload: details.payload },
|
||||||
|
});
|
||||||
|
expect(verdict.kind).toBe("defer");
|
||||||
|
expect(check).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
it("defers silently for a shell alias that re-exposes Bash", async () => {
|
it("defers silently for a shell alias that re-exposes Bash", async () => {
|
||||||
const details = bashDetails(
|
const details = bashDetails(
|
||||||
"call_1",
|
"call_1",
|
||||||
|
|||||||
Generated
+825
-37
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user