refactor(permission): consume structured bash payload

- require @gotgenes/pi-permission-system >=25.3.0 and read the complete local bash command from PromptPermissionDetails.payload instead of session-walking recovery
- remove the @sikongjueluo/pi-permission-shared package
- pass the triggering command unit to handlers via HandlerContext.unit in place of details.command
- add shadow-only AI judge modules for evidence projection, structured verdict requests, and prompt building, with vitest coverage
- record ADR 0004 and mark the ADR 0001 recovery mechanism superseded
- exclude pi-permission-system 25.3.0 from the pnpm minimumReleaseAge guard
This commit is contained in:
2026-08-16 23:08:45 +08:00
parent 6008c9e817
commit 1afcbd3118
29 changed files with 1540 additions and 587 deletions
@@ -12,12 +12,10 @@
]
},
"dependencies": {
"@gotgenes/pi-permission-system": ">=24.0.0",
"@sikongjueluo/pi-permission-shared": "workspace:*"
"@gotgenes/pi-permission-system": ">=25.3.0"
},
"bundledDependencies": [
"@gotgenes/pi-permission-system",
"@sikongjueluo/pi-permission-shared"
"@gotgenes/pi-permission-system"
],
"peerDependencies": {
"@earendil-works/pi-ai": "*",
@@ -1,14 +1,9 @@
import type { SessionEntry } from "@earendil-works/pi-coding-agent";
import type {
AuthorizerLog,
AuthorizerVerdict,
PermissionQuery,
PromptPermissionDetails,
} from "@gotgenes/pi-permission-system";
import {
NATIVE_BASH_TOOL_NAME,
recoverNativeBashCommand,
} from "@sikongjueluo/pi-permission-shared";
import { handlers } from "./handlers";
/** Convert a thrown value into a short, log-safe string. */
@@ -26,10 +21,60 @@ function toErrorString(error: unknown): string {
* stub so the decision logic stays pure and deterministic.
*/
export interface SessionProbe {
getEntries(): ReadonlyArray<SessionEntry>;
getSessionId(): string;
}
const NATIVE_BASH_TOOL_NAME = "bash";
const FULL_COMMAND_LABEL = "full command";
interface BashCommandEvidence {
readonly fullCommand: string;
readonly triggeringUnit: string;
}
function isNonBlank(value: unknown): value is string {
return typeof value === "string" && value.trim().length > 0;
}
/** Read a complete direct native-Bash ask from the structured prompt payload. */
export function extractBashCommandEvidence(
details: PromptPermissionDetails,
): BashCommandEvidence | undefined {
const payload = details.payload;
const request = payload?.request;
if (
request === undefined ||
!Array.isArray(payload.evidence) ||
details.forwarding !== undefined ||
payload.kind !== "bash" ||
request.requester?.forwarded !== false ||
details.toolName !== NATIVE_BASH_TOOL_NAME ||
request.toolName !== NATIVE_BASH_TOOL_NAME ||
request.invokedToolName !== null ||
request.surface !== NATIVE_BASH_TOOL_NAME ||
!isNonBlank(request.value) ||
(details.command !== undefined && details.command !== request.value)
) {
return undefined;
}
const fullCommands = payload.evidence.filter(
(entry) => entry.label === FULL_COMMAND_LABEL,
);
if (fullCommands.length > 1) {
return undefined;
}
const fullCommand =
fullCommands.length === 0 ? request.value : fullCommands[0]?.text;
if (!isNonBlank(fullCommand)) {
return undefined;
}
return { fullCommand, triggeringUnit: request.value };
}
/** Dependencies injected into the pure authorizer decision. */
export interface InnerCommandAuthorizerDeps {
readonly details: PromptPermissionDetails;
@@ -46,21 +91,21 @@ export interface InnerCommandAuthorizerDeps {
}
/**
* Inner-command Authorizer decision (ADR 0001).
* Inner-command Authorizer decision (ADRs 0001 and 0004).
*
* Revalidates root ownership, recovers the complete native Bash command for
* `details.toolCallId` from the captured session, then hands it to the first
* registered handler that claims it. Each handler owns its own recognition and
* verdict logic: the timeout handler unwraps one level and re-evaluates the
* inner command; the env handler defers as non-transparent.
* Revalidates root ownership, reads the complete native Bash command from the
* structured prompt payload, then hands it to the first registered handler that
* claims it. Each handler owns its own recognition and verdict logic: the
* timeout handler unwraps one level and re-evaluates the inner command; the env
* handler defers as non-transparent.
*
* Every uncertain path — forwarded requests, a session-identity mismatch,
* non-Bash tools, missing session evidence, an unrecognized command, or any
* non-Bash tools, malformed payload evidence, an unrecognized command, or any
* exception — defers to the next authority (fail-closed).
*
* Logging: handlers emit their own review/debug events for decisive and
* notable-defer outcomes; silent deferrals log nothing. Exceptions are logged
* by this engine as `inner_cmd.exception`, retaining the recovered command and
* by this engine as `inner_cmd.exception`, retaining the structured command and
* any partial evidence the active handler recorded before throwing.
*/
export async function authorizeInnerCommand(
@@ -68,7 +113,7 @@ export async function authorizeInnerCommand(
): Promise<AuthorizerVerdict> {
const { details, query, log, session, expectedSessionId } = deps;
// Track recovered evidence so an exception after recognition can retain it.
// Track structured evidence so an exception after recognition can retain it.
let command: string | undefined;
let evidence: Record<string, unknown> = {};
@@ -90,27 +135,26 @@ export async function authorizeInnerCommand(
return { kind: "defer" };
}
// Only the native Bash tool is unwrappable, and only when the ask is
// tied to a specific tool call.
if (details.toolName !== NATIVE_BASH_TOOL_NAME) {
return { kind: "defer" };
}
const toolCallId = details.toolCallId;
if (toolCallId === undefined) {
return { kind: "defer" };
}
// Recover the complete Bash input from the session, never from
// details.command or details.message.
command = recoverNativeBashCommand(session.getEntries(), toolCallId);
if (command === undefined) {
// The payload is complete by contract. A "full command" evidence entry
// exists only when it differs from request.value; otherwise that value
// is the complete command. Ambiguous or inconsistent payloads defer.
const commandEvidence = extractBashCommandEvidence(details);
if (commandEvidence === undefined) {
return { kind: "defer" };
}
command = commandEvidence.fullCommand;
// Dispatch to the first registered handler that claims the command.
for (const handler of handlers) {
evidence = {};
const verdict = handler.decide({ command, details, query, log, evidence });
const verdict = handler.decide({
command,
unit: commandEvidence.triggeringUnit,
details,
query,
log,
evidence,
});
if (verdict !== undefined) {
return verdict;
}
@@ -36,10 +36,10 @@ function stripWrapperUnit(
/**
* The simple-timeout wrapper handler (ADR 0001).
*
* Detection runs on `details.command` — the command unit the permission system
* isolated as the ask trigger — which is always wrapper-leading even when the
* full recovered command is a scaffold that starts with `cd`/`echo`/…. The
* wrapper is then stripped from the FULL command and the whole de-wrapped
* Detection runs on the payload's decision-relevant command unit, which is
* wrapper-leading even when the complete command is a scaffold that starts
* with `cd`/`echo`/…. The wrapper is then stripped from the FULL command and the
* whole de-wrapped
* compound is re-evaluated, so sibling commands (including dangerous ones) are
* still judged and cannot hide behind the wrapper's allow.
*
@@ -50,11 +50,14 @@ function stripWrapperUnit(
export const timeoutHandler: CommandHandler = {
id: "timeout",
decide(ctx) {
const { command: fullCommand, details, query, log, evidence } = ctx;
const unit = details.command;
if (unit === undefined) {
return undefined;
}
const {
command: fullCommand,
unit,
details,
query,
log,
evidence,
} = ctx;
const unitMatch = parseTimeoutWrapper(unit);
if (unitMatch === undefined) {
@@ -5,10 +5,12 @@ import type {
PromptPermissionDetails,
} from "@gotgenes/pi-permission-system";
/** Context handed to a handler for one recovered command. */
/** Context handed to a handler for one structured Bash ask. */
export interface HandlerContext {
/** The full recovered Bash command. */
/** The complete Bash tool input from the permission payload. */
readonly command: string;
/** The command unit whose deterministic rule produced the ask. */
readonly unit: string;
readonly details: PromptPermissionDetails;
readonly query: PermissionQuery;
readonly log: AuthorizerLog;
@@ -57,14 +57,14 @@ export function isRecognizedWrapper(command: string): boolean {
return parseTimeoutWrapper(command) !== undefined;
}
/** How a recovered Bash command relates to the v0.1 recognizer. */
/** How a complete Bash command relates to the v0.1 recognizer. */
export type WrapperClassification =
| { readonly kind: "recognized"; readonly match: TimeoutWrapperMatch }
| { readonly kind: "unsupportedTimeout" }
| { readonly kind: "nonTimeout" };
/**
* Classify a recovered Bash command against the v0.1 recognizer.
* Classify a complete Bash command against the v0.1 recognizer.
*
* - `recognized`: the strict simple-timeout wrapper.
* - `unsupportedTimeout`: the command invokes `timeout` but is not the
@@ -1,5 +1,4 @@
import { describe, expect, it } from "vitest";
import type { SessionEntry } from "@earendil-works/pi-coding-agent";
import type {
AuthorizerLog,
PermissionCheckResult,
@@ -58,56 +57,58 @@ function makeQuery(
return { query, calls };
}
function assistantEntry(content: unknown[]): SessionEntry {
return {
type: "message",
id: "entry-1",
parentId: null,
timestamp: "2026-08-08T00:00:00.000Z",
message: { role: "assistant", content },
} as unknown as SessionEntry;
}
function bashToolCall(id: string, command: unknown): Record<string, unknown> {
return { type: "toolCall", id, name: "bash", arguments: { command } };
}
function entriesRecovering(command: string, toolCallId = "call_1"): SessionEntry[] {
return [assistantEntry([bashToolCall(toolCallId, command)])];
}
function bashDetails(
toolCallId = "call_1",
agentName: string | null = null,
command?: string,
command = "",
fullCommand = command,
): PromptPermissionDetails {
return {
requestId: "req-1",
source: "tool_call",
agentName,
message: "May I run bash?",
payload: {
kind: "bash",
request: {
requester: {
agentName,
forwarded: false,
sessionId: null,
},
surface: "bash",
toolName: "bash",
invokedToolName: null,
value: command,
matchedPattern: null,
commandContext: null,
executedUnit: null,
},
evidence:
fullCommand === command
? []
: [
{
label: "full command",
text: fullCommand,
detail: null,
},
],
annotations: [],
},
toolCallId,
toolName: "bash",
// details.command is the winning unit the permission system isolated.
// Legacy projection; the structured payload is authoritative.
command,
};
}
function makeSessionProbe(args: {
recoveredCommand: string;
toolCallId: string;
getEntriesThrows?: boolean;
/** Live session id reported at authorize time. */
sessionId?: string;
getSessionIdThrows?: boolean;
}): SessionProbe {
} = {}): SessionProbe {
return {
getEntries: args.getEntriesThrows
? (): SessionEntry[] => {
throw new Error("session boom");
}
: (): SessionEntry[] =>
entriesRecovering(args.recoveredCommand, args.toolCallId),
getSessionId: args.getSessionIdThrows
? (): string => {
throw new Error("session id boom");
@@ -122,7 +123,6 @@ async function run(args: {
unitCommand?: string;
states?: Record<string, PermissionState>;
details?: Partial<PromptPermissionDetails>;
getEntriesThrows?: boolean;
queryThrowsOn?: string;
/** Live session id diverges from the captured provenance. */
sessionMismatch?: boolean;
@@ -138,16 +138,18 @@ async function run(args: {
throwOn: args.queryThrowsOn,
});
const session = makeSessionProbe({
recoveredCommand: args.recoveredCommand,
toolCallId,
getEntriesThrows: args.getEntriesThrows,
getSessionIdThrows: args.getSessionIdThrows,
sessionId: args.sessionMismatch ? "session-changed" : ROOT_SESSION_ID,
});
const unitCommand = args.unitCommand ?? args.recoveredCommand;
const verdict = await authorizeInnerCommand({
details: {
...bashDetails(toolCallId, null, unitCommand),
...bashDetails(
toolCallId,
null,
unitCommand,
args.recoveredCommand,
),
...args.details,
} as PromptPermissionDetails,
query,
@@ -379,32 +381,87 @@ describe("authorizeInnerCommand — fail-closed deferrals", () => {
expect(log).toEqual([]);
});
it("defers silently when toolCallId is absent", async () => {
const { verdict, log } = await run({
it("uses the structured payload when toolCallId is absent", async () => {
const { verdict } = await run({
recoveredCommand: "timeout 30s pnpm test",
states: { "pnpm test": "allow" },
details: { toolCallId: undefined },
});
expect(verdict.kind).toBe("defer");
expect(log).toEqual([]);
expect(verdict.kind).toBe("allow");
});
it("defers silently when the tool call is not in the session", async () => {
// Recover a command under a different id so recovery misses.
const { log, calls } = makeLog();
const { query, calls: check } = makeQuery({ "pnpm test": "allow" });
const verdict = await authorizeInnerCommand({
details: bashDetails("call_missing"),
query,
log,
session: makeSessionProbe({
recoveredCommand: "timeout 30s pnpm test",
toolCallId: "call_1",
}),
expectedSessionId: ROOT_SESSION_ID,
it("defers silently on duplicate full-command evidence", async () => {
const details = bashDetails(
"call_1",
null,
"timeout 30s pnpm test",
"cd /repo && timeout 30s pnpm test",
);
const duplicate = details.payload.evidence[0]!;
const { verdict, log, check } = await run({
recoveredCommand: "cd /repo && timeout 30s pnpm test",
unitCommand: "timeout 30s pnpm test",
states: { "cd /repo && pnpm test": "allow" },
details: {
payload: {
...details.payload,
evidence: [duplicate, duplicate],
},
},
});
expect(verdict.kind).toBe("defer");
expect(log).toEqual([]);
expect(check).toEqual([]);
});
it("defers silently on malformed full-command evidence", async () => {
const details = bashDetails(
"call_1",
null,
"timeout 30s pnpm test",
"cd /repo && timeout 30s pnpm test",
);
const { verdict, check } = await run({
recoveredCommand: "cd /repo && timeout 30s pnpm test",
unitCommand: "timeout 30s pnpm test",
states: { "cd /repo && pnpm test": "allow" },
details: {
payload: {
...details.payload,
evidence: [
{
label: "full command",
text: null as unknown as string,
detail: null,
},
],
},
},
});
expect(verdict.kind).toBe("defer");
expect(check).toEqual([]);
});
it("defers silently for a shell alias that re-exposes Bash", async () => {
const details = bashDetails(
"call_1",
null,
"timeout 30s pnpm test",
);
const { verdict, check } = await run({
recoveredCommand: "timeout 30s pnpm test",
states: { "pnpm test": "allow" },
details: {
payload: {
...details.payload,
request: {
...details.payload.request,
invokedToolName: "exec_command",
},
},
},
});
expect(verdict.kind).toBe("defer");
expect(calls).toEqual([]);
expect(check).toEqual([]);
});
});
@@ -543,18 +600,29 @@ describe("authorizeInnerCommand — exceptions defer with a debug log", () => {
expect(log[0]?.details).toEqual({ error: "session id boom" });
});
it("defers when reading the session throws (logs only safe data)", async () => {
it("defers when reading payload evidence throws (logs only safe data)", async () => {
const base = bashDetails(
"call_1",
null,
"timeout 30s pnpm test",
);
const payload = { ...base.payload };
Object.defineProperty(payload, "evidence", {
get(): never {
throw new Error("payload boom");
},
});
const { verdict, log } = await run({
recoveredCommand: "timeout 30s pnpm test",
states: { "pnpm test": "allow" },
getEntriesThrows: true,
details: { payload },
});
expect(verdict.kind).toBe("defer");
expect(log).toHaveLength(1);
expect(log[0]?.level).toBe("debug");
expect(log[0]?.event).toBe("inner_cmd.exception");
// Exception before recognition: only the error is available.
expect(log[0]?.details).toEqual({ error: "session boom" });
expect(log[0]?.details).toEqual({ error: "payload boom" });
});
it("retains command and innerCommand when the query throws after recognition", async () => {
@@ -229,6 +229,31 @@ describe("permissions:ready -> registerAuthorizer lifecycle", () => {
source: "tool_call",
agentName: "child",
message: "forwarded ask",
payload: {
kind: "forwarded",
request: {
requester: {
agentName: "child",
forwarded: true,
sessionId: "s1",
},
surface: "bash",
toolName: "bash",
invokedToolName: null,
value: "bash",
matchedPattern: null,
commandContext: null,
executedUnit: null,
},
evidence: [
{
label: "requested",
text: "forwarded ask",
detail: null,
},
],
annotations: [],
},
toolCallId: "call_1",
toolName: "bash",
forwarding: {