fix(permission): raise verdict output budget for reasoning tokens

This commit is contained in:
2026-08-16 23:54:58 +08:00
parent 1afcbd3118
commit f632fa34d1
3 changed files with 58 additions and 4 deletions
+6 -1
View File
@@ -102,7 +102,11 @@ export default function permissionAiJudge(pi: ExtensionAPI): void {
provider: result.metadata.provider, provider: result.metadata.provider,
model: result.metadata.model, model: result.metadata.model,
api: result.metadata.api, api: result.metadata.api,
outputTokens: result.outputTokens, // Log key deliberately avoids the substring "token":
// permission-system masks any key matching /token/i
// (structural key-name redaction), which would erase
// this usage telemetry from the review log.
outputUsage: result.outputTokens,
reasonLength: reasonLength(result.reason), reasonLength: reasonLength(result.reason),
}); });
} else { } else {
@@ -119,6 +123,7 @@ export default function permissionAiJudge(pi: ExtensionAPI): void {
provider: result.metadata?.provider ?? null, provider: result.metadata?.provider ?? null,
model: result.metadata?.model ?? null, model: result.metadata?.model ?? null,
api: result.metadata?.api ?? null, api: result.metadata?.api ?? null,
outputUsage: result.outputTokens ?? null,
}); });
} }
+13 -1
View File
@@ -8,7 +8,12 @@ import { buildJudgeContext, MAX_REASON_CODE_POINTS, REPORT_VERDICT_TOOL_NAME } f
import type { BashJudgmentEvidence } from "./evidence"; import type { BashJudgmentEvidence } from "./evidence";
const DEFAULT_TIMEOUT_MS = 15_000; const DEFAULT_TIMEOUT_MS = 15_000;
const MAX_OUTPUT_TOKENS = 256; // Reasoning-token aware cap. Providers that bill chain-of-thought inside
// completion tokens (observed on zai glm-5.2 despite `thinking: disabled`:
// 669 reasoning + 70 output for one verdict) exhausted a 256-token budget
// before emitting the forced tool call. 4096 leaves headroom over the
// observed ~740 while the no-retry timeout still bounds worst-case cost.
const MAX_OUTPUT_TOKENS = 4_096;
export type InfrastructureCode = export type InfrastructureCode =
| "no_model" | "no_model"
@@ -57,6 +62,8 @@ export type ModelAttempt =
readonly code: InfrastructureCode; readonly code: InfrastructureCode;
readonly metadata?: ModelMetadata; readonly metadata?: ModelMetadata;
readonly modelCalled: boolean; readonly modelCalled: boolean;
/** Completion-token usage when a response arrived, else null. */
readonly outputTokens?: number | null;
}; };
function forcedToolChoice(api: string): unknown | undefined { function forcedToolChoice(api: string): unknown | undefined {
@@ -155,11 +162,13 @@ export async function requestStructuredVerdict(
} }
let modelCalled = false; let modelCalled = false;
let observedOutputTokens: number | null = null;
const failure = (code: InfrastructureCode): ModelAttempt => ({ const failure = (code: InfrastructureCode): ModelAttempt => ({
kind: "infrastructure_failure", kind: "infrastructure_failure",
code, code,
metadata: availability.metadata, metadata: availability.metadata,
modelCalled, modelCalled,
outputTokens: observedOutputTokens,
}); });
const timeoutController = new AbortController(); const timeoutController = new AbortController();
const requestController = new AbortController(); const requestController = new AbortController();
@@ -193,6 +202,9 @@ export async function requestStructuredVerdict(
} }
const outputTokens = response.usage.output; const outputTokens = response.usage.output;
if (Number.isFinite(outputTokens)) {
observedOutputTokens = outputTokens;
}
if ( if (
!Number.isFinite(outputTokens) || !Number.isFinite(outputTokens) ||
outputTokens <= 0 || outputTokens <= 0 ||
@@ -118,6 +118,40 @@ describe("requestStructuredVerdict", () => {
}); });
}); });
it("accepts reasoning-heavy responses within the raised budget", async () => {
// Observed live on zai glm-5.2: 669 reasoning + 70 output tokens still
// delivered exactly one valid report_verdict call.
const complete = vi.fn(
async (_context: Context, _signal: AbortSignal) =>
response(
[
{
type: "toolCall",
id: "call-1",
name: "report_verdict",
arguments: {
verdict: "allow",
reason: "Bounded command with evident intent.",
},
},
],
739,
),
);
const result = await requestStructuredVerdict(
ready(complete),
evidence,
new AbortController().signal,
);
expect(result).toMatchObject({
kind: "judgment",
verdict: "allow",
outputTokens: 739,
});
});
it("rejects invalid arguments, verdicts, reasons, and output usage", async () => { it("rejects invalid arguments, verdicts, reasons, and output usage", async () => {
const extraArguments = await requestStructuredVerdict( const extraArguments = await requestStructuredVerdict(
ready(async () => ready(async () =>
@@ -182,13 +216,16 @@ describe("requestStructuredVerdict", () => {
arguments: { verdict: "allow", reason: "bounded" }, arguments: { verdict: "allow", reason: "bounded" },
}, },
], ],
257, 4_097,
), ),
), ),
evidence, evidence,
new AbortController().signal, new AbortController().signal,
); );
expect(excessiveUsage).toMatchObject({ code: "model_error" }); expect(excessiveUsage).toMatchObject({
code: "model_error",
outputTokens: 4_097,
});
}); });
it("maps the bounded deadline to timeout", async () => { it("maps the bounded deadline to timeout", async () => {