mirror of
https://github.com/SikongJueluo/pi-extensions.git
synced 2026-10-05 11:52:55 +08:00
fix(permission): raise verdict output budget for reasoning tokens
This commit is contained in:
@@ -102,7 +102,11 @@ export default function permissionAiJudge(pi: ExtensionAPI): void {
|
||||
provider: result.metadata.provider,
|
||||
model: result.metadata.model,
|
||||
api: result.metadata.api,
|
||||
outputTokens: result.outputTokens,
|
||||
// Log key deliberately avoids the substring "token":
|
||||
// permission-system masks any key matching /token/i
|
||||
// (structural key-name redaction), which would erase
|
||||
// this usage telemetry from the review log.
|
||||
outputUsage: result.outputTokens,
|
||||
reasonLength: reasonLength(result.reason),
|
||||
});
|
||||
} else {
|
||||
@@ -119,6 +123,7 @@ export default function permissionAiJudge(pi: ExtensionAPI): void {
|
||||
provider: result.metadata?.provider ?? null,
|
||||
model: result.metadata?.model ?? null,
|
||||
api: result.metadata?.api ?? null,
|
||||
outputUsage: result.outputTokens ?? null,
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -8,7 +8,12 @@ import { buildJudgeContext, MAX_REASON_CODE_POINTS, REPORT_VERDICT_TOOL_NAME } f
|
||||
import type { BashJudgmentEvidence } from "./evidence";
|
||||
|
||||
const DEFAULT_TIMEOUT_MS = 15_000;
|
||||
const MAX_OUTPUT_TOKENS = 256;
|
||||
// Reasoning-token aware cap. Providers that bill chain-of-thought inside
|
||||
// completion tokens (observed on zai glm-5.2 despite `thinking: disabled`:
|
||||
// 669 reasoning + 70 output for one verdict) exhausted a 256-token budget
|
||||
// before emitting the forced tool call. 4096 leaves headroom over the
|
||||
// observed ~740 while the no-retry timeout still bounds worst-case cost.
|
||||
const MAX_OUTPUT_TOKENS = 4_096;
|
||||
|
||||
export type InfrastructureCode =
|
||||
| "no_model"
|
||||
@@ -57,6 +62,8 @@ export type ModelAttempt =
|
||||
readonly code: InfrastructureCode;
|
||||
readonly metadata?: ModelMetadata;
|
||||
readonly modelCalled: boolean;
|
||||
/** Completion-token usage when a response arrived, else null. */
|
||||
readonly outputTokens?: number | null;
|
||||
};
|
||||
|
||||
function forcedToolChoice(api: string): unknown | undefined {
|
||||
@@ -155,11 +162,13 @@ export async function requestStructuredVerdict(
|
||||
}
|
||||
|
||||
let modelCalled = false;
|
||||
let observedOutputTokens: number | null = null;
|
||||
const failure = (code: InfrastructureCode): ModelAttempt => ({
|
||||
kind: "infrastructure_failure",
|
||||
code,
|
||||
metadata: availability.metadata,
|
||||
modelCalled,
|
||||
outputTokens: observedOutputTokens,
|
||||
});
|
||||
const timeoutController = new AbortController();
|
||||
const requestController = new AbortController();
|
||||
@@ -193,6 +202,9 @@ export async function requestStructuredVerdict(
|
||||
}
|
||||
|
||||
const outputTokens = response.usage.output;
|
||||
if (Number.isFinite(outputTokens)) {
|
||||
observedOutputTokens = outputTokens;
|
||||
}
|
||||
if (
|
||||
!Number.isFinite(outputTokens) ||
|
||||
outputTokens <= 0 ||
|
||||
|
||||
@@ -118,6 +118,40 @@ describe("requestStructuredVerdict", () => {
|
||||
});
|
||||
});
|
||||
|
||||
it("accepts reasoning-heavy responses within the raised budget", async () => {
|
||||
// Observed live on zai glm-5.2: 669 reasoning + 70 output tokens still
|
||||
// delivered exactly one valid report_verdict call.
|
||||
const complete = vi.fn(
|
||||
async (_context: Context, _signal: AbortSignal) =>
|
||||
response(
|
||||
[
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call-1",
|
||||
name: "report_verdict",
|
||||
arguments: {
|
||||
verdict: "allow",
|
||||
reason: "Bounded command with evident intent.",
|
||||
},
|
||||
},
|
||||
],
|
||||
739,
|
||||
),
|
||||
);
|
||||
|
||||
const result = await requestStructuredVerdict(
|
||||
ready(complete),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(result).toMatchObject({
|
||||
kind: "judgment",
|
||||
verdict: "allow",
|
||||
outputTokens: 739,
|
||||
});
|
||||
});
|
||||
|
||||
it("rejects invalid arguments, verdicts, reasons, and output usage", async () => {
|
||||
const extraArguments = await requestStructuredVerdict(
|
||||
ready(async () =>
|
||||
@@ -182,13 +216,16 @@ describe("requestStructuredVerdict", () => {
|
||||
arguments: { verdict: "allow", reason: "bounded" },
|
||||
},
|
||||
],
|
||||
257,
|
||||
4_097,
|
||||
),
|
||||
),
|
||||
evidence,
|
||||
new AbortController().signal,
|
||||
);
|
||||
expect(excessiveUsage).toMatchObject({ code: "model_error" });
|
||||
expect(excessiveUsage).toMatchObject({
|
||||
code: "model_error",
|
||||
outputTokens: 4_097,
|
||||
});
|
||||
});
|
||||
|
||||
it("maps the bounded deadline to timeout", async () => {
|
||||
|
||||
Reference in New Issue
Block a user