fix(ai-judge): classify provider aborts after timeout as timeout

This commit is contained in:
2026-08-17 16:58:10 +08:00
parent 6407b7429a
commit 92cabb6b7c
2 changed files with 37 additions and 3 deletions
+15 -2
View File
@@ -7,7 +7,13 @@ import type { ModelRegistry } from "@earendil-works/pi-coding-agent";
import { buildJudgeContext, MAX_REASON_CODE_POINTS, REPORT_VERDICT_TOOL_NAME } from "./prompt"; import { buildJudgeContext, MAX_REASON_CODE_POINTS, REPORT_VERDICT_TOOL_NAME } from "./prompt";
import type { BashJudgmentEvidence } from "./evidence"; import type { BashJudgmentEvidence } from "./evidence";
const DEFAULT_TIMEOUT_MS = 15_000; // 60s: glm-5.2 at the user's default `thinking: high` profile was observed
// both finishing in seconds and racing the former 15s deadline (one verdict
// landed at exactly 15.003s and was mislabeled `aborted`); high-variance
// reasoning latency needs the wider bound. The judge is the first chain
// link, so its wait delays the human prompt by at most this much.
// PIEXTENSIO-11 calibrates a final value from cohort data.
const DEFAULT_TIMEOUT_MS = 60_000;
// Reasoning-token aware cap. Providers that bill chain-of-thought inside // Reasoning-token aware cap. Providers that bill chain-of-thought inside
// completion tokens (observed on zai glm-5.2 despite `thinking: disabled`: // completion tokens (observed on zai glm-5.2 despite `thinking: disabled`:
// 669 reasoning + 70 output for one verdict) exhausted a 256-token budget // 669 reasoning + 70 output for one verdict) exhausted a 256-token budget
@@ -203,12 +209,19 @@ export async function requestStructuredVerdict(
requestController.signal, requestController.signal,
); );
if (shutdownSignal.aborted || response.stopReason === "aborted") { if (shutdownSignal.aborted) {
return failure("aborted"); return failure("aborted");
} }
// Check timeout before the provider's abort stopReason: a provider
// returning `aborted` after our own deadline hit is a timeout, not a
// session-shutdown abort. Mislabeling it starves PIEXTENSIO-11's
// timeout calibration of exactly the rows it needs.
if (timeoutController.signal.aborted) { if (timeoutController.signal.aborted) {
return failure("timeout"); return failure("timeout");
} }
if (response.stopReason === "aborted") {
return failure("aborted");
}
if (response.stopReason === "error") { if (response.stopReason === "error") {
return failure("model_error"); return failure("model_error");
} }
@@ -14,6 +14,7 @@ const metadata = {
function response( function response(
content: AssistantMessage["content"], content: AssistantMessage["content"],
output = 12, output = 12,
stopReason: AssistantMessage["stopReason"] = "toolUse",
): AssistantMessage { ): AssistantMessage {
return { return {
role: "assistant", role: "assistant",
@@ -21,6 +22,7 @@ function response(
api: metadata.api, api: metadata.api,
provider: metadata.provider, provider: metadata.provider,
model: metadata.model, model: metadata.model,
stopReason,
usage: { usage: {
input: 10, input: 10,
output, output,
@@ -35,7 +37,6 @@ function response(
total: 0, total: 0,
}, },
}, },
stopReason: "toolUse",
timestamp: Date.now(), timestamp: Date.now(),
}; };
} }
@@ -297,4 +298,24 @@ describe("requestStructuredVerdict", () => {
}); });
expect(complete).not.toHaveBeenCalled(); expect(complete).not.toHaveBeenCalled();
}); });
it("classifies a provider abort arriving after the deadline as timeout", async () => {
const complete = vi.fn(
(_context: Context, signal: AbortSignal) =>
new Promise<AssistantMessage>((resolve, reject) => {
signal.addEventListener("abort", () =>
// Provider surfaces the client abort as an `aborted`
// stopReason after our 1ms deadline already fired.
resolve(response([], 12, "aborted")),
);
}),
);
const timedOut = await requestStructuredVerdict(
ready(complete as never),
evidence,
new AbortController().signal,
1,
);
expect(timedOut).toMatchObject({ code: "timeout" });
});
}); });