fix(ai-judge): classify provider aborts after timeout as timeout

This commit is contained in:
2026-08-17 16:58:10 +08:00
parent 6407b7429a
commit 92cabb6b7c
2 changed files with 37 additions and 3 deletions
+15 -2
View File
@@ -7,7 +7,13 @@ import type { ModelRegistry } from "@earendil-works/pi-coding-agent";
import { buildJudgeContext, MAX_REASON_CODE_POINTS, REPORT_VERDICT_TOOL_NAME } from "./prompt";
import type { BashJudgmentEvidence } from "./evidence";
const DEFAULT_TIMEOUT_MS = 15_000;
// 60s: glm-5.2 at the user's default `thinking: high` profile was observed
// both finishing in seconds and racing the former 15s deadline (one verdict
// landed at exactly 15.003s and was mislabeled `aborted`); high-variance
// reasoning latency needs the wider bound. The judge is the first chain
// link, so its wait delays the human prompt by at most this much.
// PIEXTENSIO-11 calibrates a final value from cohort data.
const DEFAULT_TIMEOUT_MS = 60_000;
// Reasoning-token aware cap. Providers that bill chain-of-thought inside
// completion tokens (observed on zai glm-5.2 despite `thinking: disabled`:
// 669 reasoning + 70 output for one verdict) exhausted a 256-token budget
@@ -203,12 +209,19 @@ export async function requestStructuredVerdict(
requestController.signal,
);
if (shutdownSignal.aborted || response.stopReason === "aborted") {
if (shutdownSignal.aborted) {
return failure("aborted");
}
// Check timeout before the provider's abort stopReason: a provider
// returning `aborted` after our own deadline hit is a timeout, not a
// session-shutdown abort. Mislabeling it starves PIEXTENSIO-11's
// timeout calibration of exactly the rows it needs.
if (timeoutController.signal.aborted) {
return failure("timeout");
}
if (response.stopReason === "aborted") {
return failure("aborted");
}
if (response.stopReason === "error") {
return failure("model_error");
}
@@ -14,6 +14,7 @@ const metadata = {
function response(
content: AssistantMessage["content"],
output = 12,
stopReason: AssistantMessage["stopReason"] = "toolUse",
): AssistantMessage {
return {
role: "assistant",
@@ -21,6 +22,7 @@ function response(
api: metadata.api,
provider: metadata.provider,
model: metadata.model,
stopReason,
usage: {
input: 10,
output,
@@ -35,7 +37,6 @@ function response(
total: 0,
},
},
stopReason: "toolUse",
timestamp: Date.now(),
};
}
@@ -297,4 +298,24 @@ describe("requestStructuredVerdict", () => {
});
expect(complete).not.toHaveBeenCalled();
});
it("classifies a provider abort arriving after the deadline as timeout", async () => {
const complete = vi.fn(
(_context: Context, signal: AbortSignal) =>
new Promise<AssistantMessage>((resolve, reject) => {
signal.addEventListener("abort", () =>
// Provider surfaces the client abort as an `aborted`
// stopReason after our 1ms deadline already fired.
resolve(response([], 12, "aborted")),
);
}),
);
const timedOut = await requestStructuredVerdict(
ready(complete as never),
evidence,
new AbortController().signal,
1,
);
expect(timedOut).toMatchObject({ code: "timeout" });
});
});