mirror of
https://github.com/SikongJueluo/pi-extensions.git
synced 2026-10-05 11:52:55 +08:00
feat(ai-judge): conversation evidence with bounded whitelist capture
- add conversation.ts: compaction-aware active-branch capture, user-text-only whitelist, 16-item and 12,000-char bounds with latest-user preservation - bump prompt to bash-shadow-v2 with explicit-user-intent authority rules and quoted untrusted intent evidence - capture the requesting cwd and per-ask conversation state; flip evidence-quality flags from placeholders to measured values - record the candidate-identity change for prior cohorts in the scenario-set doc
This commit is contained in:
@@ -0,0 +1,111 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
buildConversationEvidence,
|
||||
type ConversationProbe,
|
||||
} from "../src/conversation";
|
||||
|
||||
function entry(text: string): unknown {
|
||||
return {
|
||||
type: "message",
|
||||
message: { role: "user", content: [{ type: "text", text }] },
|
||||
};
|
||||
}
|
||||
function assistantEntry(text: string): unknown {
|
||||
return {
|
||||
type: "message",
|
||||
message: { role: "assistant", content: [{ type: "text", text }] },
|
||||
};
|
||||
}
|
||||
function compactionEntry(): unknown {
|
||||
return { type: "compaction", summary: "derived" };
|
||||
}
|
||||
|
||||
describe("buildConversationEvidence — whitelist", () => {
|
||||
it("keeps only user text, excludes assistant and non-message entries", () => {
|
||||
const probe: ConversationProbe = {
|
||||
getActiveEntries: () => [
|
||||
entry("first user"),
|
||||
assistantEntry("assistant reasoning"),
|
||||
{ type: "label", name: "x" },
|
||||
entry("second user"),
|
||||
],
|
||||
};
|
||||
const evidence = buildConversationEvidence(probe);
|
||||
expect(evidence.items).toEqual([
|
||||
{ position: 1, role: "user", text: "first user" },
|
||||
{ position: 2, role: "user", text: "second user" },
|
||||
]);
|
||||
expect(evidence.hasCompaction).toBe(false);
|
||||
expect(evidence.truncated).toBe(false);
|
||||
expect(evidence.renderedChars).toBe("first user".length + "second user".length);
|
||||
});
|
||||
|
||||
it("accepts string message content", () => {
|
||||
const probe: ConversationProbe = {
|
||||
getActiveEntries: () => [
|
||||
{
|
||||
type: "message",
|
||||
message: { role: "user", content: "plain string" },
|
||||
},
|
||||
],
|
||||
};
|
||||
expect(buildConversationEvidence(probe).items[0]?.text).toBe("plain string");
|
||||
});
|
||||
|
||||
it("flags compaction presence without leaking summary text", () => {
|
||||
const probe: ConversationProbe = {
|
||||
getActiveEntries: () => [compactionEntry(), entry("after")],
|
||||
};
|
||||
const evidence = buildConversationEvidence(probe);
|
||||
expect(evidence.hasCompaction).toBe(true);
|
||||
expect(evidence.items).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("buildConversationEvidence — bounds", () => {
|
||||
it("keeps at most 16 items, newest window, and marks truncation", () => {
|
||||
const entries = Array.from({ length: 20 }, (_, i) => entry(`user-${i}`));
|
||||
const evidence = buildConversationEvidence({ getActiveEntries: () => entries });
|
||||
expect(evidence.items).toHaveLength(16);
|
||||
expect(evidence.truncated).toBe(true);
|
||||
expect(evidence.items[0]?.text).toBe("user-4");
|
||||
expect(evidence.items[15]?.text).toBe("user-19");
|
||||
expect(evidence.items[15]?.position).toBe(16);
|
||||
});
|
||||
|
||||
it("preserves the latest user even when it alone exceeds the char budget", () => {
|
||||
const huge = "x".repeat(15_000);
|
||||
const evidence = buildConversationEvidence({
|
||||
getActiveEntries: () => [entry("small"), entry(huge)],
|
||||
});
|
||||
expect(evidence.items).toHaveLength(1);
|
||||
expect(evidence.items[0]?.text).toBe(huge);
|
||||
expect(evidence.truncated).toBe(true);
|
||||
});
|
||||
|
||||
it("drops the oldest head when the character budget is exceeded", () => {
|
||||
const evidence = buildConversationEvidence({
|
||||
getActiveEntries: () => [
|
||||
entry("a".repeat(7_000)),
|
||||
entry("b".repeat(7_000)), // both would exceed 12,000
|
||||
entry("tail"),
|
||||
],
|
||||
});
|
||||
expect(evidence.items.map((i) => i.text)).toEqual([
|
||||
"b".repeat(7_000),
|
||||
"tail",
|
||||
]);
|
||||
expect(evidence.truncated).toBe(true);
|
||||
expect(evidence.renderedChars).toBeLessThanOrEqual(12_001);
|
||||
});
|
||||
|
||||
it("returns empty evidence for an empty branch", () => {
|
||||
const evidence = buildConversationEvidence({ getActiveEntries: () => [] });
|
||||
expect(evidence).toEqual({
|
||||
items: [],
|
||||
hasCompaction: false,
|
||||
truncated: false,
|
||||
renderedChars: 0,
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -136,6 +136,20 @@ function modelResponse(): AssistantMessage {
|
||||
};
|
||||
}
|
||||
|
||||
function fakeSessionManager(): {
|
||||
getSessionId: () => string;
|
||||
getEntries: () => unknown[];
|
||||
getLeafId: () => string | null;
|
||||
getCwd: () => string;
|
||||
} {
|
||||
return {
|
||||
getSessionId: () => "session-root",
|
||||
getEntries: () => [],
|
||||
getLeafId: () => null,
|
||||
getCwd: () => "/repo",
|
||||
};
|
||||
}
|
||||
|
||||
let publishedService: PermissionsService | undefined;
|
||||
afterEach(() => {
|
||||
if (publishedService !== undefined) {
|
||||
@@ -171,7 +185,7 @@ describe("AI judge lifecycle", () => {
|
||||
});
|
||||
const ctx = {
|
||||
hasUI: true,
|
||||
sessionManager: { getSessionId: () => "session-root" },
|
||||
sessionManager: fakeSessionManager(),
|
||||
get model() {
|
||||
return currentModel;
|
||||
},
|
||||
@@ -200,6 +214,76 @@ describe("AI judge lifecycle", () => {
|
||||
harness.shutdown();
|
||||
});
|
||||
|
||||
it("feeds conversation user text into the model prompt as quoted evidence", async () => {
|
||||
let authorize: Authorizer["authorize"] | undefined;
|
||||
const service = {
|
||||
registerAuthorizer: vi.fn((_name, callback) => {
|
||||
authorize = callback;
|
||||
return vi.fn();
|
||||
}),
|
||||
checkPermission: vi.fn(),
|
||||
getToolPermission: vi.fn(),
|
||||
} as unknown as PermissionsService;
|
||||
publishPermissionsService(service);
|
||||
publishedService = service;
|
||||
|
||||
const complete = vi.fn(
|
||||
async (
|
||||
_model: Model<any>,
|
||||
_context: Context,
|
||||
): Promise<AssistantMessage> => modelResponse(),
|
||||
);
|
||||
const sessionManager = fakeSessionManager();
|
||||
(sessionManager as { getEntries: () => unknown[] }).getEntries = () => [
|
||||
{
|
||||
type: "message",
|
||||
message: {
|
||||
role: "user",
|
||||
content: [{ type: "text", text: "please push the release tag" }],
|
||||
},
|
||||
},
|
||||
];
|
||||
const ctx = {
|
||||
hasUI: true,
|
||||
sessionManager,
|
||||
model: {
|
||||
id: "test-model",
|
||||
provider: "test-provider",
|
||||
api: "openai-codex-responses",
|
||||
} as Model<any>,
|
||||
modelRegistry: { complete },
|
||||
ui: { notify: vi.fn() },
|
||||
} as unknown as ExtensionContext;
|
||||
|
||||
const harness = createFakePi();
|
||||
extension(harness.pi);
|
||||
harness.start(ctx);
|
||||
harness.ready();
|
||||
|
||||
const log = {
|
||||
review: vi.fn(),
|
||||
debug: vi.fn(),
|
||||
};
|
||||
await authorize!(ask(), { checkPermission: vi.fn(), getToolPermission: vi.fn() }, log);
|
||||
|
||||
expect(complete).toHaveBeenCalledTimes(1);
|
||||
const promptText = JSON.stringify(complete.mock.calls[0]?.[1]);
|
||||
expect(promptText).toContain("please push the release tag");
|
||||
expect(promptText).toContain("user_intent_evidence");
|
||||
// The review row carries quality flags, not conversation content.
|
||||
expect(JSON.stringify(log.review.mock.calls)).not.toContain(
|
||||
"please push the release tag",
|
||||
);
|
||||
expect(log.review.mock.calls[0]?.[1]).toMatchObject({
|
||||
evidenceQuality: expect.objectContaining({
|
||||
explicitUserText: true,
|
||||
conversationItems: 1,
|
||||
requesterCwd: "/repo",
|
||||
}),
|
||||
});
|
||||
harness.shutdown();
|
||||
});
|
||||
|
||||
it("calls the current model once, records metadata, and still defers in Shadow", async () => {
|
||||
let authorize: Authorizer["authorize"] | undefined;
|
||||
const dispose = vi.fn();
|
||||
@@ -221,9 +305,7 @@ describe("AI judge lifecycle", () => {
|
||||
_options?: Record<string, unknown>,
|
||||
) => modelResponse(),
|
||||
);
|
||||
const sessionManager = {
|
||||
getSessionId: () => "session-root",
|
||||
};
|
||||
const sessionManager = fakeSessionManager();
|
||||
const model = {
|
||||
id: "test-model",
|
||||
provider: "test-provider",
|
||||
@@ -265,7 +347,7 @@ describe("AI judge lifecycle", () => {
|
||||
maxRetries: 0,
|
||||
toolChoice: "required",
|
||||
});
|
||||
expect(reviews).toEqual([
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
details: expect.objectContaining({
|
||||
@@ -273,7 +355,7 @@ describe("AI judge lifecycle", () => {
|
||||
mode: "shadow",
|
||||
origin: "local",
|
||||
judgeRuntimeId: expect.any(String),
|
||||
promptVersion: "bash-shadow-v1",
|
||||
promptVersion: "bash-shadow-v2",
|
||||
toolSchemaVersion: "report-verdict-v1",
|
||||
judgeLatencyMs: expect.any(Number),
|
||||
modelLatencyMs: expect.any(Number),
|
||||
@@ -313,7 +395,7 @@ describe("AI judge lifecycle", () => {
|
||||
const complete = vi.fn();
|
||||
const ctx = {
|
||||
hasUI: true,
|
||||
sessionManager: { getSessionId: () => "session-root" },
|
||||
sessionManager: fakeSessionManager(),
|
||||
model: {
|
||||
id: "test-model",
|
||||
provider: "test-provider",
|
||||
@@ -350,7 +432,7 @@ describe("AI judge lifecycle", () => {
|
||||
|
||||
expect(verdict).toEqual({ kind: "defer" });
|
||||
expect(complete).not.toHaveBeenCalled();
|
||||
expect(reviews).toEqual([
|
||||
expect(reviews).toMatchObject([
|
||||
{
|
||||
event: "ai_bash_judge.result",
|
||||
details: expect.objectContaining({
|
||||
|
||||
Reference in New Issue
Block a user