Skip to content

Commit a9062be

Browse files
spike(guardian): replace Luna with Jev
Co-Authored-By: David Cramer <david@sentry.io>
1 parent 0b0899a commit a9062be

9 files changed

Lines changed: 152 additions & 9 deletions

File tree

‎packages/junior-evals/src/guardian-harness.ts‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -41,7 +41,7 @@ function resolveGuardianModelId(): string {
4141
if (configured) {
4242
return configured;
4343
}
44-
return "openai/gpt-5.6-luna";
44+
return "typesafe-ai/jev";
4545
}
4646

4747
/** Run one Guardian proposal through the production reviewer boundary. */

‎packages/junior-evals/vitest.evals.guardian.config.ts‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -29,7 +29,7 @@ process.env.JUNIOR_STATE_ADAPTER = "redis";
2929
process.env.JUNIOR_STATE_KEY_PREFIX ??= `junior:eval-guardian:${randomUUID()}`;
3030
process.env.REDIS_URL =
3131
process.env.JUNIOR_EVAL_REDIS_URL?.trim() || "redis://127.0.0.1:6382";
32-
process.env.AI_GUARDIAN_MODEL ??= "openai/gpt-5.6-luna";
32+
process.env.AI_GUARDIAN_MODEL ??= "typesafe-ai/jev";
3333

3434
export default defineConfig({
3535
resolve: {

‎packages/junior/package.json‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -70,6 +70,7 @@
7070
"dependencies": {
7171
"@agentclientprotocol/sdk": "1.3.0",
7272
"@ai-sdk/gateway": "^3.0.119",
73+
"@ai-sdk/gateway-v4": "npm:@ai-sdk/gateway@4.0.85",
7374
"@chat-adapter/slack": "4.29.0",
7475
"@chat-adapter/state-memory": "4.29.0",
7576
"@chat-adapter/state-redis": "4.29.0",
@@ -91,6 +92,7 @@
9192
"@vercel/queue": "^0.2.0",
9293
"@vercel/sandbox": "2.8.0",
9394
"ai": "^6.0.190",
95+
"ai-v7": "npm:ai@7.0.105",
9496
"chat": "4.29.0",
9597
"commander": "^14.0.3",
9698
"drizzle-orm": "catalog:",

‎packages/junior/src/chat/config.ts‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -242,10 +242,7 @@ const DEFAULT_FAST_MODEL_ID = getModel(
242242
"vercel-ai-gateway",
243243
"openai/gpt-5.6-luna",
244244
).id;
245-
const DEFAULT_GUARDIAN_MODEL_ID = getModel(
246-
"vercel-ai-gateway",
247-
"openai/gpt-5.6-luna",
248-
).id;
245+
const DEFAULT_GUARDIAN_MODEL_ID = "typesafe-ai/jev";
249246
const DEFAULT_HANDOFF_MODEL_ID = getModel(
250247
"vercel-ai-gateway",
251248
"openai/gpt-5.6-sol",
Lines changed: 27 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
1+
import { createGatewayProvider } from "@ai-sdk/gateway-v4";
2+
import {
3+
experimental_evaluate as evaluate,
4+
type Experimental_EvaluationQuestion as EvaluationQuestion,
5+
} from "ai-v7";
6+
import { resolveGatewayCredential } from "@/chat/pi/gateway-auth";
7+
8+
/** Evaluate typed questions with an AI Gateway evaluation model. */
9+
export async function evaluateQuestions<
10+
const Questions extends Record<string, EvaluationQuestion>,
11+
>(args: {
12+
modelId: string;
13+
state: string;
14+
questions: Questions;
15+
signal?: AbortSignal;
16+
}) {
17+
const credential = await resolveGatewayCredential();
18+
const gateway = createGatewayProvider(
19+
credential ? { apiKey: credential.token } : {},
20+
);
21+
return evaluate({
22+
model: gateway.evaluationModel(args.modelId),
23+
state: args.state,
24+
questions: args.questions,
25+
abortSignal: args.signal,
26+
});
27+
}

‎packages/junior/src/chat/services/guardian-action-review.ts‎

Lines changed: 53 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,7 @@
11
import { z } from "zod";
22
import { logWarn } from "@/chat/logging";
33
import type { completeObject } from "@/chat/pi/client";
4+
import { evaluateQuestions } from "@/chat/pi/evaluate";
45
import { GUARDIAN_ACTION_POLICY } from "@/chat/services/guardian-action-policy";
56
import { ProviderError } from "@/chat/services/provider-error";
67
import type {
@@ -20,6 +21,7 @@ const guardianDecisionSchema = z
2021
type CompleteObject = typeof completeObject;
2122
const GUARDIAN_REVIEW_TIMEOUT_MS = 60_000;
2223
const GUARDIAN_REVIEW_MAX_TOKENS = 2_000;
24+
const JEV_MODEL_ID = "typesafe-ai/jev";
2325
const MAX_PROPOSAL_CHARS = 192_000;
2426

2527
/** Serialize one bounded proposal while keeping its contents untrusted. */
@@ -51,6 +53,54 @@ export function createGuardianActionReviewer(options: {
5153
const signal = reviewOptions?.signal
5254
? AbortSignal.any([reviewOptions.signal, timeoutSignal])
5355
: timeoutSignal;
56+
if (options.modelId === JEV_MODEL_ID) {
57+
const result = await evaluateQuestions({
58+
modelId: options.modelId,
59+
state: [GUARDIAN_ACTION_POLICY, guardianPrompt(proposal)].join(
60+
"\n\n",
61+
),
62+
questions: {
63+
decision: {
64+
type: "choice",
65+
instructions: "Choose the required Guardian decision.",
66+
criteria: {
67+
allow: "The policy allows the action without confirmation.",
68+
ask: "The policy requires explicit user confirmation.",
69+
deny: "The policy prohibits the action.",
70+
},
71+
},
72+
riskLevel: {
73+
type: "choice",
74+
instructions: "Choose the risk level of the planned action.",
75+
criteria: {
76+
low: "Low impact and easy to reverse.",
77+
medium: "Meaningful but limited impact.",
78+
high: "Substantial impact or hard to reverse.",
79+
critical: "Severe, broad, or irreversible impact.",
80+
},
81+
},
82+
userAuthorization: {
83+
type: "choice",
84+
instructions:
85+
"Choose how clearly the user authorized the planned action.",
86+
criteria: {
87+
high: "The user explicitly authorized this exact action.",
88+
medium: "The user intent supports the action but is not exact.",
89+
low: "The action is only weakly implied.",
90+
unknown: "The proposal shows no user authorization.",
91+
},
92+
},
93+
},
94+
signal,
95+
});
96+
return guardianDecisionSchema.parse({
97+
decision: result.answers.decision.choice,
98+
reason: "jev_evaluation",
99+
riskLevel: result.answers.riskLevel.choice,
100+
userAuthorization: result.answers.userAuthorization.choice,
101+
});
102+
}
103+
54104
const completeReview = () =>
55105
options.completeObject({
56106
modelId: options.modelId,
@@ -87,7 +137,9 @@ export function createGuardianActionReviewer(options: {
87137
}
88138
return {
89139
...guardianDecisionSchema.parse(result.object),
90-
...(result.costUsd !== undefined ? { costUsd: result.costUsd } : undefined),
140+
...(result.costUsd !== undefined
141+
? { costUsd: result.costUsd }
142+
: undefined),
91143
};
92144
},
93145
};

‎packages/junior/tests/component/config/chat-config.test.ts‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -76,13 +76,13 @@ describe("chat config", () => {
7676
expect(botConfig.fastModelId).toBe("openai/gpt-5.6-luna");
7777
});
7878

79-
it("uses Luna for Guardian when no override is configured", async () => {
79+
it("uses Jev for Guardian when no override is configured", async () => {
8080
process.env.AI_MODEL = "anthropic/claude-opus-4.6";
8181
process.env.AI_FAST_MODEL = "anthropic/claude-haiku-4.5";
8282
delete process.env.AI_GUARDIAN_MODEL;
8383

8484
const { botConfig } = await loadConfig();
85-
expect(botConfig.guardianModelId).toBe("openai/gpt-5.6-luna");
85+
expect(botConfig.guardianModelId).toBe("typesafe-ai/jev");
8686
});
8787

8888
it("uses the configured Guardian model override", async () => {

‎pnpm-lock.yaml‎

Lines changed: 60 additions & 0 deletions
Some generated files are not rendered by default. Learn more about customizing how changed files appear on GitHub.

‎pnpm-workspace.yaml‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -34,6 +34,11 @@ syncInjectedDepsAfterScripts:
3434
- build
3535
minimumReleaseAge: 1440
3636
minimumReleaseAgeExclude:
37+
# Temporary exception for the Jev evaluation model spike.
38+
- "ai@7.0.105"
39+
- "@ai-sdk/gateway@4.0.85"
40+
- "@ai-sdk/provider@4.0.17"
41+
- "@ai-sdk/provider-utils@5.0.43"
3742
- "@sentry/starlight-theme"
3843
- "@vitest-evals/core@0.16.1"
3944
- "@vitest-evals/report-ui@0.16.1"

0 commit comments

Comments
 (0)