mirror of
https://github.com/supabase/supabase.git
synced 2026-10-09 11:25:06 +03:00
## Motivation When Assistant runs a potentially destructive tool like `execute_sql`, it stops the LLM request and prompts for client-side approval and execution of the tool. After approval, a second request kicks off under a separate trace. This has made scoring and [Topics](https://www.braintrust.dev/blog/topics) classification challenging, as the generated `output` is split across stateless requests. The [span-level scoring](https://www.braintrust.dev/docs/evaluate/custom-code#score-spans) approach we've used thusfar (after the LLM call, we massage the result into an `output` payload that's stuck onto the root span) has been cumbersome and led to invalid scores / topics where only part of the assistant response is considered. It's also inefficient, as we're duplicating potentially large info (like the `search_docs` output) that already exists within the trace. An alternative to scoring spans is to [score traces](https://www.braintrust.dev/docs/evaluate/custom-code#score-traces). Braintrust [best practices](https://www.braintrust.dev/docs/evaluate/score-online#best-practices) advise: > Use span scope for evaluating individual operations or outputs. Use trace scope for evaluating multi-turn conversations, overall workflow completion, or when your scorer needs access to the full execution context. We've also received [direct guidance](https://supabase.slack.com/archives/C05QYJBLX89/p1777925770927149?thread_ts=1777905716.911979&cid=C05QYJBLX89) from their team to use this approach. ## Changes Migrates eval scorers from custom `AssistantEvalOutput` shape to trace-level scoring via `trace.getThread()` / `trace.getSpans()`, with thread parsing that scores the full latest Assistant turn and passes prior conversation separately where relevant. Moves `execute_sql` and `deploy_edge_function` from client-side execution after approval to AI SDK `needsApproval` + server-side `execute()`. SQL results returned to the model are gated by AI opt-in level, so row data is only included with `schema_and_log_and_data`; otherwise the tool returns the no-data-permissions sentinel. Adds `metadata.isFinalStep` to disambiguate multiple LLM requests within an "assistant" turn due to tool call requests/responses. For online evals, this means we should configure automations to only score traces with `metadata.isFinalStep = true` to ensure we're judging the complete generated response. Other minor kaizen changes: - Renamed `promptProviderOptions` to `systemProviderOptions` to clarify that this is associated with the "system" message and disambiguate from the root `providerOptions` - Adds `evals/trace-utils.ts` to handle Zod validation of the `unknown` span shapes from Braintrust, to more easily access typed inputs/output on tool spans. - Bumps AI SDK floor version `^6.0.116` → `^6.0.174` - Tweaked the "Conciseness" scorer to not unfairly dock points for the new `[called tool_name]` labels in serialized assistant response ## Verification In the studio staging build, I asked Assistant to create a todos table with 3 sample todos. I manually approved the `execute_sql` call and saw Assistant generate text before & after the call. In Braintrust I verified two traces were produced (see [filtered logs](https://www.braintrust.dev/app/supabase.io/p/Assistant/logs?v=Staging&tvt=trace&search={%22filter%22:[{%22text%22:%22metadata.environment%2520%253D%2520%27staging%27%22,%22label%22:%22metadata.environment%2520%253D%2520%27staging%27%22,%22originType%22:%22btql%22},{%22text%22:%22%2560Chat%2520ID%2560%2520%253D%2520%25221cb2ac45-e5e7-458c-9da4-3bf6863b8842%2522%22,%22label%22:%22Chat%2520ID%2520equals%25201cb2ac45-e5e7-458c-9da4-3bf6863b8842%22,%22originType%22:%22form%22}]})), the first with `metadata.isFinalStep = false` and the second with `metadata.isFinalStep = true`. In the Braintrust staging scorers, I ran the preview Completeness scorer on the second trace and verified it sees the complete Assistant response including markers for tool calls ([link to trace](https://www.braintrust.dev/app/supabase.io/p/Assistant%20(Staging%20Scorers)/trace?object_type=project_logs&object_id=b5214b62-ad1e-4929-9d5b-40b1daebe948&r=0ed0a4f8-8aff-4a34-bb1d-1df1d88a5070&s=ff9015f8-6bf7-4ab3-83a9-ca4e69e27e82)) <img width="1193" height="960" alt="CleanShot 2026-05-07 at 11 27 10@2x" src="https://github.com/user-attachments/assets/509d4858-c3a1-4068-986d-3aa4d5617d1a" /> I also tested the `deploy_edge_function` workflow and verified it still prompts for permission and warns on deployment of existing functions. **References** - https://www.braintrust.dev/docs/evaluate/custom-code#score-traces - https://ai-sdk.dev/docs/ai-sdk-core/tools-and-tool-calling#tool-execution-approval Supercedes https://github.com/supabase/supabase/pull/45556 and https://github.com/supabase/supabase/pull/45339 Closes AI-473 <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Tool actions (SQL execution, edge-function deploy) now require explicit user Approve/Deny before proceeding. * **Improvements** * Assistant pauses for approval responses before sending follow-ups, giving clearer control over risky actions. * Deploy/replace flows show confirmation and clearer replace warnings. * Evaluation/scoring updated to use richer trace data for more accurate assistant performance signals. <!-- end of auto-generated comment: release notes by coderabbit.ai -->
168 lines
5.3 KiB
TypeScript
168 lines
5.3 KiB
TypeScript
export type ProviderName = 'bedrock' | 'openai'
|
|
|
|
export type BedrockModel = 'anthropic.claude-3-7-sonnet-20250219-v1:0' | 'openai.gpt-oss-120b-1:0'
|
|
|
|
export type OpenAIModelId = 'gpt-5.4-nano' | 'gpt-5.3-codex'
|
|
|
|
// Source: https://developers.openai.com/api/docs/guides/reasoning + per-model pages
|
|
export type ReasoningEffort = 'none' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh'
|
|
|
|
// Per-model reasoning effort compatibility.
|
|
// Sources: https://developers.openai.com/api/docs/models/gpt-5.4-nano
|
|
// https://developers.openai.com/api/docs/models/gpt-5.3-codex
|
|
type ModelReasoningSupport = {
|
|
'gpt-5.4-nano': 'none' | 'low' | 'medium' | 'high' | 'xhigh'
|
|
'gpt-5.3-codex': 'low' | 'medium' | 'high' | 'xhigh'
|
|
}
|
|
|
|
type ReasoningEffortFor<ModelId extends OpenAIModelId> = ModelId extends keyof ModelReasoningSupport
|
|
? ModelReasoningSupport[ModelId]
|
|
: never
|
|
|
|
/** Type-safe factory for configuring OpenAI models with compatible reasoning efforts. */
|
|
export function openaiModelEntry<
|
|
ModelId extends OpenAIModelId,
|
|
RequiresAdvance extends boolean = false,
|
|
>(config: {
|
|
id: ModelId
|
|
/** When true, the model requires the `assistant.advance_model` entitlement (paid plans). Defaults to false. */
|
|
requiresAdvanceModelEntitlement?: RequiresAdvance
|
|
/**
|
|
* When omitted, OpenAI applies its own default reasoning effort for the model,
|
|
* which may not be zero. Use an explicit level to control cost and latency.
|
|
*/
|
|
reasoningEffort?: ReasoningEffortFor<ModelId>
|
|
}): {
|
|
id: ModelId
|
|
requiresAdvanceModelEntitlement: RequiresAdvance
|
|
reasoningEffort?: ReasoningEffortFor<ModelId>
|
|
} {
|
|
return {
|
|
requiresAdvanceModelEntitlement: false as RequiresAdvance,
|
|
...config,
|
|
}
|
|
}
|
|
|
|
export type OpenAIModelEntry = ReturnType<typeof openaiModelEntry>
|
|
|
|
/** Default model entry for simple completion endpoints where latency is more important than reasoning. */
|
|
export const DEFAULT_COMPLETION_MODEL = openaiModelEntry({
|
|
id: 'gpt-5.4-nano',
|
|
reasoningEffort: 'none',
|
|
})
|
|
|
|
// Single source of truth for all Assistant chat model variants and their reasoning levels.
|
|
// Models with requiresAdvanceModelEntitlement false are available to all users; true requires the assistant.advance_model entitlement.
|
|
export const ASSISTANT_MODELS = [
|
|
openaiModelEntry({
|
|
id: 'gpt-5.4-nano',
|
|
requiresAdvanceModelEntitlement: false,
|
|
reasoningEffort: 'low',
|
|
}),
|
|
openaiModelEntry({
|
|
id: 'gpt-5.3-codex',
|
|
requiresAdvanceModelEntitlement: true,
|
|
reasoningEffort: 'low',
|
|
}),
|
|
] as const
|
|
|
|
export type AssistantBaseModelId = Extract<
|
|
(typeof ASSISTANT_MODELS)[number],
|
|
{ requiresAdvanceModelEntitlement: false }
|
|
>['id']
|
|
export type AssistantModelId = (typeof ASSISTANT_MODELS)[number]['id']
|
|
|
|
const ASSISTANT_MODELS_MAP = Object.fromEntries(ASSISTANT_MODELS.map((m) => [m.id, m])) as Record<
|
|
AssistantModelId,
|
|
(typeof ASSISTANT_MODELS)[number]
|
|
>
|
|
|
|
export const DEFAULT_ASSISTANT_BASE_MODEL_ID = 'gpt-5.4-nano' satisfies AssistantBaseModelId
|
|
|
|
export const DEFAULT_ASSISTANT_ADVANCE_MODEL_ID = 'gpt-5.3-codex' satisfies AssistantModelId
|
|
|
|
export function defaultAssistantModelId(hasAccessToAdvanceModel: boolean): AssistantModelId {
|
|
return hasAccessToAdvanceModel
|
|
? DEFAULT_ASSISTANT_ADVANCE_MODEL_ID
|
|
: DEFAULT_ASSISTANT_BASE_MODEL_ID
|
|
}
|
|
|
|
export function isKnownAssistantModelId(id: string): id is AssistantModelId {
|
|
return Object.hasOwn(ASSISTANT_MODELS_MAP, id)
|
|
}
|
|
|
|
export function isAssistantBaseModelId(id: string): id is AssistantBaseModelId {
|
|
return (
|
|
id in ASSISTANT_MODELS_MAP &&
|
|
!ASSISTANT_MODELS_MAP[id as AssistantModelId].requiresAdvanceModelEntitlement
|
|
)
|
|
}
|
|
|
|
export function isAdvanceOnlyModelId(id: string): boolean {
|
|
return (
|
|
id in ASSISTANT_MODELS_MAP &&
|
|
ASSISTANT_MODELS_MAP[id as AssistantModelId].requiresAdvanceModelEntitlement
|
|
)
|
|
}
|
|
|
|
export function getAssistantModelEntry(id: AssistantModelId): (typeof ASSISTANT_MODELS)[number] {
|
|
return ASSISTANT_MODELS_MAP[id]
|
|
}
|
|
|
|
export type Model = BedrockModel | OpenAIModelId
|
|
|
|
export type ProviderModelConfig = {
|
|
/** Optional providerOptions to attach to the system message for this model */
|
|
systemProviderOptions?: Record<string, any>
|
|
/** The default model for this provider (used when limited or no preferred specified) */
|
|
default: boolean
|
|
}
|
|
|
|
export type ProviderRegistry = {
|
|
bedrock: {
|
|
models: Record<BedrockModel, ProviderModelConfig>
|
|
providerOptions?: Record<string, any>
|
|
}
|
|
openai: {
|
|
models: Record<OpenAIModelId, ProviderModelConfig>
|
|
providerOptions?: Record<string, any>
|
|
}
|
|
}
|
|
|
|
export const PROVIDERS: ProviderRegistry = {
|
|
bedrock: {
|
|
models: {
|
|
'anthropic.claude-3-7-sonnet-20250219-v1:0': {
|
|
systemProviderOptions: {
|
|
bedrock: {
|
|
// Always cache the system prompt (must not contain dynamic content)
|
|
cachePoint: { type: 'default' },
|
|
},
|
|
},
|
|
default: false,
|
|
},
|
|
'openai.gpt-oss-120b-1:0': {
|
|
default: true,
|
|
},
|
|
},
|
|
},
|
|
openai: {
|
|
models: {
|
|
'gpt-5.3-codex': { default: false },
|
|
'gpt-5.4-nano': { default: true },
|
|
},
|
|
providerOptions: {
|
|
openai: {
|
|
store: false,
|
|
},
|
|
},
|
|
},
|
|
}
|
|
|
|
export function getDefaultModelForProvider(provider: ProviderName): Model | undefined {
|
|
const models = PROVIDERS[provider]?.models as Record<Model, ProviderModelConfig>
|
|
if (!models) return undefined
|
|
|
|
return Object.keys(models).find((id) => models[id as Model]?.default) as Model | undefined
|
|
}
|