Files
supabase/apps/studio/lib/ai/generate-assistant-response.ts
Matt Rossman 517171b246 feat(assistant): online evals support and CI workflows (#43194)
Lays groundwork for online evals on Assistant chat logs.

https://www.braintrust.dev/docs/observe/score-online

### Changes

- New workflows:
- `braintrust-scorers-deploy.yml` keeps prod scorers in sync on push to
`master`
- `braintrust-preview-scorers-deploy.yml` deploys preview scorers to the
staging project for PRs labeled `preview-scorers`, posting a comment
with scorer links
([example](https://github.com/supabase/supabase/pull/43194#issuecomment-4000097222))
- `braintrust-preview-scorers-cleanup.yml` deletes preview scorers when
the PR is closed
([example](https://github.com/supabase/supabase/pull/43194#issuecomment-4000749847))
- Adds `evals/scorer-online.ts` entry point invoked with `pnpm
scorers:deploy`, registering scorers for online evals in the Braintrust
"Assistant" project
- Refactors scorer code to separate online-compatible scorers
(`scorer-online.ts`) from WASM-dependent ones (`scorer-wasm.ts`)
- "URL Validity" scorer now only checks Supabase domains to prevent
requests to untrusted origins
- Span `input` is now shaped `{ prompt: string }` instead of plain
`string` for compatibility with offline eval scorers
- Env vars `BRAINTRUST_STAGING_PROJECT_ID` and `BRAINTRUST_PROJECT_ID`
configured in GitHub repo settings
- `generateAssistantResponse` now uses `startSpan` + `withCurrent`
instead of `traced()` to manually manage the root span lifecycle — this
ensures `onFinish` logs output to the span _before_ `span.end()` is
called, which is when Braintrust triggers scoring automations

### Online Scorers

We share scoring logic across offline and online evals, but some of our
scorers aren't transferrable to an "online" setting due to runtime
challenges or ground truth requirements.

**Supported**
- Goal Completion
- Conciseness
- Completeness
- Docs Faithfulness
- URL Validity

**Unsupported**
- Correctness (requires ground truth output)
- Tool Usage (requires ground truth requiredTools)
- SQL Syntax (uses libpg-query WASM)
- SQL Identifier Quoting (uses libpg-query WASM)
 
### How to use these scorers

Going forward if you want to add/edit online eval scorers, add the
`preview-scorers` label to a PR. This deploys scorers to the [Assistant
(Staging
Scorers)](https://www.braintrust.dev/app/supabase.io/p/Assistant%20(Staging%20Scorers)?v=Overview)
project in Braintrust with branch-specific slugs, and comments on the PR
([example](https://github.com/supabase/supabase/pull/43194#issuecomment-4000097222)).
From the Braintrust dashboard you can "Test" the scorer with traces from
any project.

<img width="1866" height="528" alt="CleanShot 2026-03-05 at 15 15 00@2x"
src="https://github.com/user-attachments/assets/4f15cebc-3f2d-4e8a-9ee2-fe8ef7bf4199"
/>

Once merged, scorers are deployed to the primary
[Assistant](https://www.braintrust.dev/app/supabase.io/p/Assistant)
project, and preview scorers are deleted from the staging project. Down
the road, scorers on the Assistant project will run automatically on a
sample of production traces.

Closes AI-437
2026-03-09 13:05:26 -04:00

204 lines
6.1 KiB
TypeScript

import * as ai from 'ai'
import {
convertToModelMessages,
isToolUIPart,
stepCountIs,
type LanguageModel,
type ModelMessage,
type ToolSet,
type UIMessage,
} from 'ai'
import { startSpan, traced, withCurrent, wrapAISDK, type Span } from 'braintrust'
import { source } from 'common-tags'
import { buildAssistantEvalOutput } from 'evals/output'
import type { AssistantEvalInput, AssistantEvalOutput } from 'evals/scorer'
import type { AiOptInLevel } from 'hooks/misc/useOrgOptedIntoAi'
import { IS_TRACING_ENABLED } from 'lib/ai/braintrust-logger'
import {
CHAT_PROMPT,
EDGE_FUNCTION_PROMPT,
GENERAL_PROMPT,
LIMITATIONS_PROMPT,
PG_BEST_PRACTICES,
REALTIME_PROMPT,
RLS_PROMPT,
SECURITY_PROMPT,
} from 'lib/ai/prompts'
import { sanitizeMessagePart } from 'lib/ai/tools/tool-sanitizer'
const { streamText: tracedStreamText } = wrapAISDK(ai)
export async function generateAssistantResponse({
messages: rawMessages,
model,
tools,
aiOptInLevel = 'schema',
getSchemas,
projectRef,
chatId,
chatName,
isHipaaEnabled,
userId,
orgId,
planId,
promptProviderOptions,
providerOptions,
requestedModel,
abortSignal,
onSpanCreated,
}: {
messages: UIMessage[]
model: LanguageModel
tools: ToolSet
aiOptInLevel?: AiOptInLevel
getSchemas?: () => Promise<string>
projectRef?: string
chatId?: string
chatName?: string
isHipaaEnabled?: boolean
userId?: string
orgId?: number
planId?: string
requestedModel?: string
promptProviderOptions?: Record<string, any>
providerOptions?: Record<string, any>
abortSignal?: AbortSignal
onSpanCreated?: (spanId: string) => void
}) {
const shouldTrace = IS_TRACING_ENABLED && !isHipaaEnabled
const run = async (span?: Span) => {
// Only returns last 7 messages
// Filters out tools with invalid states
// Filters out tool outputs based on opt-in level using renderingToolOutputParser
const messages = (rawMessages || []).slice(-7).map((msg) => {
if (msg && msg.role === 'assistant' && 'results' in msg) {
const cleanedMsg = { ...msg }
delete cleanedMsg.results
return cleanedMsg
}
if (msg && msg.role === 'assistant' && msg.parts) {
const cleanedParts = msg.parts
.filter((part) => {
if (isToolUIPart(part)) {
const invalidStates = ['input-streaming', 'input-available', 'output-error']
return !invalidStates.includes(part.state)
}
return true
})
.map((part) => {
return sanitizeMessagePart(part, aiOptInLevel)
})
return { ...msg, parts: cleanedParts }
}
return msg
})
const schemasString =
aiOptInLevel !== 'disabled' && getSchemas
? await traced(async () => getSchemas(), { name: 'getSchemas', type: 'function' })
: "You don't have access to any schemas."
// Important: do not use dynamic content in the system prompt or Bedrock will not cache it
const system = source`
${GENERAL_PROMPT}
${CHAT_PROMPT}
${PG_BEST_PRACTICES}
${RLS_PROMPT}
${EDGE_FUNCTION_PROMPT}
${REALTIME_PROMPT}
${SECURITY_PROMPT}
${LIMITATIONS_PROMPT}
`
// Note: these must be of type `CoreMessage` to prevent AI SDK from stripping `providerOptions`
// https://github.com/vercel/ai/blob/81ef2511311e8af34d75e37fc8204a82e775e8c3/packages/ai/core/prompt/standardize-prompt.ts#L83-L88
const hasProjectContext =
projectRef || chatName || schemasString !== "You don't have access to any schemas."
const assistantContent = hasProjectContext
? `The user's current project is ${projectRef || 'unknown'}. Their available schemas are: ${schemasString}. The current chat name is: ${chatName || 'unnamed'}.`
: undefined
const coreMessages: ModelMessage[] = [
{
role: 'system',
content: system,
...(promptProviderOptions && {
providerOptions: promptProviderOptions,
}),
},
...(assistantContent
? [
{
role: 'assistant' as const,
// Add any dynamic context here
content: assistantContent,
},
]
: []),
...convertToModelMessages(messages),
]
const streamTextFn = shouldTrace ? tracedStreamText : ai.streamText
return streamTextFn({
model,
stopWhen: stepCountIs(5),
messages: coreMessages,
...(providerOptions && { providerOptions }),
tools,
...(abortSignal && { abortSignal }),
...(span && {
onFinish: ({ steps, finishReason }) => {
for (const step of steps) {
for (const toolCall of step.toolCalls) {
if (toolCall.toolName === 'rename_chat') {
const { newName } = toolCall.input as { newName: string }
span.log({ metadata: { chatName: newName } })
}
}
}
span.log({
output: buildAssistantEvalOutput(finishReason, steps) satisfies AssistantEvalOutput,
})
span.end()
},
}),
} satisfies Parameters<typeof ai.streamText>[0])
}
if (shouldTrace) {
// startSpan instead of traced() so we control when the span closes — onFinish logs
// output to the span before we call span.end(), ensuring online scoring sees the output.
const span = startSpan({ name: 'generateAssistantResponse', type: 'function' })
onSpanCreated?.(span.id)
const lastUserMessage = rawMessages.findLast((m) => m.role === 'user')
const lastUserText = lastUserMessage?.parts
?.filter((p): p is { type: 'text'; text: string } => p.type === 'text')
.map((p) => p.text)
.join('\n')
span.log({
input: { prompt: lastUserText ?? '' } satisfies AssistantEvalInput,
metadata: {
projectRef,
chatId,
chatName,
aiOptInLevel,
userId,
orgId,
planId,
requestedModel,
gitBranch: process.env.VERCEL_GIT_COMMIT_REF,
environment: process.env.NEXT_PUBLIC_ENVIRONMENT,
},
})
return withCurrent(span, () => run(span))
}
return run()
}