mirror of
https://github.com/supabase/supabase.git
synced 2026-10-11 20:35:07 +03:00
Lays groundwork for online evals on Assistant chat logs. https://www.braintrust.dev/docs/observe/score-online ### Changes - New workflows: - `braintrust-scorers-deploy.yml` keeps prod scorers in sync on push to `master` - `braintrust-preview-scorers-deploy.yml` deploys preview scorers to the staging project for PRs labeled `preview-scorers`, posting a comment with scorer links ([example](https://github.com/supabase/supabase/pull/43194#issuecomment-4000097222)) - `braintrust-preview-scorers-cleanup.yml` deletes preview scorers when the PR is closed ([example](https://github.com/supabase/supabase/pull/43194#issuecomment-4000749847)) - Adds `evals/scorer-online.ts` entry point invoked with `pnpm scorers:deploy`, registering scorers for online evals in the Braintrust "Assistant" project - Refactors scorer code to separate online-compatible scorers (`scorer-online.ts`) from WASM-dependent ones (`scorer-wasm.ts`) - "URL Validity" scorer now only checks Supabase domains to prevent requests to untrusted origins - Span `input` is now shaped `{ prompt: string }` instead of plain `string` for compatibility with offline eval scorers - Env vars `BRAINTRUST_STAGING_PROJECT_ID` and `BRAINTRUST_PROJECT_ID` configured in GitHub repo settings - `generateAssistantResponse` now uses `startSpan` + `withCurrent` instead of `traced()` to manually manage the root span lifecycle — this ensures `onFinish` logs output to the span _before_ `span.end()` is called, which is when Braintrust triggers scoring automations ### Online Scorers We share scoring logic across offline and online evals, but some of our scorers aren't transferrable to an "online" setting due to runtime challenges or ground truth requirements. **Supported** - Goal Completion - Conciseness - Completeness - Docs Faithfulness - URL Validity **Unsupported** - Correctness (requires ground truth output) - Tool Usage (requires ground truth requiredTools) - SQL Syntax (uses libpg-query WASM) - SQL Identifier Quoting (uses libpg-query WASM) ### How to use these scorers Going forward if you want to add/edit online eval scorers, add the `preview-scorers` label to a PR. This deploys scorers to the [Assistant (Staging Scorers)](https://www.braintrust.dev/app/supabase.io/p/Assistant%20(Staging%20Scorers)?v=Overview) project in Braintrust with branch-specific slugs, and comments on the PR ([example](https://github.com/supabase/supabase/pull/43194#issuecomment-4000097222)). From the Braintrust dashboard you can "Test" the scorer with traces from any project. <img width="1866" height="528" alt="CleanShot 2026-03-05 at 15 15 00@2x" src="https://github.com/user-attachments/assets/4f15cebc-3f2d-4e8a-9ee2-fe8ef7bf4199" /> Once merged, scorers are deployed to the primary [Assistant](https://www.braintrust.dev/app/supabase.io/p/Assistant) project, and preview scorers are deleted from the staging project. Down the road, scorers on the Assistant project will run automatically on a sample of production traces. Closes AI-437
204 lines
6.1 KiB
TypeScript
204 lines
6.1 KiB
TypeScript
import * as ai from 'ai'
|
|
import {
|
|
convertToModelMessages,
|
|
isToolUIPart,
|
|
stepCountIs,
|
|
type LanguageModel,
|
|
type ModelMessage,
|
|
type ToolSet,
|
|
type UIMessage,
|
|
} from 'ai'
|
|
import { startSpan, traced, withCurrent, wrapAISDK, type Span } from 'braintrust'
|
|
import { source } from 'common-tags'
|
|
import { buildAssistantEvalOutput } from 'evals/output'
|
|
import type { AssistantEvalInput, AssistantEvalOutput } from 'evals/scorer'
|
|
import type { AiOptInLevel } from 'hooks/misc/useOrgOptedIntoAi'
|
|
import { IS_TRACING_ENABLED } from 'lib/ai/braintrust-logger'
|
|
import {
|
|
CHAT_PROMPT,
|
|
EDGE_FUNCTION_PROMPT,
|
|
GENERAL_PROMPT,
|
|
LIMITATIONS_PROMPT,
|
|
PG_BEST_PRACTICES,
|
|
REALTIME_PROMPT,
|
|
RLS_PROMPT,
|
|
SECURITY_PROMPT,
|
|
} from 'lib/ai/prompts'
|
|
import { sanitizeMessagePart } from 'lib/ai/tools/tool-sanitizer'
|
|
|
|
const { streamText: tracedStreamText } = wrapAISDK(ai)
|
|
|
|
export async function generateAssistantResponse({
|
|
messages: rawMessages,
|
|
model,
|
|
tools,
|
|
aiOptInLevel = 'schema',
|
|
getSchemas,
|
|
projectRef,
|
|
chatId,
|
|
chatName,
|
|
isHipaaEnabled,
|
|
userId,
|
|
orgId,
|
|
planId,
|
|
promptProviderOptions,
|
|
providerOptions,
|
|
requestedModel,
|
|
abortSignal,
|
|
onSpanCreated,
|
|
}: {
|
|
messages: UIMessage[]
|
|
model: LanguageModel
|
|
tools: ToolSet
|
|
aiOptInLevel?: AiOptInLevel
|
|
getSchemas?: () => Promise<string>
|
|
projectRef?: string
|
|
chatId?: string
|
|
chatName?: string
|
|
isHipaaEnabled?: boolean
|
|
userId?: string
|
|
orgId?: number
|
|
planId?: string
|
|
requestedModel?: string
|
|
promptProviderOptions?: Record<string, any>
|
|
providerOptions?: Record<string, any>
|
|
abortSignal?: AbortSignal
|
|
onSpanCreated?: (spanId: string) => void
|
|
}) {
|
|
const shouldTrace = IS_TRACING_ENABLED && !isHipaaEnabled
|
|
|
|
const run = async (span?: Span) => {
|
|
// Only returns last 7 messages
|
|
// Filters out tools with invalid states
|
|
// Filters out tool outputs based on opt-in level using renderingToolOutputParser
|
|
const messages = (rawMessages || []).slice(-7).map((msg) => {
|
|
if (msg && msg.role === 'assistant' && 'results' in msg) {
|
|
const cleanedMsg = { ...msg }
|
|
delete cleanedMsg.results
|
|
return cleanedMsg
|
|
}
|
|
if (msg && msg.role === 'assistant' && msg.parts) {
|
|
const cleanedParts = msg.parts
|
|
.filter((part) => {
|
|
if (isToolUIPart(part)) {
|
|
const invalidStates = ['input-streaming', 'input-available', 'output-error']
|
|
return !invalidStates.includes(part.state)
|
|
}
|
|
return true
|
|
})
|
|
.map((part) => {
|
|
return sanitizeMessagePart(part, aiOptInLevel)
|
|
})
|
|
return { ...msg, parts: cleanedParts }
|
|
}
|
|
return msg
|
|
})
|
|
|
|
const schemasString =
|
|
aiOptInLevel !== 'disabled' && getSchemas
|
|
? await traced(async () => getSchemas(), { name: 'getSchemas', type: 'function' })
|
|
: "You don't have access to any schemas."
|
|
|
|
// Important: do not use dynamic content in the system prompt or Bedrock will not cache it
|
|
const system = source`
|
|
${GENERAL_PROMPT}
|
|
${CHAT_PROMPT}
|
|
${PG_BEST_PRACTICES}
|
|
${RLS_PROMPT}
|
|
${EDGE_FUNCTION_PROMPT}
|
|
${REALTIME_PROMPT}
|
|
${SECURITY_PROMPT}
|
|
${LIMITATIONS_PROMPT}
|
|
`
|
|
|
|
// Note: these must be of type `CoreMessage` to prevent AI SDK from stripping `providerOptions`
|
|
// https://github.com/vercel/ai/blob/81ef2511311e8af34d75e37fc8204a82e775e8c3/packages/ai/core/prompt/standardize-prompt.ts#L83-L88
|
|
const hasProjectContext =
|
|
projectRef || chatName || schemasString !== "You don't have access to any schemas."
|
|
|
|
const assistantContent = hasProjectContext
|
|
? `The user's current project is ${projectRef || 'unknown'}. Their available schemas are: ${schemasString}. The current chat name is: ${chatName || 'unnamed'}.`
|
|
: undefined
|
|
|
|
const coreMessages: ModelMessage[] = [
|
|
{
|
|
role: 'system',
|
|
content: system,
|
|
...(promptProviderOptions && {
|
|
providerOptions: promptProviderOptions,
|
|
}),
|
|
},
|
|
...(assistantContent
|
|
? [
|
|
{
|
|
role: 'assistant' as const,
|
|
// Add any dynamic context here
|
|
content: assistantContent,
|
|
},
|
|
]
|
|
: []),
|
|
...convertToModelMessages(messages),
|
|
]
|
|
|
|
const streamTextFn = shouldTrace ? tracedStreamText : ai.streamText
|
|
|
|
return streamTextFn({
|
|
model,
|
|
stopWhen: stepCountIs(5),
|
|
messages: coreMessages,
|
|
...(providerOptions && { providerOptions }),
|
|
tools,
|
|
...(abortSignal && { abortSignal }),
|
|
...(span && {
|
|
onFinish: ({ steps, finishReason }) => {
|
|
for (const step of steps) {
|
|
for (const toolCall of step.toolCalls) {
|
|
if (toolCall.toolName === 'rename_chat') {
|
|
const { newName } = toolCall.input as { newName: string }
|
|
span.log({ metadata: { chatName: newName } })
|
|
}
|
|
}
|
|
}
|
|
span.log({
|
|
output: buildAssistantEvalOutput(finishReason, steps) satisfies AssistantEvalOutput,
|
|
})
|
|
span.end()
|
|
},
|
|
}),
|
|
} satisfies Parameters<typeof ai.streamText>[0])
|
|
}
|
|
|
|
if (shouldTrace) {
|
|
// startSpan instead of traced() so we control when the span closes — onFinish logs
|
|
// output to the span before we call span.end(), ensuring online scoring sees the output.
|
|
const span = startSpan({ name: 'generateAssistantResponse', type: 'function' })
|
|
onSpanCreated?.(span.id)
|
|
|
|
const lastUserMessage = rawMessages.findLast((m) => m.role === 'user')
|
|
const lastUserText = lastUserMessage?.parts
|
|
?.filter((p): p is { type: 'text'; text: string } => p.type === 'text')
|
|
.map((p) => p.text)
|
|
.join('\n')
|
|
|
|
span.log({
|
|
input: { prompt: lastUserText ?? '' } satisfies AssistantEvalInput,
|
|
metadata: {
|
|
projectRef,
|
|
chatId,
|
|
chatName,
|
|
aiOptInLevel,
|
|
userId,
|
|
orgId,
|
|
planId,
|
|
requestedModel,
|
|
gitBranch: process.env.VERCEL_GIT_COMMIT_REF,
|
|
environment: process.env.NEXT_PUBLIC_ENVIRONMENT,
|
|
},
|
|
})
|
|
|
|
return withCurrent(span, () => run(span))
|
|
}
|
|
|
|
return run()
|
|
}
|