mirror of
https://github.com/supabase/supabase.git
synced 2026-10-06 01:45:10 +03:00
Lays groundwork for online evals on Assistant chat logs. https://www.braintrust.dev/docs/observe/score-online ### Changes - New workflows: - `braintrust-scorers-deploy.yml` keeps prod scorers in sync on push to `master` - `braintrust-preview-scorers-deploy.yml` deploys preview scorers to the staging project for PRs labeled `preview-scorers`, posting a comment with scorer links ([example](https://github.com/supabase/supabase/pull/43194#issuecomment-4000097222)) - `braintrust-preview-scorers-cleanup.yml` deletes preview scorers when the PR is closed ([example](https://github.com/supabase/supabase/pull/43194#issuecomment-4000749847)) - Adds `evals/scorer-online.ts` entry point invoked with `pnpm scorers:deploy`, registering scorers for online evals in the Braintrust "Assistant" project - Refactors scorer code to separate online-compatible scorers (`scorer-online.ts`) from WASM-dependent ones (`scorer-wasm.ts`) - "URL Validity" scorer now only checks Supabase domains to prevent requests to untrusted origins - Span `input` is now shaped `{ prompt: string }` instead of plain `string` for compatibility with offline eval scorers - Env vars `BRAINTRUST_STAGING_PROJECT_ID` and `BRAINTRUST_PROJECT_ID` configured in GitHub repo settings - `generateAssistantResponse` now uses `startSpan` + `withCurrent` instead of `traced()` to manually manage the root span lifecycle — this ensures `onFinish` logs output to the span _before_ `span.end()` is called, which is when Braintrust triggers scoring automations ### Online Scorers We share scoring logic across offline and online evals, but some of our scorers aren't transferrable to an "online" setting due to runtime challenges or ground truth requirements. **Supported** - Goal Completion - Conciseness - Completeness - Docs Faithfulness - URL Validity **Unsupported** - Correctness (requires ground truth output) - Tool Usage (requires ground truth requiredTools) - SQL Syntax (uses libpg-query WASM) - SQL Identifier Quoting (uses libpg-query WASM) ### How to use these scorers Going forward if you want to add/edit online eval scorers, add the `preview-scorers` label to a PR. This deploys scorers to the [Assistant (Staging Scorers)](https://www.braintrust.dev/app/supabase.io/p/Assistant%20(Staging%20Scorers)?v=Overview) project in Braintrust with branch-specific slugs, and comments on the PR ([example](https://github.com/supabase/supabase/pull/43194#issuecomment-4000097222)). From the Braintrust dashboard you can "Test" the scorer with traces from any project. <img width="1866" height="528" alt="CleanShot 2026-03-05 at 15 15 00@2x" src="https://github.com/user-attachments/assets/4f15cebc-3f2d-4e8a-9ee2-fe8ef7bf4199" /> Once merged, scorers are deployed to the primary [Assistant](https://www.braintrust.dev/app/supabase.io/p/Assistant) project, and preview scorers are deleted from the staging project. Down the road, scorers on the Assistant project will run automatically on a sample of production traces. Closes AI-437
67 lines
1.9 KiB
TypeScript
67 lines
1.9 KiB
TypeScript
import { type ToolSet, type TypedToolCall, type TypedToolResult } from 'ai'
|
|
import { type AssistantEvalOutput } from './scorer'
|
|
|
|
type Step = {
|
|
text: string
|
|
toolCalls: TypedToolCall<ToolSet>[]
|
|
toolResults: TypedToolResult<ToolSet>[]
|
|
}
|
|
|
|
type ParsedToolCall = {
|
|
/** Query generated by `execute_sql` */
|
|
sqlQuery?: string
|
|
/** Docs text pulled in from `search_docs` */
|
|
docs?: string[]
|
|
}
|
|
|
|
function parseToolCall(
|
|
toolCall: TypedToolCall<ToolSet>,
|
|
toolResult: TypedToolResult<ToolSet>
|
|
): ParsedToolCall {
|
|
switch (toolCall.toolName) {
|
|
case 'execute_sql': {
|
|
const sqlQuery = toolCall.input?.sql
|
|
if (typeof sqlQuery !== 'string') return {}
|
|
return { sqlQuery }
|
|
}
|
|
case 'search_docs': {
|
|
const content = toolResult.output?.content
|
|
if (!content || !Array.isArray(content)) return {}
|
|
const docs = content.map((item) => item?.text).filter((text) => typeof text === 'string')
|
|
if (docs.length === 0) return {}
|
|
return { docs }
|
|
}
|
|
}
|
|
return {}
|
|
}
|
|
|
|
export function buildAssistantEvalOutput(
|
|
finishReason: AssistantEvalOutput['finishReason'],
|
|
steps: Step[]
|
|
): AssistantEvalOutput {
|
|
const simplifiedSteps = steps.map((step) => ({
|
|
text: step.text,
|
|
toolCalls: step.toolCalls.map((call) => ({
|
|
toolName: call.toolName,
|
|
input: call.input,
|
|
})),
|
|
}))
|
|
|
|
const toolNames: string[] = []
|
|
const sqlQueries: string[] = []
|
|
const docs: string[] = []
|
|
|
|
for (const step of steps) {
|
|
for (const [i, toolCall] of step.toolCalls.entries()) {
|
|
toolNames.push(toolCall.toolName)
|
|
const toolResult = step.toolResults.at(i)
|
|
if (!toolResult) continue
|
|
const parsed = parseToolCall(toolCall, toolResult)
|
|
if (parsed.sqlQuery) sqlQueries.push(parsed.sqlQuery)
|
|
if (parsed.docs) docs.push(...parsed.docs)
|
|
}
|
|
}
|
|
|
|
return { finishReason, steps: simplifiedSteps, toolNames, sqlQueries, docs }
|
|
}
|