Files
supabase/apps/studio/evals/assistant.eval.ts
Matt Rossman 4b8bab4d14 feat(assistant): score URL validity and fix support ticket URL guidance (#42227)
**Logic changes**
- Adds function in `helpers.ts` to extract URLs from text via regex
- I also considering using a library like
[linkify-it](https://www.npmjs.com/package/linkify-it) for this but
figured it's not worth the extra dep
- Adds associated tests in `helpers.test.ts`
- Adds "URL Validity" scorer which performs a HEAD request for links in
Assistant response text and determins what portion of links have `.ok`
responses
- Adds eval case to check correctness of support ticket URL answers

**Prompt changes**
- Informs Assistant of https://supabase.com/dashboard/support/new being
the URL to create support tickets
- Encourages Assistant to "self-debug" issues before directing users to
create support tickets

See [Eval
Report](https://github.com/supabase/supabase/pull/42227#issuecomment-3807772871)
and
[Correctness](https://www.braintrust.dev/app/supabase.io/p/Assistant/trace?object_type=experiment&object_id=1ad0f9b0-5adb-436c-9812-a87aac62c036&r=1ef13459-a98c-4904-925e-6d81276cebb2&s=dbe5c607-a560-462b-8745-41d430744431)
analysis for new support ticket test case.

Resolves AI-384

<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

* **New Features**
* Added URL validity scoring to evaluations and helper utilities for
extracting/cleaning URLs.
* Added evaluation cases for support-ticket URL handling and OAuth
callback guidance.

* **Documentation**
* Updated assistant guidance to prefer self-resolution, include
support-ticket direction, clarified data-recovery search steps, and
added template-URL notation.

* **Tests**
* Expanded URL extraction and related utility tests to cover many
formats and edge cases.

<sub>✏️ Tip: You can customize this high-level summary in your review
settings.</sub>
<!-- end of auto-generated comment: release notes by coderabbit.ai -->
2026-01-30 09:53:21 -05:00

132 lines
3.3 KiB
TypeScript

import { openai } from '@ai-sdk/openai'
import { Eval } from 'braintrust'
import { generateAssistantResponse } from 'lib/ai/generate-assistant-response'
import { getMockTools } from 'lib/ai/tools/mock-tools'
import assert from 'node:assert'
import { dataset } from './dataset'
import {
completenessScorer,
concisenessScorer,
correctnessScorer,
docsFaithfulnessScorer,
goalCompletionScorer,
sqlIdentifierQuotingScorer,
sqlSyntaxScorer,
toolUsageScorer,
urlValidityScorer,
} from './scorer'
import { ToolSet, TypedToolCall, TypedToolResult } from 'ai'
assert(process.env.BRAINTRUST_PROJECT_ID, 'BRAINTRUST_PROJECT_ID is not set')
assert(process.env.OPENAI_API_KEY, 'OPENAI_API_KEY is not set')
Eval('Assistant', {
projectId: process.env.BRAINTRUST_PROJECT_ID,
data: () => dataset,
task: async (input) => {
const result = await generateAssistantResponse({
model: openai('gpt-5-mini'),
messages: [{ id: '1', role: 'user', parts: [{ type: 'text', text: input.prompt }] }],
tools: await getMockTools(input.mockTables ? { list_tables: input.mockTables } : undefined),
})
const finishReason = await result.finishReason
// `result.toolCalls` only shows the last step, instead aggregate tools across all steps
const steps = await result.steps
const simplifiedSteps = steps.map((step) => ({
text: step.text,
toolCalls: step.toolCalls.map((call) => ({
toolName: call.toolName,
input: call.input,
})),
}))
const toolNames: string[] = []
const sqlQueries: string[] = []
const docs: string[] = []
for (const step of steps) {
for (const [i, toolCall] of step.toolCalls.entries()) {
toolNames.push(toolCall.toolName)
const toolResult = step.toolResults.at(i)
if (!toolResult) {
continue
}
const parsed = parseToolCall(toolCall, toolResult)
if (parsed.sqlQuery) {
sqlQueries.push(parsed.sqlQuery)
}
if (parsed.docs) {
docs.push(...parsed.docs)
}
}
}
return {
finishReason,
steps: simplifiedSteps,
toolNames,
sqlQueries,
docs,
}
},
scores: [
toolUsageScorer,
sqlSyntaxScorer,
sqlIdentifierQuotingScorer,
goalCompletionScorer,
concisenessScorer,
completenessScorer,
docsFaithfulnessScorer,
correctnessScorer,
urlValidityScorer,
],
})
type ParsedToolCall = {
/** Query generated by `execute_sql` */
sqlQuery?: string
/** Docs text pulled in from `search_docs` */
docs?: string[]
}
/**
* Validate and extract relevant info from a tool call/result
*/
function parseToolCall(
toolCall: TypedToolCall<ToolSet>,
toolResult: TypedToolResult<ToolSet>
): ParsedToolCall {
switch (toolCall.toolName) {
case 'execute_sql': {
const sqlQuery = toolCall.input.sql
if (typeof sqlQuery !== 'string') {
return {}
}
return { sqlQuery }
}
case 'search_docs': {
const content = toolResult.output.content
if (!content || !Array.isArray(content)) {
return {}
}
const docs = content.map((item) => item?.text).filter((text) => typeof text === 'string')
if (docs.length === 0) {
return {}
}
return { docs }
}
}
return {}
}