mirror of
https://github.com/supabase/supabase.git
synced 2026-10-10 11:55:05 +03:00
**Logic changes** - Adds function in `helpers.ts` to extract URLs from text via regex - I also considering using a library like [linkify-it](https://www.npmjs.com/package/linkify-it) for this but figured it's not worth the extra dep - Adds associated tests in `helpers.test.ts` - Adds "URL Validity" scorer which performs a HEAD request for links in Assistant response text and determins what portion of links have `.ok` responses - Adds eval case to check correctness of support ticket URL answers **Prompt changes** - Informs Assistant of https://supabase.com/dashboard/support/new being the URL to create support tickets - Encourages Assistant to "self-debug" issues before directing users to create support tickets See [Eval Report](https://github.com/supabase/supabase/pull/42227#issuecomment-3807772871) and [Correctness](https://www.braintrust.dev/app/supabase.io/p/Assistant/trace?object_type=experiment&object_id=1ad0f9b0-5adb-436c-9812-a87aac62c036&r=1ef13459-a98c-4904-925e-6d81276cebb2&s=dbe5c607-a560-462b-8745-41d430744431) analysis for new support ticket test case. Resolves AI-384 <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Added URL validity scoring to evaluations and helper utilities for extracting/cleaning URLs. * Added evaluation cases for support-ticket URL handling and OAuth callback guidance. * **Documentation** * Updated assistant guidance to prefer self-resolution, include support-ticket direction, clarified data-recovery search steps, and added template-URL notation. * **Tests** * Expanded URL extraction and related utility tests to cover many formats and edge cases. <sub>✏️ Tip: You can customize this high-level summary in your review settings.</sub> <!-- end of auto-generated comment: release notes by coderabbit.ai -->
132 lines
3.3 KiB
TypeScript
132 lines
3.3 KiB
TypeScript
import { openai } from '@ai-sdk/openai'
|
|
import { Eval } from 'braintrust'
|
|
import { generateAssistantResponse } from 'lib/ai/generate-assistant-response'
|
|
import { getMockTools } from 'lib/ai/tools/mock-tools'
|
|
import assert from 'node:assert'
|
|
import { dataset } from './dataset'
|
|
import {
|
|
completenessScorer,
|
|
concisenessScorer,
|
|
correctnessScorer,
|
|
docsFaithfulnessScorer,
|
|
goalCompletionScorer,
|
|
sqlIdentifierQuotingScorer,
|
|
sqlSyntaxScorer,
|
|
toolUsageScorer,
|
|
urlValidityScorer,
|
|
} from './scorer'
|
|
import { ToolSet, TypedToolCall, TypedToolResult } from 'ai'
|
|
|
|
assert(process.env.BRAINTRUST_PROJECT_ID, 'BRAINTRUST_PROJECT_ID is not set')
|
|
assert(process.env.OPENAI_API_KEY, 'OPENAI_API_KEY is not set')
|
|
|
|
Eval('Assistant', {
|
|
projectId: process.env.BRAINTRUST_PROJECT_ID,
|
|
data: () => dataset,
|
|
task: async (input) => {
|
|
const result = await generateAssistantResponse({
|
|
model: openai('gpt-5-mini'),
|
|
messages: [{ id: '1', role: 'user', parts: [{ type: 'text', text: input.prompt }] }],
|
|
tools: await getMockTools(input.mockTables ? { list_tables: input.mockTables } : undefined),
|
|
})
|
|
|
|
const finishReason = await result.finishReason
|
|
|
|
// `result.toolCalls` only shows the last step, instead aggregate tools across all steps
|
|
const steps = await result.steps
|
|
|
|
const simplifiedSteps = steps.map((step) => ({
|
|
text: step.text,
|
|
toolCalls: step.toolCalls.map((call) => ({
|
|
toolName: call.toolName,
|
|
input: call.input,
|
|
})),
|
|
}))
|
|
|
|
const toolNames: string[] = []
|
|
const sqlQueries: string[] = []
|
|
const docs: string[] = []
|
|
|
|
for (const step of steps) {
|
|
for (const [i, toolCall] of step.toolCalls.entries()) {
|
|
toolNames.push(toolCall.toolName)
|
|
|
|
const toolResult = step.toolResults.at(i)
|
|
if (!toolResult) {
|
|
continue
|
|
}
|
|
|
|
const parsed = parseToolCall(toolCall, toolResult)
|
|
|
|
if (parsed.sqlQuery) {
|
|
sqlQueries.push(parsed.sqlQuery)
|
|
}
|
|
if (parsed.docs) {
|
|
docs.push(...parsed.docs)
|
|
}
|
|
}
|
|
}
|
|
|
|
return {
|
|
finishReason,
|
|
steps: simplifiedSteps,
|
|
toolNames,
|
|
sqlQueries,
|
|
docs,
|
|
}
|
|
},
|
|
scores: [
|
|
toolUsageScorer,
|
|
sqlSyntaxScorer,
|
|
sqlIdentifierQuotingScorer,
|
|
goalCompletionScorer,
|
|
concisenessScorer,
|
|
completenessScorer,
|
|
docsFaithfulnessScorer,
|
|
correctnessScorer,
|
|
urlValidityScorer,
|
|
],
|
|
})
|
|
|
|
type ParsedToolCall = {
|
|
/** Query generated by `execute_sql` */
|
|
sqlQuery?: string
|
|
|
|
/** Docs text pulled in from `search_docs` */
|
|
docs?: string[]
|
|
}
|
|
|
|
/**
|
|
* Validate and extract relevant info from a tool call/result
|
|
*/
|
|
function parseToolCall(
|
|
toolCall: TypedToolCall<ToolSet>,
|
|
toolResult: TypedToolResult<ToolSet>
|
|
): ParsedToolCall {
|
|
switch (toolCall.toolName) {
|
|
case 'execute_sql': {
|
|
const sqlQuery = toolCall.input.sql
|
|
if (typeof sqlQuery !== 'string') {
|
|
return {}
|
|
}
|
|
|
|
return { sqlQuery }
|
|
}
|
|
case 'search_docs': {
|
|
const content = toolResult.output.content
|
|
if (!content || !Array.isArray(content)) {
|
|
return {}
|
|
}
|
|
|
|
const docs = content.map((item) => item?.text).filter((text) => typeof text === 'string')
|
|
if (docs.length === 0) {
|
|
return {}
|
|
}
|
|
|
|
return { docs }
|
|
}
|
|
}
|
|
|
|
return {}
|
|
}
|