mirror of
https://github.com/supabase/supabase.git
synced 2026-10-11 20:35:07 +03:00
Runs 3 trials for Assistant evals in CI to reduce random variation. Locally, only 1 trial is run. Also adds `CI` to `studio#build` env in turbo.json. This env var is [automatically set by GitHub Actions](https://github.blog/changelog/2020-04-15-github-actions-sets-the-ci-environment-variable-to-true/). Compare number of trials: - [Assistant (mattrossman/ai-398-increase-trial-count-for-assistant-evals-1770305591)](https://www.braintrust.dev/app/supabase.io/p/Assistant/experiments/mattrossman%2Fai-398-increase-trial-count-for-assistant-evals-1770305591) - [Assistant (master)](https://www.braintrust.dev/app/supabase.io/p/Assistant/experiments/master-1770305906?c=mattrossman/ai-398-increase-trial-count-for-assistant-evals-1770305591) References: - https://www.braintrust.dev/docs/evaluate/run-evaluations#trials Closes AI-398 <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **Chores** * Updated evaluation configuration to adjust trial counts based on CI environment * Integrated CI environment variable into build system configuration <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Co-authored-by: Ali Waseem <waseema393@gmail.com>
133 lines
3.3 KiB
TypeScript
133 lines
3.3 KiB
TypeScript
import { openai } from '@ai-sdk/openai'
|
|
import { Eval } from 'braintrust'
|
|
import { generateAssistantResponse } from 'lib/ai/generate-assistant-response'
|
|
import { getMockTools } from 'lib/ai/tools/mock-tools'
|
|
import assert from 'node:assert'
|
|
import { dataset } from './dataset'
|
|
import {
|
|
completenessScorer,
|
|
concisenessScorer,
|
|
correctnessScorer,
|
|
docsFaithfulnessScorer,
|
|
goalCompletionScorer,
|
|
sqlIdentifierQuotingScorer,
|
|
sqlSyntaxScorer,
|
|
toolUsageScorer,
|
|
urlValidityScorer,
|
|
} from './scorer'
|
|
import { ToolSet, TypedToolCall, TypedToolResult } from 'ai'
|
|
|
|
assert(process.env.BRAINTRUST_PROJECT_ID, 'BRAINTRUST_PROJECT_ID is not set')
|
|
assert(process.env.OPENAI_API_KEY, 'OPENAI_API_KEY is not set')
|
|
|
|
Eval('Assistant', {
|
|
projectId: process.env.BRAINTRUST_PROJECT_ID,
|
|
trialCount: process.env.CI ? 3 : 1,
|
|
data: () => dataset,
|
|
task: async (input) => {
|
|
const result = await generateAssistantResponse({
|
|
model: openai('gpt-5-mini'),
|
|
messages: [{ id: '1', role: 'user', parts: [{ type: 'text', text: input.prompt }] }],
|
|
tools: await getMockTools(input.mockTables ? { list_tables: input.mockTables } : undefined),
|
|
})
|
|
|
|
const finishReason = await result.finishReason
|
|
|
|
// `result.toolCalls` only shows the last step, instead aggregate tools across all steps
|
|
const steps = await result.steps
|
|
|
|
const simplifiedSteps = steps.map((step) => ({
|
|
text: step.text,
|
|
toolCalls: step.toolCalls.map((call) => ({
|
|
toolName: call.toolName,
|
|
input: call.input,
|
|
})),
|
|
}))
|
|
|
|
const toolNames: string[] = []
|
|
const sqlQueries: string[] = []
|
|
const docs: string[] = []
|
|
|
|
for (const step of steps) {
|
|
for (const [i, toolCall] of step.toolCalls.entries()) {
|
|
toolNames.push(toolCall.toolName)
|
|
|
|
const toolResult = step.toolResults.at(i)
|
|
if (!toolResult) {
|
|
continue
|
|
}
|
|
|
|
const parsed = parseToolCall(toolCall, toolResult)
|
|
|
|
if (parsed.sqlQuery) {
|
|
sqlQueries.push(parsed.sqlQuery)
|
|
}
|
|
if (parsed.docs) {
|
|
docs.push(...parsed.docs)
|
|
}
|
|
}
|
|
}
|
|
|
|
return {
|
|
finishReason,
|
|
steps: simplifiedSteps,
|
|
toolNames,
|
|
sqlQueries,
|
|
docs,
|
|
}
|
|
},
|
|
scores: [
|
|
toolUsageScorer,
|
|
sqlSyntaxScorer,
|
|
sqlIdentifierQuotingScorer,
|
|
goalCompletionScorer,
|
|
concisenessScorer,
|
|
completenessScorer,
|
|
docsFaithfulnessScorer,
|
|
correctnessScorer,
|
|
urlValidityScorer,
|
|
],
|
|
})
|
|
|
|
type ParsedToolCall = {
|
|
/** Query generated by `execute_sql` */
|
|
sqlQuery?: string
|
|
|
|
/** Docs text pulled in from `search_docs` */
|
|
docs?: string[]
|
|
}
|
|
|
|
/**
|
|
* Validate and extract relevant info from a tool call/result
|
|
*/
|
|
function parseToolCall(
|
|
toolCall: TypedToolCall<ToolSet>,
|
|
toolResult: TypedToolResult<ToolSet>
|
|
): ParsedToolCall {
|
|
switch (toolCall.toolName) {
|
|
case 'execute_sql': {
|
|
const sqlQuery = toolCall.input.sql
|
|
if (typeof sqlQuery !== 'string') {
|
|
return {}
|
|
}
|
|
|
|
return { sqlQuery }
|
|
}
|
|
case 'search_docs': {
|
|
const content = toolResult.output.content
|
|
if (!content || !Array.isArray(content)) {
|
|
return {}
|
|
}
|
|
|
|
const docs = content.map((item) => item?.text).filter((text) => typeof text === 'string')
|
|
if (docs.length === 0) {
|
|
return {}
|
|
}
|
|
|
|
return { docs }
|
|
}
|
|
}
|
|
|
|
return {}
|
|
}
|