mirror of
https://github.com/supabase/supabase.git
synced 2026-10-06 09:55:06 +03:00
## Summary - Added ~11 new eval cases for Notebooks AI assistant evals, including: basic notebook cell creation, multi-cell composition (markdown+database+log), log cell time ranges, row_limit defaults, chart config, destructive SQL safety warnings, and hallucination guards for nonexistent tables - Extended `toolUsageScorer` with deterministic `forbiddenTools` field to score both required and forbidden tool usage, enabling eval cases to assert tool choice (e.g., `execute_sql` vs `create_notebook`) without LLM-as-judge - Extended SQL validators (`sqlSyntaxScorer`/`sqlIdentifierQuotingScorer`) to validate SQL inside `create_notebook` database cells (log cells deliberately excluded as they use ClickHouse dialect) - Fixed two real assistant issues in `NOTEBOOKS_PROMPT`: (a) reuse `CLICKHOUSE_LOGS_COMPLETION_INSTRUCTIONS` and schema section to prevent incorrect BigQuery-style SQL in logs queries, (b) require schema verification before writing `database_cell` to prevent queries against nonexistent tables Resolves FE-4087 ## Test plan - All 11 new eval cases run live against OpenAI via Braintrust; traces inspected and validated - Existing unit tests, lint, and typecheck pass <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Added support for creating notebooks with database, Markdown, chart, and log cells. * Improved handling of reusable notebook requests versus one-off SQL queries. * Added guidance for modern ClickHouse SQL and absolute log time ranges. * **Bug Fixes** * Improved SQL validation, row-limit enforcement, and destructive-query safety. * Prevented invalid or nonexistent-table queries from being accepted. * Improved validation of notebook cell types and tool usage. * Improved handling of ClickHouse log queries and database-cell SQL. * Improved evaluation reliability by limiting concurrent test execution. <!-- end of auto-generated comment: release notes by coderabbit.ai -->
81 lines
2.7 KiB
TypeScript
81 lines
2.7 KiB
TypeScript
import assert from 'node:assert'
|
||
import { Eval } from 'braintrust'
|
||
|
||
import { dataset } from './dataset'
|
||
import {
|
||
completenessScorer,
|
||
concisenessScorer,
|
||
correctnessScorer,
|
||
docsFaithfulnessScorer,
|
||
goalCompletionScorer,
|
||
knowledgeUsageScorer,
|
||
safetyScorer,
|
||
toolUsageScorer,
|
||
urlValidityScorer,
|
||
} from './scorer'
|
||
import { sqlIdentifierQuotingScorer, sqlSyntaxScorer } from './scorer-wasm'
|
||
import { buildTranscript } from './transcript'
|
||
import { generateAssistantResponse } from '@/lib/ai/generate-assistant-response'
|
||
import { getModel } from '@/lib/ai/model'
|
||
import { DEFAULT_ASSISTANT_BASE_MODEL_ID, getAssistantModelEntry } from '@/lib/ai/model.utils'
|
||
import { getMockTools } from '@/lib/ai/tools/mock-tools'
|
||
|
||
assert(process.env.BRAINTRUST_PROJECT_ID, 'BRAINTRUST_PROJECT_ID is not set')
|
||
assert(process.env.OPENAI_API_KEY, 'OPENAI_API_KEY is not set')
|
||
|
||
Eval('Assistant', {
|
||
projectId: process.env.BRAINTRUST_PROJECT_ID,
|
||
trialCount: process.env.CI ? 3 : 1,
|
||
// Braintrust defaults to unbounded concurrency (every case × trial runs in parallel
|
||
// in one process), so memory scales linearly with dataset size. Left uncapped, this
|
||
// OOMs the CI runner once the dataset grows large enough — cap it so the suite keeps
|
||
// scaling safely instead of racing the runner's heap ceiling.
|
||
maxConcurrency: 10,
|
||
data: () => dataset,
|
||
task: async (input) => {
|
||
const modelEntry = getAssistantModelEntry(DEFAULT_ASSISTANT_BASE_MODEL_ID)
|
||
const modelResponse = await getModel({ provider: 'openai', modelEntry })
|
||
if (modelResponse.error) throw modelResponse.error
|
||
|
||
// Owns the lifecycle of the remote MCP client opened inside getMockTools:
|
||
// aborting once generation is done closes that connection.
|
||
const toolsAbortController = new AbortController()
|
||
try {
|
||
const result = await generateAssistantResponse({
|
||
...modelResponse.modelParams,
|
||
isExplorerEnabled: true,
|
||
messages: [
|
||
{
|
||
id: '1',
|
||
role: 'user',
|
||
parts: [{ type: 'text', text: input.prompt }],
|
||
},
|
||
],
|
||
tools: await getMockTools(
|
||
input.mockTables ? { list_tables: input.mockTables } : undefined,
|
||
toolsAbortController.signal
|
||
),
|
||
})
|
||
|
||
const finishReason = await result.finishReason
|
||
const steps = await result.steps
|
||
return { finishReason, transcript: buildTranscript(input.prompt, steps) }
|
||
} finally {
|
||
toolsAbortController.abort()
|
||
}
|
||
},
|
||
scores: [
|
||
toolUsageScorer,
|
||
knowledgeUsageScorer,
|
||
sqlSyntaxScorer,
|
||
sqlIdentifierQuotingScorer,
|
||
goalCompletionScorer,
|
||
concisenessScorer,
|
||
completenessScorer,
|
||
docsFaithfulnessScorer,
|
||
correctnessScorer,
|
||
safetyScorer,
|
||
urlValidityScorer,
|
||
],
|
||
})
|