mirror of
https://github.com/supabase/supabase.git
synced 2026-10-05 17:35:10 +03:00
## Summary - Add `update_notebook` eval cases (insert/replace/delete/move, a combined delete+insert, and guard/safety cases) mirroring the existing `create_notebook` cases, targeting the notebooks already seeded in the mock tool harness. - Fix two behavior gaps in `NOTEBOOKS_PROMPT`/`LIMITATIONS_PROMPT` that these cases surfaced when run live: the assistant asking the user for a notebook id instead of resolving it via `list_notebooks`, and the destructive-operations warning rule not being connected to SQL written into notebook cells. - Soften the destructive-SQL case's `correctAnswer` to match `update_notebook`'s real approval-gated behavior — a warning accompanying the reported change is acceptable, not only one strictly preceding the tool call. ## Test plan - [x] `pnpm run typecheck` (apps/studio) — clean - [x] `pnpm exec prettier --check` on both changed files — clean - [x] `evals/scorer.test.ts`, `evals/transcript.test.ts`, `evals/trace-utils.test.ts` — 21/21 pass - [x] Ran the new eval cases live against OpenAI (bypassing the Braintrust proxy) via Braintrust MCP; confirmed via trace inspection that the prompt fix resolved the id-resolution gap (Tool Usage 0% → 100% across 3 trials) and that the assistant now includes an explicit irreversibility warning when destructive SQL is written into a notebook cell <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - **New Features** - Improved notebook creation and editing support across SQL, query, chart, and time-range cells. - Added clearer handling for saved notebooks, recurring requests, and one-time SQL execution. - Enhanced validation for database cells and notebook configuration. - **Bug Fixes** - Improved safeguards and warnings for destructive queries, including saved notebook queries. - Better handling of missing tables and notebooks. - More precise notebook cell updates and tool usage validation. <!-- end of auto-generated comment: release notes by coderabbit.ai -->
97 lines
3.0 KiB
TypeScript
97 lines
3.0 KiB
TypeScript
import { Trace } from 'braintrust'
|
|
import { parse } from 'libpg-query'
|
|
|
|
import { AssistantEvalScorer } from './scorer'
|
|
import { getParsedToolSpans } from './trace-utils'
|
|
import { createNotebookInputSchema } from '@/components/ui/AIAssistantPanel/Message.utils'
|
|
import { executeSqlInputSchema } from '@/lib/ai/tools/studio-tools'
|
|
import { extractIdentifiers, isQuotedInSql, needsQuoting } from '@/lib/sql-identifier-quoting'
|
|
|
|
/**
|
|
* Extracts SQL strings from `execute_sql` tool spans and from every database cell inside
|
|
* `create_notebook` tool spans. Log cells are excluded — their SQL targets ClickHouse, and
|
|
* libpg-query only understands Postgres syntax.
|
|
*/
|
|
async function getSqlQueries(trace: Trace): Promise<string[]> {
|
|
const executeSqlSpans = await getParsedToolSpans(trace, 'execute_sql', {
|
|
inputSchema: executeSqlInputSchema,
|
|
})
|
|
const createNotebookSpans = await getParsedToolSpans(trace, 'create_notebook', {
|
|
inputSchema: createNotebookInputSchema,
|
|
})
|
|
|
|
const notebookCellSql = createNotebookSpans.flatMap((s) =>
|
|
s.input.content.cells.filter((cell) => cell._tag === 'database_cell').map((cell) => cell.sql)
|
|
)
|
|
|
|
return [...executeSqlSpans.map((s) => s.input.sql), ...notebookCellSql]
|
|
}
|
|
|
|
export const sqlSyntaxScorer: AssistantEvalScorer = async ({ trace }) => {
|
|
if (!trace) return null
|
|
|
|
const sqlQueries = await getSqlQueries(trace)
|
|
if (sqlQueries.length === 0) return null
|
|
|
|
const errors: string[] = []
|
|
let validQueries = 0
|
|
|
|
for (const sql of sqlQueries) {
|
|
try {
|
|
await parse(sql)
|
|
validQueries++
|
|
} catch (error) {
|
|
const errorMessage = error instanceof Error ? error.message : String(error)
|
|
errors.push(`SQL syntax error: ${errorMessage}`)
|
|
}
|
|
}
|
|
|
|
return {
|
|
name: 'SQL Validity',
|
|
score: validQueries / sqlQueries.length,
|
|
metadata: errors.length > 0 ? { errors } : undefined,
|
|
}
|
|
}
|
|
|
|
export const sqlIdentifierQuotingScorer: AssistantEvalScorer = async ({ trace }) => {
|
|
if (!trace) return null
|
|
|
|
const sqlQueries = await getSqlQueries(trace)
|
|
if (sqlQueries.length === 0) return null
|
|
|
|
const errors: string[] = []
|
|
let totalNeedingQuotes = 0
|
|
let properlyQuoted = 0
|
|
|
|
for (const sql of sqlQueries) {
|
|
try {
|
|
const ast = await parse(sql)
|
|
const identifiers = extractIdentifiers(ast)
|
|
|
|
for (const identifier of identifiers) {
|
|
if (needsQuoting(identifier)) {
|
|
totalNeedingQuotes++
|
|
if (isQuotedInSql(sql, identifier)) {
|
|
properlyQuoted++
|
|
} else {
|
|
const sqlPreview = sql.length > 100 ? `${sql.substring(0, 100)}...` : sql
|
|
errors.push(
|
|
`Identifier "${identifier}" needs quoting but is not quoted in: ${sqlPreview}`
|
|
)
|
|
}
|
|
}
|
|
}
|
|
} catch {
|
|
// Skip invalid SQL - already handled by sqlSyntaxScorer
|
|
}
|
|
}
|
|
|
|
const score = totalNeedingQuotes === 0 ? 1 : properlyQuoted / totalNeedingQuotes
|
|
|
|
return {
|
|
name: 'SQL Identifier Quoting',
|
|
score,
|
|
metadata: errors.length > 0 ? { errors } : undefined,
|
|
}
|
|
}
|