Files
supabase/apps/studio/evals/scorer-wasm.ts
Charis 7a2e237892 Add update_notebook evals; fix prompt gaps they surfaced (#49324)
## Summary
- Add `update_notebook` eval cases (insert/replace/delete/move, a
combined delete+insert, and guard/safety cases) mirroring the existing
`create_notebook` cases, targeting the notebooks already seeded in the
mock tool harness.
- Fix two behavior gaps in `NOTEBOOKS_PROMPT`/`LIMITATIONS_PROMPT` that
these cases surfaced when run live: the assistant asking the user for a
notebook id instead of resolving it via `list_notebooks`, and the
destructive-operations warning rule not being connected to SQL written
into notebook cells.
- Soften the destructive-SQL case's `correctAnswer` to match
`update_notebook`'s real approval-gated behavior — a warning
accompanying the reported change is acceptable, not only one strictly
preceding the tool call.

## Test plan
- [x] `pnpm run typecheck` (apps/studio) — clean
- [x] `pnpm exec prettier --check` on both changed files — clean
- [x] `evals/scorer.test.ts`, `evals/transcript.test.ts`,
`evals/trace-utils.test.ts` — 21/21 pass
- [x] Ran the new eval cases live against OpenAI (bypassing the
Braintrust proxy) via Braintrust MCP; confirmed via trace inspection
that the prompt fix resolved the id-resolution gap (Tool Usage 0% → 100%
across 3 trials) and that the assistant now includes an explicit
irreversibility warning when destructive SQL is written into a notebook
cell

<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

- **New Features**
- Improved notebook creation and editing support across SQL, query,
chart, and time-range cells.
- Added clearer handling for saved notebooks, recurring requests, and
one-time SQL execution.
  - Enhanced validation for database cells and notebook configuration.

- **Bug Fixes**
- Improved safeguards and warnings for destructive queries, including
saved notebook queries.
  - Better handling of missing tables and notebooks.
  - More precise notebook cell updates and tool usage validation.
<!-- end of auto-generated comment: release notes by coderabbit.ai -->
2026-08-21 08:39:46 -04:00

97 lines
3.0 KiB
TypeScript

import { Trace } from 'braintrust'
import { parse } from 'libpg-query'
import { AssistantEvalScorer } from './scorer'
import { getParsedToolSpans } from './trace-utils'
import { createNotebookInputSchema } from '@/components/ui/AIAssistantPanel/Message.utils'
import { executeSqlInputSchema } from '@/lib/ai/tools/studio-tools'
import { extractIdentifiers, isQuotedInSql, needsQuoting } from '@/lib/sql-identifier-quoting'
/**
* Extracts SQL strings from `execute_sql` tool spans and from every database cell inside
* `create_notebook` tool spans. Log cells are excluded — their SQL targets ClickHouse, and
* libpg-query only understands Postgres syntax.
*/
async function getSqlQueries(trace: Trace): Promise<string[]> {
const executeSqlSpans = await getParsedToolSpans(trace, 'execute_sql', {
inputSchema: executeSqlInputSchema,
})
const createNotebookSpans = await getParsedToolSpans(trace, 'create_notebook', {
inputSchema: createNotebookInputSchema,
})
const notebookCellSql = createNotebookSpans.flatMap((s) =>
s.input.content.cells.filter((cell) => cell._tag === 'database_cell').map((cell) => cell.sql)
)
return [...executeSqlSpans.map((s) => s.input.sql), ...notebookCellSql]
}
export const sqlSyntaxScorer: AssistantEvalScorer = async ({ trace }) => {
if (!trace) return null
const sqlQueries = await getSqlQueries(trace)
if (sqlQueries.length === 0) return null
const errors: string[] = []
let validQueries = 0
for (const sql of sqlQueries) {
try {
await parse(sql)
validQueries++
} catch (error) {
const errorMessage = error instanceof Error ? error.message : String(error)
errors.push(`SQL syntax error: ${errorMessage}`)
}
}
return {
name: 'SQL Validity',
score: validQueries / sqlQueries.length,
metadata: errors.length > 0 ? { errors } : undefined,
}
}
export const sqlIdentifierQuotingScorer: AssistantEvalScorer = async ({ trace }) => {
if (!trace) return null
const sqlQueries = await getSqlQueries(trace)
if (sqlQueries.length === 0) return null
const errors: string[] = []
let totalNeedingQuotes = 0
let properlyQuoted = 0
for (const sql of sqlQueries) {
try {
const ast = await parse(sql)
const identifiers = extractIdentifiers(ast)
for (const identifier of identifiers) {
if (needsQuoting(identifier)) {
totalNeedingQuotes++
if (isQuotedInSql(sql, identifier)) {
properlyQuoted++
} else {
const sqlPreview = sql.length > 100 ? `${sql.substring(0, 100)}...` : sql
errors.push(
`Identifier "${identifier}" needs quoting but is not quoted in: ${sqlPreview}`
)
}
}
}
} catch {
// Skip invalid SQL - already handled by sqlSyntaxScorer
}
}
const score = totalNeedingQuotes === 0 ? 1 : properlyQuoted / totalNeedingQuotes
return {
name: 'SQL Identifier Quoting',
score,
metadata: errors.length > 0 ? { errors } : undefined,
}
}