mirror of
https://github.com/supabase/supabase.git
synced 2026-10-05 17:35:10 +03:00
- Eval harness's only live tool, `search_docs`, no longer needs the in-process MCP client or its dummy token — it now calls the public docs GraphQL API (`https://supabase.com/docs/api/graphql`) directly. Low risk as this is an eval-harness change only. Production assistant path (`mcp-tools.ts`) untouched. **Update:** per [@mattrossman's review](https://github.com/supabase/supabase/pull/50092#discussion_r3980396341), the eval tool's description embeds the Content API's own GraphQL schema (fetched via a `{ schema }` query and minified with `gqlmin`), mirroring how `@supabase/mcp-server-supabase`'s `docs-tools.ts`/`loadSchema` populates production's `search_docs` description. Without it, the model had no schema to work from and issued malformed queries, which caused the 218 `search_docs` errors and the -25pp Docs Faithfulness regression in the first eval run on this PR. Schema loading is required: `createSearchDocsTool()` rejects if the schema fetch fails, so preflight and the gated eval job fail loudly instead of producing untrustworthy fallback results. `createSearchDocsTool` is async because the `ai` package's `tool()` only accepts a plain string `description`, unlike the MCP SDK's async description support; both callers (`getMockTools`, `evals/preflight.ts`) await it. `gqlmin` is a direct `apps/studio` dependency and was already transitive via `@supabase/mcp-server-supabase`. ### Verification - `pnpm -C apps/studio exec -- tsc --noEmit` reaches the compiler; it reports only the pre-existing unrelated `packages/ui-patterns/src/McpUrlBuilder/components/InstructionBlocks.tsx` `StaticImageData` error. - `pnpm -C apps/studio exec -- vitest run lib/ai/tools/mock-tools.test.ts lib/ai/tools/mcp-tools.test.ts` — 21/21 passed. - `pnpm exec tsx evals/preflight.ts` — live docs API schema fetch and search_docs call passed. - `NEXT_PUBLIC_CONTENT_API_URL=http://127.0.0.1:1/graphql pnpm -C apps/studio exec -- tsx evals/preflight.ts` — failed fast as expected, proving schema/API failures gate evals. - Fresh `run-evals` pass: Docs Faithfulness 55.7% (0pp), with no systemic `search_docs` regression. Risk: eval-harness-only; schema/API outage now fails the eval job before scoring rather than allowing fallback descriptions. <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Added documentation search powered by the public Supabase documentation GraphQL API. * Documentation search results now include live schema information and clearer error handling for failed or invalid requests. * **Bug Fixes** * Improved evaluation tooling reliability by removing unnecessary connection-abort behavior. * Updated validation to detect missing search tools and malformed documentation responses. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
312 lines
13 KiB
TypeScript
312 lines
13 KiB
TypeScript
import { describe, expect, it, vi } from 'vitest'
|
|
|
|
import { getMockTools, MOCK_NOTEBOOKS_DATA } from './mock-tools'
|
|
import { getNotebookTools } from './notebook-tools'
|
|
import type { AgentNotebook } from '@/data/content/notebooks/notebook-schema'
|
|
import { createSearchDocsTool } from '@/lib/ai/tools/search-docs-tool'
|
|
import type * as SearchDocsToolModule from '@/lib/ai/tools/search-docs-tool'
|
|
|
|
// search_docs normally fetches the live docs GraphQL schema on construction
|
|
// (see search-docs-tool.ts), which would make every test in this file depend
|
|
// on the public docs API. Mock it with a deterministic local fixture instead
|
|
// — live connectivity (including the schema fetch) is covered separately by
|
|
// evals/preflight.ts. The regression-guard test below overrides this default
|
|
// to verify the missing-tool guard fires.
|
|
vi.mock('@/lib/ai/tools/search-docs-tool', () => ({
|
|
createSearchDocsTool: vi.fn().mockResolvedValue({
|
|
description: 'Search the Supabase documentation using GraphQL.',
|
|
execute: async () => ({ content: [{ type: 'text' as const, text: '{}' }] }),
|
|
} as unknown as SearchDocsToolModule.SearchDocsTool),
|
|
}))
|
|
|
|
describe('ai/tools/mock-tools getMockTools', () => {
|
|
it('wires the mocked search_docs tool through from the shared module, alongside the deterministic mocks', async () => {
|
|
const result = await getMockTools(undefined)
|
|
|
|
// The mocked tool, wired through from the shared module
|
|
expect(result.search_docs).toBeDefined()
|
|
expect(result.search_docs.description).toContain('Search the Supabase documentation')
|
|
expect(typeof result.search_docs.execute).toBe('function')
|
|
// A couple of the deterministic mocks, to confirm the merge
|
|
expect(result).toHaveProperty('list_tables')
|
|
expect(result).toHaveProperty('query_logs')
|
|
})
|
|
|
|
// This is the regression guard: if a refactor stops sourcing search_docs
|
|
// (drops it from getMockTools, or createSearchDocsTool breaks), fail loudly
|
|
// in normal CI instead of only surfacing during an opt-in Braintrust eval run.
|
|
it('throws a clear error when search_docs is missing from the harness tools', async () => {
|
|
vi.mocked(createSearchDocsTool).mockResolvedValueOnce(
|
|
undefined as unknown as SearchDocsToolModule.SearchDocsTool
|
|
)
|
|
|
|
await expect(getMockTools(undefined)).rejects.toThrow(
|
|
'search_docs tool is missing from the eval harness'
|
|
)
|
|
})
|
|
|
|
describe('notebook tools', () => {
|
|
const AUTH_HEALTH_NOTEBOOK_ID = MOCK_NOTEBOOKS_DATA[0].id
|
|
const EDGE_FUNCTION_NOTEBOOK_ID = MOCK_NOTEBOOKS_DATA[1].id
|
|
|
|
it('list_notebooks reflects the two seeded fixtures', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined')
|
|
|
|
const result = await mockTools.list_notebooks.execute(
|
|
{ limit: 20 },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
|
|
expect(result.notebooks.map((notebook) => notebook.name)).toEqual([
|
|
'Auth health check',
|
|
'Edge function error triage',
|
|
])
|
|
expect(result.notebooks.map((notebook) => notebook.cell_count)).toEqual([3, 2])
|
|
expect(result.notebooks[1].description).toBeUndefined()
|
|
expect(result.cursor).toBeUndefined()
|
|
})
|
|
|
|
it('get_notebook resolves cells in order and rejects an unknown id', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
if (!mockTools.get_notebook.execute) throw new Error('execute is undefined')
|
|
|
|
const result = await mockTools.get_notebook.execute(
|
|
{ id: AUTH_HEALTH_NOTEBOOK_ID },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
|
|
expect(result.cells.map((cell) => cell._tag)).toEqual([
|
|
'markdown_cell',
|
|
'database_cell',
|
|
'log_cell',
|
|
])
|
|
|
|
const [, databaseCell, logCell] = result.cells
|
|
if (databaseCell._tag !== 'database_cell') throw new Error('expected database_cell')
|
|
if (logCell._tag !== 'log_cell') throw new Error('expected log_cell')
|
|
|
|
expect(databaseCell.sql).toContain('signups')
|
|
expect(databaseCell.row_limit).toBe(30)
|
|
expect(logCell.time_range).toEqual({ _tag: 'relative_time_range', unit: 'hour', amount: 1 })
|
|
|
|
await expect(
|
|
mockTools.get_notebook.execute(
|
|
{ id: 'unknown-notebook-id' },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
).rejects.toThrow(/not found/i)
|
|
})
|
|
|
|
it('shares deterministic run_notebook output with eval models', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
if (!mockTools.run_notebook.execute) throw new Error('execute is undefined')
|
|
if (!mockTools.run_notebook.toModelOutput) throw new Error('toModelOutput is undefined')
|
|
|
|
const output = await mockTools.run_notebook.execute(
|
|
{ id: AUTH_HEALTH_NOTEBOOK_ID, expected_updated_at: MOCK_NOTEBOOKS_DATA[0].updated_at },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
const modelOutput = mockTools.run_notebook.toModelOutput({
|
|
toolCallId: 'test',
|
|
input: { id: AUTH_HEALTH_NOTEBOOK_ID, expected_updated_at: output.updated_at },
|
|
output,
|
|
})
|
|
|
|
expect(modelOutput).toMatchObject({
|
|
type: 'json',
|
|
value: {
|
|
cells: expect.arrayContaining([expect.objectContaining({ rows: [] })]),
|
|
},
|
|
})
|
|
})
|
|
|
|
it('overrides create_notebook needsApproval to false, unlike the real tool', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
|
|
expect(getNotebookTools().create_notebook.needsApproval).toBe(true)
|
|
expect(mockTools.create_notebook.needsApproval).toBe(false)
|
|
})
|
|
|
|
it('create_notebook stores a new notebook visible via get_notebook and list_notebooks', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
if (!mockTools.create_notebook.execute) throw new Error('execute is undefined')
|
|
if (!mockTools.get_notebook.execute) throw new Error('execute is undefined')
|
|
if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined')
|
|
|
|
const content: AgentNotebook = {
|
|
schema_version: 1,
|
|
cells: [
|
|
{ _tag: 'markdown_cell', text: '# New notebook' },
|
|
{ _tag: 'database_cell', sql: 'select 1', row_limit: 10 },
|
|
],
|
|
}
|
|
|
|
const created = await mockTools.create_notebook.execute(
|
|
{ name: 'New notebook', content },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
expect(created).toEqual({ id: expect.any(String), name: 'New notebook' })
|
|
|
|
const fetched = await mockTools.get_notebook.execute(
|
|
{ id: created.id },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
expect(fetched.cells).toHaveLength(2)
|
|
expect(fetched.cells.every((cell) => typeof cell._id === 'string')).toBe(true)
|
|
|
|
const listed = await mockTools.list_notebooks.execute(
|
|
{ limit: 20 },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
expect(listed.notebooks).toHaveLength(3)
|
|
const newEntry = listed.notebooks.find((notebook) => notebook.id === created.id)
|
|
expect(newEntry?.cell_count).toBe(2)
|
|
})
|
|
|
|
it('overrides update_notebook needsApproval to false, unlike the real tool', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
|
|
expect(getNotebookTools().update_notebook.needsApproval).toBe(true)
|
|
expect(mockTools.update_notebook.needsApproval).toBe(false)
|
|
})
|
|
|
|
it('update_notebook inserts and deletes cells, and list_notebooks reflects the new cell count', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
if (!mockTools.get_notebook.execute) throw new Error('execute is undefined')
|
|
if (!mockTools.update_notebook.execute) throw new Error('execute is undefined')
|
|
if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined')
|
|
|
|
const before = await mockTools.get_notebook.execute(
|
|
{ id: AUTH_HEALTH_NOTEBOOK_ID },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
const [markdownCell, , logCell] = before.cells
|
|
|
|
const result = await mockTools.update_notebook.execute(
|
|
{
|
|
id: AUTH_HEALTH_NOTEBOOK_ID,
|
|
expected_updated_at: before.updated_at,
|
|
operations: [
|
|
{
|
|
_tag: 'insert_cell',
|
|
after_cell_id: markdownCell._id,
|
|
cell: { _tag: 'database_cell', sql: 'select 1', row_limit: 10 },
|
|
},
|
|
{ _tag: 'delete_cell', cell_id: logCell._id },
|
|
],
|
|
},
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
expect(result).toEqual({ id: AUTH_HEALTH_NOTEBOOK_ID, name: 'Auth health check' })
|
|
|
|
const after = await mockTools.get_notebook.execute(
|
|
{ id: AUTH_HEALTH_NOTEBOOK_ID },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
expect(after.cells.map((cell) => cell._tag)).toEqual([
|
|
'markdown_cell',
|
|
'database_cell',
|
|
'database_cell',
|
|
])
|
|
|
|
const listed = await mockTools.list_notebooks.execute(
|
|
{ limit: 20 },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
const entry = listed.notebooks.find((notebook) => notebook.id === AUTH_HEALTH_NOTEBOOK_ID)
|
|
expect(entry?.cell_count).toBe(3)
|
|
})
|
|
|
|
it('update_notebook rejects an unknown cell_id without mutating the notebook', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
if (!mockTools.get_notebook.execute) throw new Error('execute is undefined')
|
|
if (!mockTools.update_notebook.execute) throw new Error('execute is undefined')
|
|
|
|
const before = await mockTools.get_notebook.execute(
|
|
{ id: EDGE_FUNCTION_NOTEBOOK_ID },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
|
|
await expect(
|
|
mockTools.update_notebook.execute(
|
|
{
|
|
id: EDGE_FUNCTION_NOTEBOOK_ID,
|
|
expected_updated_at: before.updated_at,
|
|
operations: [{ _tag: 'delete_cell', cell_id: 'does-not-exist' }],
|
|
},
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
).rejects.toThrow(/does-not-exist/)
|
|
|
|
const after = await mockTools.get_notebook.execute(
|
|
{ id: EDGE_FUNCTION_NOTEBOOK_ID },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
expect(after.cells.map((cell) => cell._tag)).toEqual(['markdown_cell', 'log_cell'])
|
|
})
|
|
|
|
it('overrides delete_notebook needsApproval to false, unlike the real tool', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
|
|
expect(getNotebookTools().delete_notebook.needsApproval).toBe(true)
|
|
expect(mockTools.delete_notebook.needsApproval).toBe(false)
|
|
})
|
|
|
|
it('delete_notebook removes the notebook, and list_notebooks no longer returns it', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
if (!mockTools.delete_notebook.execute) throw new Error('execute is undefined')
|
|
if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined')
|
|
|
|
const result = await mockTools.delete_notebook.execute(
|
|
{ id: AUTH_HEALTH_NOTEBOOK_ID },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
expect(result).toEqual({ id: AUTH_HEALTH_NOTEBOOK_ID, name: 'Auth health check' })
|
|
|
|
const listed = await mockTools.list_notebooks.execute(
|
|
{ limit: 20 },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
expect(listed.notebooks.map((notebook) => notebook.id)).toEqual([EDGE_FUNCTION_NOTEBOOK_ID])
|
|
})
|
|
|
|
it('delete_notebook rejects an unknown id', async () => {
|
|
const mockTools = await getMockTools(undefined)
|
|
if (!mockTools.delete_notebook.execute) throw new Error('execute is undefined')
|
|
|
|
await expect(
|
|
mockTools.delete_notebook.execute(
|
|
{ id: 'unknown-notebook-id' },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
).rejects.toThrow(/unknown-notebook-id/)
|
|
})
|
|
|
|
it('is isolated per call to getMockTools', async () => {
|
|
const firstCall = await getMockTools(undefined)
|
|
if (!firstCall.create_notebook.execute) throw new Error('execute is undefined')
|
|
|
|
await firstCall.create_notebook.execute(
|
|
{
|
|
name: 'Ephemeral notebook',
|
|
content: { schema_version: 1, cells: [{ _tag: 'markdown_cell', text: 'hi' }] },
|
|
},
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
|
|
const secondCall = await getMockTools(undefined)
|
|
if (!secondCall.list_notebooks.execute) throw new Error('execute is undefined')
|
|
|
|
const result = await secondCall.list_notebooks.execute(
|
|
{ limit: 20 },
|
|
{ toolCallId: 'test', messages: [], context: {} }
|
|
)
|
|
expect(result.notebooks.map((notebook) => notebook.name)).toEqual([
|
|
'Auth health check',
|
|
'Edge function error triage',
|
|
])
|
|
})
|
|
})
|
|
})
|