Files
supabase/apps/studio/lib/ai/tools/mock-tools.test.ts
Pedro RodriguesandClaude Sonnet 5 22d7bc0cfd feat(studio-evals): custom search_docs tool for the eval harness (no token / no PAT) (#50092)
- Eval harness's only live tool, `search_docs`, no longer needs the
in-process MCP client or its dummy token — it now calls the public docs
GraphQL API (`https://supabase.com/docs/api/graphql`) directly. Low risk
as this is an eval-harness change only. Production assistant path
(`mcp-tools.ts`) untouched.

**Update:** per [@mattrossman's
review](https://github.com/supabase/supabase/pull/50092#discussion_r3980396341),
the eval tool's description embeds the Content API's own GraphQL schema
(fetched via a `{ schema }` query and minified with `gqlmin`), mirroring
how `@supabase/mcp-server-supabase`'s `docs-tools.ts`/`loadSchema`
populates production's `search_docs` description. Without it, the model
had no schema to work from and issued malformed queries, which caused
the 218 `search_docs` errors and the -25pp Docs Faithfulness regression
in the first eval run on this PR. Schema loading is required:
`createSearchDocsTool()` rejects if the schema fetch fails, so preflight
and the gated eval job fail loudly instead of producing untrustworthy
fallback results. `createSearchDocsTool` is async because the `ai`
package's `tool()` only accepts a plain string `description`, unlike the
MCP SDK's async description support; both callers (`getMockTools`,
`evals/preflight.ts`) await it. `gqlmin` is a direct `apps/studio`
dependency and was already transitive via
`@supabase/mcp-server-supabase`.

### Verification
- `pnpm -C apps/studio exec -- tsc --noEmit` reaches the compiler; it
reports only the pre-existing unrelated
`packages/ui-patterns/src/McpUrlBuilder/components/InstructionBlocks.tsx`
`StaticImageData` error.
- `pnpm -C apps/studio exec -- vitest run
lib/ai/tools/mock-tools.test.ts lib/ai/tools/mcp-tools.test.ts` — 21/21
passed.
- `pnpm exec tsx evals/preflight.ts` — live docs API schema fetch and
search_docs call passed.
- `NEXT_PUBLIC_CONTENT_API_URL=http://127.0.0.1:1/graphql pnpm -C
apps/studio exec -- tsx evals/preflight.ts` — failed fast as expected,
proving schema/API failures gate evals.
- Fresh `run-evals` pass: Docs Faithfulness 55.7% (0pp), with no
systemic `search_docs` regression.

Risk: eval-harness-only; schema/API outage now fails the eval job before
scoring rather than allowing fallback descriptions.

<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->

## Summary by CodeRabbit

* **New Features**
* Added documentation search powered by the public Supabase
documentation GraphQL API.
* Documentation search results now include live schema information and
clearer error handling for failed or invalid requests.

* **Bug Fixes**
* Improved evaluation tooling reliability by removing unnecessary
connection-abort behavior.
* Updated validation to detect missing search tools and malformed
documentation responses.

<!-- end of auto-generated comment: release notes by coderabbit.ai -->

---------

Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-14 12:51:14 +02:00

312 lines
13 KiB
TypeScript

import { describe, expect, it, vi } from 'vitest'
import { getMockTools, MOCK_NOTEBOOKS_DATA } from './mock-tools'
import { getNotebookTools } from './notebook-tools'
import type { AgentNotebook } from '@/data/content/notebooks/notebook-schema'
import { createSearchDocsTool } from '@/lib/ai/tools/search-docs-tool'
import type * as SearchDocsToolModule from '@/lib/ai/tools/search-docs-tool'
// search_docs normally fetches the live docs GraphQL schema on construction
// (see search-docs-tool.ts), which would make every test in this file depend
// on the public docs API. Mock it with a deterministic local fixture instead
// — live connectivity (including the schema fetch) is covered separately by
// evals/preflight.ts. The regression-guard test below overrides this default
// to verify the missing-tool guard fires.
vi.mock('@/lib/ai/tools/search-docs-tool', () => ({
createSearchDocsTool: vi.fn().mockResolvedValue({
description: 'Search the Supabase documentation using GraphQL.',
execute: async () => ({ content: [{ type: 'text' as const, text: '{}' }] }),
} as unknown as SearchDocsToolModule.SearchDocsTool),
}))
describe('ai/tools/mock-tools getMockTools', () => {
it('wires the mocked search_docs tool through from the shared module, alongside the deterministic mocks', async () => {
const result = await getMockTools(undefined)
// The mocked tool, wired through from the shared module
expect(result.search_docs).toBeDefined()
expect(result.search_docs.description).toContain('Search the Supabase documentation')
expect(typeof result.search_docs.execute).toBe('function')
// A couple of the deterministic mocks, to confirm the merge
expect(result).toHaveProperty('list_tables')
expect(result).toHaveProperty('query_logs')
})
// This is the regression guard: if a refactor stops sourcing search_docs
// (drops it from getMockTools, or createSearchDocsTool breaks), fail loudly
// in normal CI instead of only surfacing during an opt-in Braintrust eval run.
it('throws a clear error when search_docs is missing from the harness tools', async () => {
vi.mocked(createSearchDocsTool).mockResolvedValueOnce(
undefined as unknown as SearchDocsToolModule.SearchDocsTool
)
await expect(getMockTools(undefined)).rejects.toThrow(
'search_docs tool is missing from the eval harness'
)
})
describe('notebook tools', () => {
const AUTH_HEALTH_NOTEBOOK_ID = MOCK_NOTEBOOKS_DATA[0].id
const EDGE_FUNCTION_NOTEBOOK_ID = MOCK_NOTEBOOKS_DATA[1].id
it('list_notebooks reflects the two seeded fixtures', async () => {
const mockTools = await getMockTools(undefined)
if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined')
const result = await mockTools.list_notebooks.execute(
{ limit: 20 },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(result.notebooks.map((notebook) => notebook.name)).toEqual([
'Auth health check',
'Edge function error triage',
])
expect(result.notebooks.map((notebook) => notebook.cell_count)).toEqual([3, 2])
expect(result.notebooks[1].description).toBeUndefined()
expect(result.cursor).toBeUndefined()
})
it('get_notebook resolves cells in order and rejects an unknown id', async () => {
const mockTools = await getMockTools(undefined)
if (!mockTools.get_notebook.execute) throw new Error('execute is undefined')
const result = await mockTools.get_notebook.execute(
{ id: AUTH_HEALTH_NOTEBOOK_ID },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(result.cells.map((cell) => cell._tag)).toEqual([
'markdown_cell',
'database_cell',
'log_cell',
])
const [, databaseCell, logCell] = result.cells
if (databaseCell._tag !== 'database_cell') throw new Error('expected database_cell')
if (logCell._tag !== 'log_cell') throw new Error('expected log_cell')
expect(databaseCell.sql).toContain('signups')
expect(databaseCell.row_limit).toBe(30)
expect(logCell.time_range).toEqual({ _tag: 'relative_time_range', unit: 'hour', amount: 1 })
await expect(
mockTools.get_notebook.execute(
{ id: 'unknown-notebook-id' },
{ toolCallId: 'test', messages: [], context: {} }
)
).rejects.toThrow(/not found/i)
})
it('shares deterministic run_notebook output with eval models', async () => {
const mockTools = await getMockTools(undefined)
if (!mockTools.run_notebook.execute) throw new Error('execute is undefined')
if (!mockTools.run_notebook.toModelOutput) throw new Error('toModelOutput is undefined')
const output = await mockTools.run_notebook.execute(
{ id: AUTH_HEALTH_NOTEBOOK_ID, expected_updated_at: MOCK_NOTEBOOKS_DATA[0].updated_at },
{ toolCallId: 'test', messages: [], context: {} }
)
const modelOutput = mockTools.run_notebook.toModelOutput({
toolCallId: 'test',
input: { id: AUTH_HEALTH_NOTEBOOK_ID, expected_updated_at: output.updated_at },
output,
})
expect(modelOutput).toMatchObject({
type: 'json',
value: {
cells: expect.arrayContaining([expect.objectContaining({ rows: [] })]),
},
})
})
it('overrides create_notebook needsApproval to false, unlike the real tool', async () => {
const mockTools = await getMockTools(undefined)
expect(getNotebookTools().create_notebook.needsApproval).toBe(true)
expect(mockTools.create_notebook.needsApproval).toBe(false)
})
it('create_notebook stores a new notebook visible via get_notebook and list_notebooks', async () => {
const mockTools = await getMockTools(undefined)
if (!mockTools.create_notebook.execute) throw new Error('execute is undefined')
if (!mockTools.get_notebook.execute) throw new Error('execute is undefined')
if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined')
const content: AgentNotebook = {
schema_version: 1,
cells: [
{ _tag: 'markdown_cell', text: '# New notebook' },
{ _tag: 'database_cell', sql: 'select 1', row_limit: 10 },
],
}
const created = await mockTools.create_notebook.execute(
{ name: 'New notebook', content },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(created).toEqual({ id: expect.any(String), name: 'New notebook' })
const fetched = await mockTools.get_notebook.execute(
{ id: created.id },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(fetched.cells).toHaveLength(2)
expect(fetched.cells.every((cell) => typeof cell._id === 'string')).toBe(true)
const listed = await mockTools.list_notebooks.execute(
{ limit: 20 },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(listed.notebooks).toHaveLength(3)
const newEntry = listed.notebooks.find((notebook) => notebook.id === created.id)
expect(newEntry?.cell_count).toBe(2)
})
it('overrides update_notebook needsApproval to false, unlike the real tool', async () => {
const mockTools = await getMockTools(undefined)
expect(getNotebookTools().update_notebook.needsApproval).toBe(true)
expect(mockTools.update_notebook.needsApproval).toBe(false)
})
it('update_notebook inserts and deletes cells, and list_notebooks reflects the new cell count', async () => {
const mockTools = await getMockTools(undefined)
if (!mockTools.get_notebook.execute) throw new Error('execute is undefined')
if (!mockTools.update_notebook.execute) throw new Error('execute is undefined')
if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined')
const before = await mockTools.get_notebook.execute(
{ id: AUTH_HEALTH_NOTEBOOK_ID },
{ toolCallId: 'test', messages: [], context: {} }
)
const [markdownCell, , logCell] = before.cells
const result = await mockTools.update_notebook.execute(
{
id: AUTH_HEALTH_NOTEBOOK_ID,
expected_updated_at: before.updated_at,
operations: [
{
_tag: 'insert_cell',
after_cell_id: markdownCell._id,
cell: { _tag: 'database_cell', sql: 'select 1', row_limit: 10 },
},
{ _tag: 'delete_cell', cell_id: logCell._id },
],
},
{ toolCallId: 'test', messages: [], context: {} }
)
expect(result).toEqual({ id: AUTH_HEALTH_NOTEBOOK_ID, name: 'Auth health check' })
const after = await mockTools.get_notebook.execute(
{ id: AUTH_HEALTH_NOTEBOOK_ID },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(after.cells.map((cell) => cell._tag)).toEqual([
'markdown_cell',
'database_cell',
'database_cell',
])
const listed = await mockTools.list_notebooks.execute(
{ limit: 20 },
{ toolCallId: 'test', messages: [], context: {} }
)
const entry = listed.notebooks.find((notebook) => notebook.id === AUTH_HEALTH_NOTEBOOK_ID)
expect(entry?.cell_count).toBe(3)
})
it('update_notebook rejects an unknown cell_id without mutating the notebook', async () => {
const mockTools = await getMockTools(undefined)
if (!mockTools.get_notebook.execute) throw new Error('execute is undefined')
if (!mockTools.update_notebook.execute) throw new Error('execute is undefined')
const before = await mockTools.get_notebook.execute(
{ id: EDGE_FUNCTION_NOTEBOOK_ID },
{ toolCallId: 'test', messages: [], context: {} }
)
await expect(
mockTools.update_notebook.execute(
{
id: EDGE_FUNCTION_NOTEBOOK_ID,
expected_updated_at: before.updated_at,
operations: [{ _tag: 'delete_cell', cell_id: 'does-not-exist' }],
},
{ toolCallId: 'test', messages: [], context: {} }
)
).rejects.toThrow(/does-not-exist/)
const after = await mockTools.get_notebook.execute(
{ id: EDGE_FUNCTION_NOTEBOOK_ID },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(after.cells.map((cell) => cell._tag)).toEqual(['markdown_cell', 'log_cell'])
})
it('overrides delete_notebook needsApproval to false, unlike the real tool', async () => {
const mockTools = await getMockTools(undefined)
expect(getNotebookTools().delete_notebook.needsApproval).toBe(true)
expect(mockTools.delete_notebook.needsApproval).toBe(false)
})
it('delete_notebook removes the notebook, and list_notebooks no longer returns it', async () => {
const mockTools = await getMockTools(undefined)
if (!mockTools.delete_notebook.execute) throw new Error('execute is undefined')
if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined')
const result = await mockTools.delete_notebook.execute(
{ id: AUTH_HEALTH_NOTEBOOK_ID },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(result).toEqual({ id: AUTH_HEALTH_NOTEBOOK_ID, name: 'Auth health check' })
const listed = await mockTools.list_notebooks.execute(
{ limit: 20 },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(listed.notebooks.map((notebook) => notebook.id)).toEqual([EDGE_FUNCTION_NOTEBOOK_ID])
})
it('delete_notebook rejects an unknown id', async () => {
const mockTools = await getMockTools(undefined)
if (!mockTools.delete_notebook.execute) throw new Error('execute is undefined')
await expect(
mockTools.delete_notebook.execute(
{ id: 'unknown-notebook-id' },
{ toolCallId: 'test', messages: [], context: {} }
)
).rejects.toThrow(/unknown-notebook-id/)
})
it('is isolated per call to getMockTools', async () => {
const firstCall = await getMockTools(undefined)
if (!firstCall.create_notebook.execute) throw new Error('execute is undefined')
await firstCall.create_notebook.execute(
{
name: 'Ephemeral notebook',
content: { schema_version: 1, cells: [{ _tag: 'markdown_cell', text: 'hi' }] },
},
{ toolCallId: 'test', messages: [], context: {} }
)
const secondCall = await getMockTools(undefined)
if (!secondCall.list_notebooks.execute) throw new Error('execute is undefined')
const result = await secondCall.list_notebooks.execute(
{ limit: 20 },
{ toolCallId: 'test', messages: [], context: {} }
)
expect(result.notebooks.map((notebook) => notebook.name)).toEqual([
'Auth health check',
'Edge function error triage',
])
})
})
})