diff --git a/apps/studio/evals/assistant.eval.ts b/apps/studio/evals/assistant.eval.ts index 2259b5d8bf9..b20500e7345 100644 --- a/apps/studio/evals/assistant.eval.ts +++ b/apps/studio/evals/assistant.eval.ts @@ -37,32 +37,22 @@ Eval('Assistant', { const modelResponse = await getModel({ provider: 'openai', modelEntry }) if (modelResponse.error) throw modelResponse.error - // Owns the lifecycle of the remote MCP client opened inside getMockTools: - // aborting once generation is done closes that connection. - const toolsAbortController = new AbortController() - try { - const result = await generateAssistantResponse({ - ...modelResponse.modelParams, - isExplorerEnabled: true, - messages: [ - { - id: '1', - role: 'user', - parts: [{ type: 'text', text: input.prompt }], - }, - ], - tools: await getMockTools( - input.mockTables ? { list_tables: input.mockTables } : undefined, - toolsAbortController.signal - ), - }) + const result = await generateAssistantResponse({ + ...modelResponse.modelParams, + isExplorerEnabled: true, + messages: [ + { + id: '1', + role: 'user', + parts: [{ type: 'text', text: input.prompt }], + }, + ], + tools: await getMockTools(input.mockTables ? { list_tables: input.mockTables } : undefined), + }) - const finishReason = await result.finishReason - const steps = await result.steps - return { finishReason, transcript: buildTranscript(input.prompt, steps) } - } finally { - toolsAbortController.abort() - } + const finishReason = await result.finishReason + const steps = await result.steps + return { finishReason, transcript: buildTranscript(input.prompt, steps) } }, scores: [ toolUsageScorer, diff --git a/apps/studio/evals/preflight.ts b/apps/studio/evals/preflight.ts index a8889540e2c..a0aef2dce70 100644 --- a/apps/studio/evals/preflight.ts +++ b/apps/studio/evals/preflight.ts @@ -1,48 +1,58 @@ /** - * Eval preflight — MCP connectivity check. + * Eval preflight — search_docs connectivity check. * * The assistant eval harness (`getMockTools`) mocks every tool except - * `search_docs`, which it sources from a real MCP server. If that connection is - * broken (endpoint down, bad/expired token, contract drift, missing package), - * evals fail deep inside a Braintrust run with an opaque per-case error. + * `search_docs`, which is a self-contained tool that calls the public Supabase + * docs GraphQL API directly (no MCP server, no access token). If that call is + * broken (endpoint down, contract drift, missing tool), evals fail deep inside + * a Braintrust run with an opaque per-case error. * - * This preflight exercises the exact same path and fails fast with an - * actionable message, so a broken MCP connection is caught up front when the - * eval job runs (e.g. on push). Keep it in lockstep with how `getMockTools` - * obtains `search_docs` — if that switches to the remote client (see AI-897), - * switch this too. + * This preflight exercises the exact same tool and fails fast with an + * actionable message, so a broken docs API connection is caught up front when + * the eval job runs (e.g. on push). Keep it in lockstep with how `getMockTools` + * obtains `search_docs` — both use `createSearchDocsTool` from + * `lib/ai/tools/search-docs-tool`. */ -import { createInProcessSupabaseMCPClient } from '@/lib/ai/supabase-mcp' +import { createSearchDocsTool } from '@/lib/ai/tools/search-docs-tool' async function runPreflight() { - let client: Awaited> | undefined + const searchDocs = await createSearchDocsTool() - try { - client = await createInProcessSupabaseMCPClient({ - accessToken: 'mock-access-token', - projectRef: 'mock-project-ref', - }) - - const tools = await client.tools() - - if (!tools || !('search_docs' in tools)) { - throw new Error( - 'Connected to the MCP server but `search_docs` was not returned. ' + - 'The tool contract may have drifted, or the server is misconfigured.' - ) - } - - console.log('✅ Eval MCP preflight OK — connected and `search_docs` is available.') - } finally { - await client?.close().catch(() => {}) + if (!searchDocs?.execute) { + throw new Error( + '`search_docs` is missing from the eval harness. The tool contract may have ' + + 'drifted, or `createSearchDocsTool` was removed from lib/ai/tools/search-docs-tool.' + ) } + + const output = (await searchDocs.execute( + { + graphql_query: + '{ searchDocs(query: "row level security", limit: 1) { nodes { title href } } }', + }, + { toolCallId: 'preflight', messages: [], context: {} } + )) as { content: Array<{ type?: 'text'; text: string }> } + + // Validate the MCP text-content shape the scorers parse + // (mcpTextContentSpanOutputSchema / docsFaithfulnessScorer). + const content = output?.content + const text = content?.[0]?.text + if (!Array.isArray(content) || content[0]?.type !== 'text' || typeof text !== 'string' || !text) { + throw new Error( + '`search_docs` returned an unexpected shape. Expected MCP text content ' + + '({ content: [{ type: "text", text: string }] }) but got: ' + + JSON.stringify(output ?? null) + ) + } + + console.log('✅ Eval preflight OK — `search_docs` reaches the public docs GraphQL API.') } runPreflight().catch((error) => { console.error( - '❌ Eval MCP preflight failed — the eval harness cannot reach the MCP server, ' + - 'so evals would fail. Check NEXT_PUBLIC_MCP_URL, the access token, and the ' + - '@supabase/mcp-server-supabase dependency.' + '❌ Eval preflight failed — `search_docs` cannot reach the public docs GraphQL API, ' + + 'so evals would fail. Check NEXT_PUBLIC_CONTENT_API_URL (if set) and the default ' + + 'endpoint https://supabase.com/docs/api/graphql.' ) console.error(error instanceof Error ? error.message : error) process.exit(1) diff --git a/apps/studio/lib/ai/supabase-mcp.ts b/apps/studio/lib/ai/supabase-mcp.ts index f29f4d07356..afca5f6fc5f 100644 --- a/apps/studio/lib/ai/supabase-mcp.ts +++ b/apps/studio/lib/ai/supabase-mcp.ts @@ -79,54 +79,3 @@ export async function createSupabaseMCPClient({ return client } - -/** - * In-process MCP client used by the eval harness (`getMockTools`, - * `evals/preflight.ts`) so evals stay hermetic — no live remote endpoint or - * real access token needed. Not used by the production assistant, which always - * talks to the remote MCP server (`createSupabaseMCPClient`). - * - * Instantiates `@supabase/mcp-server-supabase` in-process and connects to it over - * an in-memory transport. The heavy server package is imported dynamically so it - * is code-split into its own chunk and stays out of the remote path's bundle. - * - * TODO(AI-897): point evals at the remote MCP server instead and delete this. - */ -export async function createInProcessSupabaseMCPClient({ - accessToken, - projectRef, -}: { - accessToken: string - projectRef: string -}) { - // Dynamic imports keep the in-process server + its transport out of the - // production assistant bundle, so they're loaded only when the eval harness - // actually calls this function. - // `.js` is required for esbuild ESM resolution. - const { InMemoryTransport } = await import('@modelcontextprotocol/sdk/inMemory.js') - const { createSupabaseMcpServer } = await import('@supabase/mcp-server-supabase') - const { createSupabaseApiPlatform } = await import('@supabase/mcp-server-supabase/platform/api') - const { API_URL } = await import('@/lib/constants') - - const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair() - - // Instantiate the MCP server and connect to its transport - const apiUrl = API_URL?.replace('/platform', '') - const server = createSupabaseMcpServer({ - platform: createSupabaseApiPlatform({ - accessToken, - apiUrl, - }), - contentApiUrl: process.env.NEXT_PUBLIC_CONTENT_API_URL, - projectId: projectRef, - readOnly: true, - }) - await server.connect(serverTransport) - - const client = await createMCPClient({ - name: SOURCE_NAME, - transport: clientTransport, - }) - - return client -} diff --git a/apps/studio/lib/ai/tools/mcp-tools.ts b/apps/studio/lib/ai/tools/mcp-tools.ts index 8435e3ecbd4..d14043528a7 100644 --- a/apps/studio/lib/ai/tools/mcp-tools.ts +++ b/apps/studio/lib/ai/tools/mcp-tools.ts @@ -50,11 +50,9 @@ export const getMcpTools = async ({ // when the request ends. The caller owns that lifecycle via this signal. signal: AbortSignal }) => { - // Connect to the remote MCP server over HTTP and fetch its tools, which - // replace the old local tools. The legacy in-process server is no longer a - // production transport (eval-only now, see `createInProcessSupabaseMCPClient`), - // so this is unconditional. A remote failure (outage, timeout, auth) degrades - // to the remaining tools in `getTools` rather than breaking the assistant. + // Connect to the remote MCP server over HTTP and fetch its tools, replacing + // the local tools. A remote failure (outage, timeout, auth) degrades to the + // remaining tools in `getTools` rather than breaking the assistant. const mcpClient = await createSupabaseMCPClient({ accessToken, projectRef, diff --git a/apps/studio/lib/ai/tools/mock-tools.test.ts b/apps/studio/lib/ai/tools/mock-tools.test.ts index 52e3f82c90a..f560fb00ce6 100644 --- a/apps/studio/lib/ai/tools/mock-tools.test.ts +++ b/apps/studio/lib/ai/tools/mock-tools.test.ts @@ -1,62 +1,48 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import { getMockTools, MOCK_NOTEBOOKS_DATA } from './mock-tools' import { getNotebookTools } from './notebook-tools' import type { AgentNotebook } from '@/data/content/notebooks/notebook-schema' -import { createInProcessSupabaseMCPClient } from '@/lib/ai/supabase-mcp' +import { createSearchDocsTool } from '@/lib/ai/tools/search-docs-tool' +import type * as SearchDocsToolModule from '@/lib/ai/tools/search-docs-tool' -// The one real tool in the eval harness (search_docs) is sourced from an -// in-process MCP client. Mock that client so this test stays hermetic and -// guards the wiring, not a live connection. -vi.mock('@/lib/ai/supabase-mcp', () => ({ - createInProcessSupabaseMCPClient: vi.fn(), +// search_docs normally fetches the live docs GraphQL schema on construction +// (see search-docs-tool.ts), which would make every test in this file depend +// on the public docs API. Mock it with a deterministic local fixture instead +// — live connectivity (including the schema fetch) is covered separately by +// evals/preflight.ts. The regression-guard test below overrides this default +// to verify the missing-tool guard fires. +vi.mock('@/lib/ai/tools/search-docs-tool', () => ({ + createSearchDocsTool: vi.fn().mockResolvedValue({ + description: 'Search the Supabase documentation using GraphQL.', + execute: async () => ({ content: [{ type: 'text' as const, text: '{}' }] }), + } as unknown as SearchDocsToolModule.SearchDocsTool), })) -const SEARCH_DOCS = { description: 'search the docs' } - describe('ai/tools/mock-tools getMockTools', () => { - let close: ReturnType - let tools: ReturnType + it('wires the mocked search_docs tool through from the shared module, alongside the deterministic mocks', async () => { + const result = await getMockTools(undefined) - beforeEach(() => { - close = vi.fn().mockResolvedValue(undefined) - tools = vi.fn().mockResolvedValue({ search_docs: SEARCH_DOCS }) - vi.mocked(createInProcessSupabaseMCPClient).mockResolvedValue({ tools, close } as any) - }) - - it('sources the real search_docs from the in-process MCP server alongside the deterministic mocks', async () => { - const result = await getMockTools(undefined, new AbortController().signal) - - expect(createInProcessSupabaseMCPClient).toHaveBeenCalledTimes(1) - // The real tool, wired through from the MCP client - expect(result).toHaveProperty('search_docs', SEARCH_DOCS) + // The mocked tool, wired through from the shared module + expect(result.search_docs).toBeDefined() + expect(result.search_docs.description).toContain('Search the Supabase documentation') + expect(typeof result.search_docs.execute).toBe('function') // A couple of the deterministic mocks, to confirm the merge expect(result).toHaveProperty('list_tables') expect(result).toHaveProperty('query_logs') }) - // This is the regression guard: if the eval's MCP wiring breaks (contract - // drift, or a refactor that stops sourcing search_docs — e.g. the future - // AI-897 removal of the in-process client), fail loudly in normal CI instead - // of only surfacing during an opt-in Braintrust eval run. - it('throws a clear error when the MCP server does not expose search_docs', async () => { - tools.mockResolvedValueOnce({}) - - await expect(getMockTools(undefined, new AbortController().signal)).rejects.toThrow( - 'search_docs tool not available from MCP server' + // This is the regression guard: if a refactor stops sourcing search_docs + // (drops it from getMockTools, or createSearchDocsTool breaks), fail loudly + // in normal CI instead of only surfacing during an opt-in Braintrust eval run. + it('throws a clear error when search_docs is missing from the harness tools', async () => { + vi.mocked(createSearchDocsTool).mockResolvedValueOnce( + undefined as unknown as SearchDocsToolModule.SearchDocsTool ) - }) - it('closes the MCP client when the caller aborts the signal', async () => { - const controller = new AbortController() - - await getMockTools(undefined, controller.signal) - // Connection stays open until generation ends (search_docs runs during it) - expect(close).not.toHaveBeenCalled() - - controller.abort() - await Promise.resolve() - expect(close).toHaveBeenCalledTimes(1) + await expect(getMockTools(undefined)).rejects.toThrow( + 'search_docs tool is missing from the eval harness' + ) }) describe('notebook tools', () => { @@ -64,7 +50,7 @@ describe('ai/tools/mock-tools getMockTools', () => { const EDGE_FUNCTION_NOTEBOOK_ID = MOCK_NOTEBOOKS_DATA[1].id it('list_notebooks reflects the two seeded fixtures', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined') const result = await mockTools.list_notebooks.execute( @@ -82,7 +68,7 @@ describe('ai/tools/mock-tools getMockTools', () => { }) it('get_notebook resolves cells in order and rejects an unknown id', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) if (!mockTools.get_notebook.execute) throw new Error('execute is undefined') const result = await mockTools.get_notebook.execute( @@ -113,7 +99,7 @@ describe('ai/tools/mock-tools getMockTools', () => { }) it('shares deterministic run_notebook output with eval models', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) if (!mockTools.run_notebook.execute) throw new Error('execute is undefined') if (!mockTools.run_notebook.toModelOutput) throw new Error('toModelOutput is undefined') @@ -136,14 +122,14 @@ describe('ai/tools/mock-tools getMockTools', () => { }) it('overrides create_notebook needsApproval to false, unlike the real tool', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) expect(getNotebookTools().create_notebook.needsApproval).toBe(true) expect(mockTools.create_notebook.needsApproval).toBe(false) }) it('create_notebook stores a new notebook visible via get_notebook and list_notebooks', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) if (!mockTools.create_notebook.execute) throw new Error('execute is undefined') if (!mockTools.get_notebook.execute) throw new Error('execute is undefined') if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined') @@ -179,14 +165,14 @@ describe('ai/tools/mock-tools getMockTools', () => { }) it('overrides update_notebook needsApproval to false, unlike the real tool', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) expect(getNotebookTools().update_notebook.needsApproval).toBe(true) expect(mockTools.update_notebook.needsApproval).toBe(false) }) it('update_notebook inserts and deletes cells, and list_notebooks reflects the new cell count', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) if (!mockTools.get_notebook.execute) throw new Error('execute is undefined') if (!mockTools.update_notebook.execute) throw new Error('execute is undefined') if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined') @@ -233,7 +219,7 @@ describe('ai/tools/mock-tools getMockTools', () => { }) it('update_notebook rejects an unknown cell_id without mutating the notebook', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) if (!mockTools.get_notebook.execute) throw new Error('execute is undefined') if (!mockTools.update_notebook.execute) throw new Error('execute is undefined') @@ -261,14 +247,14 @@ describe('ai/tools/mock-tools getMockTools', () => { }) it('overrides delete_notebook needsApproval to false, unlike the real tool', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) expect(getNotebookTools().delete_notebook.needsApproval).toBe(true) expect(mockTools.delete_notebook.needsApproval).toBe(false) }) it('delete_notebook removes the notebook, and list_notebooks no longer returns it', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) if (!mockTools.delete_notebook.execute) throw new Error('execute is undefined') if (!mockTools.list_notebooks.execute) throw new Error('execute is undefined') @@ -286,7 +272,7 @@ describe('ai/tools/mock-tools getMockTools', () => { }) it('delete_notebook rejects an unknown id', async () => { - const mockTools = await getMockTools(undefined, new AbortController().signal) + const mockTools = await getMockTools(undefined) if (!mockTools.delete_notebook.execute) throw new Error('execute is undefined') await expect( @@ -298,7 +284,7 @@ describe('ai/tools/mock-tools getMockTools', () => { }) it('is isolated per call to getMockTools', async () => { - const firstCall = await getMockTools(undefined, new AbortController().signal) + const firstCall = await getMockTools(undefined) if (!firstCall.create_notebook.execute) throw new Error('execute is undefined') await firstCall.create_notebook.execute( @@ -309,7 +295,7 @@ describe('ai/tools/mock-tools getMockTools', () => { { toolCallId: 'test', messages: [], context: {} } ) - const secondCall = await getMockTools(undefined, new AbortController().signal) + const secondCall = await getMockTools(undefined) if (!secondCall.list_notebooks.execute) throw new Error('execute is undefined') const result = await secondCall.list_notebooks.execute( diff --git a/apps/studio/lib/ai/tools/mock-tools.ts b/apps/studio/lib/ai/tools/mock-tools.ts index e6c81a02448..a329cc61fa5 100644 --- a/apps/studio/lib/ai/tools/mock-tools.ts +++ b/apps/studio/lib/ai/tools/mock-tools.ts @@ -1,5 +1,5 @@ import assert from 'node:assert' -import { tool, type ToolExecutionOptions, type ToolSet } from 'ai' +import { tool, type ToolExecutionOptions } from 'ai' import { z } from 'zod' import { getStudioTools } from '../tools/studio-tools' @@ -15,7 +15,7 @@ import type { CellWire, NotebookWire, } from '@/data/content/notebooks/notebook-schema' -import { createInProcessSupabaseMCPClient } from '@/lib/ai/supabase-mcp' +import { createSearchDocsTool } from '@/lib/ai/tools/search-docs-tool' const listTablesInputSchema = z.object({ schemas: z.array(z.string()).describe('The schema names to list.'), @@ -591,32 +591,15 @@ export type MockToolOverrides = { * These mirror tool names used in prompts so the model can call them, * but return stable, static data for repeatable tests. * - * Note: search_docs uses the real implementation + * Note: search_docs uses the real implementation. */ -export async function getMockTools(overrides: MockToolOverrides | undefined, signal: AbortSignal) { +export async function getMockTools(overrides: MockToolOverrides | undefined) { const mockedStudioTools = createMockedStudioTools() const notebookStore = createMockNotebookStore() - // Every tool here is a deterministic mock except `search_docs`, which uses the - // real implementation. We source it from an in-process MCP server directly - // (rather than `getMcpTools`, which always talks to the remote server) so the - // eval harness stays hermetic: the in-process server needs no live remote - // endpoint or real access token. See AI-897 for how to point evals at the - // remote MCP server instead. - const mcpClient = await createInProcessSupabaseMCPClient({ - accessToken: 'mock-access-token', - projectRef: 'mock-project-ref', - }) - // The caller owns this signal and aborts it once generation is done, which - // closes the client opened here (search_docs executes during generation, so - // the connection must stay open until then). - signal.addEventListener('abort', () => void mcpClient.close().catch(() => {}), { once: true }) + const search_docs = await createSearchDocsTool() - const { search_docs } = (await mcpClient.tools()) as ToolSet - - assert(search_docs, 'search_docs tool not available from MCP server') - - return { + const tools = { ...mockedStudioTools, search_docs, list_tables: createMockListTablesTool(overrides?.list_tables), @@ -627,4 +610,8 @@ export async function getMockTools(overrides: MockToolOverrides | undefined, sig list_policies: createMockListPoliciesTool(), ...createMockNotebookTools(notebookStore), } + + assert(tools.search_docs, 'search_docs tool is missing from the eval harness') + + return tools } diff --git a/apps/studio/lib/ai/tools/search-docs-tool.ts b/apps/studio/lib/ai/tools/search-docs-tool.ts new file mode 100644 index 00000000000..b39a0401ef6 --- /dev/null +++ b/apps/studio/lib/ai/tools/search-docs-tool.ts @@ -0,0 +1,117 @@ +import { tool, type Tool } from 'ai' +import gqlmin from 'gqlmin' +import { z } from 'zod' + +const searchDocsInputSchema = z.object({ + graphql_query: z.string().describe('A valid GraphQL query against the Supabase docs API.'), +}) + +const CONTENT_API_URL = + process.env.NEXT_PUBLIC_CONTENT_API_URL ?? 'https://supabase.com/docs/api/graphql' + +/** + * Sends a GraphQL query to the public Supabase docs API. + * + * Mirrors the @supabase/mcp-server-supabase content API client: GET + * `?query=` with `Accept: application/json`, returning the + * GraphQL envelope's `data` field. + */ +async function queryContentApiGraphQL(graphqlQuery: string): Promise { + const url = new URL(CONTENT_API_URL) + url.searchParams.set('query', graphqlQuery) + + const response = await fetch(url, { + method: 'GET', + headers: { + Accept: 'application/json', + 'User-Agent': 'supabase-studio-evals', + }, + // A stalled connection or response body would otherwise hang getMockTools + // and preflight indefinitely. + signal: AbortSignal.timeout(10_000), + }) + if (!response.ok) { + throw new Error(`Failed to fetch Supabase Content API: HTTP status ${response.status}`) + } + + const body = (await response.json()) as { + data?: unknown + errors?: Array<{ message: string; locations?: Array<{ line: number; column: number }> }> + } + if (body.errors?.length) { + throw new Error( + `Supabase Content API GraphQL error: ${body.errors + .map((error) => { + const location = error.locations?.[0] + return `${error.message} (line ${location?.line ?? 'unknown'}, column ${location?.column ?? 'unknown'})` + }) + .join(', ')}` + ) + } + if (!body.data) { + throw new Error('Supabase Content API returned no data') + } + + return body.data +} + +const STATIC_DESCRIPTION = + 'Search the Supabase documentation using GraphQL. Must be a valid GraphQL query. ' + + 'You should default to calling this even if you think you already know the answer, ' + + 'since the documentation is always being updated.' + +/** + * Fetches and minifies the Content API's own GraphQL schema (via the `{ + * schema }` query it exposes), mirroring + * `@supabase/mcp-server-supabase`'s `loadSchema` so the eval tool's + * description is as close as practical to what production Assistant sees. + */ +async function loadContentApiSchema(): Promise { + const data = (await queryContentApiGraphQL('{ schema }')) as { schema?: unknown } + if (typeof data.schema !== 'string' || !data.schema) { + throw new Error('Supabase Content API `{ schema }` query returned no schema string') + } + return gqlmin(data.schema) +} + +/** + * Builds the tool description with the live GraphQL schema embedded, so the + * model has the same schema context production's `search_docs` gives it (see + * `@supabase/mcp-server-supabase`'s `docs-tools.ts`). + * + * Schema loading is required: running an eval without the schema makes the + * model's GraphQL queries untrustworthy and can hide a real docs-search + * regression behind fallback results. + */ +async function buildDescription(): Promise { + const schema = await loadContentApiSchema() + return `${STATIC_DESCRIPTION}\n\nBelow is the GraphQL schema for this tool:\n\n${schema}` +} + +/** + * Self-contained `search_docs` tool for the eval harness: calls the public + * docs GraphQL API directly, so no MCP client or access token is needed. + * Emits the MCP text-content shape the scorers parse + * (mcpTextContentSpanOutputSchema / docsFaithfulnessScorer): + * `{ content: [{ type: 'text', text: JSON.stringify({ result }) }] }`. + * + * `description` is resolved before construction (the `ai` package's `tool()` + * only accepts a plain string, not an async function like the MCP SDK's + * `docs-tools.ts` uses), so this factory is async. + */ +export type SearchDocsTool = Tool< + z.infer, + { content: Array<{ type: 'text'; text: string }> } +> + +export async function createSearchDocsTool(): Promise { + const description = await buildDescription() + return tool({ + description, + inputSchema: searchDocsInputSchema, + execute: async ({ graphql_query }: { graphql_query: string }) => { + const result = await queryContentApiGraphQL(graphql_query) + return { content: [{ type: 'text' as const, text: JSON.stringify({ result }) }] } + }, + }) +} diff --git a/apps/studio/package.json b/apps/studio/package.json index 0c6d2b9042d..c077f594cc8 100644 --- a/apps/studio/package.json +++ b/apps/studio/package.json @@ -108,6 +108,7 @@ "framer-motion": "^11.18.2", "fuse.js": "^7.4.0", "generate-password-browser": "^1.1.0", + "gqlmin": "^0.3.1", "graphiql": "^5.2.2", "html-to-image": "^1.11.13", "http-status": "^2.1.0", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index ebe7509dc76..3574029d7b7 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -1155,6 +1155,9 @@ importers: generate-password-browser: specifier: ^1.1.0 version: 1.1.0 + gqlmin: + specifier: ^0.3.1 + version: 0.3.1 graphiql: specifier: ^5.2.2 version: 5.2.2(@emotion/is-prop-valid@1.4.0)(@types/node@22.13.14)(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(graphql-ws@5.14.1(graphql@16.11.0))(graphql@16.11.0)(immer@10.1.1)(react-dom@19.2.6(react@19.2.6))(react@19.2.6)(use-sync-external-store@1.6.0(react@19.2.6))