feat(assistant): lazy load topic knowledge via load_knowledge tool (#44296)

Moves knowledge (RLS, Edge Functions, PostgreSQL best practices,
Realtime) out of the static system prompt and into a `load_knowledge`
tool the model calls on demand, reducing prompt bloat. This is a
temporary stopgap until the [standard Supabase
agent-skills](https://github.com/supabase/agent-skills) are ready for
integration in Assistant.

- New always-available `load_knowledge` tool added to
`rendering-tools.ts`
- Updated `Message.Parts.tsx` so the "Ran load_knowledge" chip renders
in chat
- System prompt replaces the four knowledge blobs with an `## Available
Knowledge` block and is hardened to load knowledge for given topics
- New "Knowledge Usage" scorer and `requiredKnowledge` assertions check
that knowledge loads as expected in test scenarios
- Filters GraphQL error responses out of `output.docs` before
faithfulness scoring to reduce noise


See "Knowledge Usage" scoring 100% in evals with no major regressions:
https://github.com/supabase/supabase/pull/44296#issuecomment-4145760236

Sample trace showing the tool in action
([Braintrust](https://www.braintrust.dev/app/supabase.io/p/Assistant/trace?object_type=project_logs&object_id=5a8d02e5-b3b6-40cc-ba76-ecee286478f4&r=351a11c8-9cb7-4945-93ad-d11e8cc2e3e1&s=351a11c8-9cb7-4945-93ad-d11e8cc2e3e1))

<img width="2192" height="1730" alt="CleanShot 2026-03-30 at 13 53
59@2x"
src="https://github.com/user-attachments/assets/f483767c-34e0-401c-8089-5b9834fe696a"
/>


**References**
- https://ai-sdk.dev/cookbook/guides/agent-skills

Closes AI-508

<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->

## Summary by CodeRabbit

* **New Features**
* Added dynamic knowledge loading capability enabling the AI assistant
to retrieve on-demand information about PostgreSQL best practices, Row
Level Security, Edge Functions, and Realtime.

* **Bug Fixes**
* Improved search results filtering to exclude error responses in tool
outputs.

* **Tests**
  * Enhanced evaluation metrics with knowledge usage scoring.
* Expanded test dataset cases to validate knowledge requirement
handling.

<!-- end of auto-generated comment: release notes by coderabbit.ai -->
This commit is contained in:
Matt Rossman authored and GitHub committed 2026-04-02 16:09:06 -04:00
1 parent 83ccf44461
commit 82deff37de
12 files changed
+150 -46

No files matched your search

@@ -236,7 +236,8 @@ export function MessagePartSwitcher({
}
case 'tool-list_policies':
case 'tool-search_docs':
case 'tool-get_active_incidents': {
case 'tool-get_active_incidents':
case 'tool-load_knowledge': {
return <MessagePart.Tool toolPart={part} />
}
case 'reasoning':
+2
View File
@@ -13,6 +13,7 @@ import {
correctnessScorer,
docsFaithfulnessScorer,
goalCompletionScorer,
knowledgeUsageScorer,
toolUsageScorer,
urlValidityScorer,
} from './scorer'
@@ -49,6 +50,7 @@ Eval('Assistant', {
},
scores: [
toolUsageScorer,
knowledgeUsageScorer,
sqlSyntaxScorer,
sqlIdentifierQuotingScorer,
goalCompletionScorer,
+33
View File
@@ -19,6 +19,7 @@ export const dataset: AssistantEvalCase[] = [
input: { prompt: 'Create a new table "foods" with columns for "name" and "color"' },
expected: {
requiredTools: ['execute_sql'],
requiredKnowledge: ['pg_best_practices'],
},
metadata: { category: ['sql_generation', 'schema_design'] },
},
@@ -29,6 +30,7 @@ export const dataset: AssistantEvalCase[] = [
},
expected: {
requiredTools: ['execute_sql'],
requiredKnowledge: ['pg_best_practices'],
},
metadata: { category: ['sql_generation'] },
},
@@ -36,6 +38,7 @@ export const dataset: AssistantEvalCase[] = [
input: { prompt: 'Create an index on the projects table for the name column' },
expected: {
requiredTools: ['execute_sql'],
requiredKnowledge: ['pg_best_practices'],
},
metadata: { category: ['sql_generation', 'database_optimization'] },
},
@@ -87,6 +90,7 @@ export const dataset: AssistantEvalCase[] = [
},
expected: {
requiredTools: ['execute_sql'],
requiredKnowledge: ['pg_best_practices'],
},
metadata: {
category: ['sql_generation'],
@@ -100,6 +104,7 @@ export const dataset: AssistantEvalCase[] = [
},
expected: {
requiredTools: ['execute_sql'],
requiredKnowledge: ['pg_best_practices'],
},
metadata: {
category: ['sql_generation', 'schema_design'],
@@ -128,6 +133,34 @@ export const dataset: AssistantEvalCase[] = [
'Verifies template URLs like https://<project-ref>.supabase.co/auth/v1/callback are excluded from URL validity scoring',
},
},
{
input: { prompt: "How do I write an RLS policy to restrict access to a user's own rows?" },
expected: {
requiredTools: ['list_tables', 'list_policies', 'execute_sql'],
requiredKnowledge: ['rls'],
},
metadata: { category: ['rls_policies'] },
},
{
input: { prompt: 'Write an edge function that sends a welcome email when a user signs up' },
expected: {
requiredTools: ['deploy_edge_function'],
requiredKnowledge: ['edge_functions'],
},
metadata: { category: ['edge_functions'] },
},
{
input: { prompt: 'What indexes should I add to improve query performance?' },
expected: { requiredKnowledge: ['pg_best_practices'] },
metadata: { category: ['database_optimization'] },
},
{
input: { prompt: 'How do I subscribe to realtime changes on a table?' },
expected: {
requiredKnowledge: ['realtime'],
},
metadata: { category: ['general_help'] },
},
{
input: {
prompt:
+10 -1
View File
@@ -27,7 +27,16 @@ function parseToolCall(
case 'search_docs': {
const content = toolResult.output?.content
if (!content || !Array.isArray(content)) return {}
const docs = content.map((item) => item?.text).filter((text) => typeof text === 'string')
const docs = content
.map((item) => item?.text)
.filter((text) => {
if (typeof text !== 'string') return false
try {
return !JSON.parse(text)?.error
} catch {
return true
}
})
if (docs.length === 0) return {}
return { docs }
}
+35
View File
@@ -28,6 +28,7 @@ export type AssistantEvalOutput = {
export type Expected = {
requiredTools?: string[]
requiredKnowledge?: string[]
correctAnswer?: string
}
@@ -92,6 +93,40 @@ export const toolUsageScorer: EvalScorer<
}
}
export const knowledgeUsageScorer: EvalScorer<
AssistantEvalInput,
AssistantEvalOutput,
Expected
> = async ({ output, expected }) => {
if (!expected.requiredKnowledge) return null
const loadedKnowledge = output.steps
.flatMap((step) => step.toolCalls)
.filter((call) => call.toolName === 'load_knowledge')
.flatMap((call) => {
const input = call.input
if (
typeof input !== 'object' ||
input === null ||
!('name' in input) ||
typeof input.name !== 'string'
)
return []
return [input.name]
})
const presentCount = expected.requiredKnowledge.filter((knowledge) =>
loadedKnowledge.includes(knowledge)
).length
const totalCount = expected.requiredKnowledge.length
const ratio = totalCount === 0 ? 1 : presentCount / totalCount
return {
name: 'Knowledge Usage',
score: ratio,
}
}
const concisenessEvaluator = LLMClassifierFromTemplate<{ input: string }>({
name: 'Conciseness',
promptTemplate: stripIndent`
@@ -14,16 +14,7 @@ import { buildAssistantEvalOutput } from 'evals/output'
import type { AssistantEvalInput, AssistantEvalOutput } from 'evals/scorer'
import type { AiOptInLevel } from 'hooks/misc/useOrgOptedIntoAi'
import { IS_TRACING_ENABLED } from 'lib/ai/braintrust-logger'
import {
CHAT_PROMPT,
EDGE_FUNCTION_PROMPT,
GENERAL_PROMPT,
LIMITATIONS_PROMPT,
PG_BEST_PRACTICES,
REALTIME_PROMPT,
RLS_PROMPT,
SECURITY_PROMPT,
} from 'lib/ai/prompts'
import { CHAT_PROMPT, GENERAL_PROMPT, LIMITATIONS_PROMPT, SECURITY_PROMPT } from 'lib/ai/prompts'
import { sanitizeMessagePart } from 'lib/ai/tools/tool-sanitizer'
const { streamText: tracedStreamText } = wrapAISDK(ai)
@@ -70,7 +61,7 @@ export async function generateAssistantResponse({
const run = async (span?: Span) => {
// Only returns last 7 messages
// Filters out tools with invalid states
// Filters out tool outputs based on opt-in level using renderingToolOutputParser
// Filters out tool outputs based on opt-in level
const messages = (rawMessages || []).slice(-7).map((msg) => {
if (msg && msg.role === 'assistant' && 'results' in msg) {
const cleanedMsg = { ...msg }
@@ -103,12 +94,16 @@ export async function generateAssistantResponse({
const system = source`
${GENERAL_PROMPT}
${CHAT_PROMPT}
${PG_BEST_PRACTICES}
${RLS_PROMPT}
${EDGE_FUNCTION_PROMPT}
${REALTIME_PROMPT}
${SECURITY_PROMPT}
${LIMITATIONS_PROMPT}
## Available Knowledge
Before writing SQL or answering questions about the following topics, call \`load_knowledge\` to load detailed knowledge:
- \`pg_best_practices\` — PostgreSQL best practices. Always load before writing any SQL, even simple queries.
- \`rls\` — Row Level Security policies
- \`edge_functions\` — Supabase Edge Functions
- \`realtime\` — Supabase Realtime
`
// Note: these must be of type `CoreMessage` to prevent AI SDK from stripping `providerOptions`
+3 -3
View File
@@ -577,7 +577,7 @@ Support the user by:
Before using tools, determine the task type (not exhaustive):
**For questions about Supabase features/capabilities/limitations, or tasks**
- Use \`search_docs\` FIRST before making claims or gathering database context
- Use \`search_docs\` and/or \`load_knowledge\` FIRST before making claims or gathering database context
- Examples: "How do I...", "Can Supabase...", "Is it possible to..."
**For database interactions:**
@@ -598,7 +598,7 @@ Before using tools, determine the task type (not exhaustive):
- Never use tables in responses and use emojis minimally.
If a tool output should be summarized, integrate the information clearly into the Markdown response. When a tool call returns an error, provide a concise inline explanation or summary of the error. Quote large error messages only if essential to user action. Upon each tool call or code edit, validate the result in 1–2 lines and proceed or self-correct if validation fails.
## Documentation Search
- When users ask about Supabase features, limitations, or capabilities, use \`search_docs\` BEFORE attempting database operations or making claims
- When users ask about Supabase features, limitations, or capabilities, use \`search_docs\` BEFORE attempting database operations or making claims. This DOES NOT replace the need for \`load_knowledge\`.
- If \`search_docs\` reveals a limitation, inform the user immediately without gathering database context
- Do not make claims unsupported by documentation
`
@@ -653,7 +653,7 @@ export const OUTPUT_ONLY_PROMPT = `
- **CRITICAL: Final message must be only raw code needed to fulfill the request.**
- **If you lack privelages to use a tool, do your best to generate the code without it. No need to explain why you couldn't use the tool.**
- **No explanations, no commentary, no markdown**. Do not wrap output in backticks.
- **Do not call UI display tools** (no \`display_query\`, no \`display_edge_function\").
- **Do not call UI display tools** (no \`execute_sql\`, no \`deploy_edge_function\`).
`
export const SECURITY_PROMPT = `
+3
View File
@@ -40,6 +40,8 @@ export const toolSetValidationSchema = z.record(
'getRlsKnowledge',
'getFunctions',
'getEdgeFunctionKnowledge',
'load_knowledge',
]),
basicToolSchema
)
@@ -71,6 +73,7 @@ export const TOOL_CATEGORY_MAP: Record<string, ToolCategory> = {
rename_chat: TOOL_CATEGORIES.UI,
search_docs: TOOL_CATEGORIES.UI,
get_active_incidents: TOOL_CATEGORIES.UI,
load_knowledge: TOOL_CATEGORIES.UI,
// Schema tools - MCP
list_tables: TOOL_CATEGORIES.SCHEMA,
+3 -3
View File
@@ -6,7 +6,7 @@ import { IS_PLATFORM } from 'common'
import { getIncidentTools } from './incident-tools'
import { getMcpTools } from './mcp-tools'
import { getSchemaTools } from './schema-tools'
import { getRenderingTools } from './rendering-tools'
import { getStudioTools } from './studio-tools'
export const getTools = async ({
projectRef,
@@ -23,8 +23,8 @@ export const getTools = async ({
accessToken?: string
baseUrl?: string
}) => {
// Always include rendering tools
let tools: ToolSet = getRenderingTools()
// Always include studio tools
let tools: ToolSet = getStudioTools()
// If self-hosted, only add fallback tools
if (!IS_PLATFORM) {
+7 -7
View File
@@ -1,5 +1,5 @@
import { tool, type ToolSet } from 'ai'
import { getRenderingTools } from '../tools/rendering-tools'
import { getStudioTools } from '../tools/studio-tools'
import { z } from 'zod'
import { getMcpTools } from 'lib/ai/tools/mcp-tools'
import assert from 'node:assert'
@@ -142,11 +142,11 @@ const MOCK_LOGS_DATA = [
},
]
function createMockedRenderingTools() {
const renderingTools = getRenderingTools()
function createMockedStudioTools() {
const studioTools = getStudioTools()
return Object.fromEntries(
Object.entries(renderingTools).map(([name, baseTool]) => {
Object.entries(studioTools).map(([name, baseTool]) => {
if (typeof baseTool.execute === 'function') {
return [name, baseTool]
}
@@ -166,7 +166,7 @@ function createMockedRenderingTools() {
},
]
})
) as typeof renderingTools
) as typeof studioTools
}
function createMockListTablesTool(overrideData?: Record<string, typeof MOCK_TABLES_DATA>) {
@@ -305,7 +305,7 @@ export type MockToolOverrides = {
* Note: search_docs uses the real implementation
*/
export async function getMockTools(overrides?: MockToolOverrides) {
const mockedRenderingTools = createMockedRenderingTools()
const mockedStudioTools = createMockedStudioTools()
const { search_docs } = await getMcpTools({
accessToken: 'mock-access-token',
@@ -316,7 +316,7 @@ export async function getMockTools(overrides?: MockToolOverrides) {
assert(search_docs, 'search_docs tool not available from MCP server')
return {
...mockedRenderingTools,
...mockedStudioTools,
search_docs,
list_tables: createMockListTablesTool(overrides?.list_tables),
list_extensions: createMockListExtensionsTool(),
@@ -1,49 +1,50 @@
import { describe, expect, it } from 'vitest'
import { getRenderingTools } from './rendering-tools'
import { getStudioTools } from './studio-tools'
describe('ai/tools/rendering-tools', () => {
describe('getRenderingTools', () => {
describe('ai/tools/studio-tools', () => {
describe('getStudioTools', () => {
it('should return an object with tool definitions', () => {
const tools = getRenderingTools()
const tools = getStudioTools()
expect(tools).toBeDefined()
expect(typeof tools).toBe('object')
})
it('should include execute_sql tool', () => {
const tools = getRenderingTools()
const tools = getStudioTools()
expect(tools.execute_sql).toBeDefined()
expect(tools.execute_sql.description).toContain('execute a SQL statement')
})
it('should include deploy_edge_function tool', () => {
const tools = getRenderingTools()
const tools = getStudioTools()
expect(tools.deploy_edge_function).toBeDefined()
expect(tools.deploy_edge_function.description).toContain('deploy a Supabase Edge Function')
})
it('should include rename_chat tool', () => {
const tools = getRenderingTools()
const tools = getStudioTools()
expect(tools.rename_chat).toBeDefined()
expect(tools.rename_chat.description).toContain('Rename the current chat session')
})
it('should have exactly 3 tools', () => {
const tools = getRenderingTools()
it('should have exactly 4 tools', () => {
const tools = getStudioTools()
const toolNames = Object.keys(tools)
expect(toolNames).toHaveLength(3)
expect(toolNames).toHaveLength(4)
expect(toolNames).toContain('load_knowledge')
expect(toolNames).toContain('execute_sql')
expect(toolNames).toContain('deploy_edge_function')
expect(toolNames).toContain('rename_chat')
})
it('should have execute_sql with correct input schema fields', () => {
const tools = getRenderingTools()
const tools = getStudioTools()
const executeSqlTool = tools.execute_sql
// Check that the tool has an input schema
@@ -56,7 +57,7 @@ describe('ai/tools/rendering-tools', () => {
})
it('should have deploy_edge_function with input schema', () => {
const tools = getRenderingTools()
const tools = getStudioTools()
const deployTool = tools.deploy_edge_function
expect(deployTool.inputSchema).toBeDefined()
@@ -67,7 +68,7 @@ describe('ai/tools/rendering-tools', () => {
})
it('should have rename_chat with execute function', async () => {
const tools = getRenderingTools()
const tools = getStudioTools()
const renameTool = tools.rename_chat
expect(renameTool.execute).toBeDefined()
@@ -83,7 +84,7 @@ describe('ai/tools/rendering-tools', () => {
})
it('should validate execute_sql input schema correctly', () => {
const tools = getRenderingTools()
const tools = getStudioTools()
const schema = tools.execute_sql.inputSchema
// Check if schema is a Zod schema with safeParse
@@ -119,7 +120,7 @@ describe('ai/tools/rendering-tools', () => {
})
it('should validate rename_chat input schema correctly', () => {
const tools = getRenderingTools()
const tools = getStudioTools()
const schema = tools.rename_chat.inputSchema
// Check if schema is a Zod schema with safeParse
@@ -1,8 +1,23 @@
import { tool } from 'ai'
import {
EDGE_FUNCTION_PROMPT,
PG_BEST_PRACTICES,
REALTIME_PROMPT,
RLS_PROMPT,
} from 'lib/ai/prompts'
import { fixSqlBackslashEscapes } from 'lib/ai/util'
import { z } from 'zod'
export const getRenderingTools = () => ({
const KNOWLEDGE = {
pg_best_practices: PG_BEST_PRACTICES,
rls: RLS_PROMPT,
edge_functions: EDGE_FUNCTION_PROMPT,
realtime: REALTIME_PROMPT,
} as const
type KnowledgeName = keyof typeof KNOWLEDGE
export const getStudioTools = () => ({
execute_sql: tool({
description: 'Asks the user to execute a SQL statement and return the results',
inputSchema: z.object({
@@ -41,4 +56,14 @@ export const getRenderingTools = () => ({
return { status: 'Chat request sent to client' }
},
}),
load_knowledge: tool({
description:
'Load detailed knowledge about a Supabase topic before answering questions about it.',
inputSchema: z.object({
name: z
.enum(Object.keys(KNOWLEDGE) as [KnowledgeName, ...KnowledgeName[]])
.describe('The knowledge to load'),
}),
execute: ({ name }) => KNOWLEDGE[name],
}),
})