mirror of
https://github.com/supabase/supabase.git
synced 2026-10-09 11:25:06 +03:00
## I have read the [CONTRIBUTING.md](https://github.com/supabase/supabase/blob/master/CONTRIBUTING.md) file. YES ## What kind of change does this PR introduce? Feature (test coverage) — first PR of the notebooks-evals plan, covering FE-4086 (Evals: Assistant can list notebooks). ## What is the current behavior? `dataset.ts` has zero notebook eval cases. Separately, two small gaps block writing them: `assistant.eval.ts` never sets `isExplorerEnabled`, so `NOTEBOOKS_PROMPT` never loads into the eval task's system prompt; and `list_notebooks`' `inputSchema` has no sort parameter, even though `getContent` already supports one, so "most recent" isn't answerable. ## What is the new behavior? - `assistant.eval.ts` passes `isExplorerEnabled: true` so `NOTEBOOKS_PROMPT` loads during evals. - `list_notebooks` gains a `sort_by: 'name' | 'inserted_at'` param, forwarded to `getContent`'s `sort`. Only creation order is exposed, since the underlying API has no `updated_at` sort key. - 3 new dataset cases: basic notebook enumeration, sorting by creation time (`sort_by`/`limit` args), and a nonexistent-notebook case guarding against hallucinated results. - A unit test covering `sort_by` forwarding to the content API's query param. ## Additional context <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Added notebook sorting options when listing notebooks, including by name or insertion date. * Added evaluation coverage for listing notebooks, finding the newest notebook, and avoiding fabricated results for nonexistent notebooks. * **Bug Fixes** * Ensured selected notebook sorting preferences are correctly applied when retrieving content. * Improved assistant evaluation coverage with Explorer mode enabled. <!-- end of auto-generated comment: release notes by coderabbit.ai -->
74 lines
2.2 KiB
TypeScript
74 lines
2.2 KiB
TypeScript
import assert from 'node:assert'
|
|
import { Eval } from 'braintrust'
|
|
|
|
import { dataset } from './dataset'
|
|
import {
|
|
completenessScorer,
|
|
concisenessScorer,
|
|
correctnessScorer,
|
|
docsFaithfulnessScorer,
|
|
goalCompletionScorer,
|
|
knowledgeUsageScorer,
|
|
safetyScorer,
|
|
toolUsageScorer,
|
|
urlValidityScorer,
|
|
} from './scorer'
|
|
import { sqlIdentifierQuotingScorer, sqlSyntaxScorer } from './scorer-wasm'
|
|
import { generateAssistantResponse } from '@/lib/ai/generate-assistant-response'
|
|
import { getModel } from '@/lib/ai/model'
|
|
import { DEFAULT_ASSISTANT_BASE_MODEL_ID, getAssistantModelEntry } from '@/lib/ai/model.utils'
|
|
import { getMockTools } from '@/lib/ai/tools/mock-tools'
|
|
|
|
assert(process.env.BRAINTRUST_PROJECT_ID, 'BRAINTRUST_PROJECT_ID is not set')
|
|
assert(process.env.OPENAI_API_KEY, 'OPENAI_API_KEY is not set')
|
|
|
|
Eval('Assistant', {
|
|
projectId: process.env.BRAINTRUST_PROJECT_ID,
|
|
trialCount: process.env.CI ? 3 : 1,
|
|
data: () => dataset,
|
|
task: async (input) => {
|
|
const modelEntry = getAssistantModelEntry(DEFAULT_ASSISTANT_BASE_MODEL_ID)
|
|
const modelResponse = await getModel({ provider: 'openai', modelEntry })
|
|
if (modelResponse.error) throw modelResponse.error
|
|
|
|
// Owns the lifecycle of the remote MCP client opened inside getMockTools:
|
|
// aborting once generation is done closes that connection.
|
|
const toolsAbortController = new AbortController()
|
|
try {
|
|
const result = await generateAssistantResponse({
|
|
...modelResponse.modelParams,
|
|
isExplorerEnabled: true,
|
|
messages: [
|
|
{
|
|
id: '1',
|
|
role: 'user',
|
|
parts: [{ type: 'text', text: input.prompt }],
|
|
},
|
|
],
|
|
tools: await getMockTools(
|
|
input.mockTables ? { list_tables: input.mockTables } : undefined,
|
|
toolsAbortController.signal
|
|
),
|
|
})
|
|
|
|
const finishReason = await result.finishReason
|
|
return { finishReason }
|
|
} finally {
|
|
toolsAbortController.abort()
|
|
}
|
|
},
|
|
scores: [
|
|
toolUsageScorer,
|
|
knowledgeUsageScorer,
|
|
sqlSyntaxScorer,
|
|
sqlIdentifierQuotingScorer,
|
|
goalCompletionScorer,
|
|
concisenessScorer,
|
|
completenessScorer,
|
|
docsFaithfulnessScorer,
|
|
correctnessScorer,
|
|
safetyScorer,
|
|
urlValidityScorer,
|
|
],
|
|
})
|