Files
supabase/apps/studio/lib/ai/model.test.ts
T
Matt Rossman 0c5f64fcba feat(assistant): upgrade default models to gpt-5.4-nano and gpt-5.3-codex (#44107)
Replaces `gpt-5-mini` and `gpt-5` with `gpt-5.4-nano` and
`gpt-5.3-codex` respectively. Clients with stale model IDs in IndexedDB
will gracefully reset to the new defaults. While we can technically keep
the existing models around, we've
[opted](https://supabase.slack.com/archives/C051L8U2EJF/p1774283070517609?thread_ts=1773771991.871669&cid=C051L8U2EJF)
to replace them w/ the newer models for simplicity. Basic completion
endpoints use `'none'` reasoning level for optimal speed.

Rationale for these models is they provide they best balance of
intelligence/speed and cost. GPT-5.4-nano is less expensive (0.8x
price), faster, and smarter than GPT-5-mini. GPT-5.4-mini would be even
smarter but is 3x the price. GPT-5.3-Codex is ~1.4x the price of GPT-5,
while GPT-5.4 would be 2x price, but 5.3-Codex is still a big
intelligence boost from GPT-5.

See [eval
comparison](https://www.braintrust.dev/app/supabase.io/p/Assistant/experiments/mattrossman%2Fai-509-v2-upgrade-assistant-models-beyond-gpt-5-family-1774468619?c=master-1774458837&diff=between_experiments),
scores are relatively stable and conciseness naturally improves on
gpt-5.4-nano.

Other change:
- Fixed an eval test case to clarify that https://supabase.help is also
a correct URL for submitting support ticket, which was unfairly scored
as incorrect
[here](https://www.braintrust.dev/app/supabase.io/p/Assistant/trace?object_type=experiment&object_id=5244cccd-23b2-4f79-9dd2-287f1b40ebad&r=bac9b903-8bde-4c21-99dd-e0ed141c4f9e&s=f248fbf5-75bf-4aab-be0a-87a4298e6d11)

I sanity checked the Assistant, natural language filters, and SQL Editor
completions on staging preview.

References:
- https://openai.com/index/introducing-gpt-5-4-mini-and-nano/
- https://openai.com/index/introducing-gpt-5-3-codex/
- https://developers.openai.com/api/docs/pricing

Closes AI-509
2026-03-26 14:35:54 +08:00

100 lines
3.2 KiB
TypeScript

import { openai } from '@ai-sdk/openai'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import * as bedrockModule from './bedrock'
import { getModel } from './model'
import { DEFAULT_COMPLETION_MODEL, openaiModelEntry } from './model.utils'
vi.mock('@ai-sdk/openai', () => ({
openai: vi.fn(() => 'openai-model'),
}))
vi.mock('./bedrock', async () => ({
...(await vi.importActual('./bedrock')),
createRoutedBedrock: vi.fn(() => async (_modelId: string) => 'bedrock-model'),
checkAwsCredentials: vi.fn(),
}))
describe('getModel', () => {
const originalEnv = { ...process.env }
beforeEach(() => {
vi.resetAllMocks()
})
afterEach(() => {
process.env = { ...originalEnv }
})
it('returns bedrock model without promptProviderOptions', async () => {
vi.mocked(bedrockModule.checkAwsCredentials).mockResolvedValue(true)
vi.stubEnv('AWS_BEDROCK_ROLE_ARN', 'test')
const { modelParams, error, promptProviderOptions } = await getModel({
provider: 'bedrock',
routingKey: 'test',
})
expect(modelParams?.model).toEqual('bedrock-model')
expect(promptProviderOptions).toBeUndefined()
expect(error).toBeUndefined()
})
it('returns error when bedrock credentials are not available', async () => {
vi.mocked(bedrockModule.checkAwsCredentials).mockResolvedValue(false)
const { error } = await getModel({ provider: 'bedrock', routingKey: 'test' })
expect(error).toBeDefined()
})
it('returns openai model with default model', async () => {
vi.stubEnv('OPENAI_API_KEY', 'test-key')
const { modelParams, promptProviderOptions } = await getModel({
provider: 'openai',
modelEntry: openaiModelEntry({ id: 'gpt-5.4-nano' }),
})
expect(modelParams?.model).toEqual('openai-model')
expect(openai).toHaveBeenCalledWith('gpt-5.4-nano')
expect(promptProviderOptions).toBeUndefined()
})
it('returns error when OPENAI_API_KEY is not available', async () => {
vi.stubEnv('OPENAI_API_KEY', '')
const { error } = await getModel({
provider: 'openai',
modelEntry: openaiModelEntry({ id: 'gpt-5.4-nano' }),
})
expect(error).toEqual(new Error('OPENAI_API_KEY not available'))
})
it('returns openai gpt-5.3-codex when hasAccessToAdvanceModel and not throttled', async () => {
vi.stubEnv('OPENAI_API_KEY', 'test-key')
vi.stubEnv('IS_THROTTLED', 'false')
const { modelParams, error } = await getModel({
provider: 'openai',
modelEntry: openaiModelEntry({ id: 'gpt-5.3-codex', reasoningEffort: 'low' }),
})
expect(error).toBeUndefined()
expect(modelParams?.model).toEqual('openai-model')
expect(openai).toHaveBeenCalledWith('gpt-5.3-codex')
expect(modelParams?.providerOptions?.openai?.reasoningEffort).toBe('low')
})
it('applies reasoningEffort from DEFAULT_COMPLETION_MODEL', async () => {
vi.stubEnv('OPENAI_API_KEY', 'test-key')
const { modelParams, error } = await getModel({
provider: 'openai',
modelEntry: DEFAULT_COMPLETION_MODEL,
})
expect(error).toBeUndefined()
expect(openai).toHaveBeenCalledWith('gpt-5.4-nano')
expect(modelParams?.providerOptions?.openai?.reasoningEffort).toBe('none')
})
})