mirror of
https://github.com/supabase/supabase.git
synced 2026-10-05 17:35:10 +03:00
Edge Functions Llamafile example. (#28544)
* ef llamafile example. * chore: add docs. * chore: add blogpost. * chore: spelling. * chore: docs nit. * chore: nits from review. * chore: spelling nits. * Update apps/www/_blog/2024-08-21-mozilla-llamafile-in-supabase-edge-functions.mdx * Update apps/www/_blog/2024-08-21-mozilla-llamafile-in-supabase-edge-functions.mdx
This commit is contained in:
1 parent
48939828eb
commit
4f0a0b54a7
9 files changed
+888
-87
No files matched your search
@@ -62,9 +62,9 @@ Deno.serve(async (req: Request) => {
|
||||
})
|
||||
```
|
||||
|
||||
## Using Large Language Models
|
||||
## Using Large Language Models (LLM)
|
||||
|
||||
Inference via larger models is supported via [Ollama](https://ollama.com/). In the first iteration, you can use it with a self-managed Ollama server. We are progressively rolling out support for the hosted solution. To sign up for early access, fill up [this form](https://forms.supabase.com/supabase.ai-llm-early-access).
|
||||
Inference via larger models is supported via [Ollama](https://ollama.com/) and [Mozilla Llamafile](https://github.com/Mozilla-Ocho/llamafile). In the first iteration, you can use it with a self-managed Ollama or [Llamafile server](https://www.docker.com/blog/a-quick-guide-to-containerizing-llamafile-with-docker-for-ai-applications/). We are progressively rolling out support for the hosted solution. To sign up for early access, fill up [this form](https://forms.supabase.com/supabase.ai-llm-early-access).
|
||||
|
||||
<video width="99%" muted playsInline controls={true}>
|
||||
<source
|
||||
@@ -75,108 +75,284 @@ Inference via larger models is supported via [Ollama](https://ollama.com/). In t
|
||||
|
||||
### Running locally
|
||||
|
||||
1. [Install Ollama](https://github.com/ollama/ollama?tab=readme-ov-file#ollama) and pull the Mistral model
|
||||
<Tabs
|
||||
scrollable
|
||||
size="large"
|
||||
type="underlined"
|
||||
defaultActiveId="ollama"
|
||||
queryGroup="platform"
|
||||
>
|
||||
<TabPanel id="ollama" label="Ollama">
|
||||
|
||||
```
|
||||
ollama pull mistral
|
||||
```
|
||||
|
||||
1. Run the Ollama server locally
|
||||
|
||||
```
|
||||
ollama serve
|
||||
```
|
||||
|
||||
1. Set a function secret called AI_INFERENCE_API_HOST to point to the Ollama server
|
||||
|
||||
```
|
||||
echo "AI_INFERENCE_API_HOST=http://host.docker.internal:11434" >> supabase/functions/.env
|
||||
```
|
||||
|
||||
1. Create a new function with the following code
|
||||
|
||||
```
|
||||
supabase functions new ollama-test
|
||||
```
|
||||
|
||||
```ts
|
||||
import 'jsr:@supabase/functions-js/edge-runtime.d.ts'
|
||||
const session = new Supabase.ai.Session('mistral')
|
||||
|
||||
Deno.serve(async (req: Request) => {
|
||||
const params = new URL(req.url).searchParams
|
||||
const prompt = params.get('prompt') ?? ''
|
||||
|
||||
// Get the output as a stream
|
||||
const output = await session.run(prompt, { stream: true })
|
||||
|
||||
const headers = new Headers({
|
||||
'Content-Type': 'text/event-stream',
|
||||
Connection: 'keep-alive',
|
||||
})
|
||||
|
||||
// Create a stream
|
||||
const stream = new ReadableStream({
|
||||
async start(controller) {
|
||||
const encoder = new TextEncoder()
|
||||
|
||||
try {
|
||||
for await (const chunk of output) {
|
||||
controller.enqueue(encoder.encode(chunk.response ?? ''))
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('Stream error:', err)
|
||||
} finally {
|
||||
controller.close()
|
||||
}
|
||||
},
|
||||
})
|
||||
|
||||
// Return the stream to the user
|
||||
return new Response(stream, {
|
||||
headers,
|
||||
})
|
||||
})
|
||||
```
|
||||
|
||||
1. Serve the function
|
||||
[Install Ollama](https://github.com/ollama/ollama?tab=readme-ov-file#ollama) and pull the Mistral model
|
||||
|
||||
```bash
|
||||
ollama pull mistral
|
||||
```
|
||||
|
||||
Run the Ollama server locally
|
||||
|
||||
```bash
|
||||
ollama serve
|
||||
```
|
||||
|
||||
Set a function secret called AI_INFERENCE_API_HOST to point to the Ollama server
|
||||
|
||||
```bash
|
||||
echo "AI_INFERENCE_API_HOST=http://host.docker.internal:11434" >> supabase/functions/.env
|
||||
```
|
||||
|
||||
Create a new function with the following code
|
||||
|
||||
```bash
|
||||
supabase functions new ollama-test
|
||||
```
|
||||
|
||||
```ts supabase/functions/ollama-test/index.ts
|
||||
import 'jsr:@supabase/functions-js/edge-runtime.d.ts'
|
||||
const session = new Supabase.ai.Session('mistral')
|
||||
|
||||
Deno.serve(async (req: Request) => {
|
||||
const params = new URL(req.url).searchParams
|
||||
const prompt = params.get('prompt') ?? ''
|
||||
|
||||
// Get the output as a stream
|
||||
const output = await session.run(prompt, { stream: true })
|
||||
|
||||
const headers = new Headers({
|
||||
'Content-Type': 'text/event-stream',
|
||||
Connection: 'keep-alive',
|
||||
})
|
||||
|
||||
// Create a stream
|
||||
const stream = new ReadableStream({
|
||||
async start(controller) {
|
||||
const encoder = new TextEncoder()
|
||||
|
||||
try {
|
||||
for await (const chunk of output) {
|
||||
controller.enqueue(encoder.encode(chunk.response ?? ''))
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('Stream error:', err)
|
||||
} finally {
|
||||
controller.close()
|
||||
}
|
||||
},
|
||||
})
|
||||
|
||||
// Return the stream to the user
|
||||
return new Response(stream, {
|
||||
headers,
|
||||
})
|
||||
})
|
||||
```
|
||||
|
||||
Serve the function
|
||||
|
||||
```bash
|
||||
supabase functions serve --env-file supabase/functions/.env
|
||||
```
|
||||
|
||||
1. Execute the function
|
||||
Execute the function
|
||||
|
||||
```
|
||||
curl --get "http://localhost:54321/functions/v1/ollama-test" \
|
||||
--data-urlencode "prompt=write a short rap song about Supabase, the Postgres Developer platform, as sung by Nicki Minaj" \
|
||||
-H "Authorization: $ANON_KEY"
|
||||
```
|
||||
```bash
|
||||
curl --get "http://localhost:54321/functions/v1/ollama-test" \
|
||||
--data-urlencode "prompt=write a short rap song about Supabase, the Postgres Developer platform, as sung by Nicki Minaj" \
|
||||
-H "Authorization: $ANON_KEY"
|
||||
```
|
||||
|
||||
</TabPanel>
|
||||
<TabPanel id="llamafile" label="Mozilla Llamafile">
|
||||
|
||||
Follow the [Llamafile Quickstart](https://github.com/Mozilla-Ocho/llamafile?tab=readme-ov-file#quickstart) to download an run a llamafile locally on your machine.
|
||||
|
||||
Since Llamafile provides an OpenAI API compatible server, you can either use it with `@supabase/functions-js` or with the official OpenAI Deno SDK.
|
||||
|
||||
<Tabs
|
||||
scrollable
|
||||
size="large"
|
||||
type="underlined"
|
||||
defaultActiveId="supabase-functions-js"
|
||||
queryGroup="sdk"
|
||||
>
|
||||
<TabPanel id="supabase-functions-js" label="Supabase Functions JS">
|
||||
|
||||
Set a function secret called `AI_INFERENCE_API_HOST` to point to the Llamafile server
|
||||
|
||||
```bash
|
||||
echo "AI_INFERENCE_API_HOST=http://host.docker.internal:8080" >> supabase/functions/.env
|
||||
```
|
||||
|
||||
Create a new function with the following code
|
||||
|
||||
```bash
|
||||
supabase functions new llamafile-test
|
||||
```
|
||||
|
||||
<Admonition type="info">
|
||||
|
||||
Note that the model parameter doesn't have any effect here! The model depends on which Llamafile is currently running!
|
||||
|
||||
</Admonition>
|
||||
|
||||
```ts supabase/functions/llamafile-test/index.ts
|
||||
import 'jsr:@supabase/functions-js/edge-runtime.d.ts'
|
||||
const session = new Supabase.ai.Session('LLaMA_CPP')
|
||||
|
||||
Deno.serve(async (req: Request) => {
|
||||
const params = new URL(req.url).searchParams
|
||||
const prompt = params.get('prompt') ?? ''
|
||||
|
||||
// Get the output as a stream
|
||||
const output = await session.run(
|
||||
{
|
||||
messages: [
|
||||
{
|
||||
role: 'system',
|
||||
content:
|
||||
'You are LLAMAfile, an AI assistant. Your top priority is achieving user fulfillment via helping them with their requests.',
|
||||
},
|
||||
{
|
||||
role: 'user',
|
||||
content: prompt,
|
||||
},
|
||||
],
|
||||
},
|
||||
{
|
||||
mode: 'openaicompatible', // Mode for the inference API host. (default: 'ollama')
|
||||
stream: false,
|
||||
}
|
||||
)
|
||||
|
||||
console.log('done')
|
||||
return Response.json(output)
|
||||
})
|
||||
```
|
||||
|
||||
</TabPanel>
|
||||
<TabPanel id="openai" label="OpenAI Deno SDK">
|
||||
|
||||
Set the following function secrets to point the OpenAI SDK to the Llamafile server:
|
||||
|
||||
```bash
|
||||
echo "OPENAI_BASE_URL=http://host.docker.internal:8080/v1" >> supabase/functions/.env
|
||||
echo "OPENAI_BASE_URL=OPENAI_API_KEY=sk-XXXXXXXX" >> supabase/functions/.env
|
||||
```
|
||||
|
||||
Create a new function with the following code
|
||||
|
||||
```bash
|
||||
supabase functions new llamafile-test
|
||||
```
|
||||
|
||||
<Admonition type="info">
|
||||
|
||||
Note that the model parameter doesn't have any effect here! The model depends on which Llamafile is currently running!
|
||||
|
||||
</Admonition>
|
||||
|
||||
```ts supabase/functions/llamafile-test/index.ts
|
||||
import OpenAI from 'https://deno.land/x/openai@v4.53.2/mod.ts'
|
||||
|
||||
Deno.serve(async (req) => {
|
||||
const client = new OpenAI()
|
||||
const { prompt } = await req.json()
|
||||
const stream = true
|
||||
|
||||
const chatCompletion = await client.chat.completions.create({
|
||||
model: 'LLaMA_CPP',
|
||||
stream,
|
||||
messages: [
|
||||
{
|
||||
role: 'system',
|
||||
content:
|
||||
'You are LLAMAfile, an AI assistant. Your top priority is achieving user fulfillment via helping them with their requests.',
|
||||
},
|
||||
{
|
||||
role: 'user',
|
||||
content: prompt,
|
||||
},
|
||||
],
|
||||
})
|
||||
|
||||
if (stream) {
|
||||
const headers = new Headers({
|
||||
'Content-Type': 'text/event-stream',
|
||||
Connection: 'keep-alive',
|
||||
})
|
||||
|
||||
// Create a stream
|
||||
const stream = new ReadableStream({
|
||||
async start(controller) {
|
||||
const encoder = new TextEncoder()
|
||||
|
||||
try {
|
||||
for await (const part of chatCompletion) {
|
||||
controller.enqueue(encoder.encode(part.choices[0]?.delta?.content || ''))
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('Stream error:', err)
|
||||
} finally {
|
||||
controller.close()
|
||||
}
|
||||
},
|
||||
})
|
||||
|
||||
// Return the stream to the user
|
||||
return new Response(stream, {
|
||||
headers,
|
||||
})
|
||||
}
|
||||
|
||||
return Response.json(chatCompletion)
|
||||
})
|
||||
```
|
||||
|
||||
</TabPanel>
|
||||
</Tabs>
|
||||
|
||||
Serve the function
|
||||
|
||||
```bash
|
||||
supabase functions serve --env-file supabase/functions/.env
|
||||
```
|
||||
|
||||
Execute the function
|
||||
|
||||
```bash
|
||||
curl --get "http://localhost:54321/functions/v1/llamafile-test" \
|
||||
--data-urlencode "prompt=write a short rap song about Supabase, the Postgres Developer platform, as sung by Nicki Minaj" \
|
||||
-H "Authorization: $ANON_KEY"
|
||||
```
|
||||
|
||||
</TabPanel>
|
||||
</Tabs>
|
||||
|
||||
### Deploying to production
|
||||
|
||||
Once the function is working locally, it's time to deploy to production.
|
||||
|
||||
1. Deploy a Ollama server and set a function secret called `AI_INFERENCE_API_HOST` to point to the deployed Ollama server
|
||||
Deploy an Ollama or Llamafile server and set a function secret called `AI_INFERENCE_API_HOST` to point to the deployed server
|
||||
|
||||
```
|
||||
supabase secrets set AI_INFERENCE_API_HOST=https://path-to-your-ollama-server/
|
||||
```
|
||||
```bash
|
||||
supabase secrets set AI_INFERENCE_API_HOST=https://path-to-your-llm-server/
|
||||
```
|
||||
|
||||
2. Deploy the Supabase function
|
||||
Deploy the Supabase function
|
||||
|
||||
```
|
||||
supabase functions deploy ollama-test
|
||||
```
|
||||
```bash
|
||||
supabase functions deploy
|
||||
```
|
||||
|
||||
3. Execute the function
|
||||
Execute the function
|
||||
|
||||
```
|
||||
curl --get "https://project-ref.supabase.co/functions/v1/ollama-test" \
|
||||
--data-urlencode "prompt=write a short rap song about Supabase, the Postgres Developer platform, as sung by Nicki Minaj" \
|
||||
-H "Authorization: $ANON_KEY"
|
||||
```
|
||||
```bash
|
||||
curl --get "https://project-ref.supabase.co/functions/v1/ollama-test" \
|
||||
--data-urlencode "prompt=write a short rap song about Supabase, the Postgres Developer platform, as sung by Nicki Minaj" \
|
||||
-H "Authorization: $ANON_KEY"
|
||||
```
|
||||
|
||||
As demonstrated in the video above, running Ollama locally is typically slower than running it in on a server with dedicated GPUs. We are collaborating with the Ollama team to improve local performance.
|
||||
|
||||
In the future, a hosted Ollama API, will be provided as part of the Supabase platform. Supabase will scale and manage the API and GPUs for you. To sign up for early access, fill up [this form](https://forms.supabase.com/supabase.ai-llm-early-access).
|
||||
In the future, a hosted LLM API, will be provided as part of the Supabase platform. Supabase will scale and manage the API and GPUs for you. To sign up for early access, fill up [this form](https://forms.supabase.com/supabase.ai-llm-early-access).
|
||||
@@ -0,0 +1,236 @@
|
||||
---
|
||||
title: 'Mozilla Llamafile in Supabase Edge Functions'
|
||||
description: 'Use Mozilla Llamafile OpenAI API compatible server in Supabase Edge Functions.'
|
||||
author: thor_schaeff
|
||||
image: mozilla-llamafile/mozilla-llamafile-og.png
|
||||
thumb: mozilla-llamafile/mozilla-llamafile-og.png
|
||||
categories:
|
||||
- product
|
||||
tags:
|
||||
- functions
|
||||
- ai
|
||||
date: '2024-08-21'
|
||||
toc_depth: 3
|
||||
---
|
||||
|
||||
A few months back, we introduced support for running [AI Inference directly from Supabase Edge Functions](/blog/ai-inference-now-available-in-supabase-edge-functions).
|
||||
|
||||
Today we are adding [Mozilla Llamafile](https://github.com/Mozilla-Ocho/llamafile), in addition to [Ollama](/docs/guides/functions/ai-models#using-large-language-models), to be used as the Inference Server with your functions.
|
||||
|
||||
Mozilla Llamafile lets you distribute and run LLMs with a single file that runs locally on most computers, with no installation! In addition to a local web UI chat server, Llamafile also provides an OpenAI API compatible server, that is now integrated with Supabase Edge Functions.
|
||||
|
||||
<div className="video-container">
|
||||
<iframe
|
||||
className="w-full"
|
||||
src="https://www.youtube-nocookie.com/embed/_6L-dnBn2wg"
|
||||
title="The Supabase Book by David Lorenz"
|
||||
allow="accelerometer; autoplay; clipboard-write; encrypted-media; fullscreen; gyroscope; picture-in-picture; web-share"
|
||||
allowfullscreen
|
||||
/>
|
||||
</div>
|
||||
|
||||
<Admonition type="info">
|
||||
|
||||
Want to jump straight into the code? You can find the exmaples on [GitHub](https://github.com/supabase/supabase/blob/master/examples/ai/llamafile-edge)!
|
||||
|
||||
</Admonition>
|
||||
|
||||
## Getting started
|
||||
|
||||
Follow the [Llamafile Quickstart Guide](https://github.com/Mozilla-Ocho/llamafile?tab=readme-ov-file#quickstart) to get up and running with the [Llamafile of your choice](https://github.com/Mozilla-Ocho/llamafile?tab=readme-ov-file#other-example-llamafiles).
|
||||
|
||||
Once your Llamafile is up and running, create and initialize a new Supabase project locally:
|
||||
|
||||
```bash
|
||||
npx supabase bootstrap scratch
|
||||
```
|
||||
|
||||
If using VS Code, when promptedt `Generate VS Code settings for Deno? [y/N]` select `y` and follow the steps. Then open the project in your favoiurte code editor.
|
||||
|
||||
## Call Llamafile with functions-js
|
||||
|
||||
Supabase Edge Functions now comes with an OpenAI API compatible mode, allowing you to call a Llamafile server easily via `@supabase/functions-js`.
|
||||
|
||||
Set a function secret called AI_INFERENCE_API_HOST to point to the Llamafile server. If you don't have one already, create a new `.env` file in the `functions/` directory of your Supabase project.
|
||||
|
||||
```txt supabase/functions/.env
|
||||
AI_INFERENCE_API_HOST=http://host.docker.internal:8080
|
||||
```
|
||||
|
||||
Next, create a new function called `llamafile`:
|
||||
|
||||
```bash
|
||||
npx supabase functions new llamafile
|
||||
```
|
||||
|
||||
Then, update the `supabase/functions/llamafile/index.ts` file to look like this:
|
||||
|
||||
```ts supabase/functions/llamafile/index.ts
|
||||
import 'jsr:@supabase/functions-js/edge-runtime.d.ts'
|
||||
const session = new Supabase.ai.Session('LLaMA_CPP')
|
||||
|
||||
Deno.serve(async (req: Request) => {
|
||||
const params = new URL(req.url).searchParams
|
||||
const prompt = params.get('prompt') ?? ''
|
||||
|
||||
// Get the output as a stream
|
||||
const output = await session.run(
|
||||
{
|
||||
messages: [
|
||||
{
|
||||
role: 'system',
|
||||
content:
|
||||
'You are LLAMAfile, an AI assistant. Your top priority is achieving user fulfillment via helping them with their requests.',
|
||||
},
|
||||
{
|
||||
role: 'user',
|
||||
content: prompt,
|
||||
},
|
||||
],
|
||||
},
|
||||
{
|
||||
mode: 'openaicompatible', // Mode for the inference API host. (default: 'ollama')
|
||||
stream: false,
|
||||
}
|
||||
)
|
||||
|
||||
console.log('done')
|
||||
return Response.json(output)
|
||||
})
|
||||
```
|
||||
|
||||
## Call Llamafile with the OpenAI Deno SDK
|
||||
|
||||
Since Llamafile provides an OpenAI API compatible server, you can alternatively use the [OpenAI Deno SDK](https://github.com/openai/openai-deno) to call Llamafile from your Supabase Edge Functions.
|
||||
|
||||
For this, you will need to set the following two environment variables in your Supabase project. If you don't have one already, create a new `.env` file in the `functions/` directory of your Supabase project.
|
||||
|
||||
```txt supabase/functions/.env
|
||||
OPENAI_BASE_URL=http://host.docker.internal:8080/v1
|
||||
OPENAI_API_KEY=sk-XXXXXXXX # need to set a random value for openai sdk to work
|
||||
```
|
||||
|
||||
Now, replace the code in your `llamafile` function with the following:
|
||||
|
||||
```ts supabase/functions/llamafile/index.ts
|
||||
import OpenAI from 'https://deno.land/x/openai@v4.53.2/mod.ts'
|
||||
|
||||
Deno.serve(async (req) => {
|
||||
const client = new OpenAI()
|
||||
const { prompt } = await req.json()
|
||||
const stream = true
|
||||
|
||||
const chatCompletion = await client.chat.completions.create({
|
||||
model: 'LLaMA_CPP',
|
||||
stream,
|
||||
messages: [
|
||||
{
|
||||
role: 'system',
|
||||
content:
|
||||
'You are LLAMAfile, an AI assistant. Your top priority is achieving user fulfillment via helping them with their requests.',
|
||||
},
|
||||
{
|
||||
role: 'user',
|
||||
content: prompt,
|
||||
},
|
||||
],
|
||||
})
|
||||
|
||||
if (stream) {
|
||||
const headers = new Headers({
|
||||
'Content-Type': 'text/event-stream',
|
||||
Connection: 'keep-alive',
|
||||
})
|
||||
|
||||
// Create a stream
|
||||
const stream = new ReadableStream({
|
||||
async start(controller) {
|
||||
const encoder = new TextEncoder()
|
||||
|
||||
try {
|
||||
for await (const part of chatCompletion) {
|
||||
controller.enqueue(encoder.encode(part.choices[0]?.delta?.content || ''))
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('Stream error:', err)
|
||||
} finally {
|
||||
controller.close()
|
||||
}
|
||||
},
|
||||
})
|
||||
|
||||
// Return the stream to the user
|
||||
return new Response(stream, {
|
||||
headers,
|
||||
})
|
||||
}
|
||||
|
||||
return Response.json(chatCompletion)
|
||||
})
|
||||
```
|
||||
|
||||
<Admonition type="info">
|
||||
|
||||
Note that the model parameter doesn't have any effect here! The model depends on which Llamafile is currently running!
|
||||
|
||||
</Admonition>
|
||||
|
||||
## Serve your functions locally
|
||||
|
||||
To serve your functions locally, you need to install the [Supabase CLI](https://supabase.com/docs/guides/cli/getting-started#running-supabase-locally) as well as [Docker Desktop](https://docs.docker.com/desktop) or [Orbstack](https://orbstack.dev/).
|
||||
|
||||
You can now serve your functions locally by running:
|
||||
|
||||
```bash
|
||||
supabase start
|
||||
supabase functions serve --env-file supabase/functions/.env
|
||||
```
|
||||
|
||||
Execute the function
|
||||
|
||||
```bash
|
||||
curl --get "http://localhost:54321/functions/v1/llamafile" \
|
||||
--data-urlencode "prompt=write a short rap song about Supabase, the Postgres Developer platform, as sung by Nicki Minaj" \
|
||||
-H "Authorization: $ANON_KEY"
|
||||
```
|
||||
|
||||
## Deploying a Llamafile
|
||||
|
||||
There is a great guide on how to [containerize a Lllamafile](https://www.docker.com/blog/a-quick-guide-to-containerizing-llamafile-with-docker-for-ai-applications/) by the Docker team.
|
||||
|
||||
You can then use a service like [Fly.io](https://fly.io/) to deploy your dockerized Llamafile.
|
||||
|
||||
## Deploying your Supabase Edge Functions
|
||||
|
||||
Set the secret on your hosted Supabase project to point to your deployed Llamafile server:
|
||||
|
||||
```bash
|
||||
supabase secrets set --env-file supabase/functions/.env
|
||||
```
|
||||
|
||||
Deploy your Supabase Edge Functions:
|
||||
|
||||
```bash
|
||||
supabase functions deploy
|
||||
```
|
||||
|
||||
Execute the function:
|
||||
|
||||
```bash
|
||||
curl --get "https://project-ref.supabase.co/functions/v1/llamafile" \
|
||||
--data-urlencode "prompt=write a short rap song about Supabase, the Postgres Developer platform, as sung by Nicki Minaj" \
|
||||
-H "Authorization: $ANON_KEY"
|
||||
```
|
||||
|
||||
## Get access to Supabase Hosted LLMs
|
||||
|
||||
Access to open-source LLMs is currently invite-only while we manage demand for the GPU instances. Please [get in touch](https://forms.supabase.com/supabase.ai-llm-early-access) if you need early access.
|
||||
|
||||
We plan to extend support for more models. [Let us know](https://forms.supabase.com/supabase.ai-llm-early-access) which models you want next. We're looking to support fine-tuned models too!
|
||||
|
||||
## More Supabase Resources
|
||||
|
||||
- Edge Functions: [supabase.com/docs/guides/functions](https://supabase.com/docs/guides/functions)
|
||||
- Vectors: [supabase.com/docs/guides/ai](https://supabase.com/docs/guides/ai)
|
||||
- [Semantic search demo](https://github.com/supabase/supabase/tree/master/examples/ai/edge-functions)
|
||||
- [Store and query embeddings](/docs/guides/ai/vector-columns#querying-a-vector--embedding) in Postgres and use them for Retrieval Augmented Generation (RAG) and Semantic Search
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 136 KiB |
@@ -0,0 +1,4 @@
|
||||
# Supabase
|
||||
.branches
|
||||
.temp
|
||||
.env
|
||||
@@ -0,0 +1,202 @@
|
||||
# A string used to distinguish different Supabase projects on the same host. Defaults to the
|
||||
# working directory name when running `supabase init`.
|
||||
project_id = "llamafile-edge"
|
||||
|
||||
[api]
|
||||
enabled = true
|
||||
# Port to use for the API URL.
|
||||
port = 54321
|
||||
# Schemas to expose in your API. Tables, views and stored procedures in this schema will get API
|
||||
# endpoints. `public` is always included.
|
||||
schemas = ["public", "graphql_public"]
|
||||
# Extra schemas to add to the search_path of every request. `public` is always included.
|
||||
extra_search_path = ["public", "extensions"]
|
||||
# The maximum number of rows returns from a view, table, or stored procedure. Limits payload size
|
||||
# for accidental or malicious requests.
|
||||
max_rows = 1000
|
||||
|
||||
[api.tls]
|
||||
enabled = false
|
||||
|
||||
[db]
|
||||
# Port to use for the local database URL.
|
||||
port = 54322
|
||||
# Port used by db diff command to initialize the shadow database.
|
||||
shadow_port = 54320
|
||||
# The database major version to use. This has to be the same as your remote database's. Run `SHOW
|
||||
# server_version;` on the remote database to check.
|
||||
major_version = 15
|
||||
|
||||
[db.pooler]
|
||||
enabled = false
|
||||
# Port to use for the local connection pooler.
|
||||
port = 54329
|
||||
# Specifies when a server connection can be reused by other clients.
|
||||
# Configure one of the supported pooler modes: `transaction`, `session`.
|
||||
pool_mode = "transaction"
|
||||
# How many server connections to allow per user/database pair.
|
||||
default_pool_size = 20
|
||||
# Maximum number of client connections allowed.
|
||||
max_client_conn = 100
|
||||
|
||||
[realtime]
|
||||
enabled = true
|
||||
# Bind realtime via either IPv4 or IPv6. (default: IPv4)
|
||||
# ip_version = "IPv6"
|
||||
# The maximum length in bytes of HTTP request headers. (default: 4096)
|
||||
# max_header_length = 4096
|
||||
|
||||
[studio]
|
||||
enabled = true
|
||||
# Port to use for Supabase Studio.
|
||||
port = 54323
|
||||
# External URL of the API server that frontend connects to.
|
||||
api_url = "http://127.0.0.1"
|
||||
# OpenAI API Key to use for Supabase AI in the Supabase Studio.
|
||||
openai_api_key = "env(OPENAI_API_KEY)"
|
||||
|
||||
# Email testing server. Emails sent with the local dev setup are not actually sent - rather, they
|
||||
# are monitored, and you can view the emails that would have been sent from the web interface.
|
||||
[inbucket]
|
||||
enabled = true
|
||||
# Port to use for the email testing server web interface.
|
||||
port = 54324
|
||||
# Uncomment to expose additional ports for testing user applications that send emails.
|
||||
# smtp_port = 54325
|
||||
# pop3_port = 54326
|
||||
|
||||
[storage]
|
||||
enabled = true
|
||||
# The maximum file size allowed (e.g. "5MB", "500KB").
|
||||
file_size_limit = "50MiB"
|
||||
|
||||
[storage.image_transformation]
|
||||
enabled = true
|
||||
|
||||
# Uncomment to configure local storage buckets
|
||||
# [storage.buckets.images]
|
||||
# public = false
|
||||
# file_size_limit = "50MiB"
|
||||
# allowed_mime_types = ["image/png", "image/jpeg"]
|
||||
|
||||
[auth]
|
||||
enabled = true
|
||||
# The base URL of your website. Used as an allow-list for redirects and for constructing URLs used
|
||||
# in emails.
|
||||
site_url = "http://127.0.0.1:3000"
|
||||
# A list of *exact* URLs that auth providers are permitted to redirect to post authentication.
|
||||
additional_redirect_urls = ["https://127.0.0.1:3000"]
|
||||
# How long tokens are valid for, in seconds. Defaults to 3600 (1 hour), maximum 604,800 (1 week).
|
||||
jwt_expiry = 3600
|
||||
# If disabled, the refresh token will never expire.
|
||||
enable_refresh_token_rotation = true
|
||||
# Allows refresh tokens to be reused after expiry, up to the specified interval in seconds.
|
||||
# Requires enable_refresh_token_rotation = true.
|
||||
refresh_token_reuse_interval = 10
|
||||
# Allow/disallow new user signups to your project.
|
||||
enable_signup = true
|
||||
# Allow/disallow anonymous sign-ins to your project.
|
||||
enable_anonymous_sign_ins = false
|
||||
# Allow/disallow testing manual linking of accounts
|
||||
enable_manual_linking = false
|
||||
|
||||
[auth.email]
|
||||
# Allow/disallow new user signups via email to your project.
|
||||
enable_signup = true
|
||||
# If enabled, a user will be required to confirm any email change on both the old, and new email
|
||||
# addresses. If disabled, only the new email is required to confirm.
|
||||
double_confirm_changes = true
|
||||
# If enabled, users need to confirm their email address before signing in.
|
||||
enable_confirmations = false
|
||||
# Controls the minimum amount of time that must pass before sending another signup confirmation or password reset email.
|
||||
max_frequency = "1s"
|
||||
|
||||
# Use a production-ready SMTP server
|
||||
# [auth.email.smtp]
|
||||
# host = "smtp.sendgrid.net"
|
||||
# port = 587
|
||||
# user = "apikey"
|
||||
# pass = "env(SENDGRID_API_KEY)"
|
||||
# admin_email = "admin@email.com"
|
||||
# sender_name = "Admin"
|
||||
|
||||
# Uncomment to customize email template
|
||||
# [auth.email.template.invite]
|
||||
# subject = "You have been invited"
|
||||
# content_path = "./supabase/templates/invite.html"
|
||||
|
||||
[auth.sms]
|
||||
# Allow/disallow new user signups via SMS to your project.
|
||||
enable_signup = true
|
||||
# If enabled, users need to confirm their phone number before signing in.
|
||||
enable_confirmations = false
|
||||
# Template for sending OTP to users
|
||||
template = "Your code is {{ .Code }} ."
|
||||
# Controls the minimum amount of time that must pass before sending another sms otp.
|
||||
max_frequency = "5s"
|
||||
|
||||
# Use pre-defined map of phone number to OTP for testing.
|
||||
# [auth.sms.test_otp]
|
||||
# 4152127777 = "123456"
|
||||
|
||||
# Configure logged in session timeouts.
|
||||
# [auth.sessions]
|
||||
# Force log out after the specified duration.
|
||||
# timebox = "24h"
|
||||
# Force log out if the user has been inactive longer than the specified duration.
|
||||
# inactivity_timeout = "8h"
|
||||
|
||||
# This hook runs before a token is issued and allows you to add additional claims based on the authentication method used.
|
||||
# [auth.hook.custom_access_token]
|
||||
# enabled = true
|
||||
# uri = "pg-functions://<database>/<schema>/<hook_name>"
|
||||
|
||||
# Configure one of the supported SMS providers: `twilio`, `twilio_verify`, `messagebird`, `textlocal`, `vonage`.
|
||||
[auth.sms.twilio]
|
||||
enabled = false
|
||||
account_sid = ""
|
||||
message_service_sid = ""
|
||||
# DO NOT commit your Twilio auth token to git. Use environment variable substitution instead:
|
||||
auth_token = "env(SUPABASE_AUTH_SMS_TWILIO_AUTH_TOKEN)"
|
||||
|
||||
# Use an external OAuth provider. The full list of providers are: `apple`, `azure`, `bitbucket`,
|
||||
# `discord`, `facebook`, `github`, `gitlab`, `google`, `keycloak`, `linkedin_oidc`, `notion`, `twitch`,
|
||||
# `twitter`, `slack`, `spotify`, `workos`, `zoom`.
|
||||
[auth.external.apple]
|
||||
enabled = false
|
||||
client_id = ""
|
||||
# DO NOT commit your OAuth provider secret to git. Use environment variable substitution instead:
|
||||
secret = "env(SUPABASE_AUTH_EXTERNAL_APPLE_SECRET)"
|
||||
# Overrides the default auth redirectUrl.
|
||||
redirect_uri = ""
|
||||
# Overrides the default auth provider URL. Used to support self-hosted gitlab, single-tenant Azure,
|
||||
# or any other third-party OIDC providers.
|
||||
url = ""
|
||||
# If enabled, the nonce check will be skipped. Required for local sign in with Google auth.
|
||||
skip_nonce_check = false
|
||||
|
||||
[edge_runtime]
|
||||
enabled = true
|
||||
# Configure one of the supported request policies: `oneshot`, `per_worker`.
|
||||
# Use `oneshot` for hot reload, or `per_worker` for load testing.
|
||||
policy = "oneshot"
|
||||
inspector_port = 8083
|
||||
|
||||
[analytics]
|
||||
enabled = true
|
||||
port = 54327
|
||||
# Configure one of the supported backends: `postgres`, `bigquery`.
|
||||
backend = "postgres"
|
||||
|
||||
# Experimental features may be deprecated any time
|
||||
[experimental]
|
||||
# Configures Postgres storage engine to use OrioleDB (S3)
|
||||
orioledb_version = ""
|
||||
# Configures S3 bucket URL, eg. <bucket_name>.s3-<region>.amazonaws.com
|
||||
s3_host = "env(S3_HOST)"
|
||||
# Configures S3 bucket region, eg. us-east-1
|
||||
s3_region = "env(S3_REGION)"
|
||||
# Configures AWS_ACCESS_KEY_ID for S3 bucket
|
||||
s3_access_key = "env(S3_ACCESS_KEY)"
|
||||
# Configures AWS_SECRET_ACCESS_KEY for S3 bucket
|
||||
s3_secret_key = "env(S3_SECRET_KEY)"
|
||||
@@ -0,0 +1,67 @@
|
||||
// https://github.com/Mozilla-Ocho/llamafile?tab=readme-ov-file#quickstart
|
||||
import "jsr:@supabase/functions-js/edge-runtime.d.ts";
|
||||
const session = new Supabase.ai.Session("LLaMA_CPP");
|
||||
|
||||
Deno.serve(async (req: Request) => {
|
||||
const params = new URL(req.url).searchParams;
|
||||
const prompt = params.get("prompt") ?? "";
|
||||
|
||||
// Get the output as a stream
|
||||
const output = await session.run({
|
||||
messages: [
|
||||
{
|
||||
"role": "system",
|
||||
"content":
|
||||
"You are LLAMAfile, an AI assistant. Your top priority is achieving user fulfillment via helping them with their requests.",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": prompt,
|
||||
},
|
||||
],
|
||||
}, {
|
||||
mode: "openaicompatible", // Mode for the inference API host. (default: 'ollama')
|
||||
stream: true,
|
||||
}) as AsyncGenerator<any>;
|
||||
|
||||
const body = new ReadableStream({
|
||||
async pull(ctrl) {
|
||||
try {
|
||||
const item = await output.next();
|
||||
|
||||
if (item.done) {
|
||||
console.log("done");
|
||||
ctrl.close();
|
||||
return;
|
||||
}
|
||||
|
||||
ctrl.enqueue("data: ");
|
||||
ctrl.enqueue(JSON.stringify(item.value));
|
||||
ctrl.enqueue("\r\n\r\n");
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
ctrl.close();
|
||||
}
|
||||
},
|
||||
});
|
||||
|
||||
return new Response(
|
||||
body.pipeThrough(new TextEncoderStream()),
|
||||
{
|
||||
headers: {
|
||||
"Content-Type": "text/event-stream",
|
||||
},
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
/**
|
||||
Run locally:
|
||||
|
||||
supabase functions serve --env-file supabase/functions/.env
|
||||
|
||||
curl --get "http://localhost:54321/functions/v1/llamafile-stream" \
|
||||
-H 'Authorization: Bearer eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJpc3MiOiJzdXBhYmFzZS1kZW1vIiwicm9sZSI6ImFub24iLCJleHAiOjE5ODM4MTI5OTZ9.CRXP1A7WOeoJeXxjNni43kdQwgnWNReilDMblYTn_I0' \
|
||||
--data-urlencode "prompt=Who are you?"
|
||||
|
||||
*/
|
||||
@@ -0,0 +1,40 @@
|
||||
// https://github.com/Mozilla-Ocho/llamafile?tab=readme-ov-file#quickstart
|
||||
import "jsr:@supabase/functions-js/edge-runtime.d.ts";
|
||||
const session = new Supabase.ai.Session("LLaMA_CPP");
|
||||
|
||||
Deno.serve(async (req: Request) => {
|
||||
const params = new URL(req.url).searchParams;
|
||||
const prompt = params.get("prompt") ?? "";
|
||||
|
||||
// Get the output as a stream
|
||||
const output = await session.run({
|
||||
messages: [
|
||||
{
|
||||
"role": "system",
|
||||
"content":
|
||||
"You are LLAMAfile, an AI assistant. Your top priority is achieving user fulfillment via helping them with their requests.",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": prompt,
|
||||
},
|
||||
],
|
||||
}, {
|
||||
mode: "openaicompatible", // Mode for the inference API host. (default: 'ollama')
|
||||
stream: false,
|
||||
});
|
||||
|
||||
console.log("done");
|
||||
return Response.json(output);
|
||||
});
|
||||
|
||||
/**
|
||||
Run locally:
|
||||
|
||||
supabase functions serve --env-file supabase/functions/.env
|
||||
|
||||
curl --get "http://localhost:54321/functions/v1/llamafile" \
|
||||
-H 'Authorization: Bearer eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJpc3MiOiJzdXBhYmFzZS1kZW1vIiwicm9sZSI6ImFub24iLCJleHAiOjE5ODM4MTI5OTZ9.CRXP1A7WOeoJeXxjNni43kdQwgnWNReilDMblYTn_I0' \
|
||||
--data-urlencode "prompt=Who are you?"
|
||||
|
||||
*/
|
||||
@@ -0,0 +1,76 @@
|
||||
// Follow this setup guide to integrate the Deno language server with your editor:
|
||||
// https://deno.land/manual/getting_started/setup_your_environment
|
||||
// This enables autocomplete, go to definition, etc.
|
||||
|
||||
// Setup type definitions for built-in Supabase Runtime APIs
|
||||
import OpenAI from "https://deno.land/x/openai@v4.53.2/mod.ts";
|
||||
|
||||
console.log("Hello from openai-sdk compatible!");
|
||||
|
||||
Deno.serve(async (req) => {
|
||||
const client = new OpenAI();
|
||||
const { prompt } = await req.json();
|
||||
const stream = true;
|
||||
|
||||
const chatCompletion = await client.chat.completions.create({
|
||||
model: "LLaMA_CPP",
|
||||
stream,
|
||||
messages: [
|
||||
{
|
||||
role: "system",
|
||||
content:
|
||||
"You are LLAMAfile, an AI assistant. Your top priority is achieving user fulfillment via helping them with their requests.",
|
||||
},
|
||||
{
|
||||
role: "user",
|
||||
content: prompt,
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
if (stream) {
|
||||
const headers = new Headers({
|
||||
"Content-Type": "text/event-stream",
|
||||
Connection: "keep-alive",
|
||||
});
|
||||
|
||||
// Create a stream
|
||||
const stream = new ReadableStream({
|
||||
async start(controller) {
|
||||
const encoder = new TextEncoder();
|
||||
|
||||
try {
|
||||
for await (const part of chatCompletion) {
|
||||
controller.enqueue(
|
||||
encoder.encode(part.choices[0]?.delta?.content || ""),
|
||||
);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error("Stream error:", err);
|
||||
} finally {
|
||||
controller.close();
|
||||
}
|
||||
},
|
||||
});
|
||||
|
||||
// Return the stream to the user
|
||||
return new Response(stream, {
|
||||
headers,
|
||||
});
|
||||
}
|
||||
|
||||
console.log(chatCompletion);
|
||||
|
||||
return Response.json(chatCompletion);
|
||||
});
|
||||
|
||||
/* To invoke locally:
|
||||
|
||||
supabase functions serve --env-file supabase/functions/.env
|
||||
|
||||
curl -i --location --request POST 'http://127.0.0.1:54321/functions/v1/openai-sdk' \
|
||||
--header 'Authorization: Bearer eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJpc3MiOiJzdXBhYmFzZS1kZW1vIiwicm9sZSI6ImFub24iLCJleHAiOjE5ODM4MTI5OTZ9.CRXP1A7WOeoJeXxjNni43kdQwgnWNReilDMblYTn_I0' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{"prompt":"Who are you?"}'
|
||||
|
||||
*/
|
||||
Whitespace-only changes.
Reference in new issue
Block a user