mirror of
https://github.com/supabase/supabase.git
synced 2026-10-05 09:25:06 +03:00
Thor/add elevenlabs streaming example n docs (#33989)
* feat: add TTS & STT docs. * feat: add examples. * fix: docs. * code review nits. * chore: format. * Fix mdx-lint errors. --------- Co-authored-by: Ivan Vasilov <vasilov.ivan@gmail.com>
This commit is contained in:
1 parent
8120245eb7
commit
99f1d1fc16
21 files changed
+992
-1
No files matched your search
@@ -52,3 +52,6 @@ UPSTASH_REDIS_REST_TOKEN=
|
||||
# connect-supabase - https://supabase.com/docs/guides/platform/oauth-apps/publish-an-oauth-app
|
||||
SUPA_CONNECT_CLIENT_ID=
|
||||
SUPA_CONNECT_CLIENT_SECRET=
|
||||
|
||||
# elevenlabs-text-to-speech
|
||||
ELEVENLABS_API_KEY=
|
||||
@@ -0,0 +1,3 @@
|
||||
# Configuration for private npm package dependencies
|
||||
# For more information on using private registries with Edge Functions, see:
|
||||
# https://supabase.com/docs/guides/functions/import-maps#importing-from-private-registries
|
||||
@@ -0,0 +1,52 @@
|
||||
## ElevenLabs Scribe Telegram Bot
|
||||
|
||||
This is a Telegram bot that uses the ElevenLabs API to transcribe voice messages, as well as audio and video files.
|
||||
|
||||
You can find the bot here: https://t.me/ElevenLabsScribeBot
|
||||
|
||||
For a detailed tutorial, please see the [ElevenLabs Developer Docs](https://elevenlabs.io/docs/cookbooks/speech-to-text/telegram-bot).
|
||||
|
||||
## Requirements
|
||||
|
||||
- An ElevenLabs account with an [API key](/app/settings/api-keys).
|
||||
- A [Supabase](https://supabase.com) account (you can sign up for a free account via [database.new](https://database.new)).
|
||||
- The [Supabase CLI](https://supabase.com/docs/guides/local-development) installed on your machine.
|
||||
- The [Deno runtime](https://docs.deno.com/runtime/getting_started/installation/) installed on your machine and optionally [setup in your facourite IDE](https://docs.deno.com/runtime/getting_started/setup_your_environment).
|
||||
- A [Telegram](https://telegram.org) account.
|
||||
|
||||
## Setup
|
||||
|
||||
### Register the Telegram bot
|
||||
|
||||
Next, use [the BotFather](https://t.me/BotFather) to create a new Telegram bot. Run the `/newbot` command and follow the instructions to create a new bot. At the end, you will receive your secret bot token. Note it down securely for the next step.
|
||||
|
||||

|
||||
|
||||
### Set up the environment variables
|
||||
|
||||
- `cp supabase/functions/.env.example supabase/functions/.env`
|
||||
- Update the `.env` file with your values.
|
||||
|
||||
## Test locally
|
||||
|
||||
- `supabase start`
|
||||
- `supabase functions serve --no-verify-jwt --env-file supabase/functions/.env`
|
||||
- In another terminal use [ngrok](https://ngrok.com/) to tunnel webhooks to the local server: `ngrok http 54321`
|
||||
- Set the bot's webhook url to the ngrok url: `https://api.telegram.org/bot<TELEGRAM_BOT_TOKEN>/setWebhook?url=https://<NGROK_URL>/functions/v1/elevenlabs-speech-to-text?secret=<FUNCTION_SECRET>`
|
||||
|
||||
Note: For background tasks to work locally, you need to set the `per_worker` policy in the [`supabase/config.toml`](./supabase/config.toml) file.
|
||||
|
||||
```
|
||||
[edge_runtime]
|
||||
enabled = true
|
||||
policy = "per_worker"
|
||||
```
|
||||
|
||||
## Deploy
|
||||
|
||||
1. Run `supabase link` and link your local project to your Supabase account.
|
||||
2. Run `supabase db push` to push the [setup migration](./supabase/migrations/20250203045928_init.sql) to your Supabase database.
|
||||
3. Run `supabase functions deploy --no-verify-jwt elevenlabs-speech-to-text`
|
||||
4. Run `supabase secrets set --env-file supabase/functions/.env`
|
||||
5. Set your bot's webhook url to `https://<PROJECT_REFERENCE>.functions.supabase.co/telegram-bot` (Replacing `<...>` with respective values). In order to do that, run this url (in your browser, for example): `https://api.telegram.org/bot<TELEGRAM_BOT_TOKEN>/setWebhook?url=https://<PROJECT_REFERENCE>.supabase.co/functions/v1/elevenlabs-speech-to-text?secret=<FUNCTION_SECRET>`
|
||||
6. That's it, go ahead and chat with your bot 🤖💬
|
||||
@@ -0,0 +1,3 @@
|
||||
{
|
||||
"imports": {}
|
||||
}
|
||||
@@ -0,0 +1,154 @@
|
||||
// Follow this setup guide to integrate the Deno language server with your editor:
|
||||
// https://deno.land/manual/getting_started/setup_your_environment
|
||||
// This enables autocomplete, go to definition, etc.
|
||||
|
||||
// Setup type definitions for built-in Supabase Runtime APIs
|
||||
import "jsr:@supabase/functions-js/edge-runtime.d.ts";
|
||||
import { createClient } from "jsr:@supabase/supabase-js@2";
|
||||
|
||||
console.log(`Function "elevenlabs-scribe-bot" up and running!`);
|
||||
|
||||
import { ElevenLabsClient } from "npm:elevenlabs@1.50.5";
|
||||
import {
|
||||
Bot,
|
||||
webhookCallback,
|
||||
} from "https://deno.land/x/grammy@v1.34.0/mod.ts";
|
||||
|
||||
const elevenLabsClient = new ElevenLabsClient({
|
||||
apiKey: Deno.env.get("ELEVENLABS_API_KEY") || "",
|
||||
});
|
||||
|
||||
const supabase = createClient(
|
||||
Deno.env.get("SUPABASE_URL") || "",
|
||||
Deno.env.get("SUPABASE_SERVICE_ROLE_KEY") || "",
|
||||
);
|
||||
|
||||
async function scribe(
|
||||
{ fileURL, fileType, duration, chatId, messageId, username }: {
|
||||
fileURL: string;
|
||||
fileType: string;
|
||||
duration: number;
|
||||
chatId: number;
|
||||
messageId: number;
|
||||
username: string;
|
||||
},
|
||||
) {
|
||||
let transcript: string | null = null;
|
||||
let languageCode: string | null = null;
|
||||
let errorMsg: string | null = null;
|
||||
try {
|
||||
const sourceFileArrayBuffer = await fetch(fileURL).then((res) =>
|
||||
res.arrayBuffer()
|
||||
);
|
||||
const sourceBlob = new Blob([sourceFileArrayBuffer], {
|
||||
type: fileType,
|
||||
});
|
||||
|
||||
const scribeResult = await elevenLabsClient.speechToText.convert({
|
||||
file: sourceBlob,
|
||||
model_id: "scribe_v1",
|
||||
tag_audio_events: false,
|
||||
});
|
||||
// console.log({ scribeResult });
|
||||
transcript = scribeResult.text;
|
||||
languageCode = scribeResult.language_code;
|
||||
|
||||
// Reply to the user with the transcript
|
||||
await bot.api.sendMessage(chatId, transcript, {
|
||||
reply_parameters: { message_id: messageId },
|
||||
});
|
||||
} catch (error) {
|
||||
errorMsg = error.message;
|
||||
console.log(errorMsg);
|
||||
await bot.api.sendMessage(
|
||||
chatId,
|
||||
"Sorry, there was an error. Please try again.",
|
||||
{
|
||||
reply_parameters: { message_id: messageId },
|
||||
},
|
||||
);
|
||||
}
|
||||
// Write log to Supabase.
|
||||
const logLine = {
|
||||
file_type: fileType,
|
||||
duration,
|
||||
chat_id: chatId,
|
||||
message_id: messageId,
|
||||
username,
|
||||
language_code: languageCode,
|
||||
error: errorMsg,
|
||||
};
|
||||
console.log({ logLine });
|
||||
await supabase.from("transcription_logs").insert({ ...logLine, transcript });
|
||||
}
|
||||
|
||||
// Use beforeunload event handler to be notified when function is about to shutdown
|
||||
addEventListener("beforeunload", (ev) => {
|
||||
console.log("Function will be shutdown due to", ev.detail?.reason);
|
||||
|
||||
// save state or log the current progress
|
||||
});
|
||||
|
||||
const telegramBotToken = Deno.env.get("TELEGRAM_BOT_TOKEN");
|
||||
const bot = new Bot(telegramBotToken || "");
|
||||
const startMessage =
|
||||
`Welcome to the ElevenLabs Scribe Bot\\! I can transcribe speech in 80\\+ languages with super high accuracy\\!
|
||||
\nTry it out by sending or forwarding me a voice message, video, or audio file\\!
|
||||
\n[Learn more about Scribe](https://elevenlabs.io/speech-to-text) or [build your own bot](https://elevenlabs.io/docs/cookbooks/speech-to-text/telegram-bot)\\!
|
||||
`;
|
||||
bot.command(
|
||||
"start",
|
||||
(ctx) => ctx.reply(startMessage.trim(), { parse_mode: "MarkdownV2" }),
|
||||
);
|
||||
|
||||
bot.on([":voice", ":audio", ":video"], async (ctx) => {
|
||||
try {
|
||||
// console.log(ctx);
|
||||
const file = await ctx.getFile();
|
||||
const fileURL =
|
||||
`https://api.telegram.org/file/bot${telegramBotToken}/${file.file_path}`;
|
||||
const fileMeta = ctx.message?.video ?? ctx.message?.voice ??
|
||||
ctx.message?.audio;
|
||||
// console.log({ fileURL, fileMeta });
|
||||
if (!fileMeta) {
|
||||
return ctx.reply(
|
||||
"No video|audio|voice metadata found. Please try again.",
|
||||
);
|
||||
}
|
||||
|
||||
// Run the transcription in the background.
|
||||
EdgeRuntime.waitUntil(
|
||||
scribe({
|
||||
fileURL,
|
||||
fileType: fileMeta.mime_type!,
|
||||
duration: fileMeta.duration,
|
||||
chatId: ctx.chat.id,
|
||||
messageId: ctx.message?.message_id!,
|
||||
username: ctx.from?.username || "",
|
||||
}),
|
||||
);
|
||||
|
||||
// Reply to the user immediately to let them know we received their file.
|
||||
return ctx.reply("Received. Scribing...");
|
||||
} catch (error) {
|
||||
console.error(error);
|
||||
return ctx.reply(
|
||||
"Sorry, there was an error getting the file. Please try again with a smaller file!",
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
const handleUpdate = webhookCallback(bot, "std/http");
|
||||
|
||||
Deno.serve(async (req) => {
|
||||
try {
|
||||
const url = new URL(req.url);
|
||||
if (url.searchParams.get("secret") !== Deno.env.get("FUNCTION_SECRET")) {
|
||||
return new Response("not allowed", { status: 405 });
|
||||
}
|
||||
|
||||
return await handleUpdate(req);
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
}
|
||||
});
|
||||
@@ -0,0 +1,3 @@
|
||||
# Configuration for private npm package dependencies
|
||||
# For more information on using private registries with Edge Functions, see:
|
||||
# https://supabase.com/docs/guides/functions/import-maps#importing-from-private-registries
|
||||
@@ -0,0 +1,120 @@
|
||||
# Streaming and Caching Speech with ElevenLabs
|
||||
|
||||
Generate and stream speech through Supabase Edge Functions. Store speech in Supabase Storage and cache responses via built-in smart CDN.
|
||||
|
||||
## Requirements
|
||||
|
||||
- An ElevenLabs account with an [API key](/app/settings/api-keys).
|
||||
- A [Supabase](https://supabase.com) account (you can sign up for a free account via [database.new](https://database.new)).
|
||||
- The [Supabase CLI](https://supabase.com/docs/guides/local-development) installed on your machine.
|
||||
- The [Deno runtime](https://docs.deno.com/runtime/getting_started/installation/) installed on your machine and optionally [setup in your facourite IDE](https://docs.deno.com/runtime/getting_started/setup_your_environment).
|
||||
|
||||
## Setup
|
||||
|
||||
### Create a Supabase project locally
|
||||
|
||||
After installing the [Supabase CLI](https://supabase.com/docs/guides/local-development), run the following command to create a new Supabase project locally:
|
||||
|
||||
```bash
|
||||
supabase init
|
||||
```
|
||||
|
||||
### Configure the storage bucket
|
||||
|
||||
You can configure the Supabase CLI to automatically generate a storage bucket by adding this configuration in the `config.toml` file:
|
||||
|
||||
```toml ./supabase/config.toml
|
||||
[storage.buckets.audio]
|
||||
public = false
|
||||
file_size_limit = "50MiB"
|
||||
allowed_mime_types = ["audio/mp3"]
|
||||
objects_path = "./audio"
|
||||
```
|
||||
|
||||
<Note>
|
||||
Upon running `supabase start` this will create a new storage bucket in your local Supabase
|
||||
project. Should you want to push this to your hosted Supabase project, you can run `supabase seed
|
||||
buckets --linked`.
|
||||
</Note>
|
||||
|
||||
### Configure background tasks for Supabase Edge Functions
|
||||
|
||||
To use background tasks in Supabase Edge Functions when developing locally, you need to add the following configuration in the `config.toml` file:
|
||||
|
||||
```toml ./supabase/config.toml
|
||||
[edge_runtime]
|
||||
policy = "per_worker"
|
||||
```
|
||||
|
||||
<Note>
|
||||
When running with `per_worker` policy, Function won't auto-reload on edits. You will need to
|
||||
manually restart it by running `supabase functions serve`.
|
||||
</Note>
|
||||
|
||||
## Run locally
|
||||
|
||||
To run the function locally, run the following commands:
|
||||
|
||||
```bash
|
||||
supabase start
|
||||
```
|
||||
|
||||
Once the local Supabase stack is up and running, run the following command to start the function and observe the logs:
|
||||
|
||||
```bash
|
||||
supabase functions serve
|
||||
```
|
||||
|
||||
## Deploy to Supabase
|
||||
|
||||
If you haven't already, create a new Supabase account at [database.new](https://database.new) and link the local project to your Supabase account:
|
||||
|
||||
```bash
|
||||
supabase link
|
||||
```
|
||||
|
||||
Once done, run the following command to deploy the function:
|
||||
|
||||
```bash
|
||||
supabase functions deploy
|
||||
```
|
||||
|
||||
### Set the function secrets
|
||||
|
||||
Now that you have all your secrets set locally, you can run the following command to set the secrets in your Supabase project:
|
||||
|
||||
```bash
|
||||
supabase secrets set --env-file supabase/functions/.env
|
||||
```
|
||||
|
||||
## Test the function
|
||||
|
||||
The function is designed in a way that it can be used directly as a source for an `<audio>` element.
|
||||
|
||||
```html
|
||||
<audio
|
||||
src="https://${SUPABASE_PROJECT_REF}.supabase.co/functions/v1/elevenlabs-text-to-speech?text=Hello%2C%20world!&voiceId=JBFqnCBsd6RMkjVDRZzb"
|
||||
controls
|
||||
/>
|
||||
```
|
||||
|
||||
You can find an example frontend implementation in the complete code example on [GitHub](https://github.com/elevenlabs/elevenlabs-examples/tree/main/examples/text-to-speech/supabase/stream-and-cache-storage/src/pages/Index.tsx).
|
||||
|
||||
### Try it out
|
||||
|
||||
Navigate to `http://127.0.0.1:54321/functions/v1/elevenlabs-text-to-speech?text=hello%20world`.
|
||||
|
||||
Afterwards, navigate to `http://127.0.0.1:54323/project/default/storage/buckets/audio` to see the audio file in your local Supabase Storage bucket.
|
||||
|
||||
## Test the function
|
||||
|
||||
The function is designed in a way that it can be used directly as a source for an `<audio>` element.
|
||||
|
||||
```html
|
||||
<audio
|
||||
src="https://${SUPABASE_PROJECT_REF}.supabase.co/functions/v1/elevenlabs-text-to-speech?text=Hello%2C%20world!&voiceId=JBFqnCBsd6RMkjVDRZzb"
|
||||
controls
|
||||
/>
|
||||
```
|
||||
|
||||
You can find an example frontend implementation in the complete code example on [GitHub](https://github.com/elevenlabs/elevenlabs-examples/tree/main/examples/text-to-speech/supabase/stream-and-cache-storage/src/pages/Index.tsx).
|
||||
@@ -0,0 +1,3 @@
|
||||
{
|
||||
"imports": {}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
// Setup type definitions for built-in Supabase Runtime APIs
|
||||
import "jsr:@supabase/functions-js/edge-runtime.d.ts";
|
||||
import { createClient } from "jsr:@supabase/supabase-js@2";
|
||||
import { ElevenLabsClient } from "npm:elevenlabs@1.52.0";
|
||||
import * as hash from "npm:object-hash";
|
||||
|
||||
const supabase = createClient(
|
||||
Deno.env.get("SUPABASE_URL")!,
|
||||
Deno.env.get("SUPABASE_SERVICE_ROLE_KEY")!,
|
||||
);
|
||||
|
||||
const client = new ElevenLabsClient({
|
||||
apiKey: Deno.env.get("ELEVENLABS_API_KEY"),
|
||||
});
|
||||
|
||||
// Upload audio to Supabase Storage in a background task
|
||||
async function uploadAudioToStorage(
|
||||
stream: ReadableStream,
|
||||
requestHash: string,
|
||||
) {
|
||||
const { data, error } = await supabase.storage
|
||||
.from("audio")
|
||||
.upload(`${requestHash}.mp3`, stream, {
|
||||
contentType: "audio/mp3",
|
||||
});
|
||||
|
||||
console.log("Storage upload result", { data, error });
|
||||
}
|
||||
|
||||
Deno.serve(async (req) => {
|
||||
// To secure your function for production, you can for example validate the request origin,
|
||||
// or append a user access token and validate it with Supabase Auth.
|
||||
console.log("Request origin", req.headers.get("host"));
|
||||
const url = new URL(req.url);
|
||||
const params = new URLSearchParams(url.search);
|
||||
const text = params.get("text");
|
||||
const voiceId = params.get("voiceId") ?? "JBFqnCBsd6RMkjVDRZzb";
|
||||
|
||||
const requestHash = hash.MD5({ text, voiceId });
|
||||
console.log("Request hash", requestHash);
|
||||
|
||||
// Check storage for existing audio file
|
||||
const { data } = await supabase
|
||||
.storage
|
||||
.from("audio")
|
||||
.createSignedUrl(`${requestHash}.mp3`, 60);
|
||||
|
||||
if (data) {
|
||||
console.log("Audio file found in storage", data);
|
||||
const storageRes = await fetch(data.signedUrl);
|
||||
if (storageRes.ok) return storageRes;
|
||||
}
|
||||
|
||||
if (!text) {
|
||||
return new Response(
|
||||
JSON.stringify({ error: "Text parameter is required" }),
|
||||
{ status: 400, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}
|
||||
|
||||
try {
|
||||
console.log("ElevenLabs API call");
|
||||
const response = await client.textToSpeech.convertAsStream(voiceId, {
|
||||
output_format: "mp3_44100_128",
|
||||
model_id: "eleven_multilingual_v2",
|
||||
text,
|
||||
});
|
||||
|
||||
const stream = new ReadableStream({
|
||||
async start(controller) {
|
||||
for await (const chunk of response) {
|
||||
controller.enqueue(chunk);
|
||||
}
|
||||
controller.close();
|
||||
},
|
||||
});
|
||||
|
||||
// Branch stream to Supabase Storage
|
||||
const [browserStream, storageStream] = stream.tee();
|
||||
|
||||
// Upload to Supabase Storage in the background
|
||||
EdgeRuntime.waitUntil(uploadAudioToStorage(storageStream, requestHash));
|
||||
|
||||
// Return the streaming response immediately
|
||||
return new Response(browserStream, {
|
||||
headers: {
|
||||
"Content-Type": "audio/mpeg",
|
||||
},
|
||||
});
|
||||
} catch (error) {
|
||||
console.log("error", { error });
|
||||
return new Response(JSON.stringify({ error: error.message }), {
|
||||
status: 500,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
});
|
||||
Reference in new issue
Block a user