Thor/add elevenlabs streaming example n docs (#33989)

* feat: add TTS & STT docs.

* feat: add examples.

* fix: docs.

* code review nits.

* chore: format.

* Fix mdx-lint errors.

---------

Co-authored-by: Ivan Vasilov <vasilov.ivan@gmail.com>
This commit is contained in:
Thor 雷神 SchaeffandIvan Vasilov authored and GitHub committed 2025-03-07 09:08:56 +01:00
1 parent 8120245eb7
commit 99f1d1fc16
21 files changed
+992 -1

No files matched your search

@@ -52,3 +52,6 @@ UPSTASH_REDIS_REST_TOKEN=
# connect-supabase - https://supabase.com/docs/guides/platform/oauth-apps/publish-an-oauth-app
SUPA_CONNECT_CLIENT_ID=
SUPA_CONNECT_CLIENT_SECRET=
# elevenlabs-text-to-speech
ELEVENLABS_API_KEY=
@@ -0,0 +1,3 @@
# Configuration for private npm package dependencies
# For more information on using private registries with Edge Functions, see:
# https://supabase.com/docs/guides/functions/import-maps#importing-from-private-registries
@@ -0,0 +1,52 @@
## ElevenLabs Scribe Telegram Bot
This is a Telegram bot that uses the ElevenLabs API to transcribe voice messages, as well as audio and video files.
You can find the bot here: https://t.me/ElevenLabsScribeBot
For a detailed tutorial, please see the [ElevenLabs Developer Docs](https://elevenlabs.io/docs/cookbooks/speech-to-text/telegram-bot).
## Requirements
- An ElevenLabs account with an [API key](/app/settings/api-keys).
- A [Supabase](https://supabase.com) account (you can sign up for a free account via [database.new](https://database.new)).
- The [Supabase CLI](https://supabase.com/docs/guides/local-development) installed on your machine.
- The [Deno runtime](https://docs.deno.com/runtime/getting_started/installation/) installed on your machine and optionally [setup in your facourite IDE](https://docs.deno.com/runtime/getting_started/setup_your_environment).
- A [Telegram](https://telegram.org) account.
## Setup
### Register the Telegram bot
Next, use [the BotFather](https://t.me/BotFather) to create a new Telegram bot. Run the `/newbot` command and follow the instructions to create a new bot. At the end, you will receive your secret bot token. Note it down securely for the next step.
![BotFather](/assets/images/cookbooks/scribe/telegram-bot/bot-father.png)
### Set up the environment variables
- `cp supabase/functions/.env.example supabase/functions/.env`
- Update the `.env` file with your values.
## Test locally
- `supabase start`
- `supabase functions serve --no-verify-jwt --env-file supabase/functions/.env`
- In another terminal use [ngrok](https://ngrok.com/) to tunnel webhooks to the local server: `ngrok http 54321`
- Set the bot's webhook url to the ngrok url: `https://api.telegram.org/bot<TELEGRAM_BOT_TOKEN>/setWebhook?url=https://<NGROK_URL>/functions/v1/elevenlabs-speech-to-text?secret=<FUNCTION_SECRET>`
Note: For background tasks to work locally, you need to set the `per_worker` policy in the [`supabase/config.toml`](./supabase/config.toml) file.
```
[edge_runtime]
enabled = true
policy = "per_worker"
```
## Deploy
1. Run `supabase link` and link your local project to your Supabase account.
2. Run `supabase db push` to push the [setup migration](./supabase/migrations/20250203045928_init.sql) to your Supabase database.
3. Run `supabase functions deploy --no-verify-jwt elevenlabs-speech-to-text`
4. Run `supabase secrets set --env-file supabase/functions/.env`
5. Set your bot's webhook url to `https://<PROJECT_REFERENCE>.functions.supabase.co/telegram-bot` (Replacing `<...>` with respective values). In order to do that, run this url (in your browser, for example): `https://api.telegram.org/bot<TELEGRAM_BOT_TOKEN>/setWebhook?url=https://<PROJECT_REFERENCE>.supabase.co/functions/v1/elevenlabs-speech-to-text?secret=<FUNCTION_SECRET>`
6. That's it, go ahead and chat with your bot 🤖💬
@@ -0,0 +1,3 @@
{
"imports": {}
}
@@ -0,0 +1,154 @@
// Follow this setup guide to integrate the Deno language server with your editor:
// https://deno.land/manual/getting_started/setup_your_environment
// This enables autocomplete, go to definition, etc.
// Setup type definitions for built-in Supabase Runtime APIs
import "jsr:@supabase/functions-js/edge-runtime.d.ts";
import { createClient } from "jsr:@supabase/supabase-js@2";
console.log(`Function "elevenlabs-scribe-bot" up and running!`);
import { ElevenLabsClient } from "npm:elevenlabs@1.50.5";
import {
Bot,
webhookCallback,
} from "https://deno.land/x/grammy@v1.34.0/mod.ts";
const elevenLabsClient = new ElevenLabsClient({
apiKey: Deno.env.get("ELEVENLABS_API_KEY") || "",
});
const supabase = createClient(
Deno.env.get("SUPABASE_URL") || "",
Deno.env.get("SUPABASE_SERVICE_ROLE_KEY") || "",
);
async function scribe(
{ fileURL, fileType, duration, chatId, messageId, username }: {
fileURL: string;
fileType: string;
duration: number;
chatId: number;
messageId: number;
username: string;
},
) {
let transcript: string | null = null;
let languageCode: string | null = null;
let errorMsg: string | null = null;
try {
const sourceFileArrayBuffer = await fetch(fileURL).then((res) =>
res.arrayBuffer()
);
const sourceBlob = new Blob([sourceFileArrayBuffer], {
type: fileType,
});
const scribeResult = await elevenLabsClient.speechToText.convert({
file: sourceBlob,
model_id: "scribe_v1",
tag_audio_events: false,
});
// console.log({ scribeResult });
transcript = scribeResult.text;
languageCode = scribeResult.language_code;
// Reply to the user with the transcript
await bot.api.sendMessage(chatId, transcript, {
reply_parameters: { message_id: messageId },
});
} catch (error) {
errorMsg = error.message;
console.log(errorMsg);
await bot.api.sendMessage(
chatId,
"Sorry, there was an error. Please try again.",
{
reply_parameters: { message_id: messageId },
},
);
}
// Write log to Supabase.
const logLine = {
file_type: fileType,
duration,
chat_id: chatId,
message_id: messageId,
username,
language_code: languageCode,
error: errorMsg,
};
console.log({ logLine });
await supabase.from("transcription_logs").insert({ ...logLine, transcript });
}
// Use beforeunload event handler to be notified when function is about to shutdown
addEventListener("beforeunload", (ev) => {
console.log("Function will be shutdown due to", ev.detail?.reason);
// save state or log the current progress
});
const telegramBotToken = Deno.env.get("TELEGRAM_BOT_TOKEN");
const bot = new Bot(telegramBotToken || "");
const startMessage =
`Welcome to the ElevenLabs Scribe Bot\\! I can transcribe speech in 80\\+ languages with super high accuracy\\!
\nTry it out by sending or forwarding me a voice message, video, or audio file\\!
\n[Learn more about Scribe](https://elevenlabs.io/speech-to-text) or [build your own bot](https://elevenlabs.io/docs/cookbooks/speech-to-text/telegram-bot)\\!
`;
bot.command(
"start",
(ctx) => ctx.reply(startMessage.trim(), { parse_mode: "MarkdownV2" }),
);
bot.on([":voice", ":audio", ":video"], async (ctx) => {
try {
// console.log(ctx);
const file = await ctx.getFile();
const fileURL =
`https://api.telegram.org/file/bot${telegramBotToken}/${file.file_path}`;
const fileMeta = ctx.message?.video ?? ctx.message?.voice ??
ctx.message?.audio;
// console.log({ fileURL, fileMeta });
if (!fileMeta) {
return ctx.reply(
"No video|audio|voice metadata found. Please try again.",
);
}
// Run the transcription in the background.
EdgeRuntime.waitUntil(
scribe({
fileURL,
fileType: fileMeta.mime_type!,
duration: fileMeta.duration,
chatId: ctx.chat.id,
messageId: ctx.message?.message_id!,
username: ctx.from?.username || "",
}),
);
// Reply to the user immediately to let them know we received their file.
return ctx.reply("Received. Scribing...");
} catch (error) {
console.error(error);
return ctx.reply(
"Sorry, there was an error getting the file. Please try again with a smaller file!",
);
}
});
const handleUpdate = webhookCallback(bot, "std/http");
Deno.serve(async (req) => {
try {
const url = new URL(req.url);
if (url.searchParams.get("secret") !== Deno.env.get("FUNCTION_SECRET")) {
return new Response("not allowed", { status: 405 });
}
return await handleUpdate(req);
} catch (err) {
console.error(err);
}
});
@@ -0,0 +1,3 @@
# Configuration for private npm package dependencies
# For more information on using private registries with Edge Functions, see:
# https://supabase.com/docs/guides/functions/import-maps#importing-from-private-registries
@@ -0,0 +1,120 @@
# Streaming and Caching Speech with ElevenLabs
Generate and stream speech through Supabase Edge Functions. Store speech in Supabase Storage and cache responses via built-in smart CDN.
## Requirements
- An ElevenLabs account with an [API key](/app/settings/api-keys).
- A [Supabase](https://supabase.com) account (you can sign up for a free account via [database.new](https://database.new)).
- The [Supabase CLI](https://supabase.com/docs/guides/local-development) installed on your machine.
- The [Deno runtime](https://docs.deno.com/runtime/getting_started/installation/) installed on your machine and optionally [setup in your facourite IDE](https://docs.deno.com/runtime/getting_started/setup_your_environment).
## Setup
### Create a Supabase project locally
After installing the [Supabase CLI](https://supabase.com/docs/guides/local-development), run the following command to create a new Supabase project locally:
```bash
supabase init
```
### Configure the storage bucket
You can configure the Supabase CLI to automatically generate a storage bucket by adding this configuration in the `config.toml` file:
```toml ./supabase/config.toml
[storage.buckets.audio]
public = false
file_size_limit = "50MiB"
allowed_mime_types = ["audio/mp3"]
objects_path = "./audio"
```
<Note>
Upon running `supabase start` this will create a new storage bucket in your local Supabase
project. Should you want to push this to your hosted Supabase project, you can run `supabase seed
buckets --linked`.
</Note>
### Configure background tasks for Supabase Edge Functions
To use background tasks in Supabase Edge Functions when developing locally, you need to add the following configuration in the `config.toml` file:
```toml ./supabase/config.toml
[edge_runtime]
policy = "per_worker"
```
<Note>
When running with `per_worker` policy, Function won't auto-reload on edits. You will need to
manually restart it by running `supabase functions serve`.
</Note>
## Run locally
To run the function locally, run the following commands:
```bash
supabase start
```
Once the local Supabase stack is up and running, run the following command to start the function and observe the logs:
```bash
supabase functions serve
```
## Deploy to Supabase
If you haven't already, create a new Supabase account at [database.new](https://database.new) and link the local project to your Supabase account:
```bash
supabase link
```
Once done, run the following command to deploy the function:
```bash
supabase functions deploy
```
### Set the function secrets
Now that you have all your secrets set locally, you can run the following command to set the secrets in your Supabase project:
```bash
supabase secrets set --env-file supabase/functions/.env
```
## Test the function
The function is designed in a way that it can be used directly as a source for an `<audio>` element.
```html
<audio
src="https://${SUPABASE_PROJECT_REF}.supabase.co/functions/v1/elevenlabs-text-to-speech?text=Hello%2C%20world!&voiceId=JBFqnCBsd6RMkjVDRZzb"
controls
/>
```
You can find an example frontend implementation in the complete code example on [GitHub](https://github.com/elevenlabs/elevenlabs-examples/tree/main/examples/text-to-speech/supabase/stream-and-cache-storage/src/pages/Index.tsx).
### Try it out
Navigate to `http://127.0.0.1:54321/functions/v1/elevenlabs-text-to-speech?text=hello%20world`.
Afterwards, navigate to `http://127.0.0.1:54323/project/default/storage/buckets/audio` to see the audio file in your local Supabase Storage bucket.
## Test the function
The function is designed in a way that it can be used directly as a source for an `<audio>` element.
```html
<audio
src="https://${SUPABASE_PROJECT_REF}.supabase.co/functions/v1/elevenlabs-text-to-speech?text=Hello%2C%20world!&voiceId=JBFqnCBsd6RMkjVDRZzb"
controls
/>
```
You can find an example frontend implementation in the complete code example on [GitHub](https://github.com/elevenlabs/elevenlabs-examples/tree/main/examples/text-to-speech/supabase/stream-and-cache-storage/src/pages/Index.tsx).
@@ -0,0 +1,3 @@
{
"imports": {}
}
@@ -0,0 +1,97 @@
// Setup type definitions for built-in Supabase Runtime APIs
import "jsr:@supabase/functions-js/edge-runtime.d.ts";
import { createClient } from "jsr:@supabase/supabase-js@2";
import { ElevenLabsClient } from "npm:elevenlabs@1.52.0";
import * as hash from "npm:object-hash";
const supabase = createClient(
Deno.env.get("SUPABASE_URL")!,
Deno.env.get("SUPABASE_SERVICE_ROLE_KEY")!,
);
const client = new ElevenLabsClient({
apiKey: Deno.env.get("ELEVENLABS_API_KEY"),
});
// Upload audio to Supabase Storage in a background task
async function uploadAudioToStorage(
stream: ReadableStream,
requestHash: string,
) {
const { data, error } = await supabase.storage
.from("audio")
.upload(`${requestHash}.mp3`, stream, {
contentType: "audio/mp3",
});
console.log("Storage upload result", { data, error });
}
Deno.serve(async (req) => {
// To secure your function for production, you can for example validate the request origin,
// or append a user access token and validate it with Supabase Auth.
console.log("Request origin", req.headers.get("host"));
const url = new URL(req.url);
const params = new URLSearchParams(url.search);
const text = params.get("text");
const voiceId = params.get("voiceId") ?? "JBFqnCBsd6RMkjVDRZzb";
const requestHash = hash.MD5({ text, voiceId });
console.log("Request hash", requestHash);
// Check storage for existing audio file
const { data } = await supabase
.storage
.from("audio")
.createSignedUrl(`${requestHash}.mp3`, 60);
if (data) {
console.log("Audio file found in storage", data);
const storageRes = await fetch(data.signedUrl);
if (storageRes.ok) return storageRes;
}
if (!text) {
return new Response(
JSON.stringify({ error: "Text parameter is required" }),
{ status: 400, headers: { "Content-Type": "application/json" } },
);
}
try {
console.log("ElevenLabs API call");
const response = await client.textToSpeech.convertAsStream(voiceId, {
output_format: "mp3_44100_128",
model_id: "eleven_multilingual_v2",
text,
});
const stream = new ReadableStream({
async start(controller) {
for await (const chunk of response) {
controller.enqueue(chunk);
}
controller.close();
},
});
// Branch stream to Supabase Storage
const [browserStream, storageStream] = stream.tee();
// Upload to Supabase Storage in the background
EdgeRuntime.waitUntil(uploadAudioToStorage(storageStream, requestHash));
// Return the streaming response immediately
return new Response(browserStream, {
headers: {
"Content-Type": "audio/mpeg",
},
});
} catch (error) {
console.log("error", { error });
return new Response(JSON.stringify({ error: error.message }), {
status: 500,
headers: { "Content-Type": "application/json" },
});
}
});