mirror of
https://github.com/supabase/supabase.git
synced 2026-10-05 09:25:06 +03:00
Thor/add elevenlabs streaming example n docs (#33989)
* feat: add TTS & STT docs. * feat: add examples. * fix: docs. * code review nits. * chore: format. * Fix mdx-lint errors. --------- Co-authored-by: Ivan Vasilov <vasilov.ivan@gmail.com>
This commit is contained in:
1 parent
8120245eb7
commit
99f1d1fc16
21 files changed
+992
-1
No files matched your search
@@ -1529,6 +1529,14 @@ export const functions: NavMenuConstant = {
|
||||
name: 'Type-Safe SQL with Kysely',
|
||||
url: '/guides/functions/kysely-postgres',
|
||||
},
|
||||
{
|
||||
name: 'Text To Speech with ElevenLabs',
|
||||
url: '/guides/functions/examples/elevenlabs-generate-speech-stream',
|
||||
},
|
||||
{
|
||||
name: 'Speech Transcription with ElevenLabs',
|
||||
url: '/guides/functions/examples/elevenlabs-transcribe-speech',
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
|
||||
@@ -0,0 +1,245 @@
|
||||
---
|
||||
title: Streaming Speech with ElevenLabs
|
||||
subtitle: Generate and stream speech through Supabase Edge Functions. Store speech in Supabase Storage and cache responses via built-in CDN.
|
||||
tocVideo: '4Roog4PAmZ8'
|
||||
---
|
||||
|
||||
## Introduction
|
||||
|
||||
In this tutorial you will learn how to build and edge API to generate, stream, store, and cache speech from text using Supabase Edge Functions, Supabase Storage, and ElevenLabs.`
|
||||
|
||||
<Admonition type="tip">
|
||||
Find the [example project on
|
||||
GitHub](https://github.com/elevenlabs/elevenlabs-examples/tree/main/examples/text-to-speech/supabase/stream-and-cache-storage).
|
||||
</Admonition>
|
||||
|
||||
## Requirements
|
||||
|
||||
- An ElevenLabs account with an [API key](/app/settings/api-keys).
|
||||
- A [Supabase](https://supabase.com) account (you can sign up for a free account via [database.new](https://database.new)).
|
||||
- The [Supabase CLI](https://supabase.com/docs/guides/local-development) installed on your machine.
|
||||
- The [Deno runtime](https://docs.deno.com/runtime/getting_started/installation/) installed on your machine and optionally [setup in your favourite IDE](https://docs.deno.com/runtime/getting_started/setup_your_environment).
|
||||
|
||||
## Setup
|
||||
|
||||
### Create a Supabase project locally
|
||||
|
||||
After installing the [Supabase CLI](https://supabase.com/docs/guides/local-development), run the following command to create a new Supabase project locally:
|
||||
|
||||
```bash
|
||||
supabase init
|
||||
```
|
||||
|
||||
### Configure the storage bucket
|
||||
|
||||
You can configure the Supabase CLI to automatically generate a storage bucket by adding this configuration in the `config.toml` file:
|
||||
|
||||
```toml ./supabase/config.toml
|
||||
[storage.buckets.audio]
|
||||
public = false
|
||||
file_size_limit = "50MiB"
|
||||
allowed_mime_types = ["audio/mp3"]
|
||||
objects_path = "./audio"
|
||||
```
|
||||
|
||||
<Admonition type="tip">
|
||||
Upon running `supabase start` this will create a new storage bucket in your local Supabase
|
||||
project. Should you want to push this to your hosted Supabase project, you can run `supabase seed
|
||||
buckets --linked`.
|
||||
</Admonition>
|
||||
|
||||
### Configure background tasks for Supabase Edge Functions
|
||||
|
||||
To use background tasks in Supabase Edge Functions when developing locally, you need to add the following configuration in the `config.toml` file:
|
||||
|
||||
```toml ./supabase/config.toml
|
||||
[edge_runtime]
|
||||
policy = "per_worker"
|
||||
```
|
||||
|
||||
<Admonition type="tip">
|
||||
When running with `per_worker` policy, Function won't auto-reload on edits. You will need to
|
||||
manually restart it by running `supabase functions serve`.
|
||||
</Admonition>
|
||||
|
||||
### Create a Supabase Edge Function for speech generation
|
||||
|
||||
Create a new Edge Function by running the following command:
|
||||
|
||||
```bash
|
||||
supabase functions new text-to-speech
|
||||
```
|
||||
|
||||
If you're using VS Code or Cursor, select `y` when the CLI prompts "Generate VS Code settings for Deno? [y/N]"!
|
||||
|
||||
### Set up the environment variables
|
||||
|
||||
Within the `supabase/functions` directory, create a new `.env` file and add the following variables:
|
||||
|
||||
```env supabase/functions/.env
|
||||
# Find / create an API key at https://elevenlabs.io/app/settings/api-keys
|
||||
ELEVENLABS_API_KEY=your_api_key
|
||||
```
|
||||
|
||||
### Dependencies
|
||||
|
||||
The project uses a couple of dependencies:
|
||||
|
||||
- The [@supabase/supabase-js](https://supabase.com/docs/reference/javascript) library to interact with the Supabase database.
|
||||
- The ElevenLabs [JavaScript SDK](/docs/quickstart) to interact with the text-to-speech API.
|
||||
- The open-source [object-hash](https://www.npmjs.com/package/object-hash) to generate a hash from the request parameters.
|
||||
|
||||
Since Supabase Edge Function uses the [Deno runtime](https://deno.land/), you don't need to install the dependencies, rather you can [import](https://docs.deno.com/examples/npm/) them via the `npm:` prefix.
|
||||
|
||||
## Code the Supabase Edge Function
|
||||
|
||||
In your newly created `supabase/functions/text-to-speech/index.ts` file, add the following code:
|
||||
|
||||
```ts supabase/functions/text-to-speech/index.ts
|
||||
// Setup type definitions for built-in Supabase Runtime APIs
|
||||
import 'jsr:@supabase/functions-js/edge-runtime.d.ts'
|
||||
import { createClient } from 'jsr:@supabase/supabase-js@2'
|
||||
import { ElevenLabsClient } from 'npm:elevenlabs@1.52.0'
|
||||
import * as hash from 'npm:object-hash'
|
||||
|
||||
const supabase = createClient(
|
||||
Deno.env.get('SUPABASE_URL')!,
|
||||
Deno.env.get('SUPABASE_SERVICE_ROLE_KEY')!
|
||||
)
|
||||
|
||||
const client = new ElevenLabsClient({
|
||||
apiKey: Deno.env.get('ELEVENLABS_API_KEY'),
|
||||
})
|
||||
|
||||
// Upload audio to Supabase Storage in a background task
|
||||
async function uploadAudioToStorage(stream: ReadableStream, requestHash: string) {
|
||||
const { data, error } = await supabase.storage
|
||||
.from('audio')
|
||||
.upload(`${requestHash}.mp3`, stream, {
|
||||
contentType: 'audio/mp3',
|
||||
})
|
||||
|
||||
console.log('Storage upload result', { data, error })
|
||||
}
|
||||
|
||||
Deno.serve(async (req) => {
|
||||
// To secure your function for production, you can for example validate the request origin,
|
||||
// or append a user access token and validate it with Supabase Auth.
|
||||
console.log('Request origin', req.headers.get('host'))
|
||||
const url = new URL(req.url)
|
||||
const params = new URLSearchParams(url.search)
|
||||
const text = params.get('text')
|
||||
const voiceId = params.get('voiceId') ?? 'JBFqnCBsd6RMkjVDRZzb'
|
||||
|
||||
const requestHash = hash.MD5({ text, voiceId })
|
||||
console.log('Request hash', requestHash)
|
||||
|
||||
// Check storage for existing audio file
|
||||
const { data } = await supabase.storage.from('audio').createSignedUrl(`${requestHash}.mp3`, 60)
|
||||
|
||||
if (data) {
|
||||
console.log('Audio file found in storage', data)
|
||||
const storageRes = await fetch(data.signedUrl)
|
||||
if (storageRes.ok) return storageRes
|
||||
}
|
||||
|
||||
if (!text) {
|
||||
return new Response(JSON.stringify({ error: 'Text parameter is required' }), {
|
||||
status: 400,
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
})
|
||||
}
|
||||
|
||||
try {
|
||||
console.log('ElevenLabs API call')
|
||||
const response = await client.textToSpeech.convertAsStream(voiceId, {
|
||||
output_format: 'mp3_44100_128',
|
||||
model_id: 'eleven_multilingual_v2',
|
||||
text,
|
||||
})
|
||||
|
||||
const stream = new ReadableStream({
|
||||
async start(controller) {
|
||||
for await (const chunk of response) {
|
||||
controller.enqueue(chunk)
|
||||
}
|
||||
controller.close()
|
||||
},
|
||||
})
|
||||
|
||||
// Branch stream to Supabase Storage
|
||||
const [browserStream, storageStream] = stream.tee()
|
||||
|
||||
// Upload to Supabase Storage in the background
|
||||
EdgeRuntime.waitUntil(uploadAudioToStorage(storageStream, requestHash))
|
||||
|
||||
// Return the streaming response immediately
|
||||
return new Response(browserStream, {
|
||||
headers: {
|
||||
'Content-Type': 'audio/mpeg',
|
||||
},
|
||||
})
|
||||
} catch (error) {
|
||||
console.log('error', { error })
|
||||
return new Response(JSON.stringify({ error: error.message }), {
|
||||
status: 500,
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
})
|
||||
}
|
||||
})
|
||||
```
|
||||
|
||||
## Run locally
|
||||
|
||||
To run the function locally, run the following commands:
|
||||
|
||||
```bash
|
||||
supabase start
|
||||
```
|
||||
|
||||
Once the local Supabase stack is up and running, run the following command to start the function and observe the logs:
|
||||
|
||||
```bash
|
||||
supabase functions serve
|
||||
```
|
||||
|
||||
### Try it out
|
||||
|
||||
Navigate to `http://127.0.0.1:54321/functions/v1/text-to-speech?text=hello%20world` to hear the function in action.
|
||||
|
||||
Afterwards, navigate to `http://127.0.0.1:54323/project/default/storage/buckets/audio` to see the audio file in your local Supabase Storage bucket.
|
||||
|
||||
## Deploy to Supabase
|
||||
|
||||
If you haven't already, create a new Supabase account at [database.new](https://database.new) and link the local project to your Supabase account:
|
||||
|
||||
```bash
|
||||
supabase link
|
||||
```
|
||||
|
||||
Once done, run the following command to deploy the function:
|
||||
|
||||
```bash
|
||||
supabase functions deploy
|
||||
```
|
||||
|
||||
### Set the function secrets
|
||||
|
||||
Now that you have all your secrets set locally, you can run the following command to set the secrets in your Supabase project:
|
||||
|
||||
```bash
|
||||
supabase secrets set --env-file supabase/functions/.env
|
||||
```
|
||||
|
||||
## Test the function
|
||||
|
||||
The function is designed in a way that it can be used directly as a source for an `<audio>` element.
|
||||
|
||||
```html
|
||||
<audio
|
||||
src="https://${SUPABASE_PROJECT_REF}.supabase.co/functions/v1/text-to-speech?text=Hello%2C%20world!&voiceId=JBFqnCBsd6RMkjVDRZzb"
|
||||
controls
|
||||
/>
|
||||
```
|
||||
|
||||
You can find an example frontend implementation in the complete code example on [GitHub](https://github.com/elevenlabs/elevenlabs-examples/tree/main/examples/text-to-speech/supabase/stream-and-cache-storage/src/pages/Index.tsx).
|
||||
@@ -0,0 +1,297 @@
|
||||
---
|
||||
title: Transcription Telegram Bot
|
||||
subtitle: Build a Telegram bot that transcribes audio and video messages in 99 languages using TypeScript with Deno in Supabase Edge Functions.
|
||||
tocVideo: 'CE4iPp7kd7Q'
|
||||
---
|
||||
|
||||
## Introduction
|
||||
|
||||
In this tutorial you will learn how to build a Telegram bot that transcribes audio and video messages in 99 languages using TypeScript and the ElevenLabs Scribe model via the speech-to-text API.
|
||||
|
||||
To check out what the end result will look like, you can test out the [t.me/ElevenLabsScribeBot](https://t.me/ElevenLabsScribeBot)
|
||||
|
||||
<Admonition type="tip">
|
||||
Find the [example project on
|
||||
GitHub](https://github.com/elevenlabs/elevenlabs-examples/tree/main/examples/speech-to-text/telegram-transcription-bot).
|
||||
</Admonition>
|
||||
|
||||
## Requirements
|
||||
|
||||
- An ElevenLabs account with an [API key](/app/settings/api-keys).
|
||||
- A [Supabase](https://supabase.com) account (you can sign up for a free account via [database.new](https://database.new)).
|
||||
- The [Supabase CLI](https://supabase.com/docs/guides/local-development) installed on your machine.
|
||||
- The [Deno runtime](https://docs.deno.com/runtime/getting_started/installation/) installed on your machine and optionally [setup in your favourite IDE](https://docs.deno.com/runtime/getting_started/setup_your_environment).
|
||||
- A [Telegram](https://telegram.org) account.
|
||||
|
||||
## Setup
|
||||
|
||||
### Register a Telegram bot
|
||||
|
||||
Use the [BotFather](https://t.me/BotFather) to create a new Telegram bot. Run the `/newbot` command and follow the instructions to create a new bot. At the end, you will receive your secret bot token. Note it down securely for the next step.
|
||||
|
||||

|
||||
|
||||
### Create a Supabase project locally
|
||||
|
||||
After installing the [Supabase CLI](https://supabase.com/docs/guides/local-development), run the following command to create a new Supabase project locally:
|
||||
|
||||
```bash
|
||||
supabase init
|
||||
```
|
||||
|
||||
### Create a database table to log the transcription results
|
||||
|
||||
Next, create a new database table to log the transcription results:
|
||||
|
||||
```bash
|
||||
supabase migrations new init
|
||||
```
|
||||
|
||||
This will create a new migration file in the `supabase/migrations` directory. Open the file and add the following SQL:
|
||||
|
||||
```sql supabase/migrations/init.sql
|
||||
CREATE TABLE IF NOT EXISTS transcription_logs (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
file_type VARCHAR NOT NULL,
|
||||
duration INTEGER NOT NULL,
|
||||
chat_id BIGINT NOT NULL,
|
||||
message_id BIGINT NOT NULL,
|
||||
username VARCHAR,
|
||||
transcript TEXT,
|
||||
language_code VARCHAR,
|
||||
created_at TIMESTAMP WITH TIME ZONE DEFAULT CURRENT_TIMESTAMP,
|
||||
error TEXT
|
||||
);
|
||||
|
||||
ALTER TABLE transcription_logs ENABLE ROW LEVEL SECURITY;
|
||||
```
|
||||
|
||||
### Create a Supabase Edge Function to handle Telegram webhook requests
|
||||
|
||||
Next, create a new Edge Function to handle Telegram webhook requests:
|
||||
|
||||
```bash
|
||||
supabase functions new scribe-bot
|
||||
```
|
||||
|
||||
If you're using VS Code or Cursor, select `y` when the CLI prompts "Generate VS Code settings for Deno? [y/N]"!
|
||||
|
||||
### Set up the environment variables
|
||||
|
||||
Within the `supabase/functions` directory, create a new `.env` file and add the following variables:
|
||||
|
||||
```env supabase/functions/.env
|
||||
# Find / create an API key at https://elevenlabs.io/app/settings/api-keys
|
||||
ELEVENLABS_API_KEY=your_api_key
|
||||
|
||||
# The bot token you received from the BotFather.
|
||||
TELEGRAM_BOT_TOKEN=your_bot_token
|
||||
|
||||
# A random secret chosen by you to secure the function.
|
||||
FUNCTION_SECRET=random_secret
|
||||
```
|
||||
|
||||
### Dependencies
|
||||
|
||||
The project uses a couple of dependencies:
|
||||
|
||||
- The open-source [grammY Framework](https://grammy.dev/) to handle the Telegram webhook requests.
|
||||
- The [@supabase/supabase-js](https://supabase.com/docs/reference/javascript) library to interact with the Supabase database.
|
||||
- The ElevenLabs [JavaScript SDK](/docs/quickstart) to interact with the speech-to-text API.
|
||||
|
||||
Since Supabase Edge Function uses the [Deno runtime](https://deno.land/), you don't need to install the dependencies, rather you can [import](https://docs.deno.com/examples/npm/) them via the `npm:` prefix.
|
||||
|
||||
## Code the Telegram bot
|
||||
|
||||
In your newly created `scribe-bot/index.ts` file, add the following code:
|
||||
|
||||
```ts supabase/functions/scribe-bot/index.ts
|
||||
import { Bot, webhookCallback } from 'https://deno.land/x/grammy@v1.34.0/mod.ts'
|
||||
import 'jsr:@supabase/functions-js/edge-runtime.d.ts'
|
||||
import { createClient } from 'jsr:@supabase/supabase-js@2'
|
||||
import { ElevenLabsClient } from 'npm:elevenlabs@1.50.5'
|
||||
|
||||
console.log(`Function "elevenlabs-scribe-bot" up and running!`)
|
||||
|
||||
const elevenLabsClient = new ElevenLabsClient({
|
||||
apiKey: Deno.env.get('ELEVENLABS_API_KEY') || '',
|
||||
})
|
||||
|
||||
const supabase = createClient(
|
||||
Deno.env.get('SUPABASE_URL') || '',
|
||||
Deno.env.get('SUPABASE_SERVICE_ROLE_KEY') || ''
|
||||
)
|
||||
|
||||
async function scribe({
|
||||
fileURL,
|
||||
fileType,
|
||||
duration,
|
||||
chatId,
|
||||
messageId,
|
||||
username,
|
||||
}: {
|
||||
fileURL: string
|
||||
fileType: string
|
||||
duration: number
|
||||
chatId: number
|
||||
messageId: number
|
||||
username: string
|
||||
}) {
|
||||
let transcript: string | null = null
|
||||
let languageCode: string | null = null
|
||||
let errorMsg: string | null = null
|
||||
try {
|
||||
const sourceFileArrayBuffer = await fetch(fileURL).then((res) => res.arrayBuffer())
|
||||
const sourceBlob = new Blob([sourceFileArrayBuffer], {
|
||||
type: fileType,
|
||||
})
|
||||
|
||||
const scribeResult = await elevenLabsClient.speechToText.convert({
|
||||
file: sourceBlob,
|
||||
model_id: 'scribe_v1',
|
||||
tag_audio_events: false,
|
||||
})
|
||||
|
||||
transcript = scribeResult.text
|
||||
languageCode = scribeResult.language_code
|
||||
|
||||
// Reply to the user with the transcript
|
||||
await bot.api.sendMessage(chatId, transcript, {
|
||||
reply_parameters: { message_id: messageId },
|
||||
})
|
||||
} catch (error) {
|
||||
errorMsg = error.message
|
||||
console.log(errorMsg)
|
||||
await bot.api.sendMessage(chatId, 'Sorry, there was an error. Please try again.', {
|
||||
reply_parameters: { message_id: messageId },
|
||||
})
|
||||
}
|
||||
// Write log to Supabase.
|
||||
const logLine = {
|
||||
file_type: fileType,
|
||||
duration,
|
||||
chat_id: chatId,
|
||||
message_id: messageId,
|
||||
username,
|
||||
language_code: languageCode,
|
||||
error: errorMsg,
|
||||
}
|
||||
console.log({ logLine })
|
||||
await supabase.from('transcription_logs').insert({ ...logLine, transcript })
|
||||
}
|
||||
|
||||
const telegramBotToken = Deno.env.get('TELEGRAM_BOT_TOKEN')
|
||||
const bot = new Bot(telegramBotToken || '')
|
||||
const startMessage = `Welcome to the ElevenLabs Scribe Bot\\! I can transcribe speech in 99 languages with super high accuracy\\!
|
||||
\nTry it out by sending or forwarding me a voice message, video, or audio file\\!
|
||||
\n[Learn more about Scribe](https://elevenlabs.io/speech-to-text) or [build your own bot](https://elevenlabs.io/docs/cookbooks/speech-to-text/telegram-bot)\\!
|
||||
`
|
||||
bot.command('start', (ctx) => ctx.reply(startMessage.trim(), { parse_mode: 'MarkdownV2' }))
|
||||
|
||||
bot.on([':voice', ':audio', ':video'], async (ctx) => {
|
||||
try {
|
||||
const file = await ctx.getFile()
|
||||
const fileURL = `https://api.telegram.org/file/bot${telegramBotToken}/${file.file_path}`
|
||||
const fileMeta = ctx.message?.video ?? ctx.message?.voice ?? ctx.message?.audio
|
||||
|
||||
if (!fileMeta) {
|
||||
return ctx.reply('No video|audio|voice metadata found. Please try again.')
|
||||
}
|
||||
|
||||
// Run the transcription in the background.
|
||||
EdgeRuntime.waitUntil(
|
||||
scribe({
|
||||
fileURL,
|
||||
fileType: fileMeta.mime_type!,
|
||||
duration: fileMeta.duration,
|
||||
chatId: ctx.chat.id,
|
||||
messageId: ctx.message?.message_id!,
|
||||
username: ctx.from?.username || '',
|
||||
})
|
||||
)
|
||||
|
||||
// Reply to the user immediately to let them know we received their file.
|
||||
return ctx.reply('Received. Scribing...')
|
||||
} catch (error) {
|
||||
console.error(error)
|
||||
return ctx.reply(
|
||||
'Sorry, there was an error getting the file. Please try again with a smaller file!'
|
||||
)
|
||||
}
|
||||
})
|
||||
|
||||
const handleUpdate = webhookCallback(bot, 'std/http')
|
||||
|
||||
Deno.serve(async (req) => {
|
||||
try {
|
||||
const url = new URL(req.url)
|
||||
if (url.searchParams.get('secret') !== Deno.env.get('FUNCTION_SECRET')) {
|
||||
return new Response('not allowed', { status: 405 })
|
||||
}
|
||||
|
||||
return await handleUpdate(req)
|
||||
} catch (err) {
|
||||
console.error(err)
|
||||
}
|
||||
})
|
||||
```
|
||||
|
||||
## Deploy to Supabase
|
||||
|
||||
If you haven't already, create a new Supabase account at [database.new](https://database.new) and link the local project to your Supabase account:
|
||||
|
||||
```bash
|
||||
supabase link
|
||||
```
|
||||
|
||||
### Apply the database migrations
|
||||
|
||||
Run the following command to apply the database migrations from the `supabase/migrations` directory:
|
||||
|
||||
```bash
|
||||
supabase db push
|
||||
```
|
||||
|
||||
Navigate to the [table editor](https://supabase.com/dashboard/project/_/editor) in your Supabase dashboard and you should see and empty `transcription_logs` table.
|
||||
|
||||

|
||||
|
||||
Lastly, run the following command to deploy the Edge Function:
|
||||
|
||||
```bash
|
||||
supabase functions deploy --no-verify-jwt scribe-bot
|
||||
```
|
||||
|
||||
Navigate to the [Edge Functions view](https://supabase.com/dashboard/project/_/functions) in your Supabase dashboard and you should see the `scribe-bot` function deployed. Make a note of the function URL as you'll need it later, it should look something like `https://<project-ref>.functions.supabase.co/scribe-bot`.
|
||||
|
||||

|
||||
|
||||
### Set up the webhook
|
||||
|
||||
Set your bot's webhook URL to `https://<PROJECT_REFERENCE>.functions.supabase.co/telegram-bot` (Replacing `<...>` with respective values). In order to do that, run a GET request to the following URL (in your browser, for example):
|
||||
|
||||
```
|
||||
https://api.telegram.org/bot<TELEGRAM_BOT_TOKEN>/setWebhook?url=https://<PROJECT_REFERENCE>.supabase.co/functions/v1/scribe-bot?secret=<FUNCTION_SECRET>
|
||||
```
|
||||
|
||||
Note that the `FUNCTION_SECRET` is the secret you set in your `.env` file.
|
||||
|
||||

|
||||
|
||||
### Set the function secrets
|
||||
|
||||
Now that you have all your secrets set locally, you can run the following command to set the secrets in your Supabase project:
|
||||
|
||||
```bash
|
||||
supabase secrets set --env-file supabase/functions/.env
|
||||
```
|
||||
|
||||
## Test the bot
|
||||
|
||||
Finally you can test the bot by sending it a voice message, audio or video file.
|
||||
|
||||

|
||||
|
||||
After you see the transcript as a reply, navigate back to your table editor in the Supabase dashboard and you should see a new row in your `transcription_logs` table.
|
||||
|
||||

|
||||
@@ -77,7 +77,7 @@ hideToc: true
|
||||
|
||||
You can change the schema of your Laravel application by modifying the `search_path` variable `app/config/database.php`.
|
||||
|
||||
Please note the schema you specify in `search_path` has to exist on Supabase. You can create a new schema from the [Table Editor](/dashboard/project/_/editor).
|
||||
The schema you specify in `search_path` has to exist on Supabase. You can create a new schema from the [Table Editor](/dashboard/project/_/editor).
|
||||
|
||||
</StepHikeCompact.Details>
|
||||
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 2.1 MiB |
Binary file not shown.
|
After Width: | Height: | Size: 1.2 MiB |
Binary file not shown.
|
After Width: | Height: | Size: 785 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 846 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 1.3 MiB |
Binary file not shown.
|
After Width: | Height: | Size: 2.6 MiB |
@@ -52,3 +52,6 @@ UPSTASH_REDIS_REST_TOKEN=
|
||||
# connect-supabase - https://supabase.com/docs/guides/platform/oauth-apps/publish-an-oauth-app
|
||||
SUPA_CONNECT_CLIENT_ID=
|
||||
SUPA_CONNECT_CLIENT_SECRET=
|
||||
|
||||
# elevenlabs-text-to-speech
|
||||
ELEVENLABS_API_KEY=
|
||||
@@ -0,0 +1,3 @@
|
||||
# Configuration for private npm package dependencies
|
||||
# For more information on using private registries with Edge Functions, see:
|
||||
# https://supabase.com/docs/guides/functions/import-maps#importing-from-private-registries
|
||||
@@ -0,0 +1,52 @@
|
||||
## ElevenLabs Scribe Telegram Bot
|
||||
|
||||
This is a Telegram bot that uses the ElevenLabs API to transcribe voice messages, as well as audio and video files.
|
||||
|
||||
You can find the bot here: https://t.me/ElevenLabsScribeBot
|
||||
|
||||
For a detailed tutorial, please see the [ElevenLabs Developer Docs](https://elevenlabs.io/docs/cookbooks/speech-to-text/telegram-bot).
|
||||
|
||||
## Requirements
|
||||
|
||||
- An ElevenLabs account with an [API key](/app/settings/api-keys).
|
||||
- A [Supabase](https://supabase.com) account (you can sign up for a free account via [database.new](https://database.new)).
|
||||
- The [Supabase CLI](https://supabase.com/docs/guides/local-development) installed on your machine.
|
||||
- The [Deno runtime](https://docs.deno.com/runtime/getting_started/installation/) installed on your machine and optionally [setup in your facourite IDE](https://docs.deno.com/runtime/getting_started/setup_your_environment).
|
||||
- A [Telegram](https://telegram.org) account.
|
||||
|
||||
## Setup
|
||||
|
||||
### Register the Telegram bot
|
||||
|
||||
Next, use [the BotFather](https://t.me/BotFather) to create a new Telegram bot. Run the `/newbot` command and follow the instructions to create a new bot. At the end, you will receive your secret bot token. Note it down securely for the next step.
|
||||
|
||||

|
||||
|
||||
### Set up the environment variables
|
||||
|
||||
- `cp supabase/functions/.env.example supabase/functions/.env`
|
||||
- Update the `.env` file with your values.
|
||||
|
||||
## Test locally
|
||||
|
||||
- `supabase start`
|
||||
- `supabase functions serve --no-verify-jwt --env-file supabase/functions/.env`
|
||||
- In another terminal use [ngrok](https://ngrok.com/) to tunnel webhooks to the local server: `ngrok http 54321`
|
||||
- Set the bot's webhook url to the ngrok url: `https://api.telegram.org/bot<TELEGRAM_BOT_TOKEN>/setWebhook?url=https://<NGROK_URL>/functions/v1/elevenlabs-speech-to-text?secret=<FUNCTION_SECRET>`
|
||||
|
||||
Note: For background tasks to work locally, you need to set the `per_worker` policy in the [`supabase/config.toml`](./supabase/config.toml) file.
|
||||
|
||||
```
|
||||
[edge_runtime]
|
||||
enabled = true
|
||||
policy = "per_worker"
|
||||
```
|
||||
|
||||
## Deploy
|
||||
|
||||
1. Run `supabase link` and link your local project to your Supabase account.
|
||||
2. Run `supabase db push` to push the [setup migration](./supabase/migrations/20250203045928_init.sql) to your Supabase database.
|
||||
3. Run `supabase functions deploy --no-verify-jwt elevenlabs-speech-to-text`
|
||||
4. Run `supabase secrets set --env-file supabase/functions/.env`
|
||||
5. Set your bot's webhook url to `https://<PROJECT_REFERENCE>.functions.supabase.co/telegram-bot` (Replacing `<...>` with respective values). In order to do that, run this url (in your browser, for example): `https://api.telegram.org/bot<TELEGRAM_BOT_TOKEN>/setWebhook?url=https://<PROJECT_REFERENCE>.supabase.co/functions/v1/elevenlabs-speech-to-text?secret=<FUNCTION_SECRET>`
|
||||
6. That's it, go ahead and chat with your bot 🤖💬
|
||||
@@ -0,0 +1,3 @@
|
||||
{
|
||||
"imports": {}
|
||||
}
|
||||
@@ -0,0 +1,154 @@
|
||||
// Follow this setup guide to integrate the Deno language server with your editor:
|
||||
// https://deno.land/manual/getting_started/setup_your_environment
|
||||
// This enables autocomplete, go to definition, etc.
|
||||
|
||||
// Setup type definitions for built-in Supabase Runtime APIs
|
||||
import "jsr:@supabase/functions-js/edge-runtime.d.ts";
|
||||
import { createClient } from "jsr:@supabase/supabase-js@2";
|
||||
|
||||
console.log(`Function "elevenlabs-scribe-bot" up and running!`);
|
||||
|
||||
import { ElevenLabsClient } from "npm:elevenlabs@1.50.5";
|
||||
import {
|
||||
Bot,
|
||||
webhookCallback,
|
||||
} from "https://deno.land/x/grammy@v1.34.0/mod.ts";
|
||||
|
||||
const elevenLabsClient = new ElevenLabsClient({
|
||||
apiKey: Deno.env.get("ELEVENLABS_API_KEY") || "",
|
||||
});
|
||||
|
||||
const supabase = createClient(
|
||||
Deno.env.get("SUPABASE_URL") || "",
|
||||
Deno.env.get("SUPABASE_SERVICE_ROLE_KEY") || "",
|
||||
);
|
||||
|
||||
async function scribe(
|
||||
{ fileURL, fileType, duration, chatId, messageId, username }: {
|
||||
fileURL: string;
|
||||
fileType: string;
|
||||
duration: number;
|
||||
chatId: number;
|
||||
messageId: number;
|
||||
username: string;
|
||||
},
|
||||
) {
|
||||
let transcript: string | null = null;
|
||||
let languageCode: string | null = null;
|
||||
let errorMsg: string | null = null;
|
||||
try {
|
||||
const sourceFileArrayBuffer = await fetch(fileURL).then((res) =>
|
||||
res.arrayBuffer()
|
||||
);
|
||||
const sourceBlob = new Blob([sourceFileArrayBuffer], {
|
||||
type: fileType,
|
||||
});
|
||||
|
||||
const scribeResult = await elevenLabsClient.speechToText.convert({
|
||||
file: sourceBlob,
|
||||
model_id: "scribe_v1",
|
||||
tag_audio_events: false,
|
||||
});
|
||||
// console.log({ scribeResult });
|
||||
transcript = scribeResult.text;
|
||||
languageCode = scribeResult.language_code;
|
||||
|
||||
// Reply to the user with the transcript
|
||||
await bot.api.sendMessage(chatId, transcript, {
|
||||
reply_parameters: { message_id: messageId },
|
||||
});
|
||||
} catch (error) {
|
||||
errorMsg = error.message;
|
||||
console.log(errorMsg);
|
||||
await bot.api.sendMessage(
|
||||
chatId,
|
||||
"Sorry, there was an error. Please try again.",
|
||||
{
|
||||
reply_parameters: { message_id: messageId },
|
||||
},
|
||||
);
|
||||
}
|
||||
// Write log to Supabase.
|
||||
const logLine = {
|
||||
file_type: fileType,
|
||||
duration,
|
||||
chat_id: chatId,
|
||||
message_id: messageId,
|
||||
username,
|
||||
language_code: languageCode,
|
||||
error: errorMsg,
|
||||
};
|
||||
console.log({ logLine });
|
||||
await supabase.from("transcription_logs").insert({ ...logLine, transcript });
|
||||
}
|
||||
|
||||
// Use beforeunload event handler to be notified when function is about to shutdown
|
||||
addEventListener("beforeunload", (ev) => {
|
||||
console.log("Function will be shutdown due to", ev.detail?.reason);
|
||||
|
||||
// save state or log the current progress
|
||||
});
|
||||
|
||||
const telegramBotToken = Deno.env.get("TELEGRAM_BOT_TOKEN");
|
||||
const bot = new Bot(telegramBotToken || "");
|
||||
const startMessage =
|
||||
`Welcome to the ElevenLabs Scribe Bot\\! I can transcribe speech in 80\\+ languages with super high accuracy\\!
|
||||
\nTry it out by sending or forwarding me a voice message, video, or audio file\\!
|
||||
\n[Learn more about Scribe](https://elevenlabs.io/speech-to-text) or [build your own bot](https://elevenlabs.io/docs/cookbooks/speech-to-text/telegram-bot)\\!
|
||||
`;
|
||||
bot.command(
|
||||
"start",
|
||||
(ctx) => ctx.reply(startMessage.trim(), { parse_mode: "MarkdownV2" }),
|
||||
);
|
||||
|
||||
bot.on([":voice", ":audio", ":video"], async (ctx) => {
|
||||
try {
|
||||
// console.log(ctx);
|
||||
const file = await ctx.getFile();
|
||||
const fileURL =
|
||||
`https://api.telegram.org/file/bot${telegramBotToken}/${file.file_path}`;
|
||||
const fileMeta = ctx.message?.video ?? ctx.message?.voice ??
|
||||
ctx.message?.audio;
|
||||
// console.log({ fileURL, fileMeta });
|
||||
if (!fileMeta) {
|
||||
return ctx.reply(
|
||||
"No video|audio|voice metadata found. Please try again.",
|
||||
);
|
||||
}
|
||||
|
||||
// Run the transcription in the background.
|
||||
EdgeRuntime.waitUntil(
|
||||
scribe({
|
||||
fileURL,
|
||||
fileType: fileMeta.mime_type!,
|
||||
duration: fileMeta.duration,
|
||||
chatId: ctx.chat.id,
|
||||
messageId: ctx.message?.message_id!,
|
||||
username: ctx.from?.username || "",
|
||||
}),
|
||||
);
|
||||
|
||||
// Reply to the user immediately to let them know we received their file.
|
||||
return ctx.reply("Received. Scribing...");
|
||||
} catch (error) {
|
||||
console.error(error);
|
||||
return ctx.reply(
|
||||
"Sorry, there was an error getting the file. Please try again with a smaller file!",
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
const handleUpdate = webhookCallback(bot, "std/http");
|
||||
|
||||
Deno.serve(async (req) => {
|
||||
try {
|
||||
const url = new URL(req.url);
|
||||
if (url.searchParams.get("secret") !== Deno.env.get("FUNCTION_SECRET")) {
|
||||
return new Response("not allowed", { status: 405 });
|
||||
}
|
||||
|
||||
return await handleUpdate(req);
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
}
|
||||
});
|
||||
@@ -0,0 +1,3 @@
|
||||
# Configuration for private npm package dependencies
|
||||
# For more information on using private registries with Edge Functions, see:
|
||||
# https://supabase.com/docs/guides/functions/import-maps#importing-from-private-registries
|
||||
@@ -0,0 +1,120 @@
|
||||
# Streaming and Caching Speech with ElevenLabs
|
||||
|
||||
Generate and stream speech through Supabase Edge Functions. Store speech in Supabase Storage and cache responses via built-in smart CDN.
|
||||
|
||||
## Requirements
|
||||
|
||||
- An ElevenLabs account with an [API key](/app/settings/api-keys).
|
||||
- A [Supabase](https://supabase.com) account (you can sign up for a free account via [database.new](https://database.new)).
|
||||
- The [Supabase CLI](https://supabase.com/docs/guides/local-development) installed on your machine.
|
||||
- The [Deno runtime](https://docs.deno.com/runtime/getting_started/installation/) installed on your machine and optionally [setup in your facourite IDE](https://docs.deno.com/runtime/getting_started/setup_your_environment).
|
||||
|
||||
## Setup
|
||||
|
||||
### Create a Supabase project locally
|
||||
|
||||
After installing the [Supabase CLI](https://supabase.com/docs/guides/local-development), run the following command to create a new Supabase project locally:
|
||||
|
||||
```bash
|
||||
supabase init
|
||||
```
|
||||
|
||||
### Configure the storage bucket
|
||||
|
||||
You can configure the Supabase CLI to automatically generate a storage bucket by adding this configuration in the `config.toml` file:
|
||||
|
||||
```toml ./supabase/config.toml
|
||||
[storage.buckets.audio]
|
||||
public = false
|
||||
file_size_limit = "50MiB"
|
||||
allowed_mime_types = ["audio/mp3"]
|
||||
objects_path = "./audio"
|
||||
```
|
||||
|
||||
<Note>
|
||||
Upon running `supabase start` this will create a new storage bucket in your local Supabase
|
||||
project. Should you want to push this to your hosted Supabase project, you can run `supabase seed
|
||||
buckets --linked`.
|
||||
</Note>
|
||||
|
||||
### Configure background tasks for Supabase Edge Functions
|
||||
|
||||
To use background tasks in Supabase Edge Functions when developing locally, you need to add the following configuration in the `config.toml` file:
|
||||
|
||||
```toml ./supabase/config.toml
|
||||
[edge_runtime]
|
||||
policy = "per_worker"
|
||||
```
|
||||
|
||||
<Note>
|
||||
When running with `per_worker` policy, Function won't auto-reload on edits. You will need to
|
||||
manually restart it by running `supabase functions serve`.
|
||||
</Note>
|
||||
|
||||
## Run locally
|
||||
|
||||
To run the function locally, run the following commands:
|
||||
|
||||
```bash
|
||||
supabase start
|
||||
```
|
||||
|
||||
Once the local Supabase stack is up and running, run the following command to start the function and observe the logs:
|
||||
|
||||
```bash
|
||||
supabase functions serve
|
||||
```
|
||||
|
||||
## Deploy to Supabase
|
||||
|
||||
If you haven't already, create a new Supabase account at [database.new](https://database.new) and link the local project to your Supabase account:
|
||||
|
||||
```bash
|
||||
supabase link
|
||||
```
|
||||
|
||||
Once done, run the following command to deploy the function:
|
||||
|
||||
```bash
|
||||
supabase functions deploy
|
||||
```
|
||||
|
||||
### Set the function secrets
|
||||
|
||||
Now that you have all your secrets set locally, you can run the following command to set the secrets in your Supabase project:
|
||||
|
||||
```bash
|
||||
supabase secrets set --env-file supabase/functions/.env
|
||||
```
|
||||
|
||||
## Test the function
|
||||
|
||||
The function is designed in a way that it can be used directly as a source for an `<audio>` element.
|
||||
|
||||
```html
|
||||
<audio
|
||||
src="https://${SUPABASE_PROJECT_REF}.supabase.co/functions/v1/elevenlabs-text-to-speech?text=Hello%2C%20world!&voiceId=JBFqnCBsd6RMkjVDRZzb"
|
||||
controls
|
||||
/>
|
||||
```
|
||||
|
||||
You can find an example frontend implementation in the complete code example on [GitHub](https://github.com/elevenlabs/elevenlabs-examples/tree/main/examples/text-to-speech/supabase/stream-and-cache-storage/src/pages/Index.tsx).
|
||||
|
||||
### Try it out
|
||||
|
||||
Navigate to `http://127.0.0.1:54321/functions/v1/elevenlabs-text-to-speech?text=hello%20world`.
|
||||
|
||||
Afterwards, navigate to `http://127.0.0.1:54323/project/default/storage/buckets/audio` to see the audio file in your local Supabase Storage bucket.
|
||||
|
||||
## Test the function
|
||||
|
||||
The function is designed in a way that it can be used directly as a source for an `<audio>` element.
|
||||
|
||||
```html
|
||||
<audio
|
||||
src="https://${SUPABASE_PROJECT_REF}.supabase.co/functions/v1/elevenlabs-text-to-speech?text=Hello%2C%20world!&voiceId=JBFqnCBsd6RMkjVDRZzb"
|
||||
controls
|
||||
/>
|
||||
```
|
||||
|
||||
You can find an example frontend implementation in the complete code example on [GitHub](https://github.com/elevenlabs/elevenlabs-examples/tree/main/examples/text-to-speech/supabase/stream-and-cache-storage/src/pages/Index.tsx).
|
||||
@@ -0,0 +1,3 @@
|
||||
{
|
||||
"imports": {}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
// Setup type definitions for built-in Supabase Runtime APIs
|
||||
import "jsr:@supabase/functions-js/edge-runtime.d.ts";
|
||||
import { createClient } from "jsr:@supabase/supabase-js@2";
|
||||
import { ElevenLabsClient } from "npm:elevenlabs@1.52.0";
|
||||
import * as hash from "npm:object-hash";
|
||||
|
||||
const supabase = createClient(
|
||||
Deno.env.get("SUPABASE_URL")!,
|
||||
Deno.env.get("SUPABASE_SERVICE_ROLE_KEY")!,
|
||||
);
|
||||
|
||||
const client = new ElevenLabsClient({
|
||||
apiKey: Deno.env.get("ELEVENLABS_API_KEY"),
|
||||
});
|
||||
|
||||
// Upload audio to Supabase Storage in a background task
|
||||
async function uploadAudioToStorage(
|
||||
stream: ReadableStream,
|
||||
requestHash: string,
|
||||
) {
|
||||
const { data, error } = await supabase.storage
|
||||
.from("audio")
|
||||
.upload(`${requestHash}.mp3`, stream, {
|
||||
contentType: "audio/mp3",
|
||||
});
|
||||
|
||||
console.log("Storage upload result", { data, error });
|
||||
}
|
||||
|
||||
Deno.serve(async (req) => {
|
||||
// To secure your function for production, you can for example validate the request origin,
|
||||
// or append a user access token and validate it with Supabase Auth.
|
||||
console.log("Request origin", req.headers.get("host"));
|
||||
const url = new URL(req.url);
|
||||
const params = new URLSearchParams(url.search);
|
||||
const text = params.get("text");
|
||||
const voiceId = params.get("voiceId") ?? "JBFqnCBsd6RMkjVDRZzb";
|
||||
|
||||
const requestHash = hash.MD5({ text, voiceId });
|
||||
console.log("Request hash", requestHash);
|
||||
|
||||
// Check storage for existing audio file
|
||||
const { data } = await supabase
|
||||
.storage
|
||||
.from("audio")
|
||||
.createSignedUrl(`${requestHash}.mp3`, 60);
|
||||
|
||||
if (data) {
|
||||
console.log("Audio file found in storage", data);
|
||||
const storageRes = await fetch(data.signedUrl);
|
||||
if (storageRes.ok) return storageRes;
|
||||
}
|
||||
|
||||
if (!text) {
|
||||
return new Response(
|
||||
JSON.stringify({ error: "Text parameter is required" }),
|
||||
{ status: 400, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}
|
||||
|
||||
try {
|
||||
console.log("ElevenLabs API call");
|
||||
const response = await client.textToSpeech.convertAsStream(voiceId, {
|
||||
output_format: "mp3_44100_128",
|
||||
model_id: "eleven_multilingual_v2",
|
||||
text,
|
||||
});
|
||||
|
||||
const stream = new ReadableStream({
|
||||
async start(controller) {
|
||||
for await (const chunk of response) {
|
||||
controller.enqueue(chunk);
|
||||
}
|
||||
controller.close();
|
||||
},
|
||||
});
|
||||
|
||||
// Branch stream to Supabase Storage
|
||||
const [browserStream, storageStream] = stream.tee();
|
||||
|
||||
// Upload to Supabase Storage in the background
|
||||
EdgeRuntime.waitUntil(uploadAudioToStorage(storageStream, requestHash));
|
||||
|
||||
// Return the streaming response immediately
|
||||
return new Response(browserStream, {
|
||||
headers: {
|
||||
"Content-Type": "audio/mpeg",
|
||||
},
|
||||
});
|
||||
} catch (error) {
|
||||
console.log("error", { error });
|
||||
return new Response(JSON.stringify({ error: error.message }), {
|
||||
status: 500,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
});
|
||||
@@ -186,6 +186,7 @@ may_uppercase = [
|
||||
"Swift",
|
||||
"SwiftUI",
|
||||
"Team Plan",
|
||||
"Telegram",
|
||||
"Third-Party Auth",
|
||||
"TimescaleDB",
|
||||
"Transformers.js",
|
||||
|
||||
@@ -111,6 +111,7 @@ allow_list = [
|
||||
"BigQuery",
|
||||
"Bitbucket",
|
||||
"Bitwarden",
|
||||
"BotFather",
|
||||
"Brevo",
|
||||
"CAPTCHA",
|
||||
"Cartes Bancaires",
|
||||
@@ -133,6 +134,7 @@ allow_list = [
|
||||
"Django",
|
||||
"Docker",
|
||||
"Drizzle",
|
||||
"ElevenLabs",
|
||||
"EnterpriseDB",
|
||||
"Entra",
|
||||
"ePHI",
|
||||
|
||||
Reference in new issue
Block a user