diff --git a/apps/docs/components/Navigation/NavigationMenu/NavigationMenu.constants.ts b/apps/docs/components/Navigation/NavigationMenu/NavigationMenu.constants.ts index c2eb655c5b9..8858b9d9f92 100644 --- a/apps/docs/components/Navigation/NavigationMenu/NavigationMenu.constants.ts +++ b/apps/docs/components/Navigation/NavigationMenu/NavigationMenu.constants.ts @@ -946,6 +946,10 @@ export const functions: NavMenuConstant = { name: 'Slack Bot responding to mentions', url: '/guides/functions/examples/slack-bot-mention', }, + { + name: 'Convert PDF to HTML', + url: '/guides/functions/examples/pdf-upload-extract', + }, ], }, { diff --git a/apps/docs/pages/guides/functions/examples/pdf-upload-extract.mdx b/apps/docs/pages/guides/functions/examples/pdf-upload-extract.mdx new file mode 100644 index 00000000000..38a5c4578a6 --- /dev/null +++ b/apps/docs/pages/guides/functions/examples/pdf-upload-extract.mdx @@ -0,0 +1,177 @@ +import Layout from '~/layouts/DefaultGuideLayout' + +export const meta = { + id: 'pdf-upload-extract', + title: 'Upload PDF and extract data', + description: 'Building function to upload a PDF file and extract data.', +} + +## Upload a PDF file to storage and extract its content + +An Example to deploy Supabase Edge Functions to read incoming `multipart/form-data` request. Then, write pdf files to Supabase Storage, and extract the pure text content for further use. + +```ts index.ts +// Follow this setup guide to integrate the Deno language server with your editor: +// https://deno.land/manual/getting_started/setup_your_environment +// This enables autocomplete, go to definition, etc. + +import 'https://deno.land/x/xhr@0.3.0/mod.ts' + +import { createClient } from 'https://esm.sh/@supabase/supabase-js@2' +import Api2Pdf from 'https://esm.sh/api2pdf@2' + +console.log('PDF Function is now running...') + +// Supabase API URL & Anon Key - env values exported by default +const supabaseUrl = Deno.env.get('SUPABASE_URL') ?? '' +const supabaseAnonKey = Deno.env.get('SUPABASE_ANON_KEY') ?? '' + +// Get Api2Pdf api key via https://www.api2pdf.com/ +// Run this command via Supabase CLI to set key: > supabase secrets set API2PDF_KEY= +// Remember to include secret in your deployment using the --secrets flag. +// For example: > supabase functions deploy --secrets API2PDF_KEY +const a2pAPIKey = Deno.env.get('API2PDF_KEY') + +// initialize new Api2Pdf instance +const a2pClient = new Api2Pdf(a2pAPIKey) + +const corsHeaders = { + 'Access-Control-Allow-Origin': '*', + 'Access-Control-Allow-Headers': 'authorization, x-client-info, apikey, content-type', +} + +class ApplicationError extends Error { + constructor(message: string, public data: Record = {}) { + super(message) + } +} + +class UserError extends ApplicationError {} + +Deno.serve(async (req) => { + try { + if (req.method === 'OPTIONS') { + return new Response('ok', { headers: corsHeaders }) + } + + if (!supabaseUrl) { + throw new ApplicationError('Missing environment variable: SUPABASE_URL') + } + + if (!supabaseAnonKey) { + throw new ApplicationError('Missing environment variable: SUPABASE_ANON_KEY') + } + + if (!a2pAPIKey) { + throw new ApplicationError('Missing environment variable: API2PDF_KEY') + } + + const supabaseClient = createClient( + supabaseUrl, + supabaseAnonKey, + // Add user Auth context to apply row-level-security (RLS) policies + { + global: { headers: { Authorization: req.headers.get('Authorization')! } }, + } + ) + + // Read multipart/form-data from request. + // Note: Use req.json() if request contains a JSON body + const reqFormData = await req.formData() + + if (!reqFormData) { + throw new UserError('Missing or invalid request data') + } + + // Check required data from request body + if (!reqFormData.has('pdfFile') || !reqFormData.has('uploadPath')) { + throw new UserError('Missing required (pdfFile or uploadPath) property from request data.') + } + + const pdfFile = reqFormData.get('pdfFile') + + const timestamp = +new Date() + const filePath = `${reqFormData.get('uploadPath')}-${timestamp}` + + if (!pdfFile || !(pdfFile instanceof File)) { + throw new UserError('Invalid file field or value') + } + + // Upload pdf file to supabase storage bucket + const { data: uploadedFile, error: uploadError } = await supabaseClient.storage + .from('') + .upload(filePath, pdfFile, { + contentType: 'application/pdf', + cacheControl: '3600', + }) + + if (uploadError) throw uploadError + + // Typical Supabase storage file public url pattern + const filePublicUrl = `${supabaseUrl}/storage/v1/object/public//${filePath}` + + let finalText: string = '' + + if (uploadedFile) { + // Convert PDF file to Html and remove Html tags to get plain text content + await a2pClient + .libreOfficePdfToHtml(filePublicUrl) + .then(async function (htmlResult: { FileUrl: string }) { + if (htmlResult) { + const htmlDataResponse = await fetch(htmlResult.FileUrl) + const htmlData = await htmlDataResponse.text() + + // Regular expression pattern to match HTML tags + const regex = /<[^>]+>/g + + // Remove HTML tags from the document + finalText = htmlData.replace(regex, '') + + console.log(`Final pdf text content: \n ${finalText}`) + } + }) + } + + const data = { + pdfText: finalText, + pdfStorageUrl: filePublicUrl, + } + + return new Response(JSON.stringify(data), { + headers: { ...corsHeaders, 'Content-Type': 'application/json' }, + }) + } catch (err) { + if (err instanceof UserError) { + return new Response( + JSON.stringify({ + error: err.message, + data: err.data, + }), + { + status: 400, + headers: { ...corsHeaders, 'Content-Type': 'application/json' }, + } + ) + } else if (err instanceof ApplicationError) { + console.error(`${err.message}: ${JSON.stringify(err.data)}`) + } else { + // Print out unexpected errors as is to help with debugging + console.error(err) + } + + return new Response( + JSON.stringify({ + error: 'There was an error processing your request', + }), + { + status: 500, + headers: { ...corsHeaders, 'Content-Type': 'application/json' }, + } + ) + } +}) +``` + +export const Page = ({ children }) => + +export default Page diff --git a/examples/edge-functions/supabase/functions/upload-pdf-file-and-extract-content/README.md b/examples/edge-functions/supabase/functions/upload-pdf-file-and-extract-content/README.md new file mode 100644 index 00000000000..8521fabcaa2 --- /dev/null +++ b/examples/edge-functions/supabase/functions/upload-pdf-file-and-extract-content/README.md @@ -0,0 +1,3 @@ +# Upload a PDF file to storage and extract its content + +This example shows how to use Supabase Edge Functions to read incoming `multipart/form-data` request, write pdf files to Supabase Storage, and extract the pure text content for further use. diff --git a/examples/edge-functions/supabase/functions/upload-pdf-file-and-extract-content/index.ts b/examples/edge-functions/supabase/functions/upload-pdf-file-and-extract-content/index.ts new file mode 100644 index 00000000000..0cf783f1f09 --- /dev/null +++ b/examples/edge-functions/supabase/functions/upload-pdf-file-and-extract-content/index.ts @@ -0,0 +1,157 @@ +// Follow this setup guide to integrate the Deno language server with your editor: +// https://deno.land/manual/getting_started/setup_your_environment +// This enables autocomplete, go to definition, etc. +import 'https://deno.land/x/xhr@0.3.0/mod.ts' +import { createClient } from 'https://esm.sh/@supabase/supabase-js@2' +import Api2Pdf from 'https://esm.sh/api2pdf@2' + +console.log('PDF Function is now running...') + +// Supabase API URL & Anon Key - env values exported by default +const supabaseUrl = Deno.env.get('SUPABASE_URL') ?? '' +const supabaseAnonKey = Deno.env.get('SUPABASE_ANON_KEY') ?? '' + +// Get Api2Pdf api key via https://www.api2pdf.com/ +// Run this command via Supabase CLI to set key: > supabase secrets set API2PDF_KEY= +// Remember to include secret in your deployment using the --secrets flag. +// For example: > supabase functions deploy --secrets API2PDF_KEY +const a2pAPIKey = Deno.env.get('API2PDF_KEY') + +// initialize new Api2Pdf instance +const a2pClient = new Api2Pdf(a2pAPIKey) + +const corsHeaders = { + 'Access-Control-Allow-Origin': '*', + 'Access-Control-Allow-Headers': 'authorization, x-client-info, apikey, content-type', +} + +class ApplicationError extends Error { + constructor(message: string, public data: Record = {}) { + super(message) + } +} + +class UserError extends ApplicationError {} + +Deno.serve(async (req) => { + try { + if (req.method === 'OPTIONS') { + return new Response('ok', { headers: corsHeaders }) + } + + if (!supabaseUrl) { + throw new ApplicationError('Missing environment variable: SUPABASE_URL') + } + + if (!supabaseAnonKey) { + throw new ApplicationError('Missing environment variable: SUPABASE_ANON_KEY') + } + + if (!a2pAPIKey) { + throw new ApplicationError('Missing environment variable: API2PDF_KEY') + } + + const supabaseClient = createClient( + supabaseUrl, + supabaseAnonKey, + // Add user Auth context to apply row-level-security (RLS) policies + { + global: { headers: { Authorization: req.headers.get('Authorization')! } }, + } + ) + + // Read multipart/form-data from request. + // Note: Use req.json() if request contains a JSON body + const reqFormData = await req.formData() + + if (!reqFormData) { + throw new UserError('Missing or invalid request data') + } + + // Check required data from request body + if (!reqFormData.has('pdfFile') || !reqFormData.has('uploadPath')) { + throw new UserError('Missing required (pdfFile or uploadPath) property from request data.') + } + + const pdfFile = reqFormData.get('pdfFile') + + const timestamp = +new Date() + const filePath = `${reqFormData.get('uploadPath')}-${timestamp}` + + if (!pdfFile || !(pdfFile instanceof File)) { + throw new UserError('Invalid file field or value') + } + + // Upload pdf file to supabase storage bucket + const { data: uploadedFile, error: uploadError } = await supabaseClient.storage + .from('') + .upload(filePath, pdfFile, { + contentType: 'application/pdf', + cacheControl: '3600', + }) + + if (uploadError) throw uploadError + + // Typical Supabase storage file public url pattern + const filePublicUrl = `${supabaseUrl}/storage/v1/object/public//${filePath}` + + let finalText: string = '' + + if (uploadedFile) { + // Convert PDF file to Html and remove Html tags to get plain text content + await a2pClient + .libreOfficePdfToHtml(filePublicUrl) + .then(async function (htmlResult: { FileUrl: string }) { + if (htmlResult) { + const htmlDataResponse = await fetch(htmlResult.FileUrl) + const htmlData = await htmlDataResponse.text() + + // Regular expression pattern to match HTML tags + const regex = /<[^>]+>/g + + // Remove HTML tags from the document + finalText = htmlData.replace(regex, '') + + console.log(`Final pdf text content: \n ${finalText}`) + } + }) + } + + const data = { + pdfText: finalText, + pdfStorageUrl: filePublicUrl, + } + + return new Response(JSON.stringify(data), { + headers: { ...corsHeaders, 'Content-Type': 'application/json' }, + }) + } catch (err) { + if (err instanceof UserError) { + return new Response( + JSON.stringify({ + error: err.message, + data: err.data, + }), + { + status: 400, + headers: { ...corsHeaders, 'Content-Type': 'application/json' }, + } + ) + } else if (err instanceof ApplicationError) { + console.error(`${err.message}: ${JSON.stringify(err.data)}`) + } else { + // Print out unexpected errors as is to help with debugging + console.error(err) + } + + return new Response( + JSON.stringify({ + error: 'There was an error processing your request', + }), + { + status: 500, + headers: { ...corsHeaders, 'Content-Type': 'application/json' }, + } + ) + } +}) \ No newline at end of file