chore: remove algolia build-search script

This commit is contained in:
Greg Richardson committed 2023-04-20 17:49:20 -06:00
1 parent f617dc06f6
commit 1ff754b879
8 files changed
+4 -311

No files matched your search

+1 -1
View File
@@ -96,6 +96,6 @@ On the Next.JS side of things, these work almost exactly the same as the client
#### Search
Search is handled using a Supabase instance. During CI, [a script](https://github.com/supabase/supabase/blob/master/apps/docs/scripts/generate-embeddings.ts) aggregates all content sources (eg. guides, reference docs, etc), indexes them using OpenAI embeddings, and stores them in a Supabase database.
Search is handled using a Supabase instance. During CI, [a script](https://github.com/supabase/supabase/blob/master/apps/docs/scripts/search/generate-embeddings.ts) aggregates all content sources (eg. guides, reference docs, etc), indexes them using OpenAI embeddings, and stores them in a Supabase database.
At runtime, an [Edge Function](https://github.com/supabase/supabase/blob/master/supabase/functions) is executed that performs a similarity search between the user's query and the above content sources using [`pgvector`](https://github.com/pgvector/pgvector) embeddings.
+2 -2
View File
@@ -9,9 +9,9 @@
"start": "next start",
"lint": "next lint",
"build:sitemap": "node ./internals/generate-sitemap.mjs",
"embeddings": "tsx scripts/generate-embeddings.ts",
"embeddings": "tsx scripts/search/generate-embeddings.ts",
"embeddings:refresh": "npm run embeddings -- --refresh",
"postbuild": "ts-node ./scripts/build-search.ts && node ./internals/generate-sitemap.mjs",
"postbuild": "node ./internals/generate-sitemap.mjs",
"generate:all": "npm-run-all --parallel gen:api gen:cli gen:gotrue gen:storage gen:supabase-dart:v0 gen:supabase-dart:v1 gen:supabase-csharp:v0 gen:supabase-js:v1 gen:supabase-js:v2 gen:realtime",
"gen:api": "npm-run-all gen:api:usage",
"gen:api:usage": "ts-node ./generator/index.ts gen --type api --url https://api.supabase.com --input ../../spec/transforms/api_v0_openapi_deparsed.json --output ./docs/reference/api/generated/usage.mdx",
-155
View File
@@ -1,155 +0,0 @@
import fs from 'fs'
import crypto from 'crypto'
import path from 'path'
import matter from 'gray-matter'
import dotenv from 'dotenv'
import algoliasearch from 'algoliasearch/lite'
import { isEmpty } from 'lodash'
// search objects
import { generateClientLibSearchObjects } from './files/client-libs'
import { generateAPISearchObjects } from './files/api'
import { generateCLISearchObjects } from './files/cli'
const cliObjects = generateCLISearchObjects()
const apiObjects = generateAPISearchObjects()
const clientLibSearchObjects = generateClientLibSearchObjects()
// @ts-ignore
// The properties of the searchObject that are specifically read
// by DocSearch are "type" and "hierarchy". The rest, though saved into Algolia (which we
// can potentially use to craft more nuanced search experiences) are not used by DocSearch
const ignoredFiles = [
'pages/404.mdx',
'pages/faq.mdx',
'pages/support.mdx',
'pages/oss.tsx',
'pages/_app.tsx',
'pages/_document.tsx',
'pages/[...slug].tsx',
'pages/handbook/contributing.mdx',
'pages/handbook/introduction.mdx',
'pages/handbook/supasquad.mdx',
]
async function walk(dir) {
let files = await fs.promises.readdir(dir)
//@ts-ignore
files = await Promise.all(
files.map(async (file) => {
const filePath = path.join(dir, file)
const stats = await fs.promises.stat(filePath)
if (stats.isDirectory()) return walk(filePath)
else if (stats.isFile()) return filePath
})
)
return files.reduce((all, folderContents) => all.concat(folderContents), [])
}
;(async function () {
// initialize environment variables
dotenv.config()
if (!process.env.NEXT_PUBLIC_ALGOLIA_APP_ID || !process.env.ALGOLIA_SEARCH_ADMIN_KEY) {
return console.log(
'Missing Algolia app ID / admin Key: skipping saving of Algolia search index'
)
}
console.log('Preparing docs indexing for Algolia')
try {
const indexName = process.env.NEXT_PUBLIC_ALGOLIA_INDEX_NAME
const client = algoliasearch(
process.env.NEXT_PUBLIC_ALGOLIA_APP_ID,
process.env.ALGOLIA_SEARCH_ADMIN_KEY
)
const index = client.initIndex(indexName)
const guidePages = (await walk('pages')).filter((slug) => !ignoredFiles.includes(slug))
// generate search objects for mdx guide pages
const guidePagesearchObjects = guidePages
.map((slug) => {
let id, title, description
const fileContents = fs.readFileSync(slug, 'utf8')
const { data, content } = matter(fileContents)
if (isEmpty(data)) {
// Guide pages do not have front-matter meta, unlike reference pages, have to manually extract
const metaIndex = fileContents.indexOf('export const meta = {')
if (metaIndex !== -1) {
const metaString =
fileContents
.slice(metaIndex + 20, fileContents.indexOf('}', metaIndex + 1) + 1)
.replace(/\n/g, '')
.slice(0, -2) + '}'
const meta = eval(`(${metaString})`)
id = meta.id
title = meta.title
description = meta.description
}
} else {
id = data.id
title = data.title
description = data.description
}
const url = (slug.includes('/generated') ? slug.replace('/generated', '') : slug)
.replace('docs', '')
.replace('pages', '')
.replace(/\.mdx$/, '')
const source = slug.includes('/reference') ? 'reference' : 'guide'
const object = {
// For Algolia
objectID: crypto.randomUUID(),
id,
title,
description,
url,
source,
//pageContent: content,
pageContent: '',
category: undefined,
version: undefined,
// Docsearch specific
type: 'lvl1',
hierarchy: {
lvl0: 'Guides',
lvl1: title,
lvl2: null,
lvl3: null,
lvl4: null,
lvl5: null,
lvl6: null,
},
}
return object
})
// Some of the reference generated files come with an 'index' page that we can ignore
.filter((object) => !object.url.endsWith('/index'))
.filter((object) => !object.url.endsWith('/.gitkeep'))
const combinedSearchObjects = guidePagesearchObjects.concat(
clientLibSearchObjects,
apiObjects,
cliObjects
)
//@ts-ignore
await index.clearObjects()
console.log(`Successfully cleared records from ${indexName}`)
//@ts-ignore
const algoliaResponse = await index.saveObjects(combinedSearchObjects)
//@ts-ignore
console.log(`Successfully saved ${algoliaResponse.objectIDs.length} records into ${indexName}.`)
} catch (error) {
console.log('Error:', error)
}
})()
-43
View File
@@ -1,43 +0,0 @@
import crypto from 'crypto'
import { flattenSections } from '../../lib/helpers'
import apiCommonSections from '~/../../spec/common-api-sections.json'
import specFile from '~/../../spec/transforms/api_v0_openapi_deparsed.json'
import { gen_v3 } from '../../lib/refGenerator/helpers'
// @ts-ignore
const generatedSpec = gen_v3(specFile, 'wat', { apiUrl: 'apiv0' })
const sections = flattenSections(apiCommonSections)
export function generateAPISearchObjects() {
let searchObjects = []
//@ts-ignore
sections.map((section) => {
const object = searchObjects.push({
objectID: crypto.randomUUID(),
id: section.id,
title: section.title,
// @ts-ignore
description: generatedSpec.operations.find((item) => item.operationId === section.id)
?.summary,
url: `/reference/api/${section.slug}`,
source: 'reference',
pageContent: '',
category: section.product,
version: '',
type: 'lvl2',
hierarchy: {
lvl0: 'References',
lvl1: 'Management API',
lvl2: section.title,
lvl3: null,
lvl4: null,
lvl5: null,
lvl6: null,
},
})
return object
})
return searchObjects
}
-43
View File
@@ -1,43 +0,0 @@
import crypto from 'crypto'
import fs from 'fs'
import { flattenSections } from '../../lib/helpers'
import cliCommonSections from '~/../../spec/common-cli-sections.json'
import yaml from 'js-yaml'
// @ts-ignore
const spec = yaml.load(fs.readFileSync(`../../spec/cli_v1_commands.yaml`, 'utf8'))
const commonSections = flattenSections(cliCommonSections)
export function generateCLISearchObjects() {
let searchObjects = []
//@ts-ignore
spec.commands.map((section) => {
const object = searchObjects.push({
objectID: crypto.randomUUID(),
id: section.id,
title: section.title,
// @ts-ignore
description: section.description?.substr(0, section.description.indexOf('\n')),
url: `/reference/cli/${commonSections.find((item) => item.id === section.id)?.slug}`,
source: 'reference',
pageContent: '',
category: section.product,
version: '',
type: 'lvl2',
hierarchy: {
lvl0: 'References',
lvl1: 'Supabase CLI',
lvl2: section.title,
lvl3: null,
lvl4: null,
lvl5: null,
lvl6: null,
},
})
return object
})
return searchObjects
}
-57
View File
@@ -1,57 +0,0 @@
import fs from 'fs'
import crypto from 'crypto'
import yaml from 'js-yaml'
import { flattenSections } from '../../lib/helpers'
import { nameMap } from '../helpers'
import commonLibSections from '~/../../spec/common-client-libs-sections.json'
const clientLibFiles = [
{ fileName: 'supabase_js_v2', label: 'javascript', version: 'v2', versionSlug: false },
{ fileName: 'supabase_js_v1', label: 'javascript', version: 'v1', versionSlug: true },
{ fileName: 'supabase_dart_v1', label: 'dart', version: 'v1', versionSlug: false },
{ fileName: 'supabase_dart_v0', label: 'dart', version: 'v0', versionSlug: true },
{ fileName: 'supabase_dart_v0', label: 'dart', version: 'v0', versionSlug: true },
]
const flatCommonLibSections = flattenSections(commonLibSections)
export function generateClientLibSearchObjects() {
// loop through each spec file, find the correspending entry in the common file and grab the title / description / slug
let clientLibSearchObjects = []
clientLibFiles.map((file) => {
const specs = yaml.load(fs.readFileSync(`../../spec/${file.fileName}.yml`, 'utf8'))
//take each function id, find it in the commonLibSections file and return { id, title, slug, description, }
//@ts-ignore
specs.functions.map((fn) => {
const item = flatCommonLibSections.find((section) => section.id === fn.id)
if (item) {
const object = clientLibSearchObjects.push({
objectID: crypto.randomUUID(),
id: item.id,
title: item.title,
description: item.title,
url: `/reference/${file.label}/${file.versionSlug ? file.version + '/' : ''}${item.slug}`,
source: 'reference',
pageContent: '',
category: item.product,
version: file.version,
type: 'lvl2',
hierarchy: {
lvl0: 'References',
lvl1: `${nameMap[file.label]} ${file.version}`,
lvl2: item.title,
lvl3: file.version,
lvl4: null,
lvl5: null,
lvl6: null,
},
})
return object
}
})
})
return clientLibSearchObjects
}
-9
View File
@@ -1,9 +0,0 @@
export const nameMap = {
api: 'Management API',
cli: 'Supabase CLI',
auth: 'Auth Server',
storage: 'Storage Server',
postgres: 'Postgres',
dart: 'Supabase Flutter Library',
javascript: 'Supabase JavaScript Library',
}
@@ -3,7 +3,7 @@ import dotenv from 'dotenv'
import 'openai'
import { Configuration, OpenAIApi } from 'openai'
import { inspect } from 'util'
import { fetchSources } from './search/sources'
import { fetchSources } from './sources'
dotenv.config()