diff --git a/apps/docs/scripts/generate-embeddings.ts b/apps/docs/scripts/generate-embeddings.ts index 520450083e3..23158f7f5bc 100644 --- a/apps/docs/scripts/generate-embeddings.ts +++ b/apps/docs/scripts/generate-embeddings.ts @@ -1,411 +1,24 @@ import { createClient } from '@supabase/supabase-js' -import { createHash } from 'crypto' import dotenv from 'dotenv' -import { ObjectExpression } from 'estree' -import { readdir, readFile, stat } from 'fs/promises' -import GithubSlugger from 'github-slugger' -import yaml from 'js-yaml' -import { Content, Root } from 'mdast' -import { fromMarkdown } from 'mdast-util-from-markdown' -import { mdxFromMarkdown, MdxjsEsm } from 'mdast-util-mdx' -import { toMarkdown } from 'mdast-util-to-markdown' -import { toString } from 'mdast-util-to-string' -import { mdxjs } from 'micromark-extension-mdxjs' import 'openai' import { Configuration, OpenAIApi } from 'openai' -import { OpenAPIV3 } from 'openapi-types' -import { basename, dirname, join } from 'path' -import { u } from 'unist-builder' -import { filter } from 'unist-util-filter' import { inspect } from 'util' -import { ICommonFunc, IFunctionDefinition, ISpec } from '../components/reference/Reference.types' -import { CliCommand, CliSpec } from '../generator/types/CliSpec' -import { flattenSections } from '../lib/helpers' -import { enrichedOperation, gen_v3 } from '../lib/refGenerator/helpers' +import { fetchSources } from './search/sources' dotenv.config() -const ignoredFiles = ['pages/404.mdx'] - -/** - * Extracts ES literals from an `estree` `ObjectExpression` - * into a plain JavaScript object. - */ -function getObjectFromExpression(node: ObjectExpression) { - return node.properties.reduce< - Record - >((object, property) => { - if (property.type !== 'Property') { - return object - } - - const key = (property.key.type === 'Identifier' && property.key.name) || undefined - const value = (property.value.type === 'Literal' && property.value.value) || undefined - - if (!key) { - return object - } - - return { - ...object, - [key]: value, - } - }, {}) -} - -/** - * Extracts the `meta` ESM export from the MDX file. - * - * This info is akin to frontmatter. - */ -function extractMetaExport(mdxTree: Root) { - const metaExportNode = mdxTree.children.find((node): node is MdxjsEsm => { - return ( - node.type === 'mdxjsEsm' && - node.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' && - node.data.estree.body[0].declaration?.type === 'VariableDeclaration' && - node.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' && - node.data.estree.body[0].declaration.declarations[0].id.name === 'meta' - ) - }) - - if (!metaExportNode) { - return undefined - } - - const objectExpression = - (metaExportNode.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' && - metaExportNode.data.estree.body[0].declaration?.type === 'VariableDeclaration' && - metaExportNode.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' && - metaExportNode.data.estree.body[0].declaration.declarations[0].id.name === 'meta' && - metaExportNode.data.estree.body[0].declaration.declarations[0].init?.type === - 'ObjectExpression' && - metaExportNode.data.estree.body[0].declaration.declarations[0].init) || - undefined - - if (!objectExpression) { - return undefined - } - - return getObjectFromExpression(objectExpression) -} - -/** - * Splits a `mdast` tree into multiple trees based on - * a predicate function. Will include the splitting node - * at the beginning of each tree. - * - * Useful to split a markdown file into smaller sections. - */ -function splitTreeBy(tree: Root, predicate: (node: Content) => boolean) { - return tree.children.reduce((trees, node) => { - const [lastTree] = trees.slice(-1) - - if (!lastTree || predicate(node)) { - const tree: Root = u('root', [node]) - return trees.concat(tree) - } - - lastTree.children.push(node) - return trees - }, []) -} - -type Meta = ReturnType - -type Section = { - content: string - heading?: string - slug?: string -} - -type ProcessedMdx = { - checksum: string - meta: Meta - sections: Section[] -} - -/** - * Processes MDX content for search indexing. - * It extracts metadata, strips it of all JSX, - * and splits it into sub-sections based on criteria. - */ -function processMdxForSearch(content: string): ProcessedMdx { - const checksum = createHash('sha256').update(content).digest('base64') - - const mdxTree = fromMarkdown(content, { - extensions: [mdxjs()], - mdastExtensions: [mdxFromMarkdown()], - }) - - const meta = extractMetaExport(mdxTree) - - // Remove all MDX elements from markdown - const mdTree = filter( - mdxTree, - (node) => - ![ - 'mdxjsEsm', - 'mdxJsxFlowElement', - 'mdxJsxTextElement', - 'mdxFlowExpression', - 'mdxTextExpression', - ].includes(node.type) - ) - - if (!mdTree) { - return { - checksum, - meta, - sections: [], - } - } - - const sectionTrees = splitTreeBy(mdTree, (node) => node.type === 'heading') - - const slugger = new GithubSlugger() - - const sections = sectionTrees.map((tree) => { - const [firstNode] = tree.children - - const heading = firstNode.type === 'heading' ? toString(firstNode) : undefined - const slug = heading ? slugger.slug(heading) : undefined - - return { - content: toMarkdown(tree), - heading, - slug, - } - }) - - return { - checksum, - meta, - sections, - } -} - -type WalkEntry = { - path: string - parentPath?: string -} - -async function walk(dir: string, parentPath?: string): Promise { - const immediateFiles = await readdir(dir) - - const recursiveFiles = await Promise.all( - immediateFiles.map(async (file) => { - const path = join(dir, file) - const stats = await stat(path) - if (stats.isDirectory()) { - // Keep track of document hierarchy (if this dir has corresponding doc file) - const docPath = `${basename(path)}.mdx` - - return walk( - path, - immediateFiles.includes(docPath) ? join(dirname(path), docPath) : parentPath - ) - } else if (stats.isFile()) { - return [ - { - path: path, - parentPath, - }, - ] - } else { - return [] - } - }) - ) - - const flattenedFiles = recursiveFiles.reduce( - (all, folderContents) => all.concat(folderContents), - [] - ) - - return flattenedFiles.sort((a, b) => a.path.localeCompare(b.path)) -} - -abstract class BaseEmbeddingSource { - checksum?: string - meta?: Meta - sections?: Section[] - - constructor(public source: string, public path: string, public parentPath?: string) {} - - abstract load(): Promise<{ checksum: string; meta?: Meta; sections: Section[] }> -} - -class MarkdownEmbeddingSource extends BaseEmbeddingSource { - type: 'markdown' = 'markdown' - - constructor(source: string, public filePath: string, public parentFilePath?: string) { - const path = filePath.replace(/^pages/, '').replace(/\.mdx?$/, '') - const parentPath = parentFilePath?.replace(/^pages/, '').replace(/\.mdx?$/, '') - - super(source, path, parentPath) - } - - async load() { - const contents = await readFile(this.filePath, 'utf8') - - const { checksum, meta, sections } = processMdxForSearch(contents) - - this.checksum = checksum - this.meta = meta - this.sections = sections - - return { - checksum, - meta, - sections, - } - } -} - -abstract class ReferenceEmbeddingSource extends BaseEmbeddingSource { - type: 'reference' = 'reference' - - constructor( - source: string, - path: string, - public meta: Meta, - public specFilePath: string, - public sectionsFilePath: string - ) { - super(source, path) - } - - async load() { - const specContents = await readFile(this.specFilePath, 'utf8') - const refSectionsContents = await readFile(this.sectionsFilePath, 'utf8') - - const refSections: ICommonFunc[] = JSON.parse(refSectionsContents) - const flattenedRefSections = flattenSections(refSections) - - const checksum = createHash('sha256') - .update(specContents + refSectionsContents) - .digest('base64') - - const specSections = this.getSpecSections(specContents) - - const sections = flattenedRefSections - .map((refSection) => { - const specSection = this.matchSpecSection(specSections, refSection.id) - - if (!specSection) { - return - } - - return { - heading: refSection.title, - slug: refSection.slug, - content: `${this.meta.title} for ${refSection.title}:\n${this.formatSection( - specSection, - refSection - )}`, - } - }) - .filter((section) => !!section) - - this.checksum = checksum - this.sections = sections - - return { - checksum, - sections, - meta: this.meta, - } - } - - abstract getSpecSections(specContents: string): SpecSection[] - abstract matchSpecSection(specSections: SpecSection[], id: string): SpecSection - abstract formatSection(specSection: SpecSection, refSection: ICommonFunc): string -} - -class OpenApiEmbeddingSource extends ReferenceEmbeddingSource { - getSpecSections(specContents: string): enrichedOperation[] { - const spec: OpenAPIV3.Document<{}> = JSON.parse(specContents) - - const generatedSpec = gen_v3(spec, '', { - apiUrl: 'apiv0', - }) - - return generatedSpec.operations - } - matchSpecSection(operations: enrichedOperation[], id: string): enrichedOperation { - return operations.find((operation) => operation.operationId === id) - } - formatSection(specOperation: enrichedOperation) { - const { summary, description, operation, path, tags } = specOperation - return JSON.stringify({ - summary, - description, - operation, - path, - tags, - }) - } -} - -class ClientLibEmbeddingSource extends ReferenceEmbeddingSource { - getSpecSections(specContents: string): IFunctionDefinition[] { - const spec = yaml.load(specContents) as ISpec - - return spec.functions - } - matchSpecSection(functionDefinitions: IFunctionDefinition[], id: string): IFunctionDefinition { - return functionDefinitions.find((functionDefinition) => functionDefinition.id === id) - } - formatSection(functionDefinition: IFunctionDefinition, refSection: ICommonFunc): string { - const { title } = refSection - const { description, title: functionName } = functionDefinition - - return JSON.stringify({ - title, - description, - functionName, - }) - } -} - -class CliEmbeddingSource extends ReferenceEmbeddingSource { - getSpecSections(specContents: string): CliCommand[] { - const spec = yaml.load(specContents) as CliSpec - - return spec.commands - } - matchSpecSection(cliCommands: CliCommand[], id: string): CliCommand { - return cliCommands.find((cliCommand) => cliCommand.id === id) - } - formatSection(cliCommand: CliCommand): string { - const { summary, description, usage } = cliCommand - return JSON.stringify({ - summary, - description, - usage, - }) - } -} - -type EmbeddingSource = - | MarkdownEmbeddingSource - | OpenApiEmbeddingSource - | ClientLibEmbeddingSource - | CliEmbeddingSource - async function generateEmbeddings() { // TODO: use better CLI lib like yargs const args = process.argv.slice(2) const shouldRefresh = args.includes('--refresh') - if ( - !process.env.NEXT_PUBLIC_SUPABASE_URL || - !process.env.SUPABASE_SERVICE_ROLE_KEY || - !process.env.OPENAI_KEY - ) { - return console.log( - 'Environment variables NEXT_PUBLIC_SUPABASE_URL, SUPABASE_SERVICE_ROLE_KEY, and OPENAI_KEY are required: skipping embeddings generation' + const requiredEnvVars = ['NEXT_PUBLIC_SUPABASE_URL', 'SUPABASE_SERVICE_ROLE_KEY', 'OPENAI_KEY'] + + if (requiredEnvVars.some((name) => !process.env[name])) { + throw new Error( + `Environment variables ${requiredEnvVars.join( + ', ' + )} are required: skipping embeddings generation` ) } @@ -420,54 +33,7 @@ async function generateEmbeddings() { } ) - const embeddingSources: EmbeddingSource[] = [ - new OpenApiEmbeddingSource( - 'api', - '/reference/api', - { title: 'Management API Reference' }, - '../../spec/transforms/api_v0_openapi_deparsed.json', - '../../spec/common-api-sections.json' - ), - new ClientLibEmbeddingSource( - 'js-lib', - '/reference/javascript', - { title: 'JavaScript Reference' }, - '../../spec/supabase_js_v2.yml', - '../../spec/common-client-libs-sections.json' - ), - new ClientLibEmbeddingSource( - 'dart-lib', - '/reference/dart', - { title: 'Dart Reference' }, - '../../spec/supabase_dart_v1.yml', - '../../spec/common-client-libs-sections.json' - ), - new ClientLibEmbeddingSource( - 'python-lib', - '/reference/python', - { title: 'Python Reference' }, - '../../spec/supabase_py_v2.yml', - '../../spec/common-client-libs-sections.json' - ), - new ClientLibEmbeddingSource( - 'csharp-lib', - '/reference/csharp', - { title: 'C# Reference' }, - '../../spec/supabase_csharp_v0.yml', - '../../spec/common-client-libs-sections.json' - ), - new CliEmbeddingSource( - 'cli', - '/reference/cli', - { title: 'CLI Reference' }, - '../../spec/cli_v1_commands.yaml', - '../../spec/common-cli-sections.json' - ), - ...(await walk('pages')) - .filter(({ path }) => /\.mdx?$/.test(path)) - .filter(({ path }) => !ignoredFiles.includes(path)) - .map((entry) => new MarkdownEmbeddingSource('guide', entry.path)), - ] + const embeddingSources = await fetchSources() console.log(`Discovered ${embeddingSources.length} pages`) diff --git a/apps/docs/scripts/search/sources/base.ts b/apps/docs/scripts/search/sources/base.ts new file mode 100644 index 00000000000..3a12513ed96 --- /dev/null +++ b/apps/docs/scripts/search/sources/base.ts @@ -0,0 +1,20 @@ +export type Json = Record< + string, + string | number | boolean | null | Json[] | { [key: string]: Json } +> + +export type Section = { + content: string + heading?: string + slug?: string +} + +export abstract class BaseSource { + checksum?: string + meta?: Json + sections?: Section[] + + constructor(public source: string, public path: string, public parentPath?: string) {} + + abstract load(): Promise<{ checksum: string; meta?: Json; sections: Section[] }> +} diff --git a/apps/docs/scripts/search/sources/index.ts b/apps/docs/scripts/search/sources/index.ts new file mode 100644 index 00000000000..66ff45b2932 --- /dev/null +++ b/apps/docs/scripts/search/sources/index.ts @@ -0,0 +1,85 @@ +import { MarkdownSource } from './markdown' +import { + CliReferenceSource, + ClientLibReferenceSource, + OpenApiReferenceSource, +} from './reference-doc' +import { walk } from './util' + +const ignoredFiles = ['pages/404.mdx'] + +export type SearchSource = + | MarkdownSource + | OpenApiReferenceSource + | ClientLibReferenceSource + | CliReferenceSource + +/** + * Fetches all the sources we want to index for search + */ +export async function fetchSources() { + const openApiReferenceSource = new OpenApiReferenceSource( + 'api', + '/reference/api', + { title: 'Management API Reference' }, + '../../spec/transforms/api_v0_openapi_deparsed.json', + '../../spec/common-api-sections.json' + ) + + const jsLibReferenceSource = new ClientLibReferenceSource( + 'js-lib', + '/reference/javascript', + { title: 'JavaScript Reference' }, + '../../spec/supabase_js_v2.yml', + '../../spec/common-client-libs-sections.json' + ) + + const dartLibReferenceSource = new ClientLibReferenceSource( + 'dart-lib', + '/reference/dart', + { title: 'Dart Reference' }, + '../../spec/supabase_dart_v1.yml', + '../../spec/common-client-libs-sections.json' + ) + + const pythonLibReferenceSource = new ClientLibReferenceSource( + 'python-lib', + '/reference/python', + { title: 'Python Reference' }, + '../../spec/supabase_py_v2.yml', + '../../spec/common-client-libs-sections.json' + ) + + const cSharpLibReferenceSource = new ClientLibReferenceSource( + 'csharp-lib', + '/reference/csharp', + { title: 'C# Reference' }, + '../../spec/supabase_csharp_v0.yml', + '../../spec/common-client-libs-sections.json' + ) + + const cliReferenceSource = new CliReferenceSource( + 'cli', + '/reference/cli', + { title: 'CLI Reference' }, + '../../spec/cli_v1_commands.yaml', + '../../spec/common-cli-sections.json' + ) + + const guideSources = (await walk('pages')) + .filter(({ path }) => /\.mdx?$/.test(path)) + .filter(({ path }) => !ignoredFiles.includes(path)) + .map((entry) => new MarkdownSource('guide', entry.path)) + + const sources: SearchSource[] = [ + openApiReferenceSource, + jsLibReferenceSource, + dartLibReferenceSource, + pythonLibReferenceSource, + cSharpLibReferenceSource, + cliReferenceSource, + ...guideSources, + ] + + return sources +} diff --git a/apps/docs/scripts/search/sources/markdown.ts b/apps/docs/scripts/search/sources/markdown.ts new file mode 100644 index 00000000000..1c7b29640c8 --- /dev/null +++ b/apps/docs/scripts/search/sources/markdown.ts @@ -0,0 +1,191 @@ +import { createHash } from 'crypto' +import { ObjectExpression } from 'estree' +import { readFile } from 'fs/promises' +import GithubSlugger from 'github-slugger' +import { Content, Root } from 'mdast' +import { fromMarkdown } from 'mdast-util-from-markdown' +import { MdxjsEsm, mdxFromMarkdown } from 'mdast-util-mdx' +import { toMarkdown } from 'mdast-util-to-markdown' +import { toString } from 'mdast-util-to-string' +import { mdxjs } from 'micromark-extension-mdxjs' +import { u } from 'unist-builder' +import { filter } from 'unist-util-filter' +import { BaseSource, Json, Section } from './base' + +/** + * Extracts ES literals from an `estree` `ObjectExpression` + * into a plain JavaScript object. + */ +export function getObjectFromExpression(node: ObjectExpression) { + return node.properties.reduce< + Record + >((object, property) => { + if (property.type !== 'Property') { + return object + } + + const key = (property.key.type === 'Identifier' && property.key.name) || undefined + const value = (property.value.type === 'Literal' && property.value.value) || undefined + + if (!key) { + return object + } + + return { + ...object, + [key]: value, + } + }, {}) +} + +/** + * Extracts the `meta` ESM export from the MDX file. + * + * This info is akin to frontmatter. + */ +export function extractMetaExport(mdxTree: Root) { + const metaExportNode = mdxTree.children.find((node): node is MdxjsEsm => { + return ( + node.type === 'mdxjsEsm' && + node.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' && + node.data.estree.body[0].declaration?.type === 'VariableDeclaration' && + node.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' && + node.data.estree.body[0].declaration.declarations[0].id.name === 'meta' + ) + }) + + if (!metaExportNode) { + return undefined + } + + const objectExpression = + (metaExportNode.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' && + metaExportNode.data.estree.body[0].declaration?.type === 'VariableDeclaration' && + metaExportNode.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' && + metaExportNode.data.estree.body[0].declaration.declarations[0].id.name === 'meta' && + metaExportNode.data.estree.body[0].declaration.declarations[0].init?.type === + 'ObjectExpression' && + metaExportNode.data.estree.body[0].declaration.declarations[0].init) || + undefined + + if (!objectExpression) { + return undefined + } + + return getObjectFromExpression(objectExpression) +} + +/** + * Splits a `mdast` tree into multiple trees based on + * a predicate function. Will include the splitting node + * at the beginning of each tree. + * + * Useful to split a markdown file into smaller sections. + */ +export function splitTreeBy(tree: Root, predicate: (node: Content) => boolean) { + return tree.children.reduce((trees, node) => { + const [lastTree] = trees.slice(-1) + + if (!lastTree || predicate(node)) { + const tree: Root = u('root', [node]) + return trees.concat(tree) + } + + lastTree.children.push(node) + return trees + }, []) +} + +/** + * Processes MDX content for search indexing. + * It extracts metadata, strips it of all JSX, + * and splits it into sub-sections based on criteria. + */ +export function processMdxForSearch(content: string): ProcessedMdx { + const checksum = createHash('sha256').update(content).digest('base64') + + const mdxTree = fromMarkdown(content, { + extensions: [mdxjs()], + mdastExtensions: [mdxFromMarkdown()], + }) + + const meta = extractMetaExport(mdxTree) + const serializableMeta: Json = JSON.parse(JSON.stringify(meta)) + + // Remove all MDX elements from markdown + const mdTree = filter( + mdxTree, + (node) => + ![ + 'mdxjsEsm', + 'mdxJsxFlowElement', + 'mdxJsxTextElement', + 'mdxFlowExpression', + 'mdxTextExpression', + ].includes(node.type) + ) + + if (!mdTree) { + return { + checksum, + meta: serializableMeta, + sections: [], + } + } + + const sectionTrees = splitTreeBy(mdTree, (node) => node.type === 'heading') + + const slugger = new GithubSlugger() + + const sections = sectionTrees.map((tree) => { + const [firstNode] = tree.children + + const heading = firstNode.type === 'heading' ? toString(firstNode) : undefined + const slug = heading ? slugger.slug(heading) : undefined + + return { + content: toMarkdown(tree), + heading, + slug, + } + }) + + return { + checksum, + meta: serializableMeta, + sections, + } +} + +export type ProcessedMdx = { + checksum: string + meta: Json + sections: Section[] +} + +export class MarkdownSource extends BaseSource { + type = 'markdown' as const + + constructor(source: string, public filePath: string, public parentFilePath?: string) { + const path = filePath.replace(/^pages/, '').replace(/\.mdx?$/, '') + const parentPath = parentFilePath?.replace(/^pages/, '').replace(/\.mdx?$/, '') + + super(source, path, parentPath) + } + + async load() { + const contents = await readFile(this.filePath, 'utf8') + + const { checksum, meta, sections } = processMdxForSearch(contents) + + this.checksum = checksum + this.meta = meta + this.sections = sections + + return { + checksum, + meta, + sections, + } + } +} diff --git a/apps/docs/scripts/search/sources/reference-doc.ts b/apps/docs/scripts/search/sources/reference-doc.ts new file mode 100644 index 00000000000..9f2304700bc --- /dev/null +++ b/apps/docs/scripts/search/sources/reference-doc.ts @@ -0,0 +1,139 @@ +import { createHash } from 'crypto' +import { readFile } from 'fs/promises' +import yaml from 'js-yaml' +import { OpenAPIV3 } from 'openapi-types' +import { Meta } from '..' +import { + ICommonFunc, + IFunctionDefinition, + ISpec, +} from '../../../components/reference/Reference.types' +import { CliCommand, CliSpec } from '../../../generator/types/CliSpec' +import { flattenSections } from '../../../lib/helpers' +import { enrichedOperation, gen_v3 } from '../../../lib/refGenerator/helpers' +import { BaseSource } from './base' + +export abstract class ReferenceSource extends BaseSource { + type = 'reference' as const + + constructor( + source: string, + path: string, + public meta: Meta, + public specFilePath: string, + public sectionsFilePath: string + ) { + super(source, path) + } + + async load() { + const specContents = await readFile(this.specFilePath, 'utf8') + const refSectionsContents = await readFile(this.sectionsFilePath, 'utf8') + + const refSections: ICommonFunc[] = JSON.parse(refSectionsContents) + const flattenedRefSections = flattenSections(refSections) + + const checksum = createHash('sha256') + .update(specContents + refSectionsContents) + .digest('base64') + + const specSections = this.getSpecSections(specContents) + + const sections = flattenedRefSections + .map((refSection) => { + const specSection = this.matchSpecSection(specSections, refSection.id) + + if (!specSection) { + return + } + + return { + heading: refSection.title, + slug: refSection.slug, + content: `${this.meta.title} for ${refSection.title}:\n${this.formatSection( + specSection, + refSection + )}`, + } + }) + .filter((section) => !!section) + + this.checksum = checksum + this.sections = sections + + return { + checksum, + sections, + meta: this.meta, + } + } + + abstract getSpecSections(specContents: string): SpecSection[] + abstract matchSpecSection(specSections: SpecSection[], id: string): SpecSection + abstract formatSection(specSection: SpecSection, refSection: ICommonFunc): string +} + +export class OpenApiReferenceSource extends ReferenceSource { + getSpecSections(specContents: string): enrichedOperation[] { + const spec: OpenAPIV3.Document<{}> = JSON.parse(specContents) + + const generatedSpec = gen_v3(spec, '', { + apiUrl: 'apiv0', + }) + + return generatedSpec.operations + } + matchSpecSection(operations: enrichedOperation[], id: string): enrichedOperation { + return operations.find((operation) => operation.operationId === id) + } + formatSection(specOperation: enrichedOperation) { + const { summary, description, operation, path, tags } = specOperation + return JSON.stringify({ + summary, + description, + operation, + path, + tags, + }) + } +} + +export class ClientLibReferenceSource extends ReferenceSource { + getSpecSections(specContents: string): IFunctionDefinition[] { + const spec = yaml.load(specContents) as ISpec + + return spec.functions + } + matchSpecSection(functionDefinitions: IFunctionDefinition[], id: string): IFunctionDefinition { + return functionDefinitions.find((functionDefinition) => functionDefinition.id === id) + } + formatSection(functionDefinition: IFunctionDefinition, refSection: ICommonFunc): string { + const { title } = refSection + const { description, title: functionName } = functionDefinition + + return JSON.stringify({ + title, + description, + functionName, + }) + } +} + +export class CliReferenceSource extends ReferenceSource { + getSpecSections(specContents: string): CliCommand[] { + const spec = yaml.load(specContents) as CliSpec + + return spec.commands + } + matchSpecSection(cliCommands: CliCommand[], id: string): CliCommand { + return cliCommands.find((cliCommand) => cliCommand.id === id) + } + formatSection(cliCommand: CliCommand): string { + const { summary, description, usage } = cliCommand + return JSON.stringify({ + summary, + description, + usage, + }) + } +} diff --git a/apps/docs/scripts/search/sources/util.ts b/apps/docs/scripts/search/sources/util.ts new file mode 100644 index 00000000000..b43bf98cdd1 --- /dev/null +++ b/apps/docs/scripts/search/sources/util.ts @@ -0,0 +1,43 @@ +import { readdir, stat } from 'fs/promises' +import { basename, dirname, join } from 'path' + +export type WalkEntry = { + path: string + parentPath?: string +} + +export async function walk(dir: string, parentPath?: string): Promise { + const immediateFiles = await readdir(dir) + + const recursiveFiles = await Promise.all( + immediateFiles.map(async (file) => { + const path = join(dir, file) + const stats = await stat(path) + if (stats.isDirectory()) { + // Keep track of document hierarchy (if this dir has corresponding doc file) + const docPath = `${basename(path)}.mdx` + + return walk( + path, + immediateFiles.includes(docPath) ? join(dirname(path), docPath) : parentPath + ) + } else if (stats.isFile()) { + return [ + { + path: path, + parentPath, + }, + ] + } else { + return [] + } + }) + ) + + const flattenedFiles = recursiveFiles.reduce( + (all, folderContents) => all.concat(folderContents), + [] + ) + + return flattenedFiles.sort((a, b) => a.path.localeCompare(b.path)) +}