mirror of
https://github.com/supabase/supabase.git
synced 2026-10-05 09:25:06 +03:00
refactor(search): split sources into their own files
This commit is contained in:
1 parent
fe3c31264c
commit
0ff1759dd2
6 files changed
+487
-443
No files matched your search
@@ -1,411 +1,24 @@
|
||||
import { createClient } from '@supabase/supabase-js'
|
||||
import { createHash } from 'crypto'
|
||||
import dotenv from 'dotenv'
|
||||
import { ObjectExpression } from 'estree'
|
||||
import { readdir, readFile, stat } from 'fs/promises'
|
||||
import GithubSlugger from 'github-slugger'
|
||||
import yaml from 'js-yaml'
|
||||
import { Content, Root } from 'mdast'
|
||||
import { fromMarkdown } from 'mdast-util-from-markdown'
|
||||
import { mdxFromMarkdown, MdxjsEsm } from 'mdast-util-mdx'
|
||||
import { toMarkdown } from 'mdast-util-to-markdown'
|
||||
import { toString } from 'mdast-util-to-string'
|
||||
import { mdxjs } from 'micromark-extension-mdxjs'
|
||||
import 'openai'
|
||||
import { Configuration, OpenAIApi } from 'openai'
|
||||
import { OpenAPIV3 } from 'openapi-types'
|
||||
import { basename, dirname, join } from 'path'
|
||||
import { u } from 'unist-builder'
|
||||
import { filter } from 'unist-util-filter'
|
||||
import { inspect } from 'util'
|
||||
import { ICommonFunc, IFunctionDefinition, ISpec } from '../components/reference/Reference.types'
|
||||
import { CliCommand, CliSpec } from '../generator/types/CliSpec'
|
||||
import { flattenSections } from '../lib/helpers'
|
||||
import { enrichedOperation, gen_v3 } from '../lib/refGenerator/helpers'
|
||||
import { fetchSources } from './search/sources'
|
||||
|
||||
dotenv.config()
|
||||
|
||||
const ignoredFiles = ['pages/404.mdx']
|
||||
|
||||
/**
|
||||
* Extracts ES literals from an `estree` `ObjectExpression`
|
||||
* into a plain JavaScript object.
|
||||
*/
|
||||
function getObjectFromExpression(node: ObjectExpression) {
|
||||
return node.properties.reduce<
|
||||
Record<string, string | number | bigint | true | RegExp | undefined>
|
||||
>((object, property) => {
|
||||
if (property.type !== 'Property') {
|
||||
return object
|
||||
}
|
||||
|
||||
const key = (property.key.type === 'Identifier' && property.key.name) || undefined
|
||||
const value = (property.value.type === 'Literal' && property.value.value) || undefined
|
||||
|
||||
if (!key) {
|
||||
return object
|
||||
}
|
||||
|
||||
return {
|
||||
...object,
|
||||
[key]: value,
|
||||
}
|
||||
}, {})
|
||||
}
|
||||
|
||||
/**
|
||||
* Extracts the `meta` ESM export from the MDX file.
|
||||
*
|
||||
* This info is akin to frontmatter.
|
||||
*/
|
||||
function extractMetaExport(mdxTree: Root) {
|
||||
const metaExportNode = mdxTree.children.find((node): node is MdxjsEsm => {
|
||||
return (
|
||||
node.type === 'mdxjsEsm' &&
|
||||
node.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' &&
|
||||
node.data.estree.body[0].declaration?.type === 'VariableDeclaration' &&
|
||||
node.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' &&
|
||||
node.data.estree.body[0].declaration.declarations[0].id.name === 'meta'
|
||||
)
|
||||
})
|
||||
|
||||
if (!metaExportNode) {
|
||||
return undefined
|
||||
}
|
||||
|
||||
const objectExpression =
|
||||
(metaExportNode.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' &&
|
||||
metaExportNode.data.estree.body[0].declaration?.type === 'VariableDeclaration' &&
|
||||
metaExportNode.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' &&
|
||||
metaExportNode.data.estree.body[0].declaration.declarations[0].id.name === 'meta' &&
|
||||
metaExportNode.data.estree.body[0].declaration.declarations[0].init?.type ===
|
||||
'ObjectExpression' &&
|
||||
metaExportNode.data.estree.body[0].declaration.declarations[0].init) ||
|
||||
undefined
|
||||
|
||||
if (!objectExpression) {
|
||||
return undefined
|
||||
}
|
||||
|
||||
return getObjectFromExpression(objectExpression)
|
||||
}
|
||||
|
||||
/**
|
||||
* Splits a `mdast` tree into multiple trees based on
|
||||
* a predicate function. Will include the splitting node
|
||||
* at the beginning of each tree.
|
||||
*
|
||||
* Useful to split a markdown file into smaller sections.
|
||||
*/
|
||||
function splitTreeBy(tree: Root, predicate: (node: Content) => boolean) {
|
||||
return tree.children.reduce<Root[]>((trees, node) => {
|
||||
const [lastTree] = trees.slice(-1)
|
||||
|
||||
if (!lastTree || predicate(node)) {
|
||||
const tree: Root = u('root', [node])
|
||||
return trees.concat(tree)
|
||||
}
|
||||
|
||||
lastTree.children.push(node)
|
||||
return trees
|
||||
}, [])
|
||||
}
|
||||
|
||||
type Meta = ReturnType<typeof extractMetaExport>
|
||||
|
||||
type Section = {
|
||||
content: string
|
||||
heading?: string
|
||||
slug?: string
|
||||
}
|
||||
|
||||
type ProcessedMdx = {
|
||||
checksum: string
|
||||
meta: Meta
|
||||
sections: Section[]
|
||||
}
|
||||
|
||||
/**
|
||||
* Processes MDX content for search indexing.
|
||||
* It extracts metadata, strips it of all JSX,
|
||||
* and splits it into sub-sections based on criteria.
|
||||
*/
|
||||
function processMdxForSearch(content: string): ProcessedMdx {
|
||||
const checksum = createHash('sha256').update(content).digest('base64')
|
||||
|
||||
const mdxTree = fromMarkdown(content, {
|
||||
extensions: [mdxjs()],
|
||||
mdastExtensions: [mdxFromMarkdown()],
|
||||
})
|
||||
|
||||
const meta = extractMetaExport(mdxTree)
|
||||
|
||||
// Remove all MDX elements from markdown
|
||||
const mdTree = filter(
|
||||
mdxTree,
|
||||
(node) =>
|
||||
![
|
||||
'mdxjsEsm',
|
||||
'mdxJsxFlowElement',
|
||||
'mdxJsxTextElement',
|
||||
'mdxFlowExpression',
|
||||
'mdxTextExpression',
|
||||
].includes(node.type)
|
||||
)
|
||||
|
||||
if (!mdTree) {
|
||||
return {
|
||||
checksum,
|
||||
meta,
|
||||
sections: [],
|
||||
}
|
||||
}
|
||||
|
||||
const sectionTrees = splitTreeBy(mdTree, (node) => node.type === 'heading')
|
||||
|
||||
const slugger = new GithubSlugger()
|
||||
|
||||
const sections = sectionTrees.map((tree) => {
|
||||
const [firstNode] = tree.children
|
||||
|
||||
const heading = firstNode.type === 'heading' ? toString(firstNode) : undefined
|
||||
const slug = heading ? slugger.slug(heading) : undefined
|
||||
|
||||
return {
|
||||
content: toMarkdown(tree),
|
||||
heading,
|
||||
slug,
|
||||
}
|
||||
})
|
||||
|
||||
return {
|
||||
checksum,
|
||||
meta,
|
||||
sections,
|
||||
}
|
||||
}
|
||||
|
||||
type WalkEntry = {
|
||||
path: string
|
||||
parentPath?: string
|
||||
}
|
||||
|
||||
async function walk(dir: string, parentPath?: string): Promise<WalkEntry[]> {
|
||||
const immediateFiles = await readdir(dir)
|
||||
|
||||
const recursiveFiles = await Promise.all(
|
||||
immediateFiles.map(async (file) => {
|
||||
const path = join(dir, file)
|
||||
const stats = await stat(path)
|
||||
if (stats.isDirectory()) {
|
||||
// Keep track of document hierarchy (if this dir has corresponding doc file)
|
||||
const docPath = `${basename(path)}.mdx`
|
||||
|
||||
return walk(
|
||||
path,
|
||||
immediateFiles.includes(docPath) ? join(dirname(path), docPath) : parentPath
|
||||
)
|
||||
} else if (stats.isFile()) {
|
||||
return [
|
||||
{
|
||||
path: path,
|
||||
parentPath,
|
||||
},
|
||||
]
|
||||
} else {
|
||||
return []
|
||||
}
|
||||
})
|
||||
)
|
||||
|
||||
const flattenedFiles = recursiveFiles.reduce(
|
||||
(all, folderContents) => all.concat(folderContents),
|
||||
[]
|
||||
)
|
||||
|
||||
return flattenedFiles.sort((a, b) => a.path.localeCompare(b.path))
|
||||
}
|
||||
|
||||
abstract class BaseEmbeddingSource {
|
||||
checksum?: string
|
||||
meta?: Meta
|
||||
sections?: Section[]
|
||||
|
||||
constructor(public source: string, public path: string, public parentPath?: string) {}
|
||||
|
||||
abstract load(): Promise<{ checksum: string; meta?: Meta; sections: Section[] }>
|
||||
}
|
||||
|
||||
class MarkdownEmbeddingSource extends BaseEmbeddingSource {
|
||||
type: 'markdown' = 'markdown'
|
||||
|
||||
constructor(source: string, public filePath: string, public parentFilePath?: string) {
|
||||
const path = filePath.replace(/^pages/, '').replace(/\.mdx?$/, '')
|
||||
const parentPath = parentFilePath?.replace(/^pages/, '').replace(/\.mdx?$/, '')
|
||||
|
||||
super(source, path, parentPath)
|
||||
}
|
||||
|
||||
async load() {
|
||||
const contents = await readFile(this.filePath, 'utf8')
|
||||
|
||||
const { checksum, meta, sections } = processMdxForSearch(contents)
|
||||
|
||||
this.checksum = checksum
|
||||
this.meta = meta
|
||||
this.sections = sections
|
||||
|
||||
return {
|
||||
checksum,
|
||||
meta,
|
||||
sections,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
abstract class ReferenceEmbeddingSource<SpecSection> extends BaseEmbeddingSource {
|
||||
type: 'reference' = 'reference'
|
||||
|
||||
constructor(
|
||||
source: string,
|
||||
path: string,
|
||||
public meta: Meta,
|
||||
public specFilePath: string,
|
||||
public sectionsFilePath: string
|
||||
) {
|
||||
super(source, path)
|
||||
}
|
||||
|
||||
async load() {
|
||||
const specContents = await readFile(this.specFilePath, 'utf8')
|
||||
const refSectionsContents = await readFile(this.sectionsFilePath, 'utf8')
|
||||
|
||||
const refSections: ICommonFunc[] = JSON.parse(refSectionsContents)
|
||||
const flattenedRefSections = flattenSections(refSections)
|
||||
|
||||
const checksum = createHash('sha256')
|
||||
.update(specContents + refSectionsContents)
|
||||
.digest('base64')
|
||||
|
||||
const specSections = this.getSpecSections(specContents)
|
||||
|
||||
const sections = flattenedRefSections
|
||||
.map((refSection) => {
|
||||
const specSection = this.matchSpecSection(specSections, refSection.id)
|
||||
|
||||
if (!specSection) {
|
||||
return
|
||||
}
|
||||
|
||||
return {
|
||||
heading: refSection.title,
|
||||
slug: refSection.slug,
|
||||
content: `${this.meta.title} for ${refSection.title}:\n${this.formatSection(
|
||||
specSection,
|
||||
refSection
|
||||
)}`,
|
||||
}
|
||||
})
|
||||
.filter((section) => !!section)
|
||||
|
||||
this.checksum = checksum
|
||||
this.sections = sections
|
||||
|
||||
return {
|
||||
checksum,
|
||||
sections,
|
||||
meta: this.meta,
|
||||
}
|
||||
}
|
||||
|
||||
abstract getSpecSections(specContents: string): SpecSection[]
|
||||
abstract matchSpecSection(specSections: SpecSection[], id: string): SpecSection
|
||||
abstract formatSection(specSection: SpecSection, refSection: ICommonFunc): string
|
||||
}
|
||||
|
||||
class OpenApiEmbeddingSource extends ReferenceEmbeddingSource<enrichedOperation> {
|
||||
getSpecSections(specContents: string): enrichedOperation[] {
|
||||
const spec: OpenAPIV3.Document<{}> = JSON.parse(specContents)
|
||||
|
||||
const generatedSpec = gen_v3(spec, '', {
|
||||
apiUrl: 'apiv0',
|
||||
})
|
||||
|
||||
return generatedSpec.operations
|
||||
}
|
||||
matchSpecSection(operations: enrichedOperation[], id: string): enrichedOperation {
|
||||
return operations.find((operation) => operation.operationId === id)
|
||||
}
|
||||
formatSection(specOperation: enrichedOperation) {
|
||||
const { summary, description, operation, path, tags } = specOperation
|
||||
return JSON.stringify({
|
||||
summary,
|
||||
description,
|
||||
operation,
|
||||
path,
|
||||
tags,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
class ClientLibEmbeddingSource extends ReferenceEmbeddingSource<IFunctionDefinition> {
|
||||
getSpecSections(specContents: string): IFunctionDefinition[] {
|
||||
const spec = yaml.load(specContents) as ISpec
|
||||
|
||||
return spec.functions
|
||||
}
|
||||
matchSpecSection(functionDefinitions: IFunctionDefinition[], id: string): IFunctionDefinition {
|
||||
return functionDefinitions.find((functionDefinition) => functionDefinition.id === id)
|
||||
}
|
||||
formatSection(functionDefinition: IFunctionDefinition, refSection: ICommonFunc): string {
|
||||
const { title } = refSection
|
||||
const { description, title: functionName } = functionDefinition
|
||||
|
||||
return JSON.stringify({
|
||||
title,
|
||||
description,
|
||||
functionName,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
class CliEmbeddingSource extends ReferenceEmbeddingSource<CliCommand> {
|
||||
getSpecSections(specContents: string): CliCommand[] {
|
||||
const spec = yaml.load(specContents) as CliSpec
|
||||
|
||||
return spec.commands
|
||||
}
|
||||
matchSpecSection(cliCommands: CliCommand[], id: string): CliCommand {
|
||||
return cliCommands.find((cliCommand) => cliCommand.id === id)
|
||||
}
|
||||
formatSection(cliCommand: CliCommand): string {
|
||||
const { summary, description, usage } = cliCommand
|
||||
return JSON.stringify({
|
||||
summary,
|
||||
description,
|
||||
usage,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
type EmbeddingSource =
|
||||
| MarkdownEmbeddingSource
|
||||
| OpenApiEmbeddingSource
|
||||
| ClientLibEmbeddingSource
|
||||
| CliEmbeddingSource
|
||||
|
||||
async function generateEmbeddings() {
|
||||
// TODO: use better CLI lib like yargs
|
||||
const args = process.argv.slice(2)
|
||||
const shouldRefresh = args.includes('--refresh')
|
||||
|
||||
if (
|
||||
!process.env.NEXT_PUBLIC_SUPABASE_URL ||
|
||||
!process.env.SUPABASE_SERVICE_ROLE_KEY ||
|
||||
!process.env.OPENAI_KEY
|
||||
) {
|
||||
return console.log(
|
||||
'Environment variables NEXT_PUBLIC_SUPABASE_URL, SUPABASE_SERVICE_ROLE_KEY, and OPENAI_KEY are required: skipping embeddings generation'
|
||||
const requiredEnvVars = ['NEXT_PUBLIC_SUPABASE_URL', 'SUPABASE_SERVICE_ROLE_KEY', 'OPENAI_KEY']
|
||||
|
||||
if (requiredEnvVars.some((name) => !process.env[name])) {
|
||||
throw new Error(
|
||||
`Environment variables ${requiredEnvVars.join(
|
||||
', '
|
||||
)} are required: skipping embeddings generation`
|
||||
)
|
||||
}
|
||||
|
||||
@@ -420,54 +33,7 @@ async function generateEmbeddings() {
|
||||
}
|
||||
)
|
||||
|
||||
const embeddingSources: EmbeddingSource[] = [
|
||||
new OpenApiEmbeddingSource(
|
||||
'api',
|
||||
'/reference/api',
|
||||
{ title: 'Management API Reference' },
|
||||
'../../spec/transforms/api_v0_openapi_deparsed.json',
|
||||
'../../spec/common-api-sections.json'
|
||||
),
|
||||
new ClientLibEmbeddingSource(
|
||||
'js-lib',
|
||||
'/reference/javascript',
|
||||
{ title: 'JavaScript Reference' },
|
||||
'../../spec/supabase_js_v2.yml',
|
||||
'../../spec/common-client-libs-sections.json'
|
||||
),
|
||||
new ClientLibEmbeddingSource(
|
||||
'dart-lib',
|
||||
'/reference/dart',
|
||||
{ title: 'Dart Reference' },
|
||||
'../../spec/supabase_dart_v1.yml',
|
||||
'../../spec/common-client-libs-sections.json'
|
||||
),
|
||||
new ClientLibEmbeddingSource(
|
||||
'python-lib',
|
||||
'/reference/python',
|
||||
{ title: 'Python Reference' },
|
||||
'../../spec/supabase_py_v2.yml',
|
||||
'../../spec/common-client-libs-sections.json'
|
||||
),
|
||||
new ClientLibEmbeddingSource(
|
||||
'csharp-lib',
|
||||
'/reference/csharp',
|
||||
{ title: 'C# Reference' },
|
||||
'../../spec/supabase_csharp_v0.yml',
|
||||
'../../spec/common-client-libs-sections.json'
|
||||
),
|
||||
new CliEmbeddingSource(
|
||||
'cli',
|
||||
'/reference/cli',
|
||||
{ title: 'CLI Reference' },
|
||||
'../../spec/cli_v1_commands.yaml',
|
||||
'../../spec/common-cli-sections.json'
|
||||
),
|
||||
...(await walk('pages'))
|
||||
.filter(({ path }) => /\.mdx?$/.test(path))
|
||||
.filter(({ path }) => !ignoredFiles.includes(path))
|
||||
.map((entry) => new MarkdownEmbeddingSource('guide', entry.path)),
|
||||
]
|
||||
const embeddingSources = await fetchSources()
|
||||
|
||||
console.log(`Discovered ${embeddingSources.length} pages`)
|
||||
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
export type Json = Record<
|
||||
string,
|
||||
string | number | boolean | null | Json[] | { [key: string]: Json }
|
||||
>
|
||||
|
||||
export type Section = {
|
||||
content: string
|
||||
heading?: string
|
||||
slug?: string
|
||||
}
|
||||
|
||||
export abstract class BaseSource {
|
||||
checksum?: string
|
||||
meta?: Json
|
||||
sections?: Section[]
|
||||
|
||||
constructor(public source: string, public path: string, public parentPath?: string) {}
|
||||
|
||||
abstract load(): Promise<{ checksum: string; meta?: Json; sections: Section[] }>
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
import { MarkdownSource } from './markdown'
|
||||
import {
|
||||
CliReferenceSource,
|
||||
ClientLibReferenceSource,
|
||||
OpenApiReferenceSource,
|
||||
} from './reference-doc'
|
||||
import { walk } from './util'
|
||||
|
||||
const ignoredFiles = ['pages/404.mdx']
|
||||
|
||||
export type SearchSource =
|
||||
| MarkdownSource
|
||||
| OpenApiReferenceSource
|
||||
| ClientLibReferenceSource
|
||||
| CliReferenceSource
|
||||
|
||||
/**
|
||||
* Fetches all the sources we want to index for search
|
||||
*/
|
||||
export async function fetchSources() {
|
||||
const openApiReferenceSource = new OpenApiReferenceSource(
|
||||
'api',
|
||||
'/reference/api',
|
||||
{ title: 'Management API Reference' },
|
||||
'../../spec/transforms/api_v0_openapi_deparsed.json',
|
||||
'../../spec/common-api-sections.json'
|
||||
)
|
||||
|
||||
const jsLibReferenceSource = new ClientLibReferenceSource(
|
||||
'js-lib',
|
||||
'/reference/javascript',
|
||||
{ title: 'JavaScript Reference' },
|
||||
'../../spec/supabase_js_v2.yml',
|
||||
'../../spec/common-client-libs-sections.json'
|
||||
)
|
||||
|
||||
const dartLibReferenceSource = new ClientLibReferenceSource(
|
||||
'dart-lib',
|
||||
'/reference/dart',
|
||||
{ title: 'Dart Reference' },
|
||||
'../../spec/supabase_dart_v1.yml',
|
||||
'../../spec/common-client-libs-sections.json'
|
||||
)
|
||||
|
||||
const pythonLibReferenceSource = new ClientLibReferenceSource(
|
||||
'python-lib',
|
||||
'/reference/python',
|
||||
{ title: 'Python Reference' },
|
||||
'../../spec/supabase_py_v2.yml',
|
||||
'../../spec/common-client-libs-sections.json'
|
||||
)
|
||||
|
||||
const cSharpLibReferenceSource = new ClientLibReferenceSource(
|
||||
'csharp-lib',
|
||||
'/reference/csharp',
|
||||
{ title: 'C# Reference' },
|
||||
'../../spec/supabase_csharp_v0.yml',
|
||||
'../../spec/common-client-libs-sections.json'
|
||||
)
|
||||
|
||||
const cliReferenceSource = new CliReferenceSource(
|
||||
'cli',
|
||||
'/reference/cli',
|
||||
{ title: 'CLI Reference' },
|
||||
'../../spec/cli_v1_commands.yaml',
|
||||
'../../spec/common-cli-sections.json'
|
||||
)
|
||||
|
||||
const guideSources = (await walk('pages'))
|
||||
.filter(({ path }) => /\.mdx?$/.test(path))
|
||||
.filter(({ path }) => !ignoredFiles.includes(path))
|
||||
.map((entry) => new MarkdownSource('guide', entry.path))
|
||||
|
||||
const sources: SearchSource[] = [
|
||||
openApiReferenceSource,
|
||||
jsLibReferenceSource,
|
||||
dartLibReferenceSource,
|
||||
pythonLibReferenceSource,
|
||||
cSharpLibReferenceSource,
|
||||
cliReferenceSource,
|
||||
...guideSources,
|
||||
]
|
||||
|
||||
return sources
|
||||
}
|
||||
@@ -0,0 +1,191 @@
|
||||
import { createHash } from 'crypto'
|
||||
import { ObjectExpression } from 'estree'
|
||||
import { readFile } from 'fs/promises'
|
||||
import GithubSlugger from 'github-slugger'
|
||||
import { Content, Root } from 'mdast'
|
||||
import { fromMarkdown } from 'mdast-util-from-markdown'
|
||||
import { MdxjsEsm, mdxFromMarkdown } from 'mdast-util-mdx'
|
||||
import { toMarkdown } from 'mdast-util-to-markdown'
|
||||
import { toString } from 'mdast-util-to-string'
|
||||
import { mdxjs } from 'micromark-extension-mdxjs'
|
||||
import { u } from 'unist-builder'
|
||||
import { filter } from 'unist-util-filter'
|
||||
import { BaseSource, Json, Section } from './base'
|
||||
|
||||
/**
|
||||
* Extracts ES literals from an `estree` `ObjectExpression`
|
||||
* into a plain JavaScript object.
|
||||
*/
|
||||
export function getObjectFromExpression(node: ObjectExpression) {
|
||||
return node.properties.reduce<
|
||||
Record<string, string | number | bigint | true | RegExp | undefined>
|
||||
>((object, property) => {
|
||||
if (property.type !== 'Property') {
|
||||
return object
|
||||
}
|
||||
|
||||
const key = (property.key.type === 'Identifier' && property.key.name) || undefined
|
||||
const value = (property.value.type === 'Literal' && property.value.value) || undefined
|
||||
|
||||
if (!key) {
|
||||
return object
|
||||
}
|
||||
|
||||
return {
|
||||
...object,
|
||||
[key]: value,
|
||||
}
|
||||
}, {})
|
||||
}
|
||||
|
||||
/**
|
||||
* Extracts the `meta` ESM export from the MDX file.
|
||||
*
|
||||
* This info is akin to frontmatter.
|
||||
*/
|
||||
export function extractMetaExport(mdxTree: Root) {
|
||||
const metaExportNode = mdxTree.children.find((node): node is MdxjsEsm => {
|
||||
return (
|
||||
node.type === 'mdxjsEsm' &&
|
||||
node.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' &&
|
||||
node.data.estree.body[0].declaration?.type === 'VariableDeclaration' &&
|
||||
node.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' &&
|
||||
node.data.estree.body[0].declaration.declarations[0].id.name === 'meta'
|
||||
)
|
||||
})
|
||||
|
||||
if (!metaExportNode) {
|
||||
return undefined
|
||||
}
|
||||
|
||||
const objectExpression =
|
||||
(metaExportNode.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' &&
|
||||
metaExportNode.data.estree.body[0].declaration?.type === 'VariableDeclaration' &&
|
||||
metaExportNode.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' &&
|
||||
metaExportNode.data.estree.body[0].declaration.declarations[0].id.name === 'meta' &&
|
||||
metaExportNode.data.estree.body[0].declaration.declarations[0].init?.type ===
|
||||
'ObjectExpression' &&
|
||||
metaExportNode.data.estree.body[0].declaration.declarations[0].init) ||
|
||||
undefined
|
||||
|
||||
if (!objectExpression) {
|
||||
return undefined
|
||||
}
|
||||
|
||||
return getObjectFromExpression(objectExpression)
|
||||
}
|
||||
|
||||
/**
|
||||
* Splits a `mdast` tree into multiple trees based on
|
||||
* a predicate function. Will include the splitting node
|
||||
* at the beginning of each tree.
|
||||
*
|
||||
* Useful to split a markdown file into smaller sections.
|
||||
*/
|
||||
export function splitTreeBy(tree: Root, predicate: (node: Content) => boolean) {
|
||||
return tree.children.reduce<Root[]>((trees, node) => {
|
||||
const [lastTree] = trees.slice(-1)
|
||||
|
||||
if (!lastTree || predicate(node)) {
|
||||
const tree: Root = u('root', [node])
|
||||
return trees.concat(tree)
|
||||
}
|
||||
|
||||
lastTree.children.push(node)
|
||||
return trees
|
||||
}, [])
|
||||
}
|
||||
|
||||
/**
|
||||
* Processes MDX content for search indexing.
|
||||
* It extracts metadata, strips it of all JSX,
|
||||
* and splits it into sub-sections based on criteria.
|
||||
*/
|
||||
export function processMdxForSearch(content: string): ProcessedMdx {
|
||||
const checksum = createHash('sha256').update(content).digest('base64')
|
||||
|
||||
const mdxTree = fromMarkdown(content, {
|
||||
extensions: [mdxjs()],
|
||||
mdastExtensions: [mdxFromMarkdown()],
|
||||
})
|
||||
|
||||
const meta = extractMetaExport(mdxTree)
|
||||
const serializableMeta: Json = JSON.parse(JSON.stringify(meta))
|
||||
|
||||
// Remove all MDX elements from markdown
|
||||
const mdTree = filter(
|
||||
mdxTree,
|
||||
(node) =>
|
||||
![
|
||||
'mdxjsEsm',
|
||||
'mdxJsxFlowElement',
|
||||
'mdxJsxTextElement',
|
||||
'mdxFlowExpression',
|
||||
'mdxTextExpression',
|
||||
].includes(node.type)
|
||||
)
|
||||
|
||||
if (!mdTree) {
|
||||
return {
|
||||
checksum,
|
||||
meta: serializableMeta,
|
||||
sections: [],
|
||||
}
|
||||
}
|
||||
|
||||
const sectionTrees = splitTreeBy(mdTree, (node) => node.type === 'heading')
|
||||
|
||||
const slugger = new GithubSlugger()
|
||||
|
||||
const sections = sectionTrees.map((tree) => {
|
||||
const [firstNode] = tree.children
|
||||
|
||||
const heading = firstNode.type === 'heading' ? toString(firstNode) : undefined
|
||||
const slug = heading ? slugger.slug(heading) : undefined
|
||||
|
||||
return {
|
||||
content: toMarkdown(tree),
|
||||
heading,
|
||||
slug,
|
||||
}
|
||||
})
|
||||
|
||||
return {
|
||||
checksum,
|
||||
meta: serializableMeta,
|
||||
sections,
|
||||
}
|
||||
}
|
||||
|
||||
export type ProcessedMdx = {
|
||||
checksum: string
|
||||
meta: Json
|
||||
sections: Section[]
|
||||
}
|
||||
|
||||
export class MarkdownSource extends BaseSource {
|
||||
type = 'markdown' as const
|
||||
|
||||
constructor(source: string, public filePath: string, public parentFilePath?: string) {
|
||||
const path = filePath.replace(/^pages/, '').replace(/\.mdx?$/, '')
|
||||
const parentPath = parentFilePath?.replace(/^pages/, '').replace(/\.mdx?$/, '')
|
||||
|
||||
super(source, path, parentPath)
|
||||
}
|
||||
|
||||
async load() {
|
||||
const contents = await readFile(this.filePath, 'utf8')
|
||||
|
||||
const { checksum, meta, sections } = processMdxForSearch(contents)
|
||||
|
||||
this.checksum = checksum
|
||||
this.meta = meta
|
||||
this.sections = sections
|
||||
|
||||
return {
|
||||
checksum,
|
||||
meta,
|
||||
sections,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,139 @@
|
||||
import { createHash } from 'crypto'
|
||||
import { readFile } from 'fs/promises'
|
||||
import yaml from 'js-yaml'
|
||||
import { OpenAPIV3 } from 'openapi-types'
|
||||
import { Meta } from '..'
|
||||
import {
|
||||
ICommonFunc,
|
||||
IFunctionDefinition,
|
||||
ISpec,
|
||||
} from '../../../components/reference/Reference.types'
|
||||
import { CliCommand, CliSpec } from '../../../generator/types/CliSpec'
|
||||
import { flattenSections } from '../../../lib/helpers'
|
||||
import { enrichedOperation, gen_v3 } from '../../../lib/refGenerator/helpers'
|
||||
import { BaseSource } from './base'
|
||||
|
||||
export abstract class ReferenceSource<SpecSection> extends BaseSource {
|
||||
type = 'reference' as const
|
||||
|
||||
constructor(
|
||||
source: string,
|
||||
path: string,
|
||||
public meta: Meta,
|
||||
public specFilePath: string,
|
||||
public sectionsFilePath: string
|
||||
) {
|
||||
super(source, path)
|
||||
}
|
||||
|
||||
async load() {
|
||||
const specContents = await readFile(this.specFilePath, 'utf8')
|
||||
const refSectionsContents = await readFile(this.sectionsFilePath, 'utf8')
|
||||
|
||||
const refSections: ICommonFunc[] = JSON.parse(refSectionsContents)
|
||||
const flattenedRefSections = flattenSections(refSections)
|
||||
|
||||
const checksum = createHash('sha256')
|
||||
.update(specContents + refSectionsContents)
|
||||
.digest('base64')
|
||||
|
||||
const specSections = this.getSpecSections(specContents)
|
||||
|
||||
const sections = flattenedRefSections
|
||||
.map((refSection) => {
|
||||
const specSection = this.matchSpecSection(specSections, refSection.id)
|
||||
|
||||
if (!specSection) {
|
||||
return
|
||||
}
|
||||
|
||||
return {
|
||||
heading: refSection.title,
|
||||
slug: refSection.slug,
|
||||
content: `${this.meta.title} for ${refSection.title}:\n${this.formatSection(
|
||||
specSection,
|
||||
refSection
|
||||
)}`,
|
||||
}
|
||||
})
|
||||
.filter((section) => !!section)
|
||||
|
||||
this.checksum = checksum
|
||||
this.sections = sections
|
||||
|
||||
return {
|
||||
checksum,
|
||||
sections,
|
||||
meta: this.meta,
|
||||
}
|
||||
}
|
||||
|
||||
abstract getSpecSections(specContents: string): SpecSection[]
|
||||
abstract matchSpecSection(specSections: SpecSection[], id: string): SpecSection
|
||||
abstract formatSection(specSection: SpecSection, refSection: ICommonFunc): string
|
||||
}
|
||||
|
||||
export class OpenApiReferenceSource extends ReferenceSource<enrichedOperation> {
|
||||
getSpecSections(specContents: string): enrichedOperation[] {
|
||||
const spec: OpenAPIV3.Document<{}> = JSON.parse(specContents)
|
||||
|
||||
const generatedSpec = gen_v3(spec, '', {
|
||||
apiUrl: 'apiv0',
|
||||
})
|
||||
|
||||
return generatedSpec.operations
|
||||
}
|
||||
matchSpecSection(operations: enrichedOperation[], id: string): enrichedOperation {
|
||||
return operations.find((operation) => operation.operationId === id)
|
||||
}
|
||||
formatSection(specOperation: enrichedOperation) {
|
||||
const { summary, description, operation, path, tags } = specOperation
|
||||
return JSON.stringify({
|
||||
summary,
|
||||
description,
|
||||
operation,
|
||||
path,
|
||||
tags,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
export class ClientLibReferenceSource extends ReferenceSource<IFunctionDefinition> {
|
||||
getSpecSections(specContents: string): IFunctionDefinition[] {
|
||||
const spec = yaml.load(specContents) as ISpec
|
||||
|
||||
return spec.functions
|
||||
}
|
||||
matchSpecSection(functionDefinitions: IFunctionDefinition[], id: string): IFunctionDefinition {
|
||||
return functionDefinitions.find((functionDefinition) => functionDefinition.id === id)
|
||||
}
|
||||
formatSection(functionDefinition: IFunctionDefinition, refSection: ICommonFunc): string {
|
||||
const { title } = refSection
|
||||
const { description, title: functionName } = functionDefinition
|
||||
|
||||
return JSON.stringify({
|
||||
title,
|
||||
description,
|
||||
functionName,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
export class CliReferenceSource extends ReferenceSource<CliCommand> {
|
||||
getSpecSections(specContents: string): CliCommand[] {
|
||||
const spec = yaml.load(specContents) as CliSpec
|
||||
|
||||
return spec.commands
|
||||
}
|
||||
matchSpecSection(cliCommands: CliCommand[], id: string): CliCommand {
|
||||
return cliCommands.find((cliCommand) => cliCommand.id === id)
|
||||
}
|
||||
formatSection(cliCommand: CliCommand): string {
|
||||
const { summary, description, usage } = cliCommand
|
||||
return JSON.stringify({
|
||||
summary,
|
||||
description,
|
||||
usage,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
import { readdir, stat } from 'fs/promises'
|
||||
import { basename, dirname, join } from 'path'
|
||||
|
||||
export type WalkEntry = {
|
||||
path: string
|
||||
parentPath?: string
|
||||
}
|
||||
|
||||
export async function walk(dir: string, parentPath?: string): Promise<WalkEntry[]> {
|
||||
const immediateFiles = await readdir(dir)
|
||||
|
||||
const recursiveFiles = await Promise.all(
|
||||
immediateFiles.map(async (file) => {
|
||||
const path = join(dir, file)
|
||||
const stats = await stat(path)
|
||||
if (stats.isDirectory()) {
|
||||
// Keep track of document hierarchy (if this dir has corresponding doc file)
|
||||
const docPath = `${basename(path)}.mdx`
|
||||
|
||||
return walk(
|
||||
path,
|
||||
immediateFiles.includes(docPath) ? join(dirname(path), docPath) : parentPath
|
||||
)
|
||||
} else if (stats.isFile()) {
|
||||
return [
|
||||
{
|
||||
path: path,
|
||||
parentPath,
|
||||
},
|
||||
]
|
||||
} else {
|
||||
return []
|
||||
}
|
||||
})
|
||||
)
|
||||
|
||||
const flattenedFiles = recursiveFiles.reduce(
|
||||
(all, folderContents) => all.concat(folderContents),
|
||||
[]
|
||||
)
|
||||
|
||||
return flattenedFiles.sort((a, b) => a.path.localeCompare(b.path))
|
||||
}
|
||||
Reference in new issue
Block a user