refactor(search): split sources into their own files

This commit is contained in:
Greg Richardson committed 2023-04-20 17:37:01 -06:00
1 parent fe3c31264c
commit 0ff1759dd2
6 files changed
+487 -443

No files matched your search

+9 -443
View File
@@ -1,411 +1,24 @@
import { createClient } from '@supabase/supabase-js'
import { createHash } from 'crypto'
import dotenv from 'dotenv'
import { ObjectExpression } from 'estree'
import { readdir, readFile, stat } from 'fs/promises'
import GithubSlugger from 'github-slugger'
import yaml from 'js-yaml'
import { Content, Root } from 'mdast'
import { fromMarkdown } from 'mdast-util-from-markdown'
import { mdxFromMarkdown, MdxjsEsm } from 'mdast-util-mdx'
import { toMarkdown } from 'mdast-util-to-markdown'
import { toString } from 'mdast-util-to-string'
import { mdxjs } from 'micromark-extension-mdxjs'
import 'openai'
import { Configuration, OpenAIApi } from 'openai'
import { OpenAPIV3 } from 'openapi-types'
import { basename, dirname, join } from 'path'
import { u } from 'unist-builder'
import { filter } from 'unist-util-filter'
import { inspect } from 'util'
import { ICommonFunc, IFunctionDefinition, ISpec } from '../components/reference/Reference.types'
import { CliCommand, CliSpec } from '../generator/types/CliSpec'
import { flattenSections } from '../lib/helpers'
import { enrichedOperation, gen_v3 } from '../lib/refGenerator/helpers'
import { fetchSources } from './search/sources'
dotenv.config()
const ignoredFiles = ['pages/404.mdx']
/**
* Extracts ES literals from an `estree` `ObjectExpression`
* into a plain JavaScript object.
*/
function getObjectFromExpression(node: ObjectExpression) {
return node.properties.reduce<
Record<string, string | number | bigint | true | RegExp | undefined>
>((object, property) => {
if (property.type !== 'Property') {
return object
}
const key = (property.key.type === 'Identifier' && property.key.name) || undefined
const value = (property.value.type === 'Literal' && property.value.value) || undefined
if (!key) {
return object
}
return {
...object,
[key]: value,
}
}, {})
}
/**
* Extracts the `meta` ESM export from the MDX file.
*
* This info is akin to frontmatter.
*/
function extractMetaExport(mdxTree: Root) {
const metaExportNode = mdxTree.children.find((node): node is MdxjsEsm => {
return (
node.type === 'mdxjsEsm' &&
node.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' &&
node.data.estree.body[0].declaration?.type === 'VariableDeclaration' &&
node.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' &&
node.data.estree.body[0].declaration.declarations[0].id.name === 'meta'
)
})
if (!metaExportNode) {
return undefined
}
const objectExpression =
(metaExportNode.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' &&
metaExportNode.data.estree.body[0].declaration?.type === 'VariableDeclaration' &&
metaExportNode.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' &&
metaExportNode.data.estree.body[0].declaration.declarations[0].id.name === 'meta' &&
metaExportNode.data.estree.body[0].declaration.declarations[0].init?.type ===
'ObjectExpression' &&
metaExportNode.data.estree.body[0].declaration.declarations[0].init) ||
undefined
if (!objectExpression) {
return undefined
}
return getObjectFromExpression(objectExpression)
}
/**
* Splits a `mdast` tree into multiple trees based on
* a predicate function. Will include the splitting node
* at the beginning of each tree.
*
* Useful to split a markdown file into smaller sections.
*/
function splitTreeBy(tree: Root, predicate: (node: Content) => boolean) {
return tree.children.reduce<Root[]>((trees, node) => {
const [lastTree] = trees.slice(-1)
if (!lastTree || predicate(node)) {
const tree: Root = u('root', [node])
return trees.concat(tree)
}
lastTree.children.push(node)
return trees
}, [])
}
type Meta = ReturnType<typeof extractMetaExport>
type Section = {
content: string
heading?: string
slug?: string
}
type ProcessedMdx = {
checksum: string
meta: Meta
sections: Section[]
}
/**
* Processes MDX content for search indexing.
* It extracts metadata, strips it of all JSX,
* and splits it into sub-sections based on criteria.
*/
function processMdxForSearch(content: string): ProcessedMdx {
const checksum = createHash('sha256').update(content).digest('base64')
const mdxTree = fromMarkdown(content, {
extensions: [mdxjs()],
mdastExtensions: [mdxFromMarkdown()],
})
const meta = extractMetaExport(mdxTree)
// Remove all MDX elements from markdown
const mdTree = filter(
mdxTree,
(node) =>
![
'mdxjsEsm',
'mdxJsxFlowElement',
'mdxJsxTextElement',
'mdxFlowExpression',
'mdxTextExpression',
].includes(node.type)
)
if (!mdTree) {
return {
checksum,
meta,
sections: [],
}
}
const sectionTrees = splitTreeBy(mdTree, (node) => node.type === 'heading')
const slugger = new GithubSlugger()
const sections = sectionTrees.map((tree) => {
const [firstNode] = tree.children
const heading = firstNode.type === 'heading' ? toString(firstNode) : undefined
const slug = heading ? slugger.slug(heading) : undefined
return {
content: toMarkdown(tree),
heading,
slug,
}
})
return {
checksum,
meta,
sections,
}
}
type WalkEntry = {
path: string
parentPath?: string
}
async function walk(dir: string, parentPath?: string): Promise<WalkEntry[]> {
const immediateFiles = await readdir(dir)
const recursiveFiles = await Promise.all(
immediateFiles.map(async (file) => {
const path = join(dir, file)
const stats = await stat(path)
if (stats.isDirectory()) {
// Keep track of document hierarchy (if this dir has corresponding doc file)
const docPath = `${basename(path)}.mdx`
return walk(
path,
immediateFiles.includes(docPath) ? join(dirname(path), docPath) : parentPath
)
} else if (stats.isFile()) {
return [
{
path: path,
parentPath,
},
]
} else {
return []
}
})
)
const flattenedFiles = recursiveFiles.reduce(
(all, folderContents) => all.concat(folderContents),
[]
)
return flattenedFiles.sort((a, b) => a.path.localeCompare(b.path))
}
abstract class BaseEmbeddingSource {
checksum?: string
meta?: Meta
sections?: Section[]
constructor(public source: string, public path: string, public parentPath?: string) {}
abstract load(): Promise<{ checksum: string; meta?: Meta; sections: Section[] }>
}
class MarkdownEmbeddingSource extends BaseEmbeddingSource {
type: 'markdown' = 'markdown'
constructor(source: string, public filePath: string, public parentFilePath?: string) {
const path = filePath.replace(/^pages/, '').replace(/\.mdx?$/, '')
const parentPath = parentFilePath?.replace(/^pages/, '').replace(/\.mdx?$/, '')
super(source, path, parentPath)
}
async load() {
const contents = await readFile(this.filePath, 'utf8')
const { checksum, meta, sections } = processMdxForSearch(contents)
this.checksum = checksum
this.meta = meta
this.sections = sections
return {
checksum,
meta,
sections,
}
}
}
abstract class ReferenceEmbeddingSource<SpecSection> extends BaseEmbeddingSource {
type: 'reference' = 'reference'
constructor(
source: string,
path: string,
public meta: Meta,
public specFilePath: string,
public sectionsFilePath: string
) {
super(source, path)
}
async load() {
const specContents = await readFile(this.specFilePath, 'utf8')
const refSectionsContents = await readFile(this.sectionsFilePath, 'utf8')
const refSections: ICommonFunc[] = JSON.parse(refSectionsContents)
const flattenedRefSections = flattenSections(refSections)
const checksum = createHash('sha256')
.update(specContents + refSectionsContents)
.digest('base64')
const specSections = this.getSpecSections(specContents)
const sections = flattenedRefSections
.map((refSection) => {
const specSection = this.matchSpecSection(specSections, refSection.id)
if (!specSection) {
return
}
return {
heading: refSection.title,
slug: refSection.slug,
content: `${this.meta.title} for ${refSection.title}:\n${this.formatSection(
specSection,
refSection
)}`,
}
})
.filter((section) => !!section)
this.checksum = checksum
this.sections = sections
return {
checksum,
sections,
meta: this.meta,
}
}
abstract getSpecSections(specContents: string): SpecSection[]
abstract matchSpecSection(specSections: SpecSection[], id: string): SpecSection
abstract formatSection(specSection: SpecSection, refSection: ICommonFunc): string
}
class OpenApiEmbeddingSource extends ReferenceEmbeddingSource<enrichedOperation> {
getSpecSections(specContents: string): enrichedOperation[] {
const spec: OpenAPIV3.Document<{}> = JSON.parse(specContents)
const generatedSpec = gen_v3(spec, '', {
apiUrl: 'apiv0',
})
return generatedSpec.operations
}
matchSpecSection(operations: enrichedOperation[], id: string): enrichedOperation {
return operations.find((operation) => operation.operationId === id)
}
formatSection(specOperation: enrichedOperation) {
const { summary, description, operation, path, tags } = specOperation
return JSON.stringify({
summary,
description,
operation,
path,
tags,
})
}
}
class ClientLibEmbeddingSource extends ReferenceEmbeddingSource<IFunctionDefinition> {
getSpecSections(specContents: string): IFunctionDefinition[] {
const spec = yaml.load(specContents) as ISpec
return spec.functions
}
matchSpecSection(functionDefinitions: IFunctionDefinition[], id: string): IFunctionDefinition {
return functionDefinitions.find((functionDefinition) => functionDefinition.id === id)
}
formatSection(functionDefinition: IFunctionDefinition, refSection: ICommonFunc): string {
const { title } = refSection
const { description, title: functionName } = functionDefinition
return JSON.stringify({
title,
description,
functionName,
})
}
}
class CliEmbeddingSource extends ReferenceEmbeddingSource<CliCommand> {
getSpecSections(specContents: string): CliCommand[] {
const spec = yaml.load(specContents) as CliSpec
return spec.commands
}
matchSpecSection(cliCommands: CliCommand[], id: string): CliCommand {
return cliCommands.find((cliCommand) => cliCommand.id === id)
}
formatSection(cliCommand: CliCommand): string {
const { summary, description, usage } = cliCommand
return JSON.stringify({
summary,
description,
usage,
})
}
}
type EmbeddingSource =
| MarkdownEmbeddingSource
| OpenApiEmbeddingSource
| ClientLibEmbeddingSource
| CliEmbeddingSource
async function generateEmbeddings() {
// TODO: use better CLI lib like yargs
const args = process.argv.slice(2)
const shouldRefresh = args.includes('--refresh')
if (
!process.env.NEXT_PUBLIC_SUPABASE_URL ||
!process.env.SUPABASE_SERVICE_ROLE_KEY ||
!process.env.OPENAI_KEY
) {
return console.log(
'Environment variables NEXT_PUBLIC_SUPABASE_URL, SUPABASE_SERVICE_ROLE_KEY, and OPENAI_KEY are required: skipping embeddings generation'
const requiredEnvVars = ['NEXT_PUBLIC_SUPABASE_URL', 'SUPABASE_SERVICE_ROLE_KEY', 'OPENAI_KEY']
if (requiredEnvVars.some((name) => !process.env[name])) {
throw new Error(
`Environment variables ${requiredEnvVars.join(
', '
)} are required: skipping embeddings generation`
)
}
@@ -420,54 +33,7 @@ async function generateEmbeddings() {
}
)
const embeddingSources: EmbeddingSource[] = [
new OpenApiEmbeddingSource(
'api',
'/reference/api',
{ title: 'Management API Reference' },
'../../spec/transforms/api_v0_openapi_deparsed.json',
'../../spec/common-api-sections.json'
),
new ClientLibEmbeddingSource(
'js-lib',
'/reference/javascript',
{ title: 'JavaScript Reference' },
'../../spec/supabase_js_v2.yml',
'../../spec/common-client-libs-sections.json'
),
new ClientLibEmbeddingSource(
'dart-lib',
'/reference/dart',
{ title: 'Dart Reference' },
'../../spec/supabase_dart_v1.yml',
'../../spec/common-client-libs-sections.json'
),
new ClientLibEmbeddingSource(
'python-lib',
'/reference/python',
{ title: 'Python Reference' },
'../../spec/supabase_py_v2.yml',
'../../spec/common-client-libs-sections.json'
),
new ClientLibEmbeddingSource(
'csharp-lib',
'/reference/csharp',
{ title: 'C# Reference' },
'../../spec/supabase_csharp_v0.yml',
'../../spec/common-client-libs-sections.json'
),
new CliEmbeddingSource(
'cli',
'/reference/cli',
{ title: 'CLI Reference' },
'../../spec/cli_v1_commands.yaml',
'../../spec/common-cli-sections.json'
),
...(await walk('pages'))
.filter(({ path }) => /\.mdx?$/.test(path))
.filter(({ path }) => !ignoredFiles.includes(path))
.map((entry) => new MarkdownEmbeddingSource('guide', entry.path)),
]
const embeddingSources = await fetchSources()
console.log(`Discovered ${embeddingSources.length} pages`)
+20
View File
@@ -0,0 +1,20 @@
export type Json = Record<
string,
string | number | boolean | null | Json[] | { [key: string]: Json }
>
export type Section = {
content: string
heading?: string
slug?: string
}
export abstract class BaseSource {
checksum?: string
meta?: Json
sections?: Section[]
constructor(public source: string, public path: string, public parentPath?: string) {}
abstract load(): Promise<{ checksum: string; meta?: Json; sections: Section[] }>
}
+85
View File
@@ -0,0 +1,85 @@
import { MarkdownSource } from './markdown'
import {
CliReferenceSource,
ClientLibReferenceSource,
OpenApiReferenceSource,
} from './reference-doc'
import { walk } from './util'
const ignoredFiles = ['pages/404.mdx']
export type SearchSource =
| MarkdownSource
| OpenApiReferenceSource
| ClientLibReferenceSource
| CliReferenceSource
/**
* Fetches all the sources we want to index for search
*/
export async function fetchSources() {
const openApiReferenceSource = new OpenApiReferenceSource(
'api',
'/reference/api',
{ title: 'Management API Reference' },
'../../spec/transforms/api_v0_openapi_deparsed.json',
'../../spec/common-api-sections.json'
)
const jsLibReferenceSource = new ClientLibReferenceSource(
'js-lib',
'/reference/javascript',
{ title: 'JavaScript Reference' },
'../../spec/supabase_js_v2.yml',
'../../spec/common-client-libs-sections.json'
)
const dartLibReferenceSource = new ClientLibReferenceSource(
'dart-lib',
'/reference/dart',
{ title: 'Dart Reference' },
'../../spec/supabase_dart_v1.yml',
'../../spec/common-client-libs-sections.json'
)
const pythonLibReferenceSource = new ClientLibReferenceSource(
'python-lib',
'/reference/python',
{ title: 'Python Reference' },
'../../spec/supabase_py_v2.yml',
'../../spec/common-client-libs-sections.json'
)
const cSharpLibReferenceSource = new ClientLibReferenceSource(
'csharp-lib',
'/reference/csharp',
{ title: 'C# Reference' },
'../../spec/supabase_csharp_v0.yml',
'../../spec/common-client-libs-sections.json'
)
const cliReferenceSource = new CliReferenceSource(
'cli',
'/reference/cli',
{ title: 'CLI Reference' },
'../../spec/cli_v1_commands.yaml',
'../../spec/common-cli-sections.json'
)
const guideSources = (await walk('pages'))
.filter(({ path }) => /\.mdx?$/.test(path))
.filter(({ path }) => !ignoredFiles.includes(path))
.map((entry) => new MarkdownSource('guide', entry.path))
const sources: SearchSource[] = [
openApiReferenceSource,
jsLibReferenceSource,
dartLibReferenceSource,
pythonLibReferenceSource,
cSharpLibReferenceSource,
cliReferenceSource,
...guideSources,
]
return sources
}
@@ -0,0 +1,191 @@
import { createHash } from 'crypto'
import { ObjectExpression } from 'estree'
import { readFile } from 'fs/promises'
import GithubSlugger from 'github-slugger'
import { Content, Root } from 'mdast'
import { fromMarkdown } from 'mdast-util-from-markdown'
import { MdxjsEsm, mdxFromMarkdown } from 'mdast-util-mdx'
import { toMarkdown } from 'mdast-util-to-markdown'
import { toString } from 'mdast-util-to-string'
import { mdxjs } from 'micromark-extension-mdxjs'
import { u } from 'unist-builder'
import { filter } from 'unist-util-filter'
import { BaseSource, Json, Section } from './base'
/**
* Extracts ES literals from an `estree` `ObjectExpression`
* into a plain JavaScript object.
*/
export function getObjectFromExpression(node: ObjectExpression) {
return node.properties.reduce<
Record<string, string | number | bigint | true | RegExp | undefined>
>((object, property) => {
if (property.type !== 'Property') {
return object
}
const key = (property.key.type === 'Identifier' && property.key.name) || undefined
const value = (property.value.type === 'Literal' && property.value.value) || undefined
if (!key) {
return object
}
return {
...object,
[key]: value,
}
}, {})
}
/**
* Extracts the `meta` ESM export from the MDX file.
*
* This info is akin to frontmatter.
*/
export function extractMetaExport(mdxTree: Root) {
const metaExportNode = mdxTree.children.find((node): node is MdxjsEsm => {
return (
node.type === 'mdxjsEsm' &&
node.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' &&
node.data.estree.body[0].declaration?.type === 'VariableDeclaration' &&
node.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' &&
node.data.estree.body[0].declaration.declarations[0].id.name === 'meta'
)
})
if (!metaExportNode) {
return undefined
}
const objectExpression =
(metaExportNode.data?.estree?.body[0]?.type === 'ExportNamedDeclaration' &&
metaExportNode.data.estree.body[0].declaration?.type === 'VariableDeclaration' &&
metaExportNode.data.estree.body[0].declaration.declarations[0]?.id.type === 'Identifier' &&
metaExportNode.data.estree.body[0].declaration.declarations[0].id.name === 'meta' &&
metaExportNode.data.estree.body[0].declaration.declarations[0].init?.type ===
'ObjectExpression' &&
metaExportNode.data.estree.body[0].declaration.declarations[0].init) ||
undefined
if (!objectExpression) {
return undefined
}
return getObjectFromExpression(objectExpression)
}
/**
* Splits a `mdast` tree into multiple trees based on
* a predicate function. Will include the splitting node
* at the beginning of each tree.
*
* Useful to split a markdown file into smaller sections.
*/
export function splitTreeBy(tree: Root, predicate: (node: Content) => boolean) {
return tree.children.reduce<Root[]>((trees, node) => {
const [lastTree] = trees.slice(-1)
if (!lastTree || predicate(node)) {
const tree: Root = u('root', [node])
return trees.concat(tree)
}
lastTree.children.push(node)
return trees
}, [])
}
/**
* Processes MDX content for search indexing.
* It extracts metadata, strips it of all JSX,
* and splits it into sub-sections based on criteria.
*/
export function processMdxForSearch(content: string): ProcessedMdx {
const checksum = createHash('sha256').update(content).digest('base64')
const mdxTree = fromMarkdown(content, {
extensions: [mdxjs()],
mdastExtensions: [mdxFromMarkdown()],
})
const meta = extractMetaExport(mdxTree)
const serializableMeta: Json = JSON.parse(JSON.stringify(meta))
// Remove all MDX elements from markdown
const mdTree = filter(
mdxTree,
(node) =>
![
'mdxjsEsm',
'mdxJsxFlowElement',
'mdxJsxTextElement',
'mdxFlowExpression',
'mdxTextExpression',
].includes(node.type)
)
if (!mdTree) {
return {
checksum,
meta: serializableMeta,
sections: [],
}
}
const sectionTrees = splitTreeBy(mdTree, (node) => node.type === 'heading')
const slugger = new GithubSlugger()
const sections = sectionTrees.map((tree) => {
const [firstNode] = tree.children
const heading = firstNode.type === 'heading' ? toString(firstNode) : undefined
const slug = heading ? slugger.slug(heading) : undefined
return {
content: toMarkdown(tree),
heading,
slug,
}
})
return {
checksum,
meta: serializableMeta,
sections,
}
}
export type ProcessedMdx = {
checksum: string
meta: Json
sections: Section[]
}
export class MarkdownSource extends BaseSource {
type = 'markdown' as const
constructor(source: string, public filePath: string, public parentFilePath?: string) {
const path = filePath.replace(/^pages/, '').replace(/\.mdx?$/, '')
const parentPath = parentFilePath?.replace(/^pages/, '').replace(/\.mdx?$/, '')
super(source, path, parentPath)
}
async load() {
const contents = await readFile(this.filePath, 'utf8')
const { checksum, meta, sections } = processMdxForSearch(contents)
this.checksum = checksum
this.meta = meta
this.sections = sections
return {
checksum,
meta,
sections,
}
}
}
@@ -0,0 +1,139 @@
import { createHash } from 'crypto'
import { readFile } from 'fs/promises'
import yaml from 'js-yaml'
import { OpenAPIV3 } from 'openapi-types'
import { Meta } from '..'
import {
ICommonFunc,
IFunctionDefinition,
ISpec,
} from '../../../components/reference/Reference.types'
import { CliCommand, CliSpec } from '../../../generator/types/CliSpec'
import { flattenSections } from '../../../lib/helpers'
import { enrichedOperation, gen_v3 } from '../../../lib/refGenerator/helpers'
import { BaseSource } from './base'
export abstract class ReferenceSource<SpecSection> extends BaseSource {
type = 'reference' as const
constructor(
source: string,
path: string,
public meta: Meta,
public specFilePath: string,
public sectionsFilePath: string
) {
super(source, path)
}
async load() {
const specContents = await readFile(this.specFilePath, 'utf8')
const refSectionsContents = await readFile(this.sectionsFilePath, 'utf8')
const refSections: ICommonFunc[] = JSON.parse(refSectionsContents)
const flattenedRefSections = flattenSections(refSections)
const checksum = createHash('sha256')
.update(specContents + refSectionsContents)
.digest('base64')
const specSections = this.getSpecSections(specContents)
const sections = flattenedRefSections
.map((refSection) => {
const specSection = this.matchSpecSection(specSections, refSection.id)
if (!specSection) {
return
}
return {
heading: refSection.title,
slug: refSection.slug,
content: `${this.meta.title} for ${refSection.title}:\n${this.formatSection(
specSection,
refSection
)}`,
}
})
.filter((section) => !!section)
this.checksum = checksum
this.sections = sections
return {
checksum,
sections,
meta: this.meta,
}
}
abstract getSpecSections(specContents: string): SpecSection[]
abstract matchSpecSection(specSections: SpecSection[], id: string): SpecSection
abstract formatSection(specSection: SpecSection, refSection: ICommonFunc): string
}
export class OpenApiReferenceSource extends ReferenceSource<enrichedOperation> {
getSpecSections(specContents: string): enrichedOperation[] {
const spec: OpenAPIV3.Document<{}> = JSON.parse(specContents)
const generatedSpec = gen_v3(spec, '', {
apiUrl: 'apiv0',
})
return generatedSpec.operations
}
matchSpecSection(operations: enrichedOperation[], id: string): enrichedOperation {
return operations.find((operation) => operation.operationId === id)
}
formatSection(specOperation: enrichedOperation) {
const { summary, description, operation, path, tags } = specOperation
return JSON.stringify({
summary,
description,
operation,
path,
tags,
})
}
}
export class ClientLibReferenceSource extends ReferenceSource<IFunctionDefinition> {
getSpecSections(specContents: string): IFunctionDefinition[] {
const spec = yaml.load(specContents) as ISpec
return spec.functions
}
matchSpecSection(functionDefinitions: IFunctionDefinition[], id: string): IFunctionDefinition {
return functionDefinitions.find((functionDefinition) => functionDefinition.id === id)
}
formatSection(functionDefinition: IFunctionDefinition, refSection: ICommonFunc): string {
const { title } = refSection
const { description, title: functionName } = functionDefinition
return JSON.stringify({
title,
description,
functionName,
})
}
}
export class CliReferenceSource extends ReferenceSource<CliCommand> {
getSpecSections(specContents: string): CliCommand[] {
const spec = yaml.load(specContents) as CliSpec
return spec.commands
}
matchSpecSection(cliCommands: CliCommand[], id: string): CliCommand {
return cliCommands.find((cliCommand) => cliCommand.id === id)
}
formatSection(cliCommand: CliCommand): string {
const { summary, description, usage } = cliCommand
return JSON.stringify({
summary,
description,
usage,
})
}
}
+43
View File
@@ -0,0 +1,43 @@
import { readdir, stat } from 'fs/promises'
import { basename, dirname, join } from 'path'
export type WalkEntry = {
path: string
parentPath?: string
}
export async function walk(dir: string, parentPath?: string): Promise<WalkEntry[]> {
const immediateFiles = await readdir(dir)
const recursiveFiles = await Promise.all(
immediateFiles.map(async (file) => {
const path = join(dir, file)
const stats = await stat(path)
if (stats.isDirectory()) {
// Keep track of document hierarchy (if this dir has corresponding doc file)
const docPath = `${basename(path)}.mdx`
return walk(
path,
immediateFiles.includes(docPath) ? join(dirname(path), docPath) : parentPath
)
} else if (stats.isFile()) {
return [
{
path: path,
parentPath,
},
]
} else {
return []
}
})
)
const flattenedFiles = recursiveFiles.reduce(
(all, folderContents) => all.concat(folderContents),
[]
)
return flattenedFiles.sort((a, b) => a.path.localeCompare(b.path))
}