mirror of
https://github.com/supabase/supabase.git
synced 2026-10-05 09:25:06 +03:00
## Problem For the new search, we will scan all files present within the public/markdown directory, extract text nodes and upsert a new Supabase table in a new project to do FTS type search. ## Solution In this PR: - A new script directory is created with files to solve all the steps described above. - Unit tests added for the fundamental bits of the script. - A new workflow file is added so the action runs after push every time content is altered, added or removed. <!-- ## Preview links If relevant, include links to changed pages for easy review access. Copy the preview base URL from the Vercel bot comment on this PR. Use the following table as an example template. | Site | Live | Preview | Search for | | -------------- | ------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------ | ----------------------------- | | WWW | [/blog/your-post](https://supabase.com/blog/your-post) | [/blog/your-post](https://zone-www-dot-com-git-branch-name-supabase.vercel.app/blog/your-post) | unique phrase from the change | | Docs | [/docs/guides/your-page](https://supabase.com/docs/guides/your-page) | [/docs/guides/your-page](https://docs-git-branch-name-supabase.vercel.app/docs/guides/your-page) | unique phrase from the change | | Studio | [/dashboard](https://supabase.com/dashboard) | [/dashboard](https://studio-git-branch-name-supabase.vercel.app/dashboard) | unique phrase from the change | | Design system | [/design-system](https://supabase.com/design-system) | [/design-system](https://design-system-git-branch-name-supabase.vercel.app/design-system) | unique phrase from the change | | UI library | [/library](https://supabase.com/library) | [/library](https://ui-library-git-branch-name-supabase.vercel.app/library) | unique phrase from the change | | Knowledge base | [/kb/guides/your-page](https://supabase.com/kb/guides/your-page) | [/kb/guides/your-page](https://kb-git-branch-name-supabase.vercel.app/kb/guides/your-page) | unique phrase from the change | --> <!-- ## Additional context Optionally add any other context or screenshots. --> ## Review instructions Sadly, testing this work is quite complex, but in case someone wants to: 1. Create a new Supabase project ton your personal space 1. Copy the id of the project and a secret key and add it to the new Search V2 environment variables as shown in the example file 1. Copy the content of the newly added `setup.sql` and run it on the SQL editor of your project. 1. Fetch this branch and on the search directory, run `search-v2:ingest` 1. Your project's table should have rows corresponding to the content from docs ## Checklist Check all before review: - [x] I have read [CONTRIBUTING.md](https://github.com/supabase/supabase/blob/master/CONTRIBUTING.md) - [x] If I wrote a new docs topic or edited an existing topic, I used the `/write-the-docs` or `/edit-the-docs` skill, which references [WORD_LIST](https://github.com/supabase/supabase/blob/master/apps/docs/WORD_LIST.md) and the docs [CONTRIBUTING](https://github.com/supabase/supabase/blob/master/apps/docs/CONTRIBUTING.md) guide <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - **New Features** - Documentation pages are now indexed for full-text search at the page and section level. - Search results can show the most relevant section from each page, with its title, heading, excerpt, and relevance score. - Search content is automatically refreshed when published Markdown documentation changes. - **Tests** - Added coverage for Markdown parsing, page structure, routing, and search-content generation. <!-- end of auto-generated comment: release notes by coderabbit.ai -->
140 lines
4.4 KiB
TypeScript
140 lines
4.4 KiB
TypeScript
import { toString } from 'mdast-util-to-string'
|
|
import remarkFrontmatter from 'remark-frontmatter'
|
|
import remarkGfm from 'remark-gfm'
|
|
import remarkParse from 'remark-parse'
|
|
import { unified } from 'unified'
|
|
|
|
const processor = unified().use(remarkParse).use(remarkFrontmatter, ['yaml']).use(remarkGfm)
|
|
|
|
/**
|
|
* Types are derived from the processor's own output rather than imported from
|
|
* `mdast` directly: this app pins `@types/mdast` v3 for its wider MDX pipeline,
|
|
* while `remark-parse` v11 resolves its own (structurally incompatible) v4
|
|
* types for its nested `@types/mdast` dependency.
|
|
*/
|
|
export type Root = ReturnType<typeof processor.parse>
|
|
/** Any child of the root node (paragraph, heading, list, table, ...). */
|
|
type Content = Root['children'][number]
|
|
type Heading = Extract<Content, { type: 'heading' }>
|
|
/** Union of every node type that can appear in the tree (root or content). */
|
|
type Nodes = Root | Content
|
|
|
|
export interface Section {
|
|
/** Text of this section's heading ('' for text that appears before any heading). */
|
|
heading: string
|
|
/** 1..6, or 0 for text before any heading. */
|
|
level: number
|
|
/** Headings from the H1 down to this section, e.g. ['Users', 'Permanent and anonymous users']. */
|
|
headingPath: string[]
|
|
/** Plain text of everything under the heading until the next heading. */
|
|
content: string
|
|
}
|
|
|
|
export interface ParsedPage {
|
|
title: string
|
|
excerpt: string
|
|
sections: Section[]
|
|
}
|
|
|
|
/** Parse markdown into an mdast tree (GFM tables/lists + YAML frontmatter supported). */
|
|
export function parseMarkdownAst(markdown: string): Root {
|
|
return processor.parse(markdown)
|
|
}
|
|
|
|
/**
|
|
* Convert a node to plain text. Like mdast-util-to-string but keeps words apart
|
|
* for block containers (list items on their own line, table cells separated by " | ").
|
|
*/
|
|
export function nodeToText(node: Nodes): string {
|
|
switch (node.type) {
|
|
case 'tableRow':
|
|
return node.children.map((cell) => toString(cell).trim()).join(' | ')
|
|
case 'table':
|
|
case 'list':
|
|
case 'listItem':
|
|
case 'blockquote':
|
|
return (node.children as Nodes[]).map(nodeToText).filter(Boolean).join('\n')
|
|
default:
|
|
return toString(node).trim()
|
|
}
|
|
}
|
|
|
|
/** The first H1 of the document, or '' if there is none. */
|
|
export function extractTitle(tree: Root): string {
|
|
const h1 = tree.children.find(
|
|
(node): node is Heading => node.type === 'heading' && node.depth === 1
|
|
)
|
|
return h1 ? nodeToText(h1) : ''
|
|
}
|
|
|
|
/** The first paragraph of the document (used as the page excerpt), or ''. */
|
|
export function extractExcerpt(tree: Root): string {
|
|
const paragraph = tree.children.find((node) => node.type === 'paragraph')
|
|
return paragraph ? nodeToText(paragraph) : ''
|
|
}
|
|
|
|
/** Nodes that carry no searchable text. */
|
|
function isIgnored(node: Content): boolean {
|
|
return (
|
|
node.type === 'yaml' ||
|
|
node.type === 'thematicBreak' ||
|
|
node.type === 'html' ||
|
|
node.type === 'definition'
|
|
)
|
|
}
|
|
|
|
/**
|
|
* Split the document into sections: one per heading, each with the text below it
|
|
* (until the next heading) and the path of ancestor headings leading to it.
|
|
*/
|
|
export function extractSections(tree: Root): Section[] {
|
|
const sections: Section[] = []
|
|
const stack: Array<{ text: string; level: number }> = []
|
|
let current: Section | null = null
|
|
let body: string[] = []
|
|
|
|
const flush = () => {
|
|
if (current) {
|
|
current.content = body.join('\n\n').trim()
|
|
if (current.heading || current.content) sections.push(current)
|
|
}
|
|
body = []
|
|
}
|
|
|
|
for (const node of tree.children) {
|
|
if (isIgnored(node)) continue
|
|
|
|
if (node.type === 'heading') {
|
|
flush()
|
|
while (stack.length && stack[stack.length - 1].level >= node.depth) stack.pop()
|
|
const text = nodeToText(node)
|
|
stack.push({ text, level: node.depth })
|
|
current = {
|
|
heading: text,
|
|
level: node.depth,
|
|
headingPath: stack.map((s) => s.text),
|
|
content: '',
|
|
}
|
|
continue
|
|
}
|
|
|
|
// Body text before the first heading gets a headless section.
|
|
if (!current) current = { heading: '', level: 0, headingPath: [], content: '' }
|
|
const text = nodeToText(node)
|
|
if (text) body.push(text)
|
|
}
|
|
|
|
flush()
|
|
return sections
|
|
}
|
|
|
|
/** Everything the ingester needs from one markdown file. */
|
|
export function parsePage(markdown: string): ParsedPage {
|
|
const tree = parseMarkdownAst(markdown)
|
|
return {
|
|
title: extractTitle(tree),
|
|
excerpt: extractExcerpt(tree),
|
|
sections: extractSections(tree),
|
|
}
|
|
}
|