Files
supabase/apps/docs/scripts/search_v2/markdown.ts
T
Jeremias Menichelli 7f1df457f8 feat: Create action to upsert table for new search (#50683)
## Problem

For the new search, we will scan all files present within the
public/markdown directory, extract text nodes and upsert a new Supabase
table in a new project to do FTS type search.

## Solution

In this PR:
- A new script directory is created with files to solve all the steps
described above.
 - Unit tests added for the fundamental bits of the script.
- A new workflow file is added so the action runs after push every time
content is altered, added or removed.

<!--
## Preview links

If relevant, include links to changed pages for easy review access.

Copy the preview base URL from the Vercel bot comment on this PR. Use
the following table as an example template.

| Site | Live | Preview | Search for |
| -------------- |
-------------------------------------------------------------------------
|
------------------------------------------------------------------------------------------------------------
| ----------------------------- |
| WWW | [/blog/your-post](https://supabase.com/blog/your-post) |
[/blog/your-post](https://zone-www-dot-com-git-branch-name-supabase.vercel.app/blog/your-post)
| unique phrase from the change |
| Docs |
[/docs/guides/your-page](https://supabase.com/docs/guides/your-page) |
[/docs/guides/your-page](https://docs-git-branch-name-supabase.vercel.app/docs/guides/your-page)
| unique phrase from the change |
| Studio | [/dashboard](https://supabase.com/dashboard) |
[/dashboard](https://studio-git-branch-name-supabase.vercel.app/dashboard)
| unique phrase from the change |
| Design system | [/design-system](https://supabase.com/design-system) |
[/design-system](https://design-system-git-branch-name-supabase.vercel.app/design-system)
| unique phrase from the change |
| UI library | [/library](https://supabase.com/library) |
[/library](https://ui-library-git-branch-name-supabase.vercel.app/library)
| unique phrase from the change |
| Knowledge base |
[/kb/guides/your-page](https://supabase.com/kb/guides/your-page) |
[/kb/guides/your-page](https://kb-git-branch-name-supabase.vercel.app/kb/guides/your-page)
| unique phrase from the change |
-->

<!-- ## Additional context

Optionally add any other context or screenshots.

-->

## Review instructions

Sadly, testing this work is quite complex, but in case someone wants to:

1. Create a new Supabase project ton your personal space
1. Copy the id of the project and a secret key and add it to the new
Search V2 environment variables as shown in the example file
1. Copy the content of the newly added `setup.sql` and run it on the SQL
editor of your project.
1. Fetch this branch and on the search directory, run `search-v2:ingest`
1. Your project's table should have rows corresponding to the content
from docs


## Checklist

Check all before review:

- [x] I have read
[CONTRIBUTING.md](https://github.com/supabase/supabase/blob/master/CONTRIBUTING.md)
- [x] If I wrote a new docs topic or edited an existing topic, I used
the `/write-the-docs` or `/edit-the-docs` skill, which references
[WORD_LIST](https://github.com/supabase/supabase/blob/master/apps/docs/WORD_LIST.md)
and the docs
[CONTRIBUTING](https://github.com/supabase/supabase/blob/master/apps/docs/CONTRIBUTING.md)
guide


<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

- **New Features**
- Documentation pages are now indexed for full-text search at the page
and section level.
- Search results can show the most relevant section from each page, with
its title, heading, excerpt, and relevance score.
- Search content is automatically refreshed when published Markdown
documentation changes.
- **Tests**
- Added coverage for Markdown parsing, page structure, routing, and
search-content generation.
<!-- end of auto-generated comment: release notes by coderabbit.ai -->
2026-09-28 16:34:46 -03:00

140 lines
4.4 KiB
TypeScript

import { toString } from 'mdast-util-to-string'
import remarkFrontmatter from 'remark-frontmatter'
import remarkGfm from 'remark-gfm'
import remarkParse from 'remark-parse'
import { unified } from 'unified'
const processor = unified().use(remarkParse).use(remarkFrontmatter, ['yaml']).use(remarkGfm)
/**
* Types are derived from the processor's own output rather than imported from
* `mdast` directly: this app pins `@types/mdast` v3 for its wider MDX pipeline,
* while `remark-parse` v11 resolves its own (structurally incompatible) v4
* types for its nested `@types/mdast` dependency.
*/
export type Root = ReturnType<typeof processor.parse>
/** Any child of the root node (paragraph, heading, list, table, ...). */
type Content = Root['children'][number]
type Heading = Extract<Content, { type: 'heading' }>
/** Union of every node type that can appear in the tree (root or content). */
type Nodes = Root | Content
export interface Section {
/** Text of this section's heading ('' for text that appears before any heading). */
heading: string
/** 1..6, or 0 for text before any heading. */
level: number
/** Headings from the H1 down to this section, e.g. ['Users', 'Permanent and anonymous users']. */
headingPath: string[]
/** Plain text of everything under the heading until the next heading. */
content: string
}
export interface ParsedPage {
title: string
excerpt: string
sections: Section[]
}
/** Parse markdown into an mdast tree (GFM tables/lists + YAML frontmatter supported). */
export function parseMarkdownAst(markdown: string): Root {
return processor.parse(markdown)
}
/**
* Convert a node to plain text. Like mdast-util-to-string but keeps words apart
* for block containers (list items on their own line, table cells separated by " | ").
*/
export function nodeToText(node: Nodes): string {
switch (node.type) {
case 'tableRow':
return node.children.map((cell) => toString(cell).trim()).join(' | ')
case 'table':
case 'list':
case 'listItem':
case 'blockquote':
return (node.children as Nodes[]).map(nodeToText).filter(Boolean).join('\n')
default:
return toString(node).trim()
}
}
/** The first H1 of the document, or '' if there is none. */
export function extractTitle(tree: Root): string {
const h1 = tree.children.find(
(node): node is Heading => node.type === 'heading' && node.depth === 1
)
return h1 ? nodeToText(h1) : ''
}
/** The first paragraph of the document (used as the page excerpt), or ''. */
export function extractExcerpt(tree: Root): string {
const paragraph = tree.children.find((node) => node.type === 'paragraph')
return paragraph ? nodeToText(paragraph) : ''
}
/** Nodes that carry no searchable text. */
function isIgnored(node: Content): boolean {
return (
node.type === 'yaml' ||
node.type === 'thematicBreak' ||
node.type === 'html' ||
node.type === 'definition'
)
}
/**
* Split the document into sections: one per heading, each with the text below it
* (until the next heading) and the path of ancestor headings leading to it.
*/
export function extractSections(tree: Root): Section[] {
const sections: Section[] = []
const stack: Array<{ text: string; level: number }> = []
let current: Section | null = null
let body: string[] = []
const flush = () => {
if (current) {
current.content = body.join('\n\n').trim()
if (current.heading || current.content) sections.push(current)
}
body = []
}
for (const node of tree.children) {
if (isIgnored(node)) continue
if (node.type === 'heading') {
flush()
while (stack.length && stack[stack.length - 1].level >= node.depth) stack.pop()
const text = nodeToText(node)
stack.push({ text, level: node.depth })
current = {
heading: text,
level: node.depth,
headingPath: stack.map((s) => s.text),
content: '',
}
continue
}
// Body text before the first heading gets a headless section.
if (!current) current = { heading: '', level: 0, headingPath: [], content: '' }
const text = nodeToText(node)
if (text) body.push(text)
}
flush()
return sections
}
/** Everything the ingester needs from one markdown file. */
export function parsePage(markdown: string): ParsedPage {
const tree = parseMarkdownAst(markdown)
return {
title: extractTitle(tree),
excerpt: extractExcerpt(tree),
sections: extractSections(tree),
}
}