Files
supabase/apps/docs/scripts/search_v2/sql/setup.sql
T
Jeremias Menichelli 7f1df457f8 feat: Create action to upsert table for new search (#50683)
## Problem

For the new search, we will scan all files present within the
public/markdown directory, extract text nodes and upsert a new Supabase
table in a new project to do FTS type search.

## Solution

In this PR:
- A new script directory is created with files to solve all the steps
described above.
 - Unit tests added for the fundamental bits of the script.
- A new workflow file is added so the action runs after push every time
content is altered, added or removed.

<!--
## Preview links

If relevant, include links to changed pages for easy review access.

Copy the preview base URL from the Vercel bot comment on this PR. Use
the following table as an example template.

| Site | Live | Preview | Search for |
| -------------- |
-------------------------------------------------------------------------
|
------------------------------------------------------------------------------------------------------------
| ----------------------------- |
| WWW | [/blog/your-post](https://supabase.com/blog/your-post) |
[/blog/your-post](https://zone-www-dot-com-git-branch-name-supabase.vercel.app/blog/your-post)
| unique phrase from the change |
| Docs |
[/docs/guides/your-page](https://supabase.com/docs/guides/your-page) |
[/docs/guides/your-page](https://docs-git-branch-name-supabase.vercel.app/docs/guides/your-page)
| unique phrase from the change |
| Studio | [/dashboard](https://supabase.com/dashboard) |
[/dashboard](https://studio-git-branch-name-supabase.vercel.app/dashboard)
| unique phrase from the change |
| Design system | [/design-system](https://supabase.com/design-system) |
[/design-system](https://design-system-git-branch-name-supabase.vercel.app/design-system)
| unique phrase from the change |
| UI library | [/library](https://supabase.com/library) |
[/library](https://ui-library-git-branch-name-supabase.vercel.app/library)
| unique phrase from the change |
| Knowledge base |
[/kb/guides/your-page](https://supabase.com/kb/guides/your-page) |
[/kb/guides/your-page](https://kb-git-branch-name-supabase.vercel.app/kb/guides/your-page)
| unique phrase from the change |
-->

<!-- ## Additional context

Optionally add any other context or screenshots.

-->

## Review instructions

Sadly, testing this work is quite complex, but in case someone wants to:

1. Create a new Supabase project ton your personal space
1. Copy the id of the project and a secret key and add it to the new
Search V2 environment variables as shown in the example file
1. Copy the content of the newly added `setup.sql` and run it on the SQL
editor of your project.
1. Fetch this branch and on the search directory, run `search-v2:ingest`
1. Your project's table should have rows corresponding to the content
from docs


## Checklist

Check all before review:

- [x] I have read
[CONTRIBUTING.md](https://github.com/supabase/supabase/blob/master/CONTRIBUTING.md)
- [x] If I wrote a new docs topic or edited an existing topic, I used
the `/write-the-docs` or `/edit-the-docs` skill, which references
[WORD_LIST](https://github.com/supabase/supabase/blob/master/apps/docs/WORD_LIST.md)
and the docs
[CONTRIBUTING](https://github.com/supabase/supabase/blob/master/apps/docs/CONTRIBUTING.md)
guide


<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

- **New Features**
- Documentation pages are now indexed for full-text search at the page
and section level.
- Search results can show the most relevant section from each page, with
its title, heading, excerpt, and relevance score.
- Search content is automatically refreshed when published Markdown
documentation changes.
- **Tests**
- Added coverage for Markdown parsing, page structure, routing, and
search-content generation.
<!-- end of auto-generated comment: release notes by coderabbit.ai -->
2026-09-28 16:34:46 -03:00

93 lines
3.8 KiB
PL/PgSQL

-- Run this in the Supabase SQL editor before `pnpm search-v2:ingest`.
-- Safe to re-run: it uses IF NOT EXISTS / CREATE OR REPLACE.
-- One row per markdown *section* (a heading + the text below it, up to the next heading).
create table if not exists public.docs_sections (
id bigint generated always as identity primary key,
slug text not null, -- 'guides/auth/users' (route without base path)
file_path text not null, -- path relative to the content root
page_title text not null default '', -- the page's H1
heading text not null default '', -- this section's heading
heading_level smallint not null default 0, -- 1..6, 0 = text before any heading
heading_path text[] not null default '{}',-- ['Users', 'Permanent and anonymous users']
content text not null default '', -- plain text of the section body
excerpt text not null default '', -- first paragraph of the page
updated_at timestamptz not null default now(),
-- Weighted search document, kept up to date by Postgres automatically.
-- A = H1 (page title)
-- B = H2
-- C = H3
-- D = H4+ headings and all body text
-- ts_rank_cd() weights these A > B > C > D by default ({0.1, 0.2, 0.4, 1.0} for D, C, B, A).
fts tsvector generated always as (
setweight(to_tsvector('english', coalesce(page_title, '')), 'A')
|| setweight(
to_tsvector('english', coalesce(heading, '')),
(case heading_level when 1 then 'A' when 2 then 'B' when 3 then 'C' else 'D' end)::"char"
)
|| setweight(to_tsvector('english', coalesce(content, '')), 'D')
) stored
);
create index if not exists docs_sections_fts_idx on public.docs_sections using gin (fts);
create index if not exists docs_sections_slug_idx on public.docs_sections (slug);
-- Read-only access for anon/authenticated (the ingest script writes with the secret key, which bypasses RLS).
alter table public.docs_sections enable row level security;
drop policy if exists "docs_sections are publicly readable" on public.docs_sections;
create policy "docs_sections are publicly readable"
on public.docs_sections for select
to anon, authenticated
using (true);
-- Search RPC: called from the server as supabase.rpc('search_docs', { query_text, match_limit }).
-- Score is ts_rank_cd normalised to 0..1 (normalization flag 32 => rank / (rank + 1)) and scaled to an
-- integer 0..1000 so the API can return a plain integer.
drop function if exists public.search_docs(text, int);
create or replace function public.search_docs(query_text text, match_limit int default 10)
returns table (
slug text,
page_title text,
heading text,
heading_level smallint,
heading_path text[],
excerpt text,
score int
)
language sql
stable
as $$
with q as (
select websearch_to_tsquery('english', query_text) as tsq
),
ranked as (
select
d.slug,
d.page_title,
d.heading,
d.heading_level,
d.heading_path,
d.excerpt,
round(ts_rank_cd(d.fts, q.tsq, 32) * 1000)::int as score
from public.docs_sections d, q
where d.fts @@ q.tsq
),
-- A page can have several matching sections; keep only its best-scoring one so the
-- same route doesn't take up multiple slots in the results.
best_per_page as (
select distinct on (r.slug) r.*
from ranked r
order by r.slug, r.score desc, r.heading_level asc
)
select b.slug, b.page_title, b.heading, b.heading_level, b.heading_path, b.excerpt, b.score
from best_per_page b
order by b.score desc, b.heading_level asc, b.slug asc
limit greatest(1, least(coalesce(match_limit, 10), 100));
$$;
grant execute on function public.search_docs(text, int) to anon, authenticated, service_role;