mirror of
https://github.com/supabase/supabase.git
synced 2026-10-06 01:45:10 +03:00
## Problem For the new search, we will scan all files present within the public/markdown directory, extract text nodes and upsert a new Supabase table in a new project to do FTS type search. ## Solution In this PR: - A new script directory is created with files to solve all the steps described above. - Unit tests added for the fundamental bits of the script. - A new workflow file is added so the action runs after push every time content is altered, added or removed. <!-- ## Preview links If relevant, include links to changed pages for easy review access. Copy the preview base URL from the Vercel bot comment on this PR. Use the following table as an example template. | Site | Live | Preview | Search for | | -------------- | ------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------ | ----------------------------- | | WWW | [/blog/your-post](https://supabase.com/blog/your-post) | [/blog/your-post](https://zone-www-dot-com-git-branch-name-supabase.vercel.app/blog/your-post) | unique phrase from the change | | Docs | [/docs/guides/your-page](https://supabase.com/docs/guides/your-page) | [/docs/guides/your-page](https://docs-git-branch-name-supabase.vercel.app/docs/guides/your-page) | unique phrase from the change | | Studio | [/dashboard](https://supabase.com/dashboard) | [/dashboard](https://studio-git-branch-name-supabase.vercel.app/dashboard) | unique phrase from the change | | Design system | [/design-system](https://supabase.com/design-system) | [/design-system](https://design-system-git-branch-name-supabase.vercel.app/design-system) | unique phrase from the change | | UI library | [/library](https://supabase.com/library) | [/library](https://ui-library-git-branch-name-supabase.vercel.app/library) | unique phrase from the change | | Knowledge base | [/kb/guides/your-page](https://supabase.com/kb/guides/your-page) | [/kb/guides/your-page](https://kb-git-branch-name-supabase.vercel.app/kb/guides/your-page) | unique phrase from the change | --> <!-- ## Additional context Optionally add any other context or screenshots. --> ## Review instructions Sadly, testing this work is quite complex, but in case someone wants to: 1. Create a new Supabase project ton your personal space 1. Copy the id of the project and a secret key and add it to the new Search V2 environment variables as shown in the example file 1. Copy the content of the newly added `setup.sql` and run it on the SQL editor of your project. 1. Fetch this branch and on the search directory, run `search-v2:ingest` 1. Your project's table should have rows corresponding to the content from docs ## Checklist Check all before review: - [x] I have read [CONTRIBUTING.md](https://github.com/supabase/supabase/blob/master/CONTRIBUTING.md) - [x] If I wrote a new docs topic or edited an existing topic, I used the `/write-the-docs` or `/edit-the-docs` skill, which references [WORD_LIST](https://github.com/supabase/supabase/blob/master/apps/docs/WORD_LIST.md) and the docs [CONTRIBUTING](https://github.com/supabase/supabase/blob/master/apps/docs/CONTRIBUTING.md) guide <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit - **New Features** - Documentation pages are now indexed for full-text search at the page and section level. - Search results can show the most relevant section from each page, with its title, heading, excerpt, and relevance score. - Search content is automatically refreshed when published Markdown documentation changes. - **Tests** - Added coverage for Markdown parsing, page structure, routing, and search-content generation. <!-- end of auto-generated comment: release notes by coderabbit.ai -->
93 lines
3.8 KiB
PL/PgSQL
93 lines
3.8 KiB
PL/PgSQL
|
|
-- Run this in the Supabase SQL editor before `pnpm search-v2:ingest`.
|
|
-- Safe to re-run: it uses IF NOT EXISTS / CREATE OR REPLACE.
|
|
|
|
-- One row per markdown *section* (a heading + the text below it, up to the next heading).
|
|
create table if not exists public.docs_sections (
|
|
id bigint generated always as identity primary key,
|
|
slug text not null, -- 'guides/auth/users' (route without base path)
|
|
file_path text not null, -- path relative to the content root
|
|
page_title text not null default '', -- the page's H1
|
|
heading text not null default '', -- this section's heading
|
|
heading_level smallint not null default 0, -- 1..6, 0 = text before any heading
|
|
heading_path text[] not null default '{}',-- ['Users', 'Permanent and anonymous users']
|
|
content text not null default '', -- plain text of the section body
|
|
excerpt text not null default '', -- first paragraph of the page
|
|
updated_at timestamptz not null default now(),
|
|
|
|
-- Weighted search document, kept up to date by Postgres automatically.
|
|
-- A = H1 (page title)
|
|
-- B = H2
|
|
-- C = H3
|
|
-- D = H4+ headings and all body text
|
|
-- ts_rank_cd() weights these A > B > C > D by default ({0.1, 0.2, 0.4, 1.0} for D, C, B, A).
|
|
fts tsvector generated always as (
|
|
setweight(to_tsvector('english', coalesce(page_title, '')), 'A')
|
|
|| setweight(
|
|
to_tsvector('english', coalesce(heading, '')),
|
|
(case heading_level when 1 then 'A' when 2 then 'B' when 3 then 'C' else 'D' end)::"char"
|
|
)
|
|
|| setweight(to_tsvector('english', coalesce(content, '')), 'D')
|
|
) stored
|
|
);
|
|
|
|
create index if not exists docs_sections_fts_idx on public.docs_sections using gin (fts);
|
|
create index if not exists docs_sections_slug_idx on public.docs_sections (slug);
|
|
|
|
-- Read-only access for anon/authenticated (the ingest script writes with the secret key, which bypasses RLS).
|
|
alter table public.docs_sections enable row level security;
|
|
|
|
drop policy if exists "docs_sections are publicly readable" on public.docs_sections;
|
|
create policy "docs_sections are publicly readable"
|
|
on public.docs_sections for select
|
|
to anon, authenticated
|
|
using (true);
|
|
|
|
-- Search RPC: called from the server as supabase.rpc('search_docs', { query_text, match_limit }).
|
|
-- Score is ts_rank_cd normalised to 0..1 (normalization flag 32 => rank / (rank + 1)) and scaled to an
|
|
-- integer 0..1000 so the API can return a plain integer.
|
|
drop function if exists public.search_docs(text, int);
|
|
|
|
create or replace function public.search_docs(query_text text, match_limit int default 10)
|
|
returns table (
|
|
slug text,
|
|
page_title text,
|
|
heading text,
|
|
heading_level smallint,
|
|
heading_path text[],
|
|
excerpt text,
|
|
score int
|
|
)
|
|
language sql
|
|
stable
|
|
as $$
|
|
with q as (
|
|
select websearch_to_tsquery('english', query_text) as tsq
|
|
),
|
|
ranked as (
|
|
select
|
|
d.slug,
|
|
d.page_title,
|
|
d.heading,
|
|
d.heading_level,
|
|
d.heading_path,
|
|
d.excerpt,
|
|
round(ts_rank_cd(d.fts, q.tsq, 32) * 1000)::int as score
|
|
from public.docs_sections d, q
|
|
where d.fts @@ q.tsq
|
|
),
|
|
-- A page can have several matching sections; keep only its best-scoring one so the
|
|
-- same route doesn't take up multiple slots in the results.
|
|
best_per_page as (
|
|
select distinct on (r.slug) r.*
|
|
from ranked r
|
|
order by r.slug, r.score desc, r.heading_level asc
|
|
)
|
|
select b.slug, b.page_title, b.heading, b.heading_level, b.heading_path, b.excerpt, b.score
|
|
from best_per_page b
|
|
order by b.score desc, b.heading_level asc, b.slug asc
|
|
limit greatest(1, least(coalesce(match_limit, 10), 100));
|
|
$$;
|
|
|
|
grant execute on function public.search_docs(text, int) to anon, authenticated, service_role;
|