Michelp/add supachat example (#17445)

* Add supachat example and seed.

* remove unecessary commit

* add sequence update code.

* uncomment and move cron extension into migration, doesnt seem to work in seed file.

* spellcheck.

* add readme.
This commit is contained in:
Michel Pelletier authored and GitHub committed 2023-10-03 12:34:28 +02:00
1 parent 89efa97afd
commit 56961377d9
5 files changed
+446 -1

No files matched your search

@@ -13,7 +13,7 @@ partition by list(country);
create table customers_americas
partition of customers
for values in ('US', 'CANADA')
for values in ('US', 'CANADA');
create table customers_asia
partition of customers
@@ -0,0 +1,16 @@
# Supachat Dynamic Table Partitioning Example
This is example code for the [Supachat Dynamic Table Partitioning blog post]().
Postgres has many powerful partitioning features ranging from simple
declarative partitioning to fine-grained control over how child tables
are attached and detached.
The documentation for more advanced usage of these features [is quite
detailed](https://www.postgresql.org/docs/current/ddl-partitioning.html),
but based on some customer experience, we've noticed users often end
up considering partitioning after already having a lot of data in one
big table. We've decided to create an example chat application that
shows how to use some of the more interesting features of Postgres
partitioning for migrating from the one "Large Table" problem with
minimal down time.
@@ -0,0 +1,82 @@
# A string used to distinguish different Supabase projects on the same host. Defaults to the working
# directory name when running `supabase init`.
project_id = "supachat"
[api]
# Port to use for the API URL.
port = 65431
# Schemas to expose in your API. Tables, views and stored procedures in this schema will get API
# endpoints. public and storage are always included.
schemas = ["public", "storage", "graphql_public"]
# Extra schemas to add to the search_path of every request. public is always included.
extra_search_path = ["public", "extensions"]
# The maximum number of rows returns from a view, table, or stored procedure. Limits payload size
# for accidental or malicious requests.
max_rows = 1000
[db]
# Port to use for the local database URL.
port = 65432
# The database major version to use. This has to be the same as your remote database's. Run `SHOW
# server_version;` on the remote database to check.
major_version = 15
[studio]
# Port to use for Supabase Studio.
port = 65433
# Email testing server. Emails sent with the local dev setup are not actually sent - rather, they
# are monitored, and you can view the emails that would have been sent from the web interface.
[inbucket]
# Port to use for the email testing server web interface.
port = 65434
smtp_port = 65435
pop3_port = 65436
[storage]
# The maximum file size allowed (e.g. "5MB", "500KB").
file_size_limit = "50MiB"
[auth]
# The base URL of your website. Used as an allow-list for redirects and for constructing URLs used
# in emails.
site_url = "http://localhost:3000"
# A list of *exact* URLs that auth providers are permitted to redirect to post authentication.
additional_redirect_urls = ["https://localhost:3000"]
# How long tokens are valid for, in seconds. Defaults to 3600 (1 hour), maximum 604,800 seconds (one
# week).
jwt_expiry = 3600
# Allow/disallow new user signups to your project.
enable_signup = true
[auth.email]
# Allow/disallow new user signups via email to your project.
enable_signup = true
# If enabled, a user will be required to confirm any email change on both the old, and new email
# addresses. If disabled, only the new email is required to confirm.
double_confirm_changes = true
# If enabled, users need to confirm their email address before signing in.
enable_confirmations = false
# Use an external OAuth provider. The full list of providers are: `apple`, `azure`, `bitbucket`,
# `discord`, `facebook`, `github`, `gitlab`, `google`, `keycloak`, `linkedin`, `notion`, `twitch`,
# `twitter`, `slack`, `spotify`, `workos`, `zoom`.
[auth.external.apple]
enabled = false
client_id = ""
secret = ""
# Overrides the default auth redirectUrl.
redirect_uri = ""
# Overrides the default auth provider URL. Used to support self-hosted gitlab, single-tenant Azure,
# or any other third-party OIDC providers.
url = ""
[analytics]
enabled = false
port = 65437
vector_port = 65438
# Setup BigQuery project to enable log viewer on local development stack.
# See: https://supabase.com/docs/guides/getting-started/local-development#enabling-local-logging
gcp_project_id = ""
gcp_project_number = ""
gcp_jwt_path = "supabase/gcloud.json"
@@ -0,0 +1,312 @@
--
-- Create a fake "Large Table" problem, where one table holds many rows.
--
CREATE TABLE chats(
id bigserial,
created_at timestamptz NOT NULL DEFAULT now(),
PRIMARY KEY (id)
);
CREATE TABLE chat_messages(
id bigserial,
created_at timestamptz NOT NULL,
chat_id bigint NOT NULL,
chat_created_at timestamptz NOT NULL,
message text NOT NULL,
PRIMARY KEY (id),
FOREIGN KEY (chat_id) REFERENCES chats(id)
);
CREATE INDEX ON chats (created_at);
CREATE INDEX ON chat_messages (created_at);
--
-- below are the "new" partitioned tables, in their own schema to
-- avoid name conflicts with the "old" tables above. Let's start a
-- new transaction to setup the new tables and migration procedures.
-- In order to avoid namespace pollution and name conflicts with the
-- old tables, lets put all the new partitioned tables in a new
-- schema.
--
CREATE SCHEMA app;
CREATE EXTENSION pg_cron;
CREATE TABLE app.chats(
id bigserial,
created_at timestamptz NOT NULL DEFAULT now(),
PRIMARY KEY (id, created_at) -- the partition column must be part of pk
) PARTITION BY RANGE (created_at);
CREATE INDEX "chats_created_at" ON app.chats (created_at);
CREATE TABLE app.chat_messages(
id bigserial,
created_at timestamptz NOT NULL,
chat_id bigint NOT NULL,
chat_created_at timestamptz NOT NULL,
message text NOT NULL,
PRIMARY KEY (id, created_at),
FOREIGN KEY (chat_id, chat_created_at) -- multicolumn fk to ensure
REFERENCES app.chats(id, created_at)
) PARTITION BY RANGE (created_at);
CREATE INDEX "chat_messages_created_at" ON app.chat_messages (created_at);
--
-- need this index on the fk source to lookup messages by parent
--
CREATE INDEX "chat_messages_chat_id_chat_created_at"
ON app.chat_messages (chat_id, chat_created_at);
--
-- Function creates a chats partition for the given day argument
--
CREATE OR REPLACE PROCEDURE app.create_chats_partition(partition_day date)
LANGUAGE plpgsql AS
$$
BEGIN
EXECUTE format(
$i$
CREATE TABLE IF NOT EXISTS app."chats_%1$s"
(LIKE app.chats INCLUDING DEFAULTS INCLUDING CONSTRAINTS);
$i$, partition_day);
END;
$$;
--
-- Function creates a chat_messages partition for the given day argument
--
CREATE OR REPLACE PROCEDURE app.create_chat_messages_partition(partition_day date)
LANGUAGE plpgsql AS
$$
BEGIN
EXECUTE format(
$i$
CREATE TABLE IF NOT EXISTS app."chat_messages_%1$s"
(LIKE app.chat_messages INCLUDING DEFAULTS INCLUDING CONSTRAINTS);
-- adding these check constraints means postgres can
-- attach partitions without locking and having to scan them.
ALTER TABLE app."chat_messages_%1$s" ADD CONSTRAINT
"chat_messages_partition_by_range_check_%1$s"
CHECK ( created_at >= DATE %1$L AND created_at < DATE %2$L );
$i$, partition_day, (partition_day + interval '1 day')::date);
END;
$$;
--
-- Function copies one day's worth of chats rows from old "large"
-- table new partition. Note that the copied data is ordered by
-- created_at, this improves block cache density.
--
CREATE OR REPLACE PROCEDURE app.copy_chats_partition(partition_day date)
LANGUAGE plpgsql AS
$$
DECLARE
num_copied bigint = 0;
BEGIN
EXECUTE format(
$i$
INSERT INTO app."chats_%1$s" (id, created_at)
SELECT id, created_at FROM chats
WHERE created_at::date >= %1$L::date AND created_at::date < (%1$L::date + interval '1 day')
ORDER BY created_at
$i$, partition_day);
GET DIAGNOSTICS num_copied = ROW_COUNT;
RAISE NOTICE 'Copied % rows to %', num_copied, format('app."chats_%1$s"', partition_day);
END;
$$;
--
-- Function copies one day's worth of chat_messages rows from old
-- "large" table new partition. Note that the data is ordered by
-- chat_id then created_at, this improves block cache density.
--
CREATE OR REPLACE PROCEDURE app.copy_chat_messages_partition(partition_day date)
LANGUAGE plpgsql AS
$$
DECLARE
num_copied bigint = 0;
BEGIN
EXECUTE format(
$i$
INSERT INTO app."chat_messages_%1$s" (id, created_at, chat_id, chat_created_at, message)
SELECT m.id, m.created_at, c.id, c.created_at, m.message FROM chat_messages m JOIN chats c on (c.id = m.chat_id)
WHERE m.created_at::date >= %1$L::date AND m.created_at::date < (%1$L::date + interval '1 day')
ORDER BY chat_id, m.created_at
$i$, partition_day);
GET DIAGNOSTICS num_copied = ROW_COUNT;
RAISE NOTICE 'Copied % rows to %', num_copied, format('app."chat_messages_%1$s"', partition_day);
END;
$$;
--
-- Function indexes and attaches one day's worth of chats to parent table
--
CREATE OR REPLACE PROCEDURE app.index_and_attach_chats_partition(partition_day date)
LANGUAGE plpgsql AS
$$
BEGIN
EXECUTE format(
$i$
-- now that any bulk data is loaded, setup the new partition table's pks
ALTER TABLE app."chats_%1$s" ADD PRIMARY KEY (id, created_at);
-- adding these check constraints means postgres can
-- attach partitions without locking and having to scan them.
ALTER TABLE app."chats_%1$s" ADD CONSTRAINT
"chats_partition_by_range_check_%1$s"
CHECK ( created_at >= DATE %1$L AND created_at < DATE %2$L );
-- add more partition indexes here if necessary
CREATE INDEX "chats_%1$s_created_at"
ON app."chats_%1$s"
USING btree(created_at)
WITH (fillfactor=100);
-- by "attaching" the new tables and indexes *after* the pk,
-- indexing and check constraints verify all rows,
-- no scan checks or locks are necessary, attachment is very fast,
-- and queries to parent are not blocked.
ALTER TABLE app.chats
ATTACH PARTITION app."chats_%1$s"
FOR VALUES FROM (%1$L) TO (%2$L);
-- You now also "attach" any indexes you made at this point
ALTER INDEX app."chats_created_at"
ATTACH PARTITION app."chats_%1$s_created_at";
-- Droping the now unnecessary check constraints they were just needed
-- to prevent the attachment from forcing a scan to do the same check
ALTER TABLE app."chats_%1$s" DROP CONSTRAINT
"chats_partition_by_range_check_%1$s";
$i$,
partition_day, (partition_day + interval '1 day')::date);
END;
$$;
--
-- Function indexes and attaches one day's worth of chat_messages to parent table
--
CREATE OR REPLACE PROCEDURE app.index_and_attach_chat_messages_partition(partition_day date)
LANGUAGE plpgsql AS
$$
BEGIN
EXECUTE format(
$i$
-- now that any bulk data is loaded, setup the new partition table's pks
ALTER TABLE app."chat_messages_%1$s" ADD PRIMARY KEY (id, created_at);
-- here's where you create per-partition indexes on the partitions
CREATE INDEX "chat_messages_%1$s_created_at"
ON app."chat_messages_%1$s"
USING btree(created_at)
WITH (fillfactor=100);
CREATE INDEX "chat_messages_%1$s_chat_id_chat_created_at"
ON app."chat_messages_%1$s"
USING btree(chat_id, chat_created_at)
WITH (fillfactor=100);
-- add more partition indexes here if necessary
-- by "attaching" the new tables and indexes *after* the pk,
-- indexing and check constraints verify all rows,
-- no scan checks or locks are necessary, attachment is very fast,
-- and queries to parent are not blocked.
ALTER TABLE app.chat_messages
ATTACH PARTITION app."chat_messages_%1$s"
FOR VALUES FROM (%1$L) TO (%2$L);
-- You now also "attach" any indexes you made at this point
ALTER INDEX app."chat_messages_created_at"
ATTACH PARTITION app."chat_messages_%1$s_created_at";
ALTER INDEX app."chat_messages_chat_id_chat_created_at"
ATTACH PARTITION app."chat_messages_%1$s_chat_id_chat_created_at";
-- Droping the now unnecessary check constraints they were just needed
-- to prevent the attachment from forcing a scan to do the same check
ALTER TABLE app."chat_messages_%1$s" DROP CONSTRAINT
"chat_messages_partition_by_range_check_%1$s";
$i$,
partition_day, (partition_day + interval '1 day')::date);
END;
$$;
--
-- Wrapper functions to loop over all days in large table, creating
-- new partions, copying them, then indexing and attaching them.
--
CREATE OR REPLACE PROCEDURE app.load_chats_partition(i date)
LANGUAGE plpgsql AS
$$
BEGIN
CALL app.create_chats_partition(i);
CALL app.copy_chats_partition(i);
CALL app.index_and_attach_chats_partition(i);
COMMIT;
END;
$$;
CREATE OR REPLACE PROCEDURE app.load_chats_partitions()
LANGUAGE plpgsql AS
$$
DECLARE
start_date date;
end_date date;
i date;
BEGIN
SELECT min(created_at)::date INTO start_date FROM chats;
SELECT max(created_at)::date INTO end_date FROM chats;
FOR i IN SELECT * FROM generate_series(end_date, start_date, interval '-1 day') LOOP
CALL app.load_chats_partition(i);
END LOOP;
END;
$$;
--
-- Wrapper function loops over all days in large table, creating new
-- partions, copying them, then indexing and attaching them.
--
CREATE OR REPLACE PROCEDURE app.load_chat_messages_partition(i date)
LANGUAGE plpgsql AS
$$
BEGIN
CALL app.create_chat_messages_partition(i);
CALL app.copy_chat_messages_partition(i);
CALL app.index_and_attach_chat_messages_partition(i);
COMMIT;
END;
$$;
CREATE OR REPLACE PROCEDURE app.load_chat_messages_partitions()
LANGUAGE plpgsql AS
$$
DECLARE
start_date date;
end_date date;
i date;
BEGIN
SELECT min(created_at)::date INTO start_date FROM chat_messages;
SELECT max(created_at)::date INTO end_date FROM chat_messages;
FOR i IN SELECT * FROM generate_series(end_date, start_date, interval '-1 day') LOOP
CALL app.load_chat_messages_partition(i);
END LOOP;
END;
$$;
--
-- This procedure will be used by pg_cron to create both new
-- partitions for "today".
--
CREATE OR REPLACE PROCEDURE app.create_daily_partitions(today date = now()::date)
LANGUAGE plpgsql AS
$$
BEGIN
CALL app.create_chats_partition(today);
CALL app.create_chat_messages_partition(today);
END;
$$;
--
-- This procedure will reset the sequence for partition tables after
-- old rows have been copied into them.
--
CREATE OR REPLACE PROCEDURE app.update_chat_sequences()
LANGUAGE plpgsql AS
$$
BEGIN
PERFORM setval('app.chats_id_seq', coalesce((SELECT max(id) FROM app.chats), 1));
PERFORM setval('app.chat_messages_id_seq', coalesce((SELECT max(id) FROM app.chat_messages), 1));
END;
$$;
@@ -0,0 +1,35 @@
--
-- generate one year of fake chat data
--
INSERT INTO chats (created_at)
SELECT generate_series(
'2022-01-01'::timestamptz,
'2022-01-07 23:00:00'::timestamptz,
interval '1 hour');
INSERT INTO chat_messages (created_at, chat_id, chat_created_at, message)
SELECT
mca,
chats.id,
chats.created_at,
(SELECT ($$[0:3]={'hello','goodbye','How are you today','I am fine'}$$::text[])[trunc(random() * 4)::int])
FROM chats
CROSS JOIN LATERAL (
SELECT generate_series(
chats.created_at,
chats.created_at + interval '1 day',
interval '1 minute') AS mca) b;
CALL app.load_chats_partitions();
CALL app.load_chat_messages_partitions();
CALL app.update_chat_sequences();
--
-- Now schedule a job to create new partitions for "tomorrow" every night
--
SELECT cron.schedule('new-chat-partition', '0 0 * * *', 'CALL app.load_app_partitions(now()::date)');
--
--
-- After bulk loading data, tables should be vacuumed and analyzed.
-- This cannot be done inside a transaction block.
--
VACUUM ANALYZE app.chats, app.chat_messages;