From 56961377d905bf6d3fdb3872534edbc69ef59efb Mon Sep 17 00:00:00 2001 From: Michel Pelletier Date: Tue, 3 Oct 2023 03:34:28 -0700 Subject: [PATCH] Michelp/add supachat example (#17445) * Add supachat example and seed. * remove unecessary commit * add sequence update code. * uncomment and move cron extension into migration, doesnt seem to work in seed file. * spellcheck. * add readme. --- .../migrations/20000101003626_customers.sql | 2 +- .../enterprise-patterns/supachat/README.md | 16 + .../supachat/supabase/config.toml | 82 +++++ .../migrations/20230913205645_supachat.sql | 312 ++++++++++++++++++ .../supachat/supabase/seed.sql | 35 ++ 5 files changed, 446 insertions(+), 1 deletion(-) create mode 100644 examples/enterprise-patterns/supachat/README.md create mode 100644 examples/enterprise-patterns/supachat/supabase/config.toml create mode 100644 examples/enterprise-patterns/supachat/supabase/migrations/20230913205645_supachat.sql create mode 100644 examples/enterprise-patterns/supachat/supabase/seed.sql diff --git a/examples/enterprise-patterns/partitions/supabase/migrations/20000101003626_customers.sql b/examples/enterprise-patterns/partitions/supabase/migrations/20000101003626_customers.sql index dba097e5026..373414640e2 100644 --- a/examples/enterprise-patterns/partitions/supabase/migrations/20000101003626_customers.sql +++ b/examples/enterprise-patterns/partitions/supabase/migrations/20000101003626_customers.sql @@ -13,7 +13,7 @@ partition by list(country); create table customers_americas partition of customers - for values in ('US', 'CANADA') + for values in ('US', 'CANADA'); create table customers_asia partition of customers diff --git a/examples/enterprise-patterns/supachat/README.md b/examples/enterprise-patterns/supachat/README.md new file mode 100644 index 00000000000..bf41629d358 --- /dev/null +++ b/examples/enterprise-patterns/supachat/README.md @@ -0,0 +1,16 @@ +# Supachat Dynamic Table Partitioning Example + +This is example code for the [Supachat Dynamic Table Partitioning blog post](). + +Postgres has many powerful partitioning features ranging from simple +declarative partitioning to fine-grained control over how child tables +are attached and detached. + +The documentation for more advanced usage of these features [is quite +detailed](https://www.postgresql.org/docs/current/ddl-partitioning.html), +but based on some customer experience, we've noticed users often end +up considering partitioning after already having a lot of data in one +big table. We've decided to create an example chat application that +shows how to use some of the more interesting features of Postgres +partitioning for migrating from the one "Large Table" problem with +minimal down time. diff --git a/examples/enterprise-patterns/supachat/supabase/config.toml b/examples/enterprise-patterns/supachat/supabase/config.toml new file mode 100644 index 00000000000..116399345f6 --- /dev/null +++ b/examples/enterprise-patterns/supachat/supabase/config.toml @@ -0,0 +1,82 @@ +# A string used to distinguish different Supabase projects on the same host. Defaults to the working +# directory name when running `supabase init`. +project_id = "supachat" + +[api] +# Port to use for the API URL. +port = 65431 +# Schemas to expose in your API. Tables, views and stored procedures in this schema will get API +# endpoints. public and storage are always included. +schemas = ["public", "storage", "graphql_public"] +# Extra schemas to add to the search_path of every request. public is always included. +extra_search_path = ["public", "extensions"] +# The maximum number of rows returns from a view, table, or stored procedure. Limits payload size +# for accidental or malicious requests. +max_rows = 1000 + +[db] +# Port to use for the local database URL. +port = 65432 +# The database major version to use. This has to be the same as your remote database's. Run `SHOW +# server_version;` on the remote database to check. +major_version = 15 + +[studio] +# Port to use for Supabase Studio. +port = 65433 + +# Email testing server. Emails sent with the local dev setup are not actually sent - rather, they +# are monitored, and you can view the emails that would have been sent from the web interface. +[inbucket] +# Port to use for the email testing server web interface. +port = 65434 +smtp_port = 65435 +pop3_port = 65436 + +[storage] +# The maximum file size allowed (e.g. "5MB", "500KB"). +file_size_limit = "50MiB" + +[auth] +# The base URL of your website. Used as an allow-list for redirects and for constructing URLs used +# in emails. +site_url = "http://localhost:3000" +# A list of *exact* URLs that auth providers are permitted to redirect to post authentication. +additional_redirect_urls = ["https://localhost:3000"] +# How long tokens are valid for, in seconds. Defaults to 3600 (1 hour), maximum 604,800 seconds (one +# week). +jwt_expiry = 3600 +# Allow/disallow new user signups to your project. +enable_signup = true + +[auth.email] +# Allow/disallow new user signups via email to your project. +enable_signup = true +# If enabled, a user will be required to confirm any email change on both the old, and new email +# addresses. If disabled, only the new email is required to confirm. +double_confirm_changes = true +# If enabled, users need to confirm their email address before signing in. +enable_confirmations = false + +# Use an external OAuth provider. The full list of providers are: `apple`, `azure`, `bitbucket`, +# `discord`, `facebook`, `github`, `gitlab`, `google`, `keycloak`, `linkedin`, `notion`, `twitch`, +# `twitter`, `slack`, `spotify`, `workos`, `zoom`. +[auth.external.apple] +enabled = false +client_id = "" +secret = "" +# Overrides the default auth redirectUrl. +redirect_uri = "" +# Overrides the default auth provider URL. Used to support self-hosted gitlab, single-tenant Azure, +# or any other third-party OIDC providers. +url = "" + +[analytics] +enabled = false +port = 65437 +vector_port = 65438 +# Setup BigQuery project to enable log viewer on local development stack. +# See: https://supabase.com/docs/guides/getting-started/local-development#enabling-local-logging +gcp_project_id = "" +gcp_project_number = "" +gcp_jwt_path = "supabase/gcloud.json" diff --git a/examples/enterprise-patterns/supachat/supabase/migrations/20230913205645_supachat.sql b/examples/enterprise-patterns/supachat/supabase/migrations/20230913205645_supachat.sql new file mode 100644 index 00000000000..98ecfc84647 --- /dev/null +++ b/examples/enterprise-patterns/supachat/supabase/migrations/20230913205645_supachat.sql @@ -0,0 +1,312 @@ +-- +-- Create a fake "Large Table" problem, where one table holds many rows. +-- + +CREATE TABLE chats( + id bigserial, + created_at timestamptz NOT NULL DEFAULT now(), + PRIMARY KEY (id) + ); + +CREATE TABLE chat_messages( + id bigserial, + created_at timestamptz NOT NULL, + chat_id bigint NOT NULL, + chat_created_at timestamptz NOT NULL, + message text NOT NULL, + PRIMARY KEY (id), + FOREIGN KEY (chat_id) REFERENCES chats(id) + ); + +CREATE INDEX ON chats (created_at); +CREATE INDEX ON chat_messages (created_at); + +-- +-- below are the "new" partitioned tables, in their own schema to +-- avoid name conflicts with the "old" tables above. Let's start a +-- new transaction to setup the new tables and migration procedures. +-- In order to avoid namespace pollution and name conflicts with the +-- old tables, lets put all the new partitioned tables in a new +-- schema. +-- +CREATE SCHEMA app; +CREATE EXTENSION pg_cron; + +CREATE TABLE app.chats( + id bigserial, + created_at timestamptz NOT NULL DEFAULT now(), + PRIMARY KEY (id, created_at) -- the partition column must be part of pk + ) PARTITION BY RANGE (created_at); + +CREATE INDEX "chats_created_at" ON app.chats (created_at); + +CREATE TABLE app.chat_messages( + id bigserial, + created_at timestamptz NOT NULL, + chat_id bigint NOT NULL, + chat_created_at timestamptz NOT NULL, + message text NOT NULL, + PRIMARY KEY (id, created_at), + FOREIGN KEY (chat_id, chat_created_at) -- multicolumn fk to ensure + REFERENCES app.chats(id, created_at) + ) PARTITION BY RANGE (created_at); + +CREATE INDEX "chat_messages_created_at" ON app.chat_messages (created_at); +-- +-- need this index on the fk source to lookup messages by parent +-- +CREATE INDEX "chat_messages_chat_id_chat_created_at" + ON app.chat_messages (chat_id, chat_created_at); +-- +-- Function creates a chats partition for the given day argument +-- +CREATE OR REPLACE PROCEDURE app.create_chats_partition(partition_day date) + LANGUAGE plpgsql AS +$$ +BEGIN + EXECUTE format( + $i$ + CREATE TABLE IF NOT EXISTS app."chats_%1$s" + (LIKE app.chats INCLUDING DEFAULTS INCLUDING CONSTRAINTS); + $i$, partition_day); +END; +$$; +-- +-- Function creates a chat_messages partition for the given day argument +-- +CREATE OR REPLACE PROCEDURE app.create_chat_messages_partition(partition_day date) + LANGUAGE plpgsql AS +$$ +BEGIN + EXECUTE format( + $i$ + CREATE TABLE IF NOT EXISTS app."chat_messages_%1$s" + (LIKE app.chat_messages INCLUDING DEFAULTS INCLUDING CONSTRAINTS); + + -- adding these check constraints means postgres can + -- attach partitions without locking and having to scan them. + ALTER TABLE app."chat_messages_%1$s" ADD CONSTRAINT + "chat_messages_partition_by_range_check_%1$s" + CHECK ( created_at >= DATE %1$L AND created_at < DATE %2$L ); + + $i$, partition_day, (partition_day + interval '1 day')::date); +END; +$$; +-- +-- Function copies one day's worth of chats rows from old "large" +-- table new partition. Note that the copied data is ordered by +-- created_at, this improves block cache density. +-- +CREATE OR REPLACE PROCEDURE app.copy_chats_partition(partition_day date) + LANGUAGE plpgsql AS +$$ +DECLARE + num_copied bigint = 0; +BEGIN + EXECUTE format( + $i$ + INSERT INTO app."chats_%1$s" (id, created_at) + SELECT id, created_at FROM chats + WHERE created_at::date >= %1$L::date AND created_at::date < (%1$L::date + interval '1 day') + ORDER BY created_at + $i$, partition_day); + GET DIAGNOSTICS num_copied = ROW_COUNT; + RAISE NOTICE 'Copied % rows to %', num_copied, format('app."chats_%1$s"', partition_day); +END; +$$; +-- +-- Function copies one day's worth of chat_messages rows from old +-- "large" table new partition. Note that the data is ordered by +-- chat_id then created_at, this improves block cache density. +-- +CREATE OR REPLACE PROCEDURE app.copy_chat_messages_partition(partition_day date) + LANGUAGE plpgsql AS +$$ +DECLARE + num_copied bigint = 0; +BEGIN + EXECUTE format( + $i$ + INSERT INTO app."chat_messages_%1$s" (id, created_at, chat_id, chat_created_at, message) + SELECT m.id, m.created_at, c.id, c.created_at, m.message FROM chat_messages m JOIN chats c on (c.id = m.chat_id) + WHERE m.created_at::date >= %1$L::date AND m.created_at::date < (%1$L::date + interval '1 day') + ORDER BY chat_id, m.created_at + $i$, partition_day); + GET DIAGNOSTICS num_copied = ROW_COUNT; + RAISE NOTICE 'Copied % rows to %', num_copied, format('app."chat_messages_%1$s"', partition_day); +END; +$$; +-- +-- Function indexes and attaches one day's worth of chats to parent table +-- +CREATE OR REPLACE PROCEDURE app.index_and_attach_chats_partition(partition_day date) + LANGUAGE plpgsql AS +$$ +BEGIN + EXECUTE format( + $i$ + -- now that any bulk data is loaded, setup the new partition table's pks + ALTER TABLE app."chats_%1$s" ADD PRIMARY KEY (id, created_at); + + -- adding these check constraints means postgres can + -- attach partitions without locking and having to scan them. + ALTER TABLE app."chats_%1$s" ADD CONSTRAINT + "chats_partition_by_range_check_%1$s" + CHECK ( created_at >= DATE %1$L AND created_at < DATE %2$L ); + + -- add more partition indexes here if necessary + CREATE INDEX "chats_%1$s_created_at" + ON app."chats_%1$s" + USING btree(created_at) + WITH (fillfactor=100); + + -- by "attaching" the new tables and indexes *after* the pk, + -- indexing and check constraints verify all rows, + -- no scan checks or locks are necessary, attachment is very fast, + -- and queries to parent are not blocked. + ALTER TABLE app.chats + ATTACH PARTITION app."chats_%1$s" + FOR VALUES FROM (%1$L) TO (%2$L); + + -- You now also "attach" any indexes you made at this point + ALTER INDEX app."chats_created_at" + ATTACH PARTITION app."chats_%1$s_created_at"; + + -- Droping the now unnecessary check constraints they were just needed + -- to prevent the attachment from forcing a scan to do the same check + ALTER TABLE app."chats_%1$s" DROP CONSTRAINT + "chats_partition_by_range_check_%1$s"; + $i$, + partition_day, (partition_day + interval '1 day')::date); +END; +$$; +-- +-- Function indexes and attaches one day's worth of chat_messages to parent table +-- +CREATE OR REPLACE PROCEDURE app.index_and_attach_chat_messages_partition(partition_day date) + LANGUAGE plpgsql AS +$$ +BEGIN + EXECUTE format( + $i$ + -- now that any bulk data is loaded, setup the new partition table's pks + ALTER TABLE app."chat_messages_%1$s" ADD PRIMARY KEY (id, created_at); + + -- here's where you create per-partition indexes on the partitions + CREATE INDEX "chat_messages_%1$s_created_at" + ON app."chat_messages_%1$s" + USING btree(created_at) + WITH (fillfactor=100); + + CREATE INDEX "chat_messages_%1$s_chat_id_chat_created_at" + ON app."chat_messages_%1$s" + USING btree(chat_id, chat_created_at) + WITH (fillfactor=100); + + -- add more partition indexes here if necessary + + -- by "attaching" the new tables and indexes *after* the pk, + -- indexing and check constraints verify all rows, + -- no scan checks or locks are necessary, attachment is very fast, + -- and queries to parent are not blocked. + ALTER TABLE app.chat_messages + ATTACH PARTITION app."chat_messages_%1$s" + FOR VALUES FROM (%1$L) TO (%2$L); + + -- You now also "attach" any indexes you made at this point + ALTER INDEX app."chat_messages_created_at" + ATTACH PARTITION app."chat_messages_%1$s_created_at"; + + ALTER INDEX app."chat_messages_chat_id_chat_created_at" + ATTACH PARTITION app."chat_messages_%1$s_chat_id_chat_created_at"; + + -- Droping the now unnecessary check constraints they were just needed + -- to prevent the attachment from forcing a scan to do the same check + ALTER TABLE app."chat_messages_%1$s" DROP CONSTRAINT + "chat_messages_partition_by_range_check_%1$s"; + $i$, + partition_day, (partition_day + interval '1 day')::date); +END; +$$; +-- +-- Wrapper functions to loop over all days in large table, creating +-- new partions, copying them, then indexing and attaching them. +-- +CREATE OR REPLACE PROCEDURE app.load_chats_partition(i date) + LANGUAGE plpgsql AS +$$ +BEGIN + CALL app.create_chats_partition(i); + CALL app.copy_chats_partition(i); + CALL app.index_and_attach_chats_partition(i); + COMMIT; +END; +$$; +CREATE OR REPLACE PROCEDURE app.load_chats_partitions() + LANGUAGE plpgsql AS +$$ +DECLARE + start_date date; + end_date date; + i date; +BEGIN + SELECT min(created_at)::date INTO start_date FROM chats; + SELECT max(created_at)::date INTO end_date FROM chats; + FOR i IN SELECT * FROM generate_series(end_date, start_date, interval '-1 day') LOOP + CALL app.load_chats_partition(i); + END LOOP; +END; +$$; +-- +-- Wrapper function loops over all days in large table, creating new +-- partions, copying them, then indexing and attaching them. +-- +CREATE OR REPLACE PROCEDURE app.load_chat_messages_partition(i date) + LANGUAGE plpgsql AS +$$ +BEGIN + CALL app.create_chat_messages_partition(i); + CALL app.copy_chat_messages_partition(i); + CALL app.index_and_attach_chat_messages_partition(i); + COMMIT; +END; +$$; +CREATE OR REPLACE PROCEDURE app.load_chat_messages_partitions() + LANGUAGE plpgsql AS +$$ +DECLARE + start_date date; + end_date date; + i date; +BEGIN + SELECT min(created_at)::date INTO start_date FROM chat_messages; + SELECT max(created_at)::date INTO end_date FROM chat_messages; + FOR i IN SELECT * FROM generate_series(end_date, start_date, interval '-1 day') LOOP + CALL app.load_chat_messages_partition(i); + END LOOP; +END; +$$; +-- +-- This procedure will be used by pg_cron to create both new +-- partitions for "today". +-- +CREATE OR REPLACE PROCEDURE app.create_daily_partitions(today date = now()::date) + LANGUAGE plpgsql AS +$$ +BEGIN + CALL app.create_chats_partition(today); + CALL app.create_chat_messages_partition(today); +END; +$$; +-- +-- This procedure will reset the sequence for partition tables after +-- old rows have been copied into them. +-- +CREATE OR REPLACE PROCEDURE app.update_chat_sequences() + LANGUAGE plpgsql AS +$$ +BEGIN + PERFORM setval('app.chats_id_seq', coalesce((SELECT max(id) FROM app.chats), 1)); + PERFORM setval('app.chat_messages_id_seq', coalesce((SELECT max(id) FROM app.chat_messages), 1)); +END; +$$; diff --git a/examples/enterprise-patterns/supachat/supabase/seed.sql b/examples/enterprise-patterns/supachat/supabase/seed.sql new file mode 100644 index 00000000000..2d785cbe1b3 --- /dev/null +++ b/examples/enterprise-patterns/supachat/supabase/seed.sql @@ -0,0 +1,35 @@ +-- +-- generate one year of fake chat data +-- +INSERT INTO chats (created_at) + SELECT generate_series( + '2022-01-01'::timestamptz, + '2022-01-07 23:00:00'::timestamptz, + interval '1 hour'); + +INSERT INTO chat_messages (created_at, chat_id, chat_created_at, message) + SELECT + mca, + chats.id, + chats.created_at, + (SELECT ($$[0:3]={'hello','goodbye','How are you today','I am fine'}$$::text[])[trunc(random() * 4)::int]) + FROM chats + CROSS JOIN LATERAL ( + SELECT generate_series( + chats.created_at, + chats.created_at + interval '1 day', + interval '1 minute') AS mca) b; + +CALL app.load_chats_partitions(); +CALL app.load_chat_messages_partitions(); +CALL app.update_chat_sequences(); +-- +-- Now schedule a job to create new partitions for "tomorrow" every night +-- +SELECT cron.schedule('new-chat-partition', '0 0 * * *', 'CALL app.load_app_partitions(now()::date)'); +-- +-- +-- After bulk loading data, tables should be vacuumed and analyzed. +-- This cannot be done inside a transaction block. +-- +VACUUM ANALYZE app.chats, app.chat_messages;