From ced974e0730d91a189264c9a413997f534a0702c Mon Sep 17 00:00:00 2001 From: Riccardo Busetti Date: Mon, 21 Sep 2026 08:27:07 +0200 Subject: [PATCH] docs(pipelines): Align and streamline replication guides (#49252) --- .../billing/pricing/pricing_pipelines.mdx | 19 +- .../content/guides/database/replication.mdx | 77 ++- .../guides/database/replication/bigquery.mdx | 262 +++++---- .../database/replication/clickhouse.mdx | 140 ++--- .../guides/database/replication/ducklake.mdx | 134 ++--- .../replication/manual-replication-faq.mdx | 32 +- .../manual-replication-monitoring.mdx | 126 +++-- .../replication/manual-replication-setup.mdx | 38 +- .../database/replication/pipelines-faq.mdx | 267 +++------ .../replication/pipelines-monitoring.mdx | 297 ++++------ .../guides/database/replication/pipelines.mdx | 508 +++++++++--------- .../guides/database/replication/snowflake.mdx | 161 +++--- .../platform/manage-your-usage/pipelines.mdx | 64 ++- .../DestinationForm/AdvancedSettings.tsx | 2 +- supa-mdx-lint/Rule003Spelling.toml | 4 + 15 files changed, 968 insertions(+), 1163 deletions(-) diff --git a/apps/docs/content/_partials/billing/pricing/pricing_pipelines.mdx b/apps/docs/content/_partials/billing/pricing/pricing_pipelines.mdx index e2e8be44f74..fd0377e753f 100644 --- a/apps/docs/content/_partials/billing/pricing/pricing_pipelines.mdx +++ b/apps/docs/content/_partials/billing/pricing/pricing_pipelines.mdx @@ -1,13 +1,6 @@ -{/* prettier-ignore */} - per hour for each configured pipeline. per Gigabyte of data processed during initial sync. per Gigabyte of data processed during ongoing replication. - -| Plan | Configured Pipeline | Initial Sync Data Processed | Ongoing Replication Data Processed | -| ---------- | -------------------------- | ----------------------------- | ---------------------------------- | -| Free | - | - | - | -| Pro | /hr | per GB | per GB | -| Team | /hr | per GB | per GB | -| Enterprise | Custom | Custom | Custom | - -**Data processed** is Postgres row data successfully processed by a pipeline and accepted by its destination. It is measured from the logical row data emitted by Postgres for replication, rather than physical table storage or destination-specific encoding. - -For a detailed breakdown of how charges are calculated, refer to [Manage Pipeline usage](/docs/guides/platform/manage-your-usage/pipelines). +| Plan | Pipeline hours | Initial sync | Ongoing replication | +| ---------- | -------------------------- | ----------------------------- | ----------------------------- | +| Free | - | - | - | +| Pro | /hr | per GB | per GB | +| Team | /hr | per GB | per GB | +| Enterprise | Custom | Custom | Custom | diff --git a/apps/docs/content/guides/database/replication.mdx b/apps/docs/content/guides/database/replication.mdx index 69bee5d30d2..e1c8fbb3f6e 100644 --- a/apps/docs/content/guides/database/replication.mdx +++ b/apps/docs/content/guides/database/replication.mdx @@ -8,14 +8,6 @@ sidebar_label: 'Overview' Replication keeps data synchronized with another location. Logical replication products such as Supabase Pipelines use change data capture (CDC) to read database changes and apply them to a destination. -## Use cases - -You might use database replication for: - -- **Analytics and data warehousing**: Replicate your operational database to analytics platforms for complex analysis without impacting your application's performance. -- **Data integration**: Keep your data synchronized across different systems and services in your tech stack. -- **Operational reporting**: Maintain a copy of selected application data that you can query in another system. - ## Replication methods Supabase supports three replication methods. Choose based on whether you need another Supabase Postgres database, a managed replication pipeline to a destination system, or full control over your own logical replication setup. @@ -24,7 +16,7 @@ Supabase supports three replication methods. Choose based on whether you need an Read replicas are additional Supabase Postgres databases kept in sync with your primary database. Use them when you want read-only query capacity, lower latency in another region, or to isolate analytical reads from application writes while staying inside Supabase Postgres. -- [Set up read replicas](/docs/guides/platform/read-replicas) +See [Set up read replicas](/docs/guides/platform/read-replicas). {/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} @@ -32,33 +24,38 @@ Read replicas are additional Supabase Postgres databases kept in sync with your <$Partial path="pipelines-public-alpha.mdx" /> -Supabase Pipelines is a managed CDC product for moving data from Supabase Postgres to supported destination systems. It uses Postgres logical replication with the open-source [Supabase ETL engine](https://github.com/supabase/etl). A destination is where your replicated data is stored; a pipeline first performs an initial sync of existing rows, then uses ongoing replication (CDC) to send subsequent database changes to that destination. +Supabase Pipelines is a managed CDC product for moving data from Supabase Postgres to supported destination systems. It uses Postgres logical replication with the open-source [Supabase ETL engine](https://github.com/supabase/etl). A destination is where your replicated data is stored; a pipeline copies existing rows for the tables selected for initial sync, then uses ongoing replication (CDC) to send subsequent database changes to that destination. -- [Set up Pipelines](/docs/guides/database/replication/pipelines) - -#### Supported destinations - -{/* supa-mdx-lint-disable-next-line Rule003Spelling */} -BigQuery is currently available as a managed destination. ClickHouse, DuckLake, and Snowflake are in Early Access. [Request access](/go/supabase-pipelines-new-destinations) to these destinations. - -Managed Pipelines run in **AWS `eu-central-1` (Frankfurt)**. Choose destination resources as close as possible to Frankfurt to reduce network latency and replication lag. - -| Destination | Insert | Update | Delete | Truncate | Schema change | Data model | -| ---------------------------------------------------------- | ------------ | ----------------------- | ---------------------------- | ------------ | ---------------------- | -------------------------------------------------------------------------------- | -| [BigQuery](/docs/guides/database/replication/bigquery) | ✅ Supported | ✅ Supported | ✅ Supported | ✅ Supported | Beta (limited) | Current-state tables. | -| [ClickHouse](/docs/guides/database/replication/clickhouse) | ✅ Supported | `REPLICA IDENTITY FULL` | Primary-key or full identity | ✅ Supported | Early Access (limited) | Current-state tables (default) or append-only CDC history. | -| [DuckLake](/docs/guides/database/replication/ducklake) | ✅ Supported | Row identity required | Row identity required | ✅ Supported | Early Access (limited) | Current-state lakehouse tables backed by a SQL catalog and object storage. | -| [Snowflake](/docs/guides/database/replication/snowflake) | ✅ Supported | `REPLICA IDENTITY FULL` | Row identity required | ✅ Supported | Early Access (limited) | Append-only CDC history. Source `TRUNCATE` operations and table resets erase it. | +See [Set up Pipelines](/docs/guides/database/replication/pipelines). ### Manual replication Manual replication uses the same underlying Postgres logical replication features as Pipelines, but you configure and operate the pieces yourself. Use this path when you want to connect tools such as Airbyte, Estuary, Fivetran, Materialize, Stitch, AWS DMS, or another system that supports Postgres logical replication. -- [Set up manual replication](/docs/guides/database/replication/manual-replication-setup) +See [Set up manual replication](/docs/guides/database/replication/manual-replication-setup). + +## Supported destinations + +| Destination | Status | +| ---------------------------------------------------------- | ------------- | +| [BigQuery](/docs/guides/database/replication/bigquery) | Public alpha | +| [ClickHouse](/docs/guides/database/replication/clickhouse) | Private alpha | +| [DuckLake](/docs/guides/database/replication/ducklake) | Private alpha | +| [Snowflake](/docs/guides/database/replication/snowflake) | Private alpha | + +[Request access](/go/supabase-pipelines-new-destinations) to destinations in private alpha. + +## Use cases + +You might use database replication for: + +- **Analytics and data warehousing**: Run analytical queries on replicated data in a separate platform. Initial sync and ongoing replication still use resources on the source database. +- **Data integration**: Keep your data synchronized across different systems and services in your tech stack. +- **Operational reporting**: Maintain a copy of selected application data that you can query in another system. ## Related features -For realtime features and syncing data to clients (browsers, mobile apps), see [Realtime](/docs/guides/realtime). +For realtime features and syncing data to browsers and mobile apps, see [Realtime](/docs/guides/realtime). Realtime also uses Postgres changes, but it is intended for broadcasting database updates to clients rather than maintaining a copy of your database in another system. @@ -66,7 +63,7 @@ Realtime also uses Postgres changes, but it is intended for broadcasting databas ### Write-Ahead Log (WAL) -Postgres uses a system called the Write-Ahead Log (WAL) to manage changes to the database. As you make changes, they are appended to the WAL, which is a series of files (also called "segments") where the file size can be specified. Once one segment is full, Postgres will start appending to a new segment. After a period of time, a checkpoint occurs and Postgres synchronizes the WAL with your database. Once the checkpoint is complete, then the WAL files can be removed from disk and free up space. +Postgres records database changes in the Write-Ahead Log (WAL) before writing them to data files. WAL is stored in files called segments. A checkpoint writes modified data pages to disk so crash recovery can start from a recent position. Older WAL segments can be recycled or removed once they are no longer needed for recovery, archiving, or replication. ### Logical replication and WAL @@ -78,11 +75,11 @@ LSN is a Log Sequence Number that identifies a position in the WAL. It is often ## Logical replication architecture -When setting up logical replication, three key components are involved: +Logical replication uses these components: -- `publication` - A set of tables on your primary database that will be `published` -- `replication slot` - A slot used for replicating the data from a single publication. The slot, when created, will specify the output format of the changes -- `subscription` - A subscription is created from an external system (that is, another Postgres database) and must specify the name of the `publication`. If you do not specify a replication slot, one is automatically created +- **Publication**: Defines the source tables and change types to publish, with optional column lists and row filters. +- **Replication slot**: Tracks a consumer's progress and retains WAL it still needs. A slot uses an output plugin to decode changes; it is not tied to a single publication. +- **Consumer**: Reads decoded changes and applies them to a destination. Another Postgres database can use a subscription to manage this connection and its publications. Pipelines connects directly to the replication stream without creating a Postgres subscription. ## Logical replication output format @@ -90,13 +87,13 @@ Logical replication is typically output in two forms, `pgoutput` and `wal2json`. ## Logical replication configuration -When using logical replication, Postgres keeps WAL files around for longer than it otherwise needs them. If the files are removed too soon, then your `replication slot` can become inactive or lost if the database receives a large number of changes in a short time. +Logical replication slots retain WAL until their consumers confirm that they no longer need it. If Postgres removes required WAL before a consumer catches up, the slot can become unusable and the consumer may need a new initial copy. -In order to mitigate this, Postgres has many options and settings that can be [tweaked](/docs/guides/database/custom-postgres-config) to manage the WAL usage effectively. Not all of these settings are user configurable as they can impact the stability of your database. For those that are, these should be considered as advanced configuration and not changed without understanding that they can cause additional disk space and resources to be used, as well as incur additional costs. +Postgres settings control replication capacity and WAL retention. Higher retention limits reduce the risk of losing required WAL, but can consume more database storage. Treat these settings as advanced configuration and check available disk before changing them. On Supabase, configure the supported settings with the [Supabase CLI](/docs/guides/database/custom-postgres-config#managing-postgres-configuration-with-the-cli). -| Setting | Description | User-facing | Default | -| ---------------------------------------------------------------------------------------- | ------------------------------------------------------ | ----------- | ------- | -| [`max_replication_slots`](https://postgresqlco.nf/doc/en/param/max_replication_slots/) | Max count of replication slots allowed | No | | -| [`wal_keep_size`](https://postgresqlco.nf/doc/en/param/wal_keep_size/) | Minimum size of WAL files to keep for replication | No | | -| [`max_slot_wal_keep_size`](https://postgresqlco.nf/doc/en/param/max_slot_wal_keep_size/) | Max WAL size that can be reserved by replication slots | No | | -| [`checkpoint_timeout`](https://postgresqlco.nf/doc/en/param/checkpoint_timeout/) | Max time between WAL checkpoints | No | | +| Setting | Description | Supabase configuration | +| ------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------- | +| [`max_replication_slots`](https://www.postgresql.org/docs/current/runtime-config-replication.html#GUC-MAX-REPLICATION-SLOTS) | Maximum number of replication slots. Pipelines needs one main slot and can temporarily use one additional slot per active table-sync worker. | CLI only | +| [`wal_keep_size`](https://www.postgresql.org/docs/current/runtime-config-replication.html#GUC-WAL-KEEP-SIZE) | Minimum amount of old WAL retained for standby servers. This setting is separate from the per-slot retention limit. | CLI only | +| [`max_slot_wal_keep_size`](https://www.postgresql.org/docs/current/runtime-config-replication.html#GUC-MAX-SLOT-WAL-KEEP-SIZE) | Maximum WAL that replication slots can retain at checkpoint time. `-1` lets slots retain unlimited WAL. A finite limit can invalidate a slot when its consumer falls too far behind. | CLI only | +| [`checkpoint_timeout`](https://www.postgresql.org/docs/current/runtime-config-wal.html#GUC-CHECKPOINT-TIMEOUT) | Maximum time between automatic WAL checkpoints. Slot WAL limits are enforced at checkpoint time. | CLI only | diff --git a/apps/docs/content/guides/database/replication/bigquery.mdx b/apps/docs/content/guides/database/replication/bigquery.mdx index b0c28364c64..e30dc09bd49 100644 --- a/apps/docs/content/guides/database/replication/bigquery.mdx +++ b/apps/docs/content/guides/database/replication/bigquery.mdx @@ -8,122 +8,97 @@ sidebar_label: 'BigQuery' <$Partial path="pipelines-public-alpha.mdx" /> -[BigQuery](https://cloud.google.com/bigquery) is Google's fully managed data warehouse. You can replicate your database tables to BigQuery for analytics and reporting. - -## Prepare GCP resources - -Before configuring BigQuery as a destination, set up the following in Google Cloud Platform: - -1. **Google Cloud Platform (GCP) account**: [Sign up for GCP](https://cloud.google.com/gcp) if you don't have one. In the destination project, make sure the [BigQuery API and BigQuery Storage API](https://cloud.google.com/bigquery/docs/service-dependencies) are enabled. - -2. **BigQuery dataset**: Create a [BigQuery dataset](https://cloud.google.com/bigquery/docs/datasets-intro) in your GCP project - - Open the BigQuery console in GCP - - Select your project - - Click **Create Dataset** - - Provide a dataset ID, for example `supabase_replication` - - Choose the [dataset location](https://cloud.google.com/bigquery/docs/locations) intentionally. Supabase Pipelines run in **AWS `eu-central-1` (Frankfurt)**, so choose the closest available BigQuery location to reduce network latency. You can't change a dataset's location after it is created, and Pipelines doesn't infer it from your Supabase project or copy Postgres partitioning settings. - -3. **GCP service account key**: Create a [service account](https://cloud.google.com/iam/docs/keys-create-delete) with appropriate permissions - - Go to **IAM & Admin > Service Accounts** - - Click **Create Service Account** - - Grant **BigQuery Data Editor** on the destination dataset - - Grant **BigQuery Job User** on the GCP project - - Create and download the JSON key file - - Treat the downloaded JSON as a secret. Don't commit or share it, and [rotate or revoke the key](https://cloud.google.com/iam/docs/key-rotation) if it is exposed. - -These roles provide the permissions Pipelines needs to inspect and manage destination tables, write data through the Storage Write API, and run BigQuery jobs. If you use a custom IAM role, it must provide: - -- `bigquery.datasets.get` -- `bigquery.jobs.create` -- `bigquery.tables.create` -- `bigquery.tables.delete` -- `bigquery.tables.get` -- `bigquery.tables.getData` -- `bigquery.tables.list` -- `bigquery.tables.update` -- `bigquery.tables.updateData` - -## Configure BigQuery as a destination - -1. Navigate to the [**Database > Replication**](/dashboard/project/_/database/replication) section of the Dashboard -2. Click **Add destination** - -3. Select **BigQuery** as the destination type - -4. Configure the destination details: - - **Destination name**: A name to identify this destination, for example "BigQuery Warehouse" - - **Publication**: The publication to replicate data from - - **Region**: Managed Pipelines run in the fixed **AWS `eu-central-1` (Frankfurt)** region. This is separate from your BigQuery dataset location and can't be changed. - -5. Configure the BigQuery settings: - - **Project ID**: Your BigQuery project identifier, found in the GCP Console - - **Dataset ID**: The name of your BigQuery dataset, without the project ID - - - - In the GCP Console, the dataset is shown as `project-id.dataset-id`. Enter only the part after the dot. For example, if you see `my-project.my_dataset`, enter `my_dataset`. - - - - - **Service Account Key**: Your GCP service account key in JSON format - -BigQuery destination form with pipeline details and BigQuery credentials - -6. Optionally expand **Advanced settings** for pipeline and BigQuery-specific tuning: - - The general batch, initial-sync concurrency, and invalidated-slot settings are described in [Set up Pipelines](/docs/guides/database/replication/pipelines#step-3-configure-a-destination). BigQuery adds these settings: - - | Setting | Default | Description | - | ------------------------ | ---------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | - | **Connection pool size** | `4` connections | Number of BigQuery Storage Write API connections used for destination writes. More connections can increase write throughput, but consume more pipeline and BigQuery resources. | - | **Maximum staleness** | Freshest results | Maximum acceptable staleness, in whole minutes, of table data returned by queries while BigQuery applies CDC `UPSERT` and `DELETE` changes in the background. Leave unset for the freshest table data. A larger number of minutes allows BigQuery to return older data, which can reduce query-time CDC merge cost and latency. For example, `15` allows data to be up to 15 minutes stale. This value is applied when Pipelines creates or recreates a table; changing it doesn't alter existing destination tables. | - -7. Review the [source table requirements](#source-table-requirements), then click **Create and start pipeline** to begin replication - -The pipeline begins the initial sync from your database to BigQuery. - -Supabase Pipelines charges and Google Cloud charges are separate. BigQuery can charge for Storage Write API ingestion, storage, and the compute used to apply CDC changes. See [BigQuery CDC pricing](https://cloud.google.com/bigquery/docs/change-data-capture#pricing). - -## How it works - -Once configured, replication to BigQuery: - -1. Captures the `INSERT`, `UPDATE`, `DELETE`, and `TRUNCATE` operations included by your Postgres publication -2. Optimizes delivery automatically -3. Creates destination tables from the replicated source schema using BigQuery-compatible names and types -4. Streams data to BigQuery - -Pipelines keeps a current-state table that you can query for each replicated source table and may replace its destination data during a truncate or new initial sync. It does not provide a history of every row version that you can query. +Replicate Postgres tables to [BigQuery](https://cloud.google.com/bigquery) for analytics. [Prepare Google Cloud resources](#prepare-gcp-resources), [configure the destination](#configure-bigquery-as-a-destination), then [query replicated data](#query-replicated-data). ## Source table requirements -BigQuery replication requires each source table to have a primary key, and the publication must include the primary-key columns. Pipelines declares those columns as the BigQuery destination primary key so BigQuery change data capture (CDC) can apply `UPSERT` and `DELETE` rows. +Each source table needs a primary key of at most 16 columns, all included in the publication. Pipelines declares it as a BigQuery `NOT ENFORCED` primary key so CDC can match upserts and deletes. Keep source keys unique and non-null. -BigQuery primary keys are `NOT ENFORCED`, and BigQuery change data capture (CDC) supports composite primary keys with up to 16 columns. Your source primary key must stay unique and non-null because BigQuery uses it to match CDC rows. +Check [replica identity and complete update rows](#replica-identity), especially for tables with large text or JSON values. -Source tables must also use a BigQuery-compatible Postgres `REPLICA IDENTITY` setting. Most tables can keep the Postgres default, as long as they have a primary key and all primary-key columns are included in the publication. +## Prepare GCP resources -| Source table setting | BigQuery support | Guidance | -| ------------------------------------------------ | ---------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `REPLICA IDENTITY DEFAULT` with a primary key | Supported | Recommended for most tables. BigQuery uses the replicated source primary key to apply upserts and deletes. | -| `REPLICA IDENTITY FULL` | Supported | Recommended for tables with large `text`, `jsonb`, `bytea`, or other values that Postgres may store out-of-line using TOAST, especially when those rows update. | -| `REPLICA IDENTITY USING INDEX` | Limited | Supported only when the selected unique index contains exactly the source primary-key columns. An alternative unique-key identity is not supported. | -| `REPLICA IDENTITY NOTHING` | Insert-only | Inserts can be replicated, but updates and deletes do not include enough row identity for BigQuery to apply them safely. | -| `REPLICA IDENTITY DEFAULT` without a primary key | Not supported | BigQuery requires a source primary key. | +Prepare these Google Cloud Platform (GCP) resources: -For a general explanation of how replica identity affects update and delete events, see [How does replica identity affect updates and deletes?](/docs/guides/database/replication/pipelines-faq#how-does-replica-identity-affect-updates-and-deletes). +1. **GCP account**: [Sign up for GCP](https://cloud.google.com/gcp) if you don't have one. In the destination project, make sure the [BigQuery API and BigQuery Storage API](https://cloud.google.com/bigquery/docs/service-dependencies) are enabled. -For updates, Postgres does not always send a complete old row through logical replication. It can also mark unchanged toasted values as `unchanged toast` instead of resending the value. BigQuery change data capture (CDC) upserts require a complete new row because omitted columns are not preserved in the destination. The replication pipeline can reconstruct a complete update when the old row image contains the missing value, which is reliable with `REPLICA IDENTITY FULL`. +2. **BigQuery dataset**: Create a [BigQuery dataset](https://docs.cloud.google.com/bigquery/docs/datasets) in your GCP project + - Use a dataset ID such as `supabase_replication` + - Choose a [dataset location](https://cloud.google.com/bigquery/docs/locations) near the [pipeline region](/docs/guides/database/replication/pipelines#region). You cannot change it after creation; it is independent of your Supabase project region. -If a BigQuery pipeline fails with an error about a partial update row, set `REPLICA IDENTITY FULL` on the affected source table and restart the pipeline. Changing replica identity only affects new WAL records, so a retained update that was written before the change may still need to be skipped by recreating the pipeline or restarting the affected table's initial sync. +3. **GCP service account key**: Create a [service account](https://docs.cloud.google.com/iam/docs/service-accounts-create) with appropriate permissions + - Grant **BigQuery Data Editor** on the destination dataset + - Grant **BigQuery Job User** on the GCP project + - [Create and download the JSON key file](https://cloud.google.com/iam/docs/keys-create-delete) + + Treat the downloaded JSON as a secret. Don't commit or share it, and [delete the key](https://cloud.google.com/iam/docs/keys-create-delete#deleting) if it is exposed. + +If you use a custom IAM role, see the [required permissions](#custom-iam-permissions). + +## Configure BigQuery as a destination + +Follow [Set up Pipelines](/docs/guides/database/replication/pipelines#setup-overview) to enable Pipelines, select **BigQuery**, and configure the publication and initial sync. Then enter: + +| Field | Value | +| ----------------------- | -------------------------------------------------------------------------------- | +| **Project ID** | The Google Cloud project identifier | +| **Dataset ID** | The dataset name without the project prefix: use `dataset` for `project.dataset` | +| **Service account key** | The downloaded service account JSON | + +Optionally adjust [destination settings](#destination-settings) and [table partitioning and clustering](#table-partitioning-and-clustering) before creation. + +Click **Create and start pipeline** and complete the validation and cost confirmations. + +Supabase Pipelines charges and Google Cloud charges are separate. BigQuery can charge for Storage Write API ingestion, storage, and the compute used to apply CDC changes. See [BigQuery CDC pricing](https://cloud.google.com/bigquery/docs/change-data-capture#CDC_pricing). + +## How it works + +Pipelines creates current-state BigQuery tables using destination-compatible names and types, then applies published inserts, updates, deletes, and truncates. These tables do not retain a history of row versions to query. Truncates and table restarts replace destination data. + +## Query replicated data + +Query the generated view for each source table. Its name combines the source schema and table with an underscore, doubling any existing underscores: `public.orders` becomes `public_orders`, and `my_schema.orders` becomes `my__schema_orders`. Pipelines manages versioned physical tables behind the view and updates its target after a truncate. Queries tied directly to a physical table version can become stale or fail when that version is removed. + +## Destination settings + +Expand **Advanced settings** for BigQuery-specific options: + +| Setting | Behavior | +| ------------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| **Connection pool size** | Default: `4` connections. Storage Write API connections for destination writes. More connections can improve throughput but use more resources. | +| **Maximum staleness** | Default: Freshest results. Maximum data age in whole minutes while BigQuery applies CDC changes. For example, `15` allows results up to 15 minutes stale and can reduce query-time merge cost. Unset gives the freshest results. Applies only when a table is created or recreated. | + +## Table partitioning and clustering + +You can configure BigQuery partitioning and clustering for individual replicated tables under **Advanced settings > Table layout** to control their physical layout and improve query performance and cost. + +The publication determines [which source partitions become destination tables](/docs/guides/database/replication/pipelines#partitioned-tables). BigQuery layout is configured separately; Pipelines does not copy Postgres partition keys or bounds. + +Layout settings apply only when a destination table is created or recreated, including after a table restart or source truncate. A pipeline restart that resumes a table's saved progress does not apply new layout settings. [Restart replication for the table](/docs/guides/database/replication/pipelines-monitoring#restarting-tables) to apply them, which replaces its data. + +Set either option, both, or neither: + +- **Partitioning**: Partition by a `date`, `timestamp`, or `timestamptz` column with `hour`, `day`, `month`, or `year` granularity, by an integer range, or by ingestion time. Date columns cannot use hourly granularity; integer ranges need a start, end, and interval. +- **Clustering**: Cluster by one to four ordered, distinct replicated columns. BigQuery validates whether the clustering column types are supported. + +See the BigQuery documentation for [partition expressions](https://cloud.google.com/bigquery/docs/reference/standard-sql/data-definition-language#partition_expression) and [clustering column requirements](https://cloud.google.com/bigquery/docs/creating-clustered-tables#clustered_column_requirements). + +## Replica identity + +Choose a supported Postgres replica identity: + +| Source table setting | Guidance | +| --------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `REPLICA IDENTITY DEFAULT` with a primary key | Recommended for most tables. BigQuery uses the replicated source primary key to apply upserts and deletes. | +| `REPLICA IDENTITY FULL` | Recommended for tables with large `text`, `jsonb`, `bytea`, or other values that Postgres may store out-of-line using TOAST, especially when those rows update. | +| `REPLICA IDENTITY USING INDEX` | Supported only when the selected unique index contains exactly the source primary-key columns. An alternative unique-key identity is not supported. | +| `REPLICA IDENTITY NOTHING` | Insert-only. Inserts can be replicated, but updates and deletes do not include enough row identity for BigQuery to apply them safely. | + +### Complete update rows and TOAST + +BigQuery upserts require complete new rows. Postgres can omit unchanged out-of-line TOAST values; `REPLICA IDENTITY FULL` supplies the old row so Pipelines can reconstruct them. + +If replication fails on a partial update row, set full replica identity and restart the pipeline. The change affects only new WAL: incompatible retained updates can still require a [table restart](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). Check a table's current replica identity: @@ -146,48 +121,69 @@ Set full replica identity when a table has toasted columns and update replicatio alter table public.your_table replica identity full; ``` -`REPLICA IDENTITY FULL` increases WAL volume because Postgres logs the full old row for updates and deletes. Use it on tables where update correctness is more important than the extra replication overhead. +`REPLICA IDENTITY FULL` increases WAL volume by logging the complete old row for updates and deletes. + +## Column names + +Pipelines converts ASCII uppercase letters to lowercase and preserves other supported characters. For example, `Name` and `name` conflict; `Ä` and `ä` remain distinct. Avoid ASCII case-only differences. + +BigQuery supports [flexible column names](https://cloud.google.com/bigquery/docs/schemas#flexible-column-names), including spaces, Unicode letters, and selected punctuation. Its reserved prefixes and unsupported-character restrictions still apply. ## Schema change support -Schema change support for BigQuery is currently in beta. Pipelines supports a limited set of schema changes for BigQuery while the feature is developed further. +Pipelines supports: -Supported schema changes: +- Adding columns: scalar columns are nullable, and arrays are repeated fields +- Removing or renaming columns, provided the primary key stays unchanged +- Dropping `NOT NULL` from an existing scalar column +- Adding, replacing, or removing supported literal defaults +- Adding or removing published columns on tracked tables, provided the primary key stays unchanged and no existing array column is newly included -- Adding a scalar, top-level column (created as `NULLABLE` in BigQuery) -- Removing a column -- Renaming a column -- Dropping a `NOT NULL` constraint -- Setting or dropping supported column default metadata +New scalar columns leave historical BigQuery rows `NULL`; new array columns expose an empty array. Later row changes supply the source value. Pipelines does not backfill existing rows when adding a column. -Unsupported or limited schema changes: +Removing a column from the publication also removes its destination values. Adding it again does not restore those values. To include a previously excluded array column, select the table for initial sync and [restart its replication](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). -- Changing a column's data type -- Adding `NOT NULL` with `SET NOT NULL` -- Filling existing rows for `ADD COLUMN ... DEFAULT` -- Unsupported default expressions +Adding `NOT NULL` keeps an existing destination column nullable. For type changes, unsupported changes, and interrupted schema changes, see the shared [schema-change behavior and recovery](/docs/guides/database/replication/pipelines#schema-change-support). -When the initial sync creates a BigQuery table, Pipelines preserves whether each scalar, non-array source column allows `NULL`: Postgres `NOT NULL` columns become `REQUIRED`, and nullable columns become `NULLABLE`. BigQuery represents Postgres arrays as `REPEATED` fields instead of using `REQUIRED` or `NULLABLE` mode. +### Column defaults -After the table exists, BigQuery requires every newly added scalar, top-level column to be `NULLABLE`. If Postgres adds a `NOT NULL` column, Pipelines adds it as `NULLABLE` in BigQuery and logs a warning. Postgres remains the source of truth and rejects new `NULL` values before they reach the destination. +Pipelines supports literal defaults such as strings, numbers, dates, timestamps, JSON values, and UUIDs. It skips defaults that depend on when or where they run, including `now()`, sequences, `random()`, and generated UUID functions. -For Postgres `DROP NOT NULL`, Pipelines relaxes an existing BigQuery column from `REQUIRED` to `NULLABLE`. For Postgres `SET NOT NULL`, BigQuery cannot change an existing `NULLABLE` column to `REQUIRED`, so Pipelines leaves the destination column nullable and logs a warning. +Skipping a BigQuery default doesn't lose values from Postgres. Postgres evaluates the default, and Pipelines replicates the resulting value with future row changes. Pipelines adds a column before setting its default, so [defaults don't fill existing BigQuery rows](https://cloud.google.com/bigquery/docs/default-values#change_default_values). -Column defaults are handled independently from whether the column allows `NULL`. BigQuery does not support `ADD COLUMN ... DEFAULT` on an existing table, so Pipelines first adds the nullable column and then applies supported default metadata with a separate statement. This is destination metadata for future BigQuery writes that omit the column; pipeline writes already contain the value evaluated by Postgres, and the metadata doesn't populate existing destination rows. Unsupported defaults, including defaults on destination primary-key columns, are skipped with a warning instead of failing replication. +### Publication changes + +Column-list changes for tracked tables follow the [schema rules](#schema-change-support). Follow the guidance for [table membership](/docs/guides/database/replication/pipelines#adding-or-removing-tables) or [other publication changes](/docs/guides/database/replication/pipelines#other-publication-changes), including row filters and partition behavior. ## Limitations -- **Row size**: Limited to 10 MB per row due to BigQuery Storage Write API constraints -- **Primary keys**: Source tables must have a primary key, the replicated primary key can contain at most 16 columns, and BigQuery does not enforce key uniqueness +- **Row size**: Each serialized row must fit within the Storage Write API's [20 MB append-request limit](https://cloud.google.com/bigquery/quotas#write-api-limits), including request metadata and encoding overhead. - **Columns**: BigQuery CDC supports at most 2,000 top-level columns -- **Replica identity**: Updates and deletes require a supported primary-key identity or `REPLICA IDENTITY FULL` - **Schema and table names**: Source schema and table names can't start or end with `_` or contain `"` or `;` when replicating to BigQuery -- **Arrays and numeric values**: Arrays can't contain `NULL` elements. Numeric values with more than 38 fractional digits and exact JSON integer values outside BigQuery's supported range can't be replicated. -- **BigQuery CDC tables**: While CDC is active, BigQuery doesn't support mutating DML (`UPDATE`, `DELETE`, or `MERGE`), wildcard table queries, or search indexes on the destination table. See [BigQuery CDC limitations](https://cloud.google.com/bigquery/docs/change-data-capture#limitations) for the complete list. -- **Managed destination objects**: Don't delete or modify tables or views created by Pipelines. Doing so can stop replication and may require a new, billable initial sync. -- **Schema changes**: Limited to the supported schema changes listed above +- **Arrays**: Arrays can't contain `NULL` elements. +- **Numeric and JSON values**: BigQuery applies its destination [data-type limits](https://cloud.google.com/bigquery/docs/reference/standard-sql/data-types). Values can be rounded or rejected when they exceed the supported precision or range; Postgres values are not guaranteed to retain their exact representation. +- **BigQuery CDC tables**: While CDC is active, BigQuery doesn't support mutating DML such as `UPDATE`, `DELETE`, or `MERGE`, wildcard table queries, or search indexes on the destination table. See [BigQuery CDC limitations](https://cloud.google.com/bigquery/docs/change-data-capture#limitations) for the complete list. +- **Managed objects**: Follow the [removal procedure](/docs/guides/database/replication/pipelines#removing-tables-from-replication) before deleting destination tables or views. + +## Custom IAM permissions + +Pipelines needs permission to inspect and manage destination tables, write data through the Storage Write API, and run BigQuery jobs. A custom IAM role must provide: + +- `bigquery.datasets.get` +- `bigquery.jobs.create` +- `bigquery.tables.create` +- `bigquery.tables.delete` +- `bigquery.tables.get` +- `bigquery.tables.getData` +- `bigquery.tables.list` +- `bigquery.tables.update` +- `bigquery.tables.updateData` + +## Troubleshooting + +Use [pipeline monitoring](/docs/guides/database/replication/pipelines-monitoring) to inspect errors. For update failures, check [replica identity and TOAST](#complete-update-rows-and-toast). For schema failures, review [schema-change behavior and recovery](/docs/guides/database/replication/pipelines#schema-change-support). ## Additional resources -- [BigQuery documentation](https://cloud.google.com/bigquery/docs) - Official Google BigQuery documentation -- [BigQuery change data capture](https://cloud.google.com/bigquery/docs/change-data-capture) - BigQuery change data capture (CDC) requirements and limitations +- [BigQuery documentation](https://cloud.google.com/bigquery/docs) +- [BigQuery change data capture](https://cloud.google.com/bigquery/docs/change-data-capture) diff --git a/apps/docs/content/guides/database/replication/clickhouse.mdx b/apps/docs/content/guides/database/replication/clickhouse.mdx index 4637bd5c25f..a51eaf1033b 100644 --- a/apps/docs/content/guides/database/replication/clickhouse.mdx +++ b/apps/docs/content/guides/database/replication/clickhouse.mdx @@ -10,13 +10,28 @@ sidebar_label: 'ClickHouse' {/* supa-mdx-lint-disable Rule003Spelling */} - +The ClickHouse destination is in private alpha and available only to approved organizations. [Request access](/go/supabase-pipelines-new-destinations) before following this guide. -The ClickHouse destination is in Early Access and available only to approved organizations. [Request access](/go/supabase-pipelines-new-destinations) before following this guide. +Replicate Postgres changes to [ClickHouse](https://clickhouse.com/) as current-state tables or an append-only history. [Choose a table engine](#choose-a-table-engine), [prepare resources](#prepare-clickhouse-resources), then [configure the destination](#configure-clickhouse-as-a-destination). - +## Source table requirements -[ClickHouse](https://clickhouse.com/) is a column-oriented database for analytics. Supabase Pipelines can maintain current-state tables in ClickHouse or write an append-only change history, depending on the selected table engine. +`ReplacingMergeTree` requires a source primary key. `MergeTree` can replicate insert-only tables without one. Include all primary-key columns in the publication when the table has a key. + +Check the [replica-identity and array requirements](#replica-identity-and-arrays) for your source tables. + +### Choose a table engine + +The table engine controls how ClickHouse represents changes. It is selected for the entire destination. + +Choose the engine before creating the pipeline. Changing **Table engine** later does not convert existing destination tables; writes fail if their engine differs from the configured one. Restore the previous setting to resume using those tables. + +| Engine | Data model and query pattern | +| -------------------- | ------------------------------------------------------------------------------------------------ | +| `ReplacingMergeTree` | Current-state tables. Requires a primary key. Query the generated `__current` view. | +| `MergeTree` | Append-only CDC history. A primary key is optional for insert-only tables. Query the base table. | + +Updating a source primary-key value removes the old key from the current-state view and writes the row under its new key. Changing the primary-key definition is a separate [schema change](#schema-change-support). {/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} @@ -35,27 +50,29 @@ Before creating the destination: Keep the database otherwise empty. Pipelines manages the replicated tables and current-state views. Don't pre-create or manually alter those objects. -The default `ReplacingMergeTree` engine requires ClickHouse **23.5 or newer**. The `MergeTree` event-log engine does not have this minimum-version requirement. +The default `ReplacingMergeTree` engine requires ClickHouse 23.5 or later. The `MergeTree` event-log engine does not have this minimum-version requirement. {/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} ## Configure ClickHouse as a destination -1. Open [**Database > Replication**](/dashboard/project/_/database/replication) -2. Click **Add destination** -3. Select **ClickHouse**. If it isn't available, [request Early Access](/go/supabase-pipelines-new-destinations). -4. Select a Postgres publication and enter a destination name -5. Enter the ClickHouse settings: - - **URL**: The HTTPS endpoint, including its port when required - - **User**: The dedicated ClickHouse user - - **Password**: The user's password, if authentication requires one - - **Database**: The existing target database - - **Table engine**: Choose **ReplacingMergeTree** for current-state tables or **MergeTree** for an append-only event log -6. Click **Create and start pipeline** +Follow [Set up Pipelines](/docs/guides/database/replication/pipelines#setup-overview) and select **ClickHouse**. Enter these destination settings: -Managed Pipelines run in **AWS `eu-central-1` (Frankfurt)**. When possible, place the ClickHouse service close to Frankfurt to reduce network latency and replication lag. +| Field | Value | +| ---------------- | -------------------------------------------------------------------------- | +| **URL** | Public HTTPS endpoint, including its port when required | +| **User** | Dedicated ClickHouse user | +| **Password** | User's password, if required | +| **Database** | Existing target database | +| **Table engine** | `replacing_merge_tree` for current state or `merge_tree` for event history | -## How table names are mapped +Click **Create and start pipeline** and complete the validation and cost confirmations. + +Place ClickHouse near the [managed pipeline region](/docs/guides/database/replication/pipelines#region). + +## Query replicated data + +### How table names are mapped Pipelines maps each Postgres schema and table pair to one ClickHouse table. Existing underscores are doubled, and the schema and table names are joined with one underscore: @@ -66,15 +83,6 @@ Pipelines maps each Postgres schema and table pair to one ClickHouse table. Exis Postgres schema and table names cannot start or end with `_` or contain `"` or `;` when replicating to ClickHouse. -## Choose a table engine - -The table engine controls how ClickHouse represents changes. It is selected for the entire destination. - -| Engine | Data model | Source primary key | Query pattern | -| ------------------------------ | ----------------------------- | ------------------------------------------------------- | ------------------------------------------- | -| `ReplacingMergeTree` (default) | Current-state tables | Required | Query the generated `__current` view | -| `MergeTree` | Append-only CDC event history | Optional for insert-only tables; see requirements below | Query the base table | - ### ReplacingMergeTree `ReplacingMergeTree` is the default and is intended for current-state analytics. Pipelines: @@ -86,8 +94,6 @@ The table engine controls how ClickHouse represents changes. It is selected for The `_etl_version` and `_etl_deleted` names are reserved and can't be used by source columns. -During Early Access, source primary-key values must remain immutable when using `ReplacingMergeTree`. Updating a primary-key value can leave the old key visible in the generated current-state view. This limitation will be removed when primary-key update handling is available. - Query the generated view for the current state: ```sql @@ -105,30 +111,34 @@ Pipelines does not run `OPTIMIZE ... FINAL CLEANUP`. ClickHouse operators remain - `cdc_operation`, containing `INSERT`, `UPDATE`, or `DELETE` - `cdc_lsn`, containing the Postgres commit LSN for the change +- `cdc_tx_ordinal`, containing the change's position within that transaction -The `cdc_operation` and `cdc_lsn` names are reserved and can't be used by source columns. +The `cdc_operation`, `cdc_lsn`, and `cdc_tx_ordinal` names are reserved and can't be used by source columns. -Read the base table to analyze the event history. Multiple changes committed in one Postgres transaction can share the same `cdc_lsn`, so it is not a unique event ID or a total ordering for reconstructing current state. +Inserts and updates append the complete new row. A primary-key value update also appends a `DELETE` for the old key. Deletes append the old row when the source uses `REPLICA IDENTITY FULL`. With primary-key identity, a delete contains the key values; other fields use `NULL` for nullable scalars or placeholders such as zero, empty strings, and empty arrays. Those placeholders are not the deleted row's original values. -A source `TRUNCATE` truncates the ClickHouse table for either engine. Resetting a table also drops and recreates its table and, for `ReplacingMergeTree`, its generated view. These operations erase the destination data accumulated for that table before a new initial sync begins. +Read the base table to analyze the event history. Order source changes by `cdc_lsn` and then `cdc_tx_ordinal`. The old-key delete and new-key update from one primary-key value change share both values, so this pair is not a unique destination-row ID. -## Source table requirements +### Truncates and table restarts -ClickHouse requirements depend on the engine and the operations published for a table. +A source `TRUNCATE` truncates the ClickHouse table for either engine, then ongoing replication continues. It does not start a new initial sync. -| Source table and publication | Support | Guidance | -| --------------------------------------------------- | ------- | ------------------------------------------------------------------------------------------------------------------------------------------- | -| `ReplacingMergeTree` table without a primary key | No | Add a source primary key, include all primary-key columns in the publication, or use `MergeTree` for an event-log layout. | -| Insert-only `MergeTree` table without a primary key | Yes | No row identity is required for inserts. | -| Table with a primary key | Yes | Include every primary-key column in the publication. | -| Updates with primary-key replica identity | No | Use `REPLICA IDENTITY FULL` so unchanged out-of-line values can be reconstructed. | -| Deletes with primary-key replica identity | Yes | The publication must include every primary-key column. | -| Updates or deletes with `REPLICA IDENTITY FULL` | Yes | Full identity provides the complete row image required for updates. | -| Updates with `REPLICA IDENTITY USING INDEX` | No | Use `REPLICA IDENTITY FULL`. | -| Deletes with `REPLICA IDENTITY USING INDEX` | Limited | Supported only when the selected index resolves to the same columns as the source primary key. Alternative unique indexes aren't supported. | -| Updates or deletes with `REPLICA IDENTITY NOTHING` | No | Configure primary-key or full identity for deletes and full identity for updates. | +Restarting replication for a table drops and recreates its table and, for `ReplacingMergeTree`, its generated view. A table restart erases the accumulated destination data and copies existing source rows only if the table is [selected for initial sync](/docs/guides/database/replication/pipelines#choosing-which-tables-to-copy). -Top-level Postgres array columns can contain arrays with nullable elements, but the array column itself must not contain `NULL`. ClickHouse's RowBinary format cannot encode a top-level `NULL` array. Replace existing `NULL` values and make the source array column `NOT NULL`, or ensure producers always write an array value. Empty arrays remain supported. +## Replica identity and arrays + +| Published operation | Required replica identity | +| ------------------------------------------ | --------------------------------------------------------------------------------------- | +| Insert | None; `ReplacingMergeTree` still requires a primary key | +| Update | Primary-key identity or `REPLICA IDENTITY FULL`; updates must contain complete new rows | +| Delete | Primary-key identity or `REPLICA IDENTITY FULL` | +| Delete with `REPLICA IDENTITY USING INDEX` | The index must contain exactly the source primary-key columns | + +`REPLICA IDENTITY NOTHING` cannot support updates or deletes. + +Use `REPLICA IDENTITY FULL` for tables whose updates can omit unchanged out-of-line TOAST values. It lets Pipelines reconstruct the complete new row. Changing replica identity affects only new WAL; incompatible retained updates can still require a [table restart](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). + +Array elements can be `NULL`, but a top-level array value cannot. Replace top-level `NULL` values and enforce `NOT NULL`, or ensure producers always supply an array. Empty arrays are supported. ## Type mapping @@ -154,40 +164,32 @@ Nullable scalar columns are wrapped in `Nullable(...)`. Character, text, `numeri ## Schema change support -ClickHouse schema change support is limited during Early Access. - -Supported changes include: +Pipelines supports: - Adding columns -- Renaming columns, except moving a nested subcolumn to a different parent -- Dropping columns +- Renaming or dropping columns; nested subcolumns must stay under the same parent when renamed - Dropping `NOT NULL` from an existing scalar column - Adding, changing, or removing supported column defaults +- Adding or removing published columns on tracked tables, subject to the same restrictions -The following changes are not applied automatically: +With `ReplacingMergeTree`, renaming or dropping a primary-key column or changing the key's columns or their order is rejected. -- Changing a column's data type -- Adding `NOT NULL` to an existing nullable column -- Changing, dropping, or renaming a source primary-key column when using `ReplacingMergeTree` -- Renaming a source table or schema +New scalar columns are made nullable when ClickHouse needs a value for historical rows and the source default cannot be represented safely. Adding `NOT NULL` keeps an existing destination column nullable. Defaults that cannot be translated are skipped; Postgres still supplies the source values through replication. -New scalar columns are made nullable when ClickHouse needs a value for historical rows and the source default cannot be represented safely. Some Postgres defaults cannot be translated to ClickHouse and are skipped with a warning. +A previously excluded scalar column is added without a default, leaving historical rows `NULL`. Removing a published column drops its destination values; adding it again does not restore them. To include a previously excluded array column, select the table for initial sync and [restart its replication](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). -ClickHouse DDL is not transactional. An interrupted multi-column schema change can leave a partially applied destination schema. Don't repair managed tables or views manually. If the pipeline remains failed after a restart, [contact support](/dashboard/support/new). +For type changes, unsupported changes, and interrupted schema changes, see the shared [schema-change behavior and recovery](/docs/guides/database/replication/pipelines#schema-change-support). ## Troubleshooting -| Issue | Resolution | -| -------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| ClickHouse isn't available in the destination list | The destination is organization-gated during Early Access. [Request access](/go/supabase-pipelines-new-destinations). | -| URL validation fails | Use the public ClickHouse HTTPS endpoint, including its port when required. HTTP, localhost, and private or internal endpoints are not supported. | -| Connection fails | Confirm the endpoint is reachable from the internet and that the username and password are correct. | -| Database validation fails | Create the configured database and grant the user permission to read `system.databases`. | -| Validation succeeds but table setup or writes fail | Grant the user the target-database permissions listed above. Check for a pre-existing table or view with the generated name, and don't manually modify managed objects. | -| `ReplacingMergeTree` initialization fails | Confirm the server is ClickHouse 23.5 or newer and every source table has a published primary key. | -| Updates or deletes fail | Use `REPLICA IDENTITY FULL` for updates. Deletes can use primary-key identity or full identity. Include all identity columns in the publication. | -| A nullable array fails to replicate | Replace top-level `NULL` array values and make the source column `NOT NULL`, or ensure producers always write an array. Empty arrays are supported. | -| A schema change fails | Check the supported changes above. ClickHouse DDL can be partially applied, so don't repair managed objects manually. [Contact support](/dashboard/support/new) with the pipeline ID and error details. | +| Issue | Resolution | +| ------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------- | +| URL or connection fails | Use a reachable public HTTPS endpoint with the correct port and credentials. Private endpoints are unsupported. | +| Validation passes but setup or writes fail | Check [permissions and resource requirements](#prepare-clickhouse-resources), including table ownership and the server version. | +| Updates, deletes, or arrays fail | Check the [source table requirements](#source-table-requirements). | +| A schema change fails | Follow the shared [schema-change behavior and recovery](/docs/guides/database/replication/pipelines#schema-change-support). | + +Use [pipeline monitoring](/docs/guides/database/replication/pipelines-monitoring) to inspect errors and include the pipeline ID when [contacting support](/dashboard/support/new). ## Additional resources diff --git a/apps/docs/content/guides/database/replication/ducklake.mdx b/apps/docs/content/guides/database/replication/ducklake.mdx index 343d455b7b7..ecaf97319ac 100644 --- a/apps/docs/content/guides/database/replication/ducklake.mdx +++ b/apps/docs/content/guides/database/replication/ducklake.mdx @@ -10,19 +10,19 @@ sidebar_label: 'DuckLake' {/* supa-mdx-lint-disable Rule003Spelling */} - +The DuckLake destination is in private alpha and available only to approved organizations. [Request access](/go/supabase-pipelines-new-destinations) before following this guide. -The DuckLake destination is in Early Access and available only to approved organizations. [Request access](/go/supabase-pipelines-new-destinations) before following this guide. +Replicate Postgres tables to [DuckLake](https://ducklake.select/) for current-state lakehouse queries. [Prepare resources](#understand-the-ducklake-components), [configure the destination](#choose-a-configuration-mode), then [query replicated data](#query-the-destination). - +## Source table requirements -[DuckLake](https://ducklake.select/) is an open lakehouse format that stores metadata in a SQL catalog and table data in object storage. Supabase Pipelines keeps current-state DuckLake tables synchronized with published Postgres tables. +Insert-only tables don't require a primary key or replica identity. Updates and deletes require a published Postgres row identity. See [supported replica identities](#replica-identity). {/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} -## Understand the DuckLake components +## Prepare DuckLake resources [#understand-the-ducklake-components] -A DuckLake destination has three independent components: +Prepare a Postgres catalog, object storage, and a compatible query engine: | Component | Purpose | | ---------------- | ----------------------------------------------------------------------------------------------------------------------------------- | @@ -30,13 +30,15 @@ A DuckLake destination has three independent components: | Object storage | Stores Parquet data and delete files under an `s3://` path. | | Query engine | Reads the catalog and object storage. Pipelines doesn't include a DuckLake query endpoint; use DuckDB or another compatible engine. | -Always query through the DuckLake catalog. Reading the Parquet objects directly can miss inlined changes, delete files, and the current snapshot selected by the catalog. - You can query replicated tables, but treat them and their underlying catalog and object-storage state as read-only. Writes outside Pipelines can conflict with replication and background maintenance. -## Choose a configuration mode +{/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} -The Dashboard offers two ways to provide the catalog and storage. +## Configure DuckLake as a destination [#choose-a-configuration-mode] + +Choose a mode below for its resource requirements and configuration steps. + +Follow [Set up Pipelines](/docs/guides/database/replication/pipelines#setup-overview), select **DuckLake**, then choose **Use Supabase** or **Custom parameters** for the catalog and storage. ### Use Supabase @@ -48,27 +50,16 @@ Before you begin: - Make sure your organization role can administer SQL in the catalog project and Storage in the storage project. - Create a private standard Storage bucket, or create one from the destination form. - Choose a metadata schema unique to this DuckLake. Use only letters, numbers, and underscores. -- Keep the catalog project and storage project in the same region when possible. Managed Pipelines run in **AWS `eu-central-1` (Frankfurt)**, so resources near Frankfurt reduce network latency. +- Keep catalog and storage in the same region when possible, near the [managed pipeline region](/docs/guides/database/replication/pipelines#region). To configure the destination: -1. Open [**Database > Replication**](/dashboard/project/_/database/replication) -2. Click **Add destination** -3. Select **DuckLake**. If it isn't available, [request Early Access](/go/supabase-pipelines-new-destinations). -4. Select a Postgres publication and enter a destination name -5. Select **Use Supabase** -6. Configure the catalog: - - **Catalog project**: The project whose Postgres database stores DuckLake metadata - - **Pool size**: The number of concurrent DuckDB connections to the catalog, from `1` to `6`. The default is `4`. - - **Metadata schema**: A unique Postgres schema for this DuckLake's metadata -7. Configure object storage: - - **Storage project**: The project whose Storage service holds DuckLake files - - **Bucket**: A dedicated private bucket for the DuckLake -8. Click **Create and start pipeline** +1. Select **Use Supabase**. +2. Choose the **Catalog project**, **Pool size**, and **Metadata schema**. Pool size allows `1` to `6` concurrent DuckDB connections; the default is `4`. +3. Choose the **Storage project** and private **Bucket**. +4. Click **Create and start pipeline** and complete the validation and cost confirmations. -Validation can show warnings that Supabase catalog and Storage credentials are provisioned only when the destination is saved. This is expected for **Use Supabase** mode. Review the selected projects and proceed to create the destination. - -Pipelines-generated credentials remain hidden and are only for the managed writer. Create separate read credentials when connecting an external query engine. +Credential-provisioning warnings are expected before creation: catalog and Storage credentials are provisioned when you save the destination. Review the selected resources before proceeding. ### Custom parameters @@ -94,11 +85,21 @@ Configure these fields in the destination form: - **Use SSL**: Keep enabled for production endpoints - **Metadata schema**: A unique Postgres schema for DuckLake metadata, using only letters, numbers, and underscores +Click **Create and start pipeline** and complete the validation and cost confirmations. + The catalog URL and storage credentials are stored as secrets and aren't returned after creation. When editing the destination, leave a secret field empty to keep its stored value, or enter a new value to replace it. -## Query the destination +## How replication works -Pipelines manages replication but doesn't provide query compute for DuckLake during Early Access. Connect a DuckLake-compatible engine, such as DuckDB with its `ducklake` extension, to the same Postgres catalog and object-storage path. +Pipelines creates current-state tables, copies rows according to the [initial sync selection](/docs/guides/database/replication/pipelines#choosing-which-tables-to-copy), then applies published changes and supported schema changes. + +Source schema and table names are preserved, while ASCII uppercase letters in column names are converted to lowercase. DuckDB compares schema, table, and column identifiers without ASCII case distinctions, so don't publish names that differ only by ASCII case. Use distinct lowercase column names. Source column names matching the generated `supabase_etl_ducklake_dropped__` shape are reserved for schema-change recovery. + +A source `TRUNCATE` truncates the DuckLake table. A [table restart](/docs/guides/database/replication/pipelines-monitoring#restarting-tables) drops and recreates it. [Removing a table from the publication](/docs/guides/database/replication/pipelines#removing-tables-from-replication) leaves its destination data in place. + +## Query replicated data [#query-the-destination] + +Connect DuckDB with its `ducklake` extension, or another compatible engine, to the same catalog and storage path. Query through the catalog; reading raw Parquet files can miss inlined changes, delete files, and the current snapshot. For **Use Supabase** mode, create separate read credentials for the selected catalog and Storage projects. The writer credentials generated for Pipelines aren't exposed. For **Custom parameters**, use separate read-only credentials when your catalog and storage provider support them. @@ -109,34 +110,19 @@ select * from my_ducklake.public.orders; ``` -See the [DuckDB connection guide](https://ducklake.select/docs/stable/duckdb/usage/connecting) for the current `ducklake` extension and `ATTACH` syntax. If your query client can't read a Supabase-backed catalog during Early Access, [contact support](/dashboard/support/new). +See the [DuckDB connection guide](https://ducklake.select/docs/stable/duckdb/usage/connecting) for the current `ducklake` extension and `ATTACH` syntax. If your query client can't read a Supabase-backed catalog during private alpha, [contact support](/dashboard/support/new). -## How replication works +## Replica identity -For each published Postgres table, Pipelines: +| Source table setting | Guidance | +| ------------------------------------------------ | ---------------------------------------------------------------------------------------------------------------------- | +| `REPLICA IDENTITY DEFAULT` with a primary key | Include every primary-key column in the publication. | +| `REPLICA IDENTITY USING INDEX` | Include every column from the replica-identity index in the publication. | +| `REPLICA IDENTITY FULL` | Use when the table has no suitable key or the full old row is required. This increases source WAL volume. | +| `REPLICA IDENTITY NOTHING` | Insert-only. Inserts replicate, but updates and deletes don't contain an identity that DuckLake can match safely. | +| `REPLICA IDENTITY DEFAULT` without a primary key | Insert-only. Add a key, configure a replica-identity index, or use full identity before publishing updates or deletes. | -1. Creates the corresponding schema and table in DuckLake -2. Copies existing rows during the initial sync -3. Applies subsequent inserts, updates, deletes, and truncates to the current-state table -4. Applies supported source schema changes - -Source schema and table names are preserved, while ASCII uppercase letters in column names are converted to lowercase. DuckDB compares schema, table, and column identifiers without ASCII case distinctions, so don't publish names that differ only by ASCII case. Use distinct lowercase column names. Source column names matching the generated `supabase_etl_ducklake_dropped__` shape are reserved for schema-change recovery. - -A source `TRUNCATE` truncates the DuckLake table. Resetting a table drops and recreates the DuckLake table before running a new initial sync. Removing a source table from the publication stops new changes after the pipeline restarts but leaves the destination table in place. - -## Source table requirements - -Insert-only tables don't require a primary key or replica identity. Updates and deletes require a published Postgres row identity. - -| Source table setting | DuckLake support | Guidance | -| ------------------------------------------------ | ---------------- | --------------------------------------------------------------------------------------------------------- | -| `REPLICA IDENTITY DEFAULT` with a primary key | Supported | Include every primary-key column in the publication. | -| `REPLICA IDENTITY USING INDEX` | Supported | Include every column from the replica-identity index in the publication. | -| `REPLICA IDENTITY FULL` | Supported | Use when the table has no suitable key or the full old row is required. This increases source WAL volume. | -| `REPLICA IDENTITY NOTHING` | Insert-only | Inserts replicate, but updates and deletes don't contain an identity that DuckLake can match safely. | -| `REPLICA IDENTITY DEFAULT` without a primary key | Insert-only | Add a key, configure a replica-identity index, or use full identity before publishing updates or deletes. | - -Changing replica identity only affects new WAL records. A retained update or delete written before the change can continue to fail after restart and may require recreating the pipeline or restarting the affected table's initial sync. +Replica-identity changes affect only new WAL. If retained updates or deletes still fail after a pipeline restart, [restart replication for the affected table](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). ## Type mapping @@ -164,39 +150,35 @@ Pipelines creates DuckLake columns with these mappings: DuckDB decimals support precision from `1` to `38` and a scale between `0` and the precision. Postgres `numeric` values declared outside that range, unconstrained `numeric`, and numeric types with unsupported modifiers are stored as `varchar` to preserve their serialized value. +Numeric arrays use `varchar[]`, even when their declared precision and scale would fit a scalar DuckLake decimal. + ## Schema change support -DuckLake schema change support is limited during Early Access. +Pipelines supports: -Supported changes include: - -- Adding columns -- Renaming columns -- Dropping columns +- Adding, renaming, or dropping columns - Dropping `NOT NULL` from an existing column - Adding, changing, or removing supported column defaults +- Adding or removing published columns on tracked tables -The following changes are not applied automatically: +New columns are created as nullable so existing destination rows remain valid. Adding `NOT NULL` keeps an existing destination column nullable. Supported defaults are stored in DuckLake metadata; other defaults are skipped. Postgres still supplies the source values through replication. -- Changing a column's data type -- Adding `NOT NULL` to an existing nullable column -- Renaming a source table or schema +Previously excluded columns are added without a default, leaving historical rows `NULL`. Removing a published column drops its destination values; adding it again does not restore them. -New columns are created as nullable so existing destination rows remain valid. If a source default can be represented safely, it is stored as DuckLake metadata; unsupported defaults are skipped with a warning. DuckLake applies each planned multi-column schema change in a transaction. +For type changes, unsupported changes, and interrupted schema changes, see the shared [schema-change behavior and recovery](/docs/guides/database/replication/pipelines#schema-change-support). ## Troubleshooting -| Issue | Resolution | -| ------------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| DuckLake isn't available in the destination list | The destination is organization-gated during Early Access. [Request access](/go/supabase-pipelines-new-destinations). | -| Supabase mode shows credential-provisioning warnings | This is expected before creation. Confirm your selected projects, bucket, and metadata schema, then proceed. Credentials are provisioned when the destination is saved. | -| Catalog URL validation fails | Use a valid `postgres://` or `postgresql://` URL with credentials and an `sslmode` appropriate for your provider. Confirm the catalog is reachable from managed Pipelines. | -| Data path validation fails | Use an `s3:///` URL. Local `file://` paths aren't supported. | -| S3 validation fails | Provide both non-empty keys, the correct region, an endpoint without a protocol scheme, the correct URL style, and the provider's SSL setting. Confirm the credentials cover the configured prefix. | -| Metadata schema already exists | For a new DuckLake, choose another schema. Proceed with an existing schema only when you intentionally want to reuse that DuckLake and its corresponding data path. | -| Inserts work but updates or deletes fail | Configure a primary-key identity, replica-identity index, or `REPLICA IDENTITY FULL`, and include every identity column in the publication. | -| An external query omits recent changes or deleted rows | Attach and query the DuckLake catalog instead of scanning raw Parquet files. Confirm that the query engine has access to both the Postgres catalog and the object-storage path. | -| A schema change fails | Check the supported changes above. Don't modify managed catalog tables or files manually. [Contact support](/dashboard/support/new) with the pipeline ID and error details. | +| Issue | Resolution | +| ---------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Credential-provisioning warnings | Expected in [Use Supabase](#use-supabase) mode before saving. Review the selected resources. | +| Catalog or storage validation fails | Check [custom parameters](#custom-parameters), credentials, connectivity, and permissions for the configured prefix. Local `file://` paths are unsupported. | +| Metadata schema exists | Choose a new schema, unless intentionally reusing the same DuckLake and its corresponding data path. | +| Inserts work but updates or deletes fail | Check [replica identity and published columns](#source-table-requirements). | +| Queries omit changes or deleted rows | [Query through the catalog](#query-the-destination) with credentials for both catalog and storage. | +| A schema change fails | Review [supported changes](#schema-change-support). Don't modify catalog tables or files manually. | + +Use [pipeline monitoring](/docs/guides/database/replication/pipelines-monitoring) to inspect errors. For unresolved failures, [contact support](/dashboard/support/new) with the pipeline ID and error details. ## Additional resources diff --git a/apps/docs/content/guides/database/replication/manual-replication-faq.mdx b/apps/docs/content/guides/database/replication/manual-replication-faq.mdx index 09e28509a0a..61ff77d6a2a 100644 --- a/apps/docs/content/guides/database/replication/manual-replication-faq.mdx +++ b/apps/docs/content/guides/database/replication/manual-replication-faq.mdx @@ -8,25 +8,15 @@ sidebar_label: 'FAQ' ## Which connection string should be used? -Always use the direct connection string for logical replication. - -Connections through a pooler, such as Supavisor, will not work. +Use the direct connection string for logical replication. Connections through a pooler, such as Supavisor, will not work. ## The tool in use does not support IPv6 You can enable the [IPv4 add-on](/docs/guides/platform/ipv4-address) for your project. -## What is XMIN and should it be used? - -Xmin is a different form of replication from logical replication and should only be used if logical replication is not available for your database (i.e. older versions of Postgres). - -Xmin performs replication by checking the [xmin system column](https://www.postgresql.org/docs/current/ddl-system-columns.html) and determining if that row has already been synchronized. - -It does not capture deletion of data and is **not recommended**, particularly for larger databases. - ## Can replication be configured in the Dashboard? -You can view [publications](/dashboard/project/default/database/publications) in the Dashboard but all steps to configure replication must be done using the [SQL Editor](/dashboard/project/default/sql/new) or a CLI tool of your choice. +You can inspect [publications](/dashboard/project/_/database/publications) and change their published operations and table selection in the Dashboard. Use the [SQL Editor](/dashboard/project/_/sql/new), a CLI, or your consumer's setup flow for the remaining setup. Connect and operate the external consumer separately; the Dashboard doesn't manage a manual replication pipeline. ## How to configure database settings for replication? @@ -36,9 +26,17 @@ Using the Supabase CLI, you can [configure database settings](/docs/guides/datab Some of the more important options to be aware of are: -- `max_wal_size` - Maximum size the WAL can grow between automatic WAL checkpoints -- `max_slot_wal_keep_size` - Maximum size of WAL files that replication slots are allowed to retain -- `wal_keep_size` - Minimum number of past WAL files to keep for standby servers -- `max_wal_senders` - Maximum number of concurrent connections from standby servers or streaming backup clients +- `max_wal_size`: Soft limit for WAL written between automatic checkpoints. Postgres can exceed it under some conditions. +- `max_slot_wal_keep_size`: Per-slot limit on retained WAL at checkpoint time. `-1` lets replication slots retain unlimited WAL. A finite limit can cause a lagging slot to lose required WAL. +- `wal_keep_size`: Minimum amount of old WAL retained for standby servers. It does not replace the per-slot limit. +- `max_wal_senders`: Maximum number of concurrent WAL sender processes for standby servers, logical replication consumers, and streaming backups. -These settings help ensure your replication slots don't run out of space and that replicas can reconnect without requiring a full re-sync. +Size WAL retention for the consumer's maximum expected lag or downtime and the database's WAL generation rate. Higher values reduce the risk that a consumer needs a new initial sync, but can use more database storage. Monitor each slot's retained WAL and `wal_status` instead of treating a configured limit as a guarantee. + +## What is XMIN and should it be used? + +Xmin is a different form of replication from logical replication and should only be used if logical replication is not available for your database, such as with older versions of Postgres. + +Xmin performs replication by checking the [xmin system column](https://www.postgresql.org/docs/current/ddl-system-columns.html) and determining if that row has already been synchronized. + +It does not capture deletion of data and is **not recommended**, particularly for larger databases. diff --git a/apps/docs/content/guides/database/replication/manual-replication-monitoring.mdx b/apps/docs/content/guides/database/replication/manual-replication-monitoring.mdx index aa4d5e38e8b..139141abcfe 100644 --- a/apps/docs/content/guides/database/replication/manual-replication-monitoring.mdx +++ b/apps/docs/content/guides/database/replication/manual-replication-monitoring.mdx @@ -6,49 +6,62 @@ subtitle: 'Track replication health and performance.' sidebar_label: 'Monitoring' --- -Monitoring replication lag is important and there are 3 ways to do this: +Use source-side slot metrics to monitor any logical replication consumer. If the destination is another Postgres database, also check its subscription state. -1. Dashboard - In [Reports](/docs/guides/observability/reports), you can view the replication lag of your project -2. Database - - - pg_stat_subscription (subscriber) - if PID is null, then the subscription is not active - - pg_stat_subscription_stats - look here for error_count to see if there were issues applying or syncing (if yes, check the logs for why) - - pg_replication_slots - use this to check if the slot is active and you can also calculate the lag from here -3. [Metrics](/docs/guides/observability/metrics) - Using the prometheus endpoint for your project - - replication_slots_max_lag_bytes - this is the more important one - - pg_stat_replication_replay_lag - lag to replay WAL files from the source DB on the target DB (throttled by disk or high activity) - - pg_stat_replication_send_lag - lag in sending WAL files from the source DB (a high lag means that the publisher is not being asked to send new WAL files OR network issues) +- **Dashboard**: View replication lag in [Reports](/docs/guides/observability/reports). +- **Database**: Run the queries in [Primary](#primary) on the source and those in [Subscriber](#subscriber) on the destination Postgres database. +- **Metrics**: Use your project's [Prometheus endpoint](/docs/guides/observability/metrics) to track replication slot lag over time. ## Primary +Run these queries on the source database. + +### Replication slot status + +A replication slot has separate connection and WAL availability fields: + +- `active` is `true` when a consumer is connected to the slot. +- `wal_status` is `reserved` or `extended` while required WAL is available, `unreserved` when the slot is at risk of losing required WAL, and `lost` after required WAL has been removed. +- `safe_wal_size` estimates how many more bytes can be written before the slot is at risk. It is `NULL` when `max_slot_wal_keep_size` is unlimited or the slot is already lost. + +Check these fields and the amount of WAL retained from the slot's `restart_lsn`: + +```sql +select + slot_name, + active, + wal_status, + pg_size_pretty(safe_wal_size) as wal_retention_remaining, + pg_size_pretty(pg_wal_lsn_diff(pg_current_wal_lsn(), restart_lsn)) as retained_wal +from pg_replication_slots; +``` + +An inactive slot can still retain WAL. Investigate a slot whose retained WAL keeps growing, whose remaining retention keeps shrinking, or whose `wal_status` becomes `unreserved`. A `lost` slot can't continue from its previous position and usually requires a new initial sync. + ### Replication status and lag The `pg_stat_replication` table shows the status of any replicas connected to the primary database. ```sql -select pid, application_name, state, sent_lsn, write_lsn, flush_lsn, replay_lsn, sync_state +select + pid, + application_name, + state, + sent_lsn, + write_lsn, + flush_lsn, + replay_lsn, + sync_state from pg_stat_replication; ``` -### Replication slot status - -A replication slot can be in one of three states: - -- `active` - The slot is active and is receiving data -- `inactive` - The slot is not active and is not receiving data -- `lost` - The slot is lost and is not receiving data - -The state can be checked using the `pg_replication_slots` table: - -```sql -select slot_name, active, state from pg_replication_slots; -``` - ### WAL size The WAL size can be checked using the `pg_ls_waldir()` function: ```sql -select * from pg_ls_waldir(); +select pg_size_pretty(sum(size)) as wal_directory_size +from pg_ls_waldir(); ``` ### Check the LSN @@ -59,40 +72,53 @@ select pg_current_wal_lsn(); ## Subscriber +Run these queries on the destination only when it is a Postgres logical subscriber. For other destination systems, use their monitoring tools. + ### Subscription status The `pg_subscription` table shows the status of any subscriptions on a replica and the `pg_subscription_rel` table shows the status of each table within a subscription. -The `srsubstate` column in `pg_subscription_rel` can be one of the following: +The `srsubstate` column in [`pg_subscription_rel`](https://www.postgresql.org/docs/15/catalog-pg-subscription-rel.html) records each table's synchronization state: -- `i` - Initializing - The subscription is being initialized -- `d` - Data Synchronizing - The subscription is synchronizing data for the first time (i.e. doing the initial copy) -- `s` - Synchronized - The subscription is synchronized -- `r` - Replicating - The subscription is replicating data +- `i`: Initializing +- `d`: Copying existing data +- `f`: Finished copying; synchronization is not yet complete +- `s`: Synchronized +- `r`: Ready for normal replication ```sql -SELECT - sub.subname AS subscription_name, - relid::regclass AS table_name, - srel.srsubstate AS replication_state, - CASE srel.srsubstate - WHEN 'i' THEN 'Initializing' - WHEN 'd' THEN 'Data Synchronizing' - WHEN 's' THEN 'Synchronized' - WHEN 'r' THEN 'Replicating' - ELSE 'Unknown' - END AS state_description, - srel.srsyncedlsn AS last_synced_lsn -FROM - pg_subscription sub -JOIN - pg_subscription_rel srel ON sub.oid = srel.srsubid -ORDER BY - table_name; +select + sub.subname as subscription_name, + srel.srrelid::regclass as table_name, + srel.srsubstate as replication_state, + case srel.srsubstate + when 'i' then 'Initializing' + when 'd' then 'Copying data' + when 'f' then 'Finished copying' + when 's' then 'Synchronized' + when 'r' then 'Ready' + else 'Unknown' + end as state_description, + srel.srsublsn as synchronization_lsn +from pg_subscription sub +join pg_subscription_rel srel on sub.oid = srel.srsubid +order by table_name; ``` +`synchronization_lsn` coordinates the table's initial synchronization. It is not a continuously updated replication checkpoint. + ### Check the LSN +Use [`pg_stat_subscription`](https://www.postgresql.org/docs/15/monitoring-stats.html#MONITORING-PG-STAT-SUBSCRIPTION) to inspect the WAL positions reported by subscription workers: + ```sql -select pg_last_wal_replay_lsn(); +select + subname, + pid, + received_lsn, + latest_end_lsn, + latest_end_time +from pg_stat_subscription; ``` + +A missing `pid` means that worker is not running. `received_lsn` shows the last WAL position received, while `latest_end_lsn` and `latest_end_time` show the latest position and time reported to the publisher. These fields do not by themselves confirm that all tables have finished initial synchronization. diff --git a/apps/docs/content/guides/database/replication/manual-replication-setup.mdx b/apps/docs/content/guides/database/replication/manual-replication-setup.mdx index 194d5756ea8..dc55a726120 100644 --- a/apps/docs/content/guides/database/replication/manual-replication-setup.mdx +++ b/apps/docs/content/guides/database/replication/manual-replication-setup.mdx @@ -10,7 +10,7 @@ This guide covers setting up **manual logical replication** using your own tools -This guide is for replicating data to destination systems using your own tools. For deploying read-only databases across multiple regions, see [read replicas](/docs/guides/platform/read-replicas) instead. +For deploying read-only databases across multiple regions, see [read replicas](/docs/guides/platform/read-replicas). @@ -21,15 +21,15 @@ To set up replication, the following is recommended: - Instance size of XL or greater - [IPv4 add-on](/docs/guides/platform/ipv4-address) enabled -To create a replication slot, you will need to use the `postgres` user and follow the instructions in the [logical replication example](/docs/guides/database/postgres/setup-replication-external). - If you are running Postgres 17 or higher, you can create a new user and grant them replication permissions with the `postgres` user. For versions below 17, you will need to use the `postgres` user. -If you are replicating to a destination system and using any of the tools below, check their documentation first. Additional information is provided where the setup with Supabase can vary. +## Choose your replication tool + +Follow your tool's setup guide, using the Supabase-specific notes below. If you are building your own consumer, follow the [logical replication example](/docs/guides/database/postgres/setup-replication-external) to create a publication and replication slot. -Airbyte has the following [documentation](https://docs.airbyte.com/integrations/sources/postgres/) for setting up Postgres as a source, either in their cloud offering or by self-hosting. +Follow the [Airbyte Postgres source guide](https://docs.airbyte.com/integrations/sources/postgres/) for its cloud or self-hosted service. -You can follow those steps with the following modifications: +Apply these Supabase-specific settings: 1. Use the `postgres` user 2. Select `logical replication` as the replication method (`xmin` is possible, but not recommended) @@ -51,37 +51,35 @@ You can follow those steps with the following modifications: -Estuary has the following [documentation](https://docs.estuary.dev/reference/Connectors/capture-connectors/PostgreSQL/Supabase/) for setting up Postgres as a source. +Follow the [Estuary Supabase source guide](https://docs.estuary.dev/reference/Connectors/capture-connectors/PostgreSQL/Supabase/). -Fivetran has the following [documentation](https://fivetran.com/docs/connectors/databases/postgresql/setup-guide) for setting up Postgres as a source. +Follow the [Fivetran Postgres setup guide](https://fivetran.com/docs/connectors/databases/postgresql/setup-guide). -You can follow those steps with the following modifications: +Apply these Supabase-specific settings: 1. In Step 2, choose `logical replication` as the sync mechanism 2. In Step 3, do not create a user and use the existing `postgres` user for replication -3. In Step 5, no need to modify any WAL settings as this has been configured +3. In Step 5, don't copy generic WAL settings without sizing them for your database. Supabase supports logical replication, but you still need to size `max_slot_wal_keep_size` for your write rate and expected replication lag, then [monitor the slot](/docs/guides/database/replication/manual-replication-monitoring#replication-slot-status). -Materialize has the following [documentation](https://materialize.com/docs/sql/create-source/postgres/) on setting up Postgres as a source. +Follow the [Materialize Postgres source guide](https://materialize.com/docs/sql/create-source/postgres/). -You can follow those steps with the following modifications: - -1. Follow the steps in the [logical replication example](/docs/guides/database/postgres/setup-replication-external) to create a publication slot +Use the [logical replication example](/docs/guides/database/postgres/setup-replication-external) for the Supabase publication and replication slot setup. -Stitch has the following [documentation](https://www.stitchdata.com/docs/integrations/databases/postgresql/v2#extract-data) on configuring Postgres as a source. +Follow the [Stitch Postgres extraction guide](https://www.stitchdata.com/docs/integrations/databases/postgresql/v2#extracting-data). -You can follow those steps with the following modifications: +Apply these Supabase-specific settings: 1. Use the `postgres` user for replication 2. Skip step 3 @@ -90,11 +88,11 @@ You can follow those steps with the following modifications: -AWS DMS has the following [documentation](https://docs.aws.amazon.com/dms/latest/userguide/Welcome.html) on configuring Postgres as a source. +Follow the [AWS DMS documentation](https://docs.aws.amazon.com/dms/latest/userguide/Welcome.html) to configure Postgres as a source. DMS is notably useful if you have infrastructure running in AWS and/or you have custom networking. An additional benefit is that DMS is able to replicate schema changes. -You can follow those steps with the following modifications: +Apply these Supabase-specific settings: 1. Use the `postgres` user for replication (or create a new user with replication permissions: `ALTER USER WITH REPLICATION;`) 2. Set `pluginname` to `test-decoding` @@ -104,3 +102,7 @@ You can follow those steps with the following modifications: + +## Monitor replication + +After starting your consumer, [monitor replication slots, WAL retention, and progress](/docs/guides/database/replication/manual-replication-monitoring). diff --git a/apps/docs/content/guides/database/replication/pipelines-faq.mdx b/apps/docs/content/guides/database/replication/pipelines-faq.mdx index 3c52c5476ca..b8cc8282b7f 100644 --- a/apps/docs/content/guides/database/replication/pipelines-faq.mdx +++ b/apps/docs/content/guides/database/replication/pipelines-faq.mdx @@ -8,236 +8,97 @@ sidebar_label: 'FAQ' <$Partial path="pipelines-public-alpha.mdx" /> -## What destinations are supported? - -**BigQuery** is currently available as a managed destination. **ClickHouse**, **DuckLake**, and **Snowflake** are in Early Access. See the [ClickHouse](/docs/guides/database/replication/clickhouse), [DuckLake](/docs/guides/database/replication/ducklake), and [Snowflake](/docs/guides/database/replication/snowflake) destination guides for setup instructions and data-model details. - -{/* supa-mdx-lint-disable-next-line Rule003Spelling */} -[Request Early Access](/go/supabase-pipelines-new-destinations) for ClickHouse, DuckLake, or Snowflake. Supported destinations can be Supabase-managed or third-party systems as support expands. - -## Does the destination's region affect performance? - -Yes. The further apart your source database, the pipeline, and your destination are, the more network latency is added to replication, which increases replication lag and reduces throughput. - -Managed Pipelines run in **AWS `eu-central-1` (Frankfurt)**. For the best performance, place both your source project and destination as close to Frankfurt as possible. Choose the closest practical BigQuery dataset location, ClickHouse service region, DuckLake catalog and storage region, or Snowflake account region. - -If you can only optimize one side, prioritize placing your **destination** close to the pipeline's region. Replicated data continuously streams out to the destination, so latency on that leg has a bigger impact on overall replication lag than latency between the pipeline and the source. - {/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} ## Which plans support Pipelines? -Pipelines requires a Pro, Team, or Enterprise plan. During public alpha, access is being rolled out gradually, so an eligible plan does not guarantee that Pipelines is enabled for every organization yet. If it is not available, request access from the **Database > Replication** page or contact your account manager. +Pipelines requires a Pro, Team, or Enterprise plan. During public alpha, availability varies by organization; an eligible plan does not guarantee access. If unavailable, request access from **Database > Replication** or contact your account manager. -{/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} +## What destinations are supported? -## What happened to Analytics Buckets replication? +BigQuery is in public alpha. ClickHouse, DuckLake, and Snowflake are in private alpha and require access approval. See [supported destinations](/docs/guides/database/replication#supported-destinations) for their status and access request. -We are currently working on a new Supabase Warehouse product designed to address the limitations of the previous Analytics Buckets. Our goal is to build a solution we can confidently stand behind, rather than continuing to support an approach that does not meet the quality and flexibility we want for our users. +## Does the destination's region affect performance? -As a result, replication into Analytics Buckets via Pipelines is no longer supported. Separately, **BigQuery** is currently available as a managed Pipelines destination, and **ClickHouse**, **DuckLake**, and **Snowflake** are in Early Access. DuckLake here is a Pipelines replication destination, not the Warehouse product described above. +Yes. Distance between the source, pipeline, and destination adds network latency and can reduce throughput. Choose resources near the [managed pipeline region](/docs/guides/database/replication/pipelines#region), prioritizing the destination if you can optimize only one side. {/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} ## What does Pipelines install in the database? -When you enable Pipelines, Supabase installs database objects that help track replication state and support schema changes: +Pipelines installs objects in your project's Postgres database to track replication and support schema changes: -- An event trigger that runs on every `ALTER TABLE` statement. Supabase uses this to support schema change handling. -- A set of tables in the `etl` schema. These tables track replication state for your pipelines. +- An `etl` schema containing internal tables for replication state and progress, source table schemas, destination mappings, and migration history. +- Helper functions in `etl` that read table definitions and prepare schema-change messages. +- A database event trigger, `supabase_etl_ddl_message_trigger`, that runs after `ALTER TABLE` and `ALTER PUBLICATION`. It writes schema-change information for published tables to the write-ahead log (WAL), so Pipelines can process supported changes alongside row changes. -The replication state tables are not updated very often, especially after the initial sync is complete. +Each pipeline also uses replication slots to track its position in WAL. See [initial sync and table-sync slots](/docs/guides/database/replication/pipelines-monitoring#initial-sync-and-table-sync-slots) for how these retain changes during replication. + + + +The `etl` schema is reserved for Pipelines. If your application already uses a schema named `etl`, rename it and update references to it before enabling Pipelines. Do not add application objects to the Pipelines-managed schema or include it in a publication. + + + +To remove the installed objects, delete all pipelines, then [disable Pipelines](#what-happens-when-you-disable-pipelines). + +{/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} + +## What does Pipelines check before creating a pipeline? + +The Dashboard validates source access, replication capacity, publication tables, and destination connectivity and requirements. **Required** issues block creation; **Warnings** require review. See [creation checks](/docs/guides/database/replication/pipelines#creation-checks). ## What schema changes are supported? -Schema change support is destination-specific and limited. +Support differs by destination. Check the [schema-change guides](/docs/guides/database/replication/pipelines#schema-change-support) before changing column types, defaults, keys, or constraints. -BigQuery beta support includes: +## Can data be processed more than once? -- Adding a scalar, top-level column (created as `NULLABLE` in BigQuery) -- Removing a column -- Renaming a column -- Dropping a `NOT NULL` constraint -- Setting or dropping supported column default metadata +Yes. Pipelines provides at-least-once processing: recovery can replay acknowledged data, and consumers of append-only histories must tolerate repeated events. Destination deduplication does not provide an exactly-once processing guarantee. -Initial BigQuery table creation preserves whether each scalar, non-array Postgres column allows `NULL`. Arrays use BigQuery's `REPEATED` mode. For later changes, `DROP NOT NULL` relaxes a BigQuery `REQUIRED` column to `NULLABLE`, while `SET NOT NULL` leaves an existing BigQuery column nullable and logs a warning. Newly added scalar, top-level columns are always nullable in BigQuery. Supported defaults are applied separately as destination metadata and do not populate existing destination rows. Pipelines does not currently support changing column data types. See [BigQuery schema change support](/docs/guides/database/replication/bigquery#schema-change-support) for details. +See the data models for [BigQuery](/docs/guides/database/replication/bigquery#how-it-works), [ClickHouse](/docs/guides/database/replication/clickhouse#choose-a-table-engine), [DuckLake](/docs/guides/database/replication/ducklake#how-replication-works), and [Snowflake](/docs/guides/database/replication/snowflake#append-only-change-history), and the [billing implications](/docs/guides/platform/manage-your-usage/pipelines#how-data-processed-is-measured). -Snowflake Early Access supports adding, renaming, and dropping columns. It doesn't support column type changes or table and schema renames. Changes to whether existing columns accept `NULL` and changes to their defaults are ignored. Schema changes affect existing history. Dropping a column removes it from older events. See [Snowflake schema change support](/docs/guides/database/replication/snowflake#schema-change-support) for details. +## Why is a table not being replicated? -ClickHouse Early Access supports adding, renaming, and dropping columns, dropping `NOT NULL`, and supported default changes. It doesn't apply column type changes or `SET NOT NULL`. Changing a source primary-key column isn't supported for `ReplacingMergeTree`. ClickHouse DDL is not transactional, so an interrupted multi-column change can be partially applied. See [ClickHouse schema change support](/docs/guides/database/replication/clickhouse#schema-change-support) for details. +Check that: -DuckLake Early Access supports adding, renaming, and dropping columns, dropping `NOT NULL`, and supported default changes. It doesn't apply column type changes or `SET NOT NULL`. New columns are made nullable, and each multi-column schema change is applied in a transaction. See [DuckLake schema change support](/docs/guides/database/replication/ducklake#schema-change-support) for details. +- The table is published. If you added it after starting the pipeline, [restart the pipeline to discover it](/docs/guides/database/replication/pipelines#adding-or-removing-tables). +- The table and published columns satisfy the destination's primary-key and replica-identity requirements. +- [Publication row filters](/docs/guides/database/replication/pipelines#filtering-rows-with-a-predicate) include the expected rows. +- Missing columns aren't generated; Pipelines skips generated columns. + +If new changes arrive but existing rows are missing, check the [initial sync selection](/docs/guides/database/replication/pipelines#choosing-which-tables-to-copy). [Changing a row filter](/docs/guides/database/replication/pipelines#other-publication-changes) also does not copy newly included historical rows. + +If inserts work but updates or deletes fail, review the source table requirements for [BigQuery](/docs/guides/database/replication/bigquery#source-table-requirements), [ClickHouse](/docs/guides/database/replication/clickhouse#source-table-requirements), [DuckLake](/docs/guides/database/replication/ducklake#source-table-requirements), or [Snowflake](/docs/guides/database/replication/snowflake#source-table-requirements). + +## Why is a table in error state? + +An initial sync or ongoing replication operation failed, for example after an unsupported schema change. Some errors retry automatically; others require restarting replication for the affected tables from scratch. See [Table errors](/docs/guides/database/replication/pipelines-monitoring#table-errors) for diagnosis and [Restarting tables](/docs/guides/database/replication/pipelines-monitoring#restarting-tables) for the procedure and its effects. + +## Why is a pipeline failed or stopped? + +**Failed** reports a startup or runtime error and can recover automatically. **Stopped** requires a manual start. Follow [pipeline error recovery](/docs/guides/database/replication/pipelines-monitoring#pipeline-errors) to diagnose the cause before restarting. + +## Why is replication lag increasing? + +Postgres is producing WAL faster than Pipelines confirms progress. Follow [Investigate the lag](/docs/guides/database/replication/pipelines-monitoring#investigate-the-lag) to distinguish source activity, destination throughput, and connectivity problems. + +## What does a `Lost` slot status mean? + +Required WAL is gone, so replication cannot continue from that slot. Follow [slot recovery](/docs/guides/database/replication/pipelines-monitoring#respond-based-on-the-slot-status); recovery scope depends on whether the main slot or a table-sync slot was lost. + +## What happens if a table is deleted at the destination? + +Deleting or modifying managed objects can stop replication and require a new initial sync. To remove one safely, [remove the source table from the publication](/docs/guides/database/replication/pipelines#removing-tables-from-replication) and restart the pipeline before deleting its destination table. + +## What happens when a project becomes inactive or moves to the Free Plan? + +Project inactivity stops its pipelines. Start them manually after restarting the project. Downgrading to the Free Plan deletes its pipelines. {/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} ## What happens when you disable Pipelines? -Disabling Pipelines removes the database objects that were installed in your source database, including the replication state tables in the `etl` schema and the DDL event trigger. +Disabling Pipelines removes its database event trigger and the entire Pipelines-managed `etl` schema, including its tables and helper functions. Your source application tables and existing destination data remain. -To remove Pipelines from a project, delete all Pipelines destinations first. After all destinations are deleted, the disable action becomes available. For the Dashboard steps, see [Disabling Pipelines](/docs/guides/database/replication/pipelines#disabling-pipelines). - -Disabling Pipelines stops Supabase from managing replication for the project. It does not delete tables or data that were already written to your destination. - -## Why is a table not being replicated? - -Common reasons: - -- **Missing primary key**: BigQuery requires each source table to have a primary key and requires the publication to include its columns. This is a BigQuery Pipelines requirement, not a general requirement for publishing Postgres inserts. -- **ClickHouse requirements**: `ReplacingMergeTree` requires a source primary key. ClickHouse updates require `REPLICA IDENTITY FULL`. Deletes require primary-key identity or full identity. The publication must include all primary-key columns. -- **DuckLake requirements**: Insert-only tables don't require row identity. Updates and deletes require a primary key, `REPLICA IDENTITY USING INDEX`, or `REPLICA IDENTITY FULL`, and the publication must include every identity column. -- **Insufficient replica identity**: Snowflake updates require `REPLICA IDENTITY FULL`. Deletes require a primary key, `REPLICA IDENTITY USING INDEX`, or `REPLICA IDENTITY FULL`. The publication must include all identity columns. -- **Not in publication**: Ensure the table is included in your Postgres publication -- **Generated columns**: Generated columns are skipped during replication - -Check your publication settings and verify your table meets the requirements. - -Custom data types replicate as strings. Check that your destination can interpret those string values correctly. - -## Why are partitioned tables replicated as separate tables? - -Postgres controls this with the publication's `publish_via_partition_root` setting. If the setting is `false`, or if you created the publication manually with SQL and did not set it, Postgres publishes changes from the leaf partitions. Pipelines then creates destination tables for those leaf partitions. If `publish_via_partition_root = true`, Postgres publishes changes as the partition root, so the partition hierarchy is treated as the published partition root. - -Publications created from the Dashboard replication flow use `publish_via_partition_root = true`. - -See [Partitioned tables](/docs/guides/database/replication/pipelines#partitioned-tables) for examples and the full behavior. - -## How does replica identity affect updates and deletes? - -If inserts replicate but updates or deletes fail, the source table might not be sending enough old-row identity through Postgres logical replication. - -For BigQuery, every source table must have a primary key, every primary-key column must be included in the publication, and updates and deletes require a supported primary-key identity or full identity. - -For ClickHouse, the default `ReplacingMergeTree` engine requires a source primary key. The `MergeTree` event-log engine can replicate inserts from tables without a primary key. Updates require `REPLICA IDENTITY FULL`. Deletes require primary-key identity or full identity. The publication must include every primary-key column. - -For DuckLake, insert-only tables don't require row identity. Updates and deletes require a primary-key identity, `REPLICA IDENTITY USING INDEX`, or `REPLICA IDENTITY FULL`. The publication must include every identity column. - -For Snowflake, updates require `REPLICA IDENTITY FULL`. Deletes require a primary key, `REPLICA IDENTITY USING INDEX`, or `REPLICA IDENTITY FULL`. The publication must include all identity columns. Insert-only Snowflake tables don't require row identity. - -```sql -alter table public.your_table replica identity full; -``` - -Full replica identity increases WAL volume and only affects new WAL records. Fix the setting before generating more changes. If the failing update is already retained in WAL, restarting may fail again; recreate the pipeline or restart the affected table's initial sync. See the [BigQuery](/docs/guides/database/replication/bigquery#source-table-requirements), [ClickHouse](/docs/guides/database/replication/clickhouse#source-table-requirements), [DuckLake](/docs/guides/database/replication/ducklake#source-table-requirements), or [Snowflake](/docs/guides/database/replication/snowflake#source-table-requirements) requirements for details. - -## Why aren't publication changes reflected after adding or removing tables? - -After modifying your Postgres publication, you must restart the replication pipeline for changes to take effect. See [Adding or removing tables](/docs/guides/database/replication/pipelines#adding-or-removing-tables) for instructions. - -## Why is a pipeline in failed state? - -A pipeline enters **Failed** when it encounters a non-retryable pipeline-level error during startup or ongoing replication. The pipeline stops instead of silently skipping the failure. To recover: - -1. Check the error message by hovering over the **Failed** status -2. Click **View pipeline** for detailed information -3. Fix the underlying issue (e.g., schema mismatches, destination connectivity) -4. Restart the pipeline - -See [Handling errors](/docs/guides/database/replication/pipelines-monitoring#handling-errors) for more details. - -## Why was the pipeline stopped? - -A pipeline is stopped when you select **Stop pipeline**, when its project becomes inactive, or while an operation requires it to restart. A non-retryable configuration, schema, or data error can instead put the pipeline in **Failed** state. Transient connection and destination errors retry automatically when possible. - -To recover: - -1. Check whether the pipeline is **Stopped**, **Stopping**, or **Failed** -2. If it failed, review the error details and [replication logs](/dashboard/project/_/logs/replication-logs) -3. Fix the underlying issue, such as destination connectivity, schema mismatches, permissions, or rate limits -4. Start or restart the pipeline from [**Database > Replication**](/dashboard/project/_/database/replication) - -If the same error continues after restart, the pipeline may stop again. Review [Monitor pipeline status](/docs/guides/database/replication/pipelines-monitoring) for troubleshooting steps. - -## Why is replication lag increasing? - -Lag increases when Postgres produces WAL faster than the pipeline can confirm it has processed. Common causes include a slow or rate-limited destination, a pipeline issue, heavy source database activity, long transactions, network latency between the pipeline and source database, or a stopped/disconnected pipeline. - -Open [**Database > Replication**](/dashboard/project/_/database/replication), click **View pipeline**, and check **Waiting to sync**, **WAL retention remaining**, **Last check-in**, **Connected**, and **Slot status**. See [Dealing with replication lag](/docs/guides/database/replication/pipelines-monitoring#dealing-with-replication-lag) for the full investigation and response flow. - -## What does a `Lost` slot status mean? - -`Lost` means Postgres has already removed WAL files that the pipeline's replication slot needed. The pipeline cannot continue from that slot. - -You can recreate the pipeline, or open the pipeline's **Advanced settings**, set **Invalidated slot behavior** to **Recreate**, and start the pipeline again. The pipeline resets its saved table-sync state, creates a new replication slot, and replaces each destination table through a new initial sync. This destructive restart is required for consistency because the old slot can no longer provide every change the pipeline missed, and the data processed during the new initial sync is billed again. - -See [Slot statuses](/docs/guides/database/replication/pipelines-monitoring#slot-statuses) for all slot states and what to do next. - -## Why is a table in error state? - -Table errors occur during the initial sync. To recover, click **View pipeline**, find the affected table, and click its restart action. This restarts that table's initial sync from the beginning, deletes its existing destination data, and bills the successfully processed row data again. - -## How to verify replication is working - -Check the [**Database > Replication**](/dashboard/project/_/database/replication) section of the Dashboard: - -1. Verify your pipeline shows **Running** status -2. Click **View pipeline** to check table states -3. Ensure all tables show **Live** state (actively replicating) -4. Monitor replication lag metrics - -See [Monitor pipeline status](/docs/guides/database/replication/pipelines-monitoring) for comprehensive monitoring instructions. - -## How to stop replication - -You can manage your pipeline using the actions menu in the destinations list. See [Managing your pipeline](/docs/guides/database/replication/pipelines#managing-your-pipeline) for details on available actions. - - - -Stopping replication causes changes to queue up in the WAL. - -Stopping requests a graceful shutdown, so the pipeline can remain **Stopping** for up to five minutes while in-flight work finishes. Configured pipeline-hour billing continues while the pipeline is stopped. Delete the destination to end that charge. - - - -## What happens if a project becomes inactive? - -If your project becomes inactive, Pipelines stops any running pipelines and does not automatically resume them after the project is restarted. - -After restarting the project, restart each replication pipeline manually from the [**Database > Replication**](/dashboard/project/_/database/replication) section of the Dashboard. - -## What happens after a downgrade to the free plan? - -When a project is downgraded to the Free Plan, all replication pipelines created with Pipelines for that project are deleted. - -## What happens if a table is deleted at the destination? - -Don't delete or modify tables or views managed by Pipelines. Deleting a managed object can stop replication and may require a new, billable initial sync. Pipelines does not guarantee that it will automatically repair or fully resynchronize an object removed manually. - -**To permanently remove a table from your destination you have two options:** - -**Option 1: End replication permanently** - -1. Delete the destination to permanently end replication and pipeline-hour billing. Stopping the pipeline alone does not end pipeline-hour billing. -2. Delete the table at your destination -3. Don't restart the pipeline while the table remains in its publication - -**Option 2: Remove from publication first** - -1. Remove the table from your Postgres publication using `ALTER PUBLICATION ... DROP TABLE` -2. Restart your replication pipeline to apply the change (the table at the destination will remain but stop receiving new changes) -3. Delete the table at your destination - - - -Removing a table from the publication and restarting the pipeline does not delete the table downstream, it only stops replicating new changes to it. - - - -## Can data be processed more than once? - -Yes. Pipelines uses at-least-once processing. Failed destination write attempts that Pipelines retries are not counted. Data is counted only after the destination acknowledges successful processing. - -In rare cases, Pipelines can count an acknowledged batch but crash or be interrupted before its replication checkpoint is persisted. Recovery can then process and count the same data again. - -BigQuery and the default ClickHouse `ReplacingMergeTree` layout use the replicated source primary key and ordering metadata to converge on current table state. DuckLake applies current-state mutation batches and records destination progress in its catalog. ClickHouse `MergeTree` tables are append-only histories; `cdc_lsn` can be shared by changes in the same transaction and isn't a unique event ID. Snowflake uses committed channel offsets to suppress ordinary replay, but its append-only histories still require consumers to tolerate repeated events. `_cdc_sequence_number` is an ordering and checkpoint token, not a globally unique event ID. Pipelines doesn't guarantee exactly-once event processing. - -## Where to find replication logs - -Navigate to the [**Logs > Replication**](/dashboard/project/_/logs/replication-logs) section of the Dashboard to see all pipeline logs. Logs contain diagnostic information. If you're experiencing issues, contact support with your error details. - -## How to get help - -If you need assistance: - -1. Check [Set up Pipelines](/docs/guides/database/replication/pipelines) and [Monitor pipeline status](/docs/guides/database/replication/pipelines-monitoring) -2. Review this FAQ for common issues -3. Contact support with your error details and logs +Delete all pipelines first, then follow [Disabling Pipelines](/docs/guides/database/replication/pipelines#disabling-pipelines). Deleting the pipelines alone does not remove the shared database installation. diff --git a/apps/docs/content/guides/database/replication/pipelines-monitoring.mdx b/apps/docs/content/guides/database/replication/pipelines-monitoring.mdx index 3696dcb4268..e943b76b23c 100644 --- a/apps/docs/content/guides/database/replication/pipelines-monitoring.mdx +++ b/apps/docs/content/guides/database/replication/pipelines-monitoring.mdx @@ -8,32 +8,29 @@ sidebar_label: 'Monitoring' <$Partial path="pipelines-public-alpha.mdx" /> -After setting up Supabase Pipelines, you can monitor the status and health of your pipelines directly from the Dashboard. A pipeline first performs an initial sync of existing rows, then uses ongoing replication (CDC) to send subsequent database changes to your destination. +Use the Dashboard to check [replication progress](#viewing-detailed-pipeline-metrics), [investigate lag](#dealing-with-replication-lag), and [recover from errors](#handling-errors). ## Viewing pipeline status -To monitor your pipelines: - -1. Navigate to the [**Database > Replication**](/dashboard/project/_/database/replication) section of the Dashboard -2. You'll see a list of all your destinations with their pipeline status +Open [**Database > Replication**](/dashboard/project/_/database/replication) to see each destination's pipeline status. ### Pipeline states -Each destination shows its pipeline in one of these states: +| State | Meaning | +| -------------- | --------------------------------------------------------------- | +| **Stopped** | Not running; requires a manual start | +| **Starting** | Starting or replacing the process after a restart | +| **Running** | Process is running | +| **Stopping** | Finishing in-flight work before stopping | +| **Restarting** | Applying settings or table state before restarting | +| **Failed** | Reports a startup or runtime failure; can recover automatically | +| **Unknown** | Status is unavailable | -| State | Description | -| -------------- | ------------------------------------------------------------------------------- | -| **Stopped** | Pipeline is not running | -| **Starting** | Pipeline is being started | -| **Running** | Pipeline is actively replicating data | -| **Stopping** | Pipeline is being stopped | -| **Restarting** | Pipeline settings or table state are being applied before replication restarts | -| **Failed** | Pipeline has encountered an error (hover over the status to view error details) | -| **Unknown** | The Dashboard can't currently determine the pipeline status | +**Running** does not confirm that initial sync has finished, the source is connected, or destination writes are complete. Check table states and lag to assess progress. ## Viewing detailed pipeline metrics -For detailed information about a specific pipeline, click **View pipeline** on the destination. This opens the pipeline status page where you can monitor replication performance and table states. +Click a pipeline in the list to inspect its metrics and table states. +### Table states + +| State | Meaning | +| ----------------- | -------------------------------------------------------------------- | +| **Queued** | Waiting to begin initial sync | +| **Copying** | Copying existing rows | +| **Copied** | Copy finished; changes made during the copy still need to be applied | +| **Live** | Following WAL changes, including catch-up after copying | +| **Error** | Replication encountered an error | +| **Restarting** | Restarting replication for the table from scratch | +| **Not Available** | State is temporarily unavailable during a pipeline state change | +| **Unknown** | The Dashboard does not recognize the reported state | + ### Replication lag metrics -The status page shows replication lag metrics that help you determine how far the pipeline is behind Postgres. These metrics are loaded directly from Postgres replication slot state. +These metrics come from Postgres replication slot state. The pipeline list also shows a byte-based lag value: **Caught up** means no gap between the confirmed flush position and current WAL position at measurement time. It does not guarantee that every table has finished initial sync or that data is visible to destination queries. -The destinations list also shows a compact lag value. This value is byte-based: it shows how much WAL the pipeline has not confirmed as flushed yet. A value of **Caught up** means the pipeline has confirmed every change currently available for its slot. +| Metric | Meaning | Investigate when | +| --------------------------- | -------------------------------------------------------------------------------------------------- | -------------------------------------------------------------- | +| **Waiting to sync** | WAL bytes between the confirmed flush position and current Postgres WAL position | The value keeps growing | +| **WAL retention remaining** | WAL that can accumulate before the slot risks becoming unusable, based on `max_slot_wal_keep_size` | The value is small or shrinking | +| **Last check-in** | Time since the last replication feedback to Postgres | Feedback is old or missing | +| **Connected** | Whether the replication slot is active | It says **Not connected** while the pipeline should be running | +| **Slot status** | Whether required WAL is still retained | The slot is **Unreserved** or **Lost** | -The detailed status page shows: +The Dashboard displays **Unlimited** when Postgres returns no `safe_wal_size`. That happens both with unlimited retention and with a **Lost** slot, so always check slot status. See [Postgres replication slot metrics](https://www.postgresql.org/docs/current/view-pg-replication-slots.html). -| Metric | What it means | What to watch for | -| --------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **Waiting to sync** | Bytes of WAL between the pipeline's confirmed flush position and the current Postgres WAL position. This is the main byte-based replication lag. | A value that keeps growing means the pipeline is receiving changes more slowly than Postgres produces them. | -| **WAL retention remaining** | How much WAL can still accumulate before the replication slot is at risk of becoming unusable. This is controlled by `max_slot_wal_keep_size`. | A small or shrinking value means you should investigate before required WAL is removed. `Unlimited` means Postgres is not reporting a slot WAL retention limit. | -| **Last check-in** | How long it has been since the pipeline last sent replication feedback to Postgres. | An old value can mean the pipeline is stopped, disconnected, overloaded, or unable to make progress. | -| **Connected** | Whether the pipeline's replication slot is active and currently being used. | `Not connected` while the pipeline should be running usually means you should check pipeline status and logs. | -| **Slot status** | How safely Postgres is keeping the WAL files the pipeline still needs. | `Unreserved` and `Lost` require action. See [Slot statuses](#slot-statuses). | +Pipelines keeps sending feedback during slow writes and while WAL reading is paused. It repeats the last safe position without acknowledging unfinished writes. A recent check-in and an active connection can therefore coexist with growing lag. -Pipelines uses one main pipeline replication slot for ongoing replication. During the initial sync, it can also create temporary table-sync replication slots. These temporary slots let multiple tables sync in parallel, make large initial syncs faster, and allow individual tables to be retried without restarting the whole pipeline. - -Temporary table-sync slots show the same kind of lag and slot health metrics while they are active. After a table finishes its initial sync and catches up, its temporary slot is removed and ongoing replication continues through the main pipeline slot. For overall replication health, focus first on the main pipeline slot. +The main slot tracks ongoing replication. [Table-sync slots](#initial-sync-and-table-sync-slots) appear temporarily during initial sync and have the same metrics. ### Slot statuses -Replication slot status tells you whether Postgres is still retaining the WAL that the pipeline needs to continue from its current position. +| Status | Meaning | +| -------------- | ----------------------------------------------------------------------------------- | +| **Reserved** | Required WAL is retained within the normal WAL size limit | +| **Extended** | Required WAL is retained beyond that limit; this alone does not mean lag is growing | +| **Unreserved** | Required WAL is no longer fully reserved and can be removed | +| **Lost** | Required WAL has been removed; replication cannot continue from this slot | +| **Unknown** | Slot state is unavailable or unrecognized | -| Status | Meaning | -| -------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **Reserved** | Healthy. Postgres is keeping the WAL files this pipeline's replication slot needs, and they are within the normal WAL size limit. | -| **Extended** | Healthy, but growing. The slot is holding on to more WAL than usual, but Postgres is still keeping everything the pipeline needs. | -| **Unreserved** | At risk. Postgres is no longer reserving all WAL files this pipeline's replication slot needs. If the pipeline does not catch up soon, those files may be removed. | -| **Lost** | Broken. Some WAL files this pipeline's replication slot needs have already been removed. The pipeline can no longer continue from this slot. Recreate the pipeline, or set **Invalidated slot behavior** to **Recreate** in the pipeline's advanced settings and restart it. | -| **Unknown** | Postgres reported an unknown or unavailable state for this pipeline's replication slot. | - -### Table states - -The pipeline status page also shows the state of individual tables being replicated. Each table can be in one of these states: - -| State | Description | -| ----------------- | ------------------------------------------------------------------------ | -| **Queued** | Table is waiting for the pipeline to begin its initial sync | -| **Copying** | Existing rows are being copied during the initial sync | -| **Copied** | Initial sync is complete and the table is preparing to replicate changes | -| **Live** | Table is now receiving ongoing replication | -| **Error** | Table has experienced an error during replication | -| **Restarting** | The table's initial sync is being restarted | -| **Not Available** | Table state is temporarily unavailable while the pipeline changes state | -| **Unknown** | The Dashboard received a table state it doesn't recognize | - -## Dealing with replication lag - -Replication lag means the pipeline is behind the source database. Some lag is expected during the initial sync, after a burst of writes, or after restarting a stopped pipeline. Lag becomes a problem when it keeps increasing, when **WAL retention remaining** is running low, or when the slot status moves to **Unreserved** or **Lost**. - -Lag can come from several places: - -- **Destination throughput**: The destination is slow, rate-limited, unavailable, or rejecting writes. -- **Pipeline throughput**: The pipeline is overloaded, processing a very large transaction, or not performing as expected for the project workload. -- **Source database activity**: Postgres is producing WAL faster than the pipeline can consume it, often during bulk writes, migrations, or backfill jobs. -- **Network latency**: Latency or instability between the pipeline and source database can slow down WAL streaming. -- **Stopped or disconnected pipeline**: When a pipeline is stopped, disconnected, or failed, Postgres keeps WAL for the slot until the retention limit is reached. -- **Slow initial sync**: A temporary table-sync slot can fall behind if existing rows are copied more slowly than new changes are written to that table. - -### Initial sync and table-sync slots - -A common initial sync issue happens when a large or busy table is still in **Copying** while new rows keep being inserted or updated. The temporary table-sync slot retains changes that happen during the initial sync. If copying is too slow compared to the table's write rate, the slot can move to **Unreserved** and then **Lost** if Postgres removes changes the sync still needs. - -When a table-sync slot is lost, the affected table needs to run its initial sync again. Tune the copy settings, then retry the table: - -- Increase **Copy connections per table** when one large table is the bottleneck. This lets the pipeline copy chunks of that table over multiple source connections, up to the point where the source database, network, or destination becomes the limit. -- Increase **Table sync workers** when several tables need to copy at the same time. Each worker can copy one table, and each worker uses an additional temporary replication slot during initial sync. -- If possible, run the initial sync during a quieter write period or reduce bulk writes until the table reaches **Live**. - -After the affected table finishes copying and catches up, the temporary slot is deleted. The table then continues through the main pipeline replication slot. - -### Investigate the lag - -1. Open [**Database > Replication**](/dashboard/project/_/database/replication) and check the destination's lag column. -2. Click **View pipeline** and check **Waiting to sync**, **WAL retention remaining**, **Last check-in**, **Connected**, and **Slot status**. -3. Check table states. Tables in **Copying** can create temporary lag while the initial sync catches up to ongoing changes. If a table-sync slot is **Unreserved** or **Lost**, tune copy parallelism and retry the affected table's initial sync. -4. Open [**Logs > Replication**](/dashboard/project/_/logs/replication-logs) and look for destination errors, retries, rate limits, schema errors, or repeated restarts. -5. Compare the lag trend with recent database activity, such as imports, migrations, bulk updates, or long transactions. - -### Respond based on the slot status - -| Slot status | What to do | -| -------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **Reserved** | If **Waiting to sync** is stable or decreasing, continue monitoring. If it keeps increasing, check destination write performance, logs, and whether the publication includes more tables or write volume than expected. | -| **Extended** | Treat it as an early warning. Confirm the pipeline is connected, check logs for retries or destination slowness, and reduce avoidable write bursts if possible until the pipeline catches up. | -| **Unreserved** | Act soon. The slot is at risk of losing required WAL. Check whether the pipeline is connected and making progress, fix destination or pipeline errors, and contact support if the lag continues to grow. | -| **Lost** | The pipeline cannot continue from the existing slot because required WAL has been removed. Recreate the pipeline, or set **Invalidated slot behavior** to **Recreate** in the pipeline's advanced settings and restart the pipeline. This creates a new slot and starts replication from scratch for all tables. | -| **Unknown** | Check replication logs for errors or missing slot details. If the status remains unknown while the pipeline should be running, contact support with the pipeline ID and recent log details. | - -### Reduce future lag risk - -- Keep publications focused on the tables and operations you need at the destination. -- Avoid leaving pipelines stopped for long periods while the source database is still receiving writes. -- Schedule bulk updates, imports, and migrations during lower-traffic windows when possible. -- For BigQuery, verify that service account permissions, table requirements, and replica identity settings match the [BigQuery destination guide](/docs/guides/database/replication/bigquery). -- For ClickHouse, check the HTTPS endpoint, target-database permissions, selected table engine, server version, and source table requirements against the [ClickHouse destination guide](/docs/guides/database/replication/clickhouse). -- For DuckLake, check catalog connectivity, object-storage permissions, catalog and storage latency, and source row identity against the [DuckLake destination guide](/docs/guides/database/replication/ducklake). -- For Snowflake, check the service user's default role, schema privileges, key pair, row limits, and replica identity against the [Snowflake destination guide](/docs/guides/database/replication/snowflake). -- If the initial sync is the bottleneck, review **Table sync workers** and **Copy connections per table** in the pipeline's advanced settings. Increasing either can use more source database connections; increasing table sync workers can also use more temporary replication slots. - -## Handling errors - -Errors can occur at two levels: per table or per pipeline. - -### Table errors - -Table errors occur during the initial sync and affect individual tables. These errors can be retried without stopping the entire pipeline. - -**Viewing table error details:** - -1. Click **View pipeline** on your destination -2. Check the table states section to identify tables in **Error** state -3. Review the error message for that specific table - -**Recovering from table errors:** - -When a table encounters an error during the initial sync, you can reset the table state. This deletes the table's existing destination data and restarts its initial sync from the beginning. For ClickHouse, the reset drops and recreates the table and any generated current-state view. For DuckLake, it drops and recreates the table in the catalog; managed maintenance later cleans up unreferenced files. For Snowflake, it drops and recreates the append-only history table and its managed streaming channel. - -### Pipeline errors - -Pipeline errors can occur during startup or ongoing replication and affect the entire pipeline. If a non-retryable pipeline-level error occurs, the entire pipeline stops and enters a **Failed** state instead of silently skipping the failure. - -**Viewing pipeline error details:** - -1. Hover over the **Failed** status in the destinations list to see a quick error summary -2. Click **View pipeline** for comprehensive error information -3. Navigate to the [**Logs > Replication**](/dashboard/project/_/logs/replication-logs) section of the Dashboard for detailed error logs - -**Recovering from pipeline errors:** - -To recover from a pipeline error, you'll need to: - -1. Investigate the root cause using the error details and logs -2. Fix the underlying issue (e.g., destination connectivity, schema compatibility) -3. Restart the pipeline from the destinations list +See [Respond based on the slot status](#respond-based-on-the-slot-status) for recovery actions. ## Viewing logs -To see detailed logs for all your pipelines: +Open [**Logs > Replication**](/dashboard/project/_/logs/replication-logs) for errors, retries, and performance warnings. Include the pipeline ID and relevant errors when [contacting support](/dashboard/support/new). -1. Navigate to the [**Logs > Replication**](/dashboard/project/_/logs/replication-logs) section of the Dashboard -2. Select **Replication** from the log source filter -3. You'll see all logs from your pipelines +## Handling errors - +Table errors can pause one table while others continue replicating. Pipeline errors affect the overall process. -Logs contain diagnostic information that may be too technical for most users. If you're experiencing issues with replication, reaching out to support with your error details is recommended. +### Table errors - +Open the pipeline, inspect the table error and any scheduled retry, and fix the cause. If replication cannot safely resume from its saved state, [restart replication for the affected tables](#restarting-tables). -## Common monitoring scenarios +### Pipeline errors -### Checking if replication is healthy +A **Failed** pipeline can recover automatically. Check logs for persistent connectivity, permission, schema, or destination errors, fix the cause, and restart if needed. Supabase stops pipelines after repeated failures; a **Stopped** pipeline requires a manual start. -1. Navigate to the [**Database > Replication**](/dashboard/project/_/database/replication) section of the Dashboard -2. Verify your destination shows a "Running" status -3. Click **View pipeline** to check replication lag and table states -4. Ensure all tables show a "Live" state +## Dealing with replication lag -### Investigating errors +Lag grows when Postgres produces WAL faster than Pipelines confirms progress. Some lag is expected during initial sync, write bursts, and recovery from downtime. Sustained growth or shrinking WAL retention requires investigation. -If you see a **Failed** status: +### Investigate the lag -1. Hover over the status to see the error summary -2. Click **View pipeline** to see detailed error information -3. Check table states to identify which tables are affected -4. Navigate to the [**Logs > Replication**](/dashboard/project/_/logs/replication-logs) section of the Dashboard for full error details -5. For table errors, attempt to reset the affected tables +1. Open the pipeline and compare **Waiting to sync** with **WAL retention remaining** and **Slot status**. +2. Check **Connected** and **Last check-in** for connection or feedback problems. +3. Check table states for initial sync or table errors. +4. Open [replication logs](#viewing-logs) and look for destination errors, rate limits, retries, or repeated restarts. +5. Compare the lag trend with source activity, such as bulk imports, large transactions, or long-running writes. -### Monitoring performance +Common causes include a slow destination, source or pipeline resource limits, network latency, and stopped or disconnected pipelines. -To ensure optimal performance: +### Initial sync and table-sync slots -1. Regularly check replication lag metrics in the pipeline status view -2. Monitor table states to ensure tables are staying in a "Live" state -3. Review logs for warnings or performance issues -4. If lag is consistently high, review your publication and destination configuration +Pipelines uses one main slot plus up to one temporary slot per active table-sync worker. These slots retain changes made while rows are copied. After copying and catch-up finish, the temporary slot is removed and the table continues through the main slot. -## Troubleshooting +If a table-sync slot is lost, that table must repeat its initial sync. Automatic retries can recover individual tables; a manual [table restart](#restarting-tables) temporarily stops the whole pipeline. -If you notice issues with your replication: +To improve copy throughput: -1. **Check pipeline state**: Ensure the pipeline is in **Running** state -2. **Review table states**: Identify tables in **Error** state -3. **Check logs**: Navigate to the [**Logs > Replication**](/dashboard/project/_/logs/replication-logs) section of the Dashboard for detailed error information -4. **Verify publication**: Ensure your Postgres publication is properly configured -5. **Monitor replication lag**: High lag may indicate performance issues +- Increase **Initial sync connections per table** when one large table is the bottleneck. +- Increase **Table sync workers** when several tables need to copy concurrently. +- Reduce bulk writes until the table is **Live** and its lag has caught up. -For more troubleshooting tips, see the [Pipelines FAQ](/docs/guides/database/replication/pipelines-faq). +Higher concurrency uses more source connections, and additional workers need more slots. Increase it only while the source database, network, and destination have capacity. See [setting defaults and limits](/docs/guides/database/replication/pipelines#advanced-settings). -## Next steps +### Respond based on the slot status -- [Set up Pipelines](/docs/guides/database/replication/pipelines) -- [View the Pipelines FAQ](/docs/guides/database/replication/pipelines-faq) +| Slot status | Action | +| ---------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------- | +| **Reserved** or **Extended** | Monitor the trend. If lag keeps growing, investigate destination performance, source write volume, and logs. | +| **Unreserved** | Restore connectivity or resolve throughput and write failures before Postgres removes required WAL. Contact support if lag continues to grow. | +| **Lost** | For a table-sync slot, retry the affected table. For the main slot, use the recovery procedure that follows. | +| **Unknown** | Check logs and slot details. If the state persists, contact support with the pipeline ID and error details. | + +To recover a lost main slot, open **Edit pipeline > Advanced settings** and set **Invalidated slot behavior** to **Recreate slot**. Review the effects below, then click **Apply and restart pipeline** or **Apply and start pipeline**, depending on its current state. + +**This replaces every destination table.** Existing source rows are copied only for tables [selected for initial sync](/docs/guides/database/replication/pipelines#choosing-which-tables-to-copy); excluded tables resume with new changes only. Data processed again is [billed again](/docs/guides/platform/manage-your-usage/pipelines#how-data-processed-is-measured). + +Alternatively, stop and delete the pipeline, then [create a new one](/docs/guides/database/replication/pipelines#setup-overview). Deletion automatically removes that pipeline's main slot, table-sync slots, and saved replication state when the source database is reachable. It leaves source tables, existing destination data, and the [shared Pipelines installation](/docs/guides/database/replication/pipelines-faq#what-happens-when-you-disable-pipelines) in place. + +### Reduce future lag risk + +- Review the [WAL retention recommendation](/docs/guides/database/replication/pipelines#creation-checks) at creation and continue monitoring retention after startup. +- Publish only the tables and operations you need. +- Avoid prolonged stops while the source is receiving writes. +- Schedule bulk writes and initial sync during quieter periods when possible. + +## Restarting tables + +Restart table replication to rebuild one or more destination tables from scratch, for example after a schema change or an unrecoverable table error. **Restart pipeline** restarts the process using saved replication progress; it does not request a table rebuild. See [pipeline restart and recovery behavior](/docs/guides/database/replication/pipelines#pipeline-restarts-and-recovery) for exceptions during initial sync or lost-slot recovery. + +When you restart table replication: + +- **All existing data in the affected destination tables is deleted.** The original Postgres source tables and their rows are unchanged. +- For tables [selected for initial sync](/docs/guides/database/replication/pipelines#choosing-which-tables-to-copy), Pipelines copies all current source rows included by the publication again, then resumes ongoing replication. Tables excluded from initial sync resume with new changes only; their existing source rows are not copied. +- The table restart resets replication progress only for the selected tables. +- Data processed again is [billed again](/docs/guides/platform/manage-your-usage/pipelines#how-data-processed-is-measured). + +The pipeline stops temporarily to apply the table restart. The Dashboard restarts it automatically, or starts it if it was stopped. Replication for other tables is paused during this process, and their unfinished initial syncs can restart from scratch. + +Before restarting, use **Edit pipeline** to check the **Initial sync** selection. Include any affected tables whose existing source rows you need to copy again, then apply your changes. Open the pipeline from the list on **Database > Replication**. + +### Restart a single table + +1. Click the restart icon on the table's row. +2. Review the table name, data replacement, and cost details in the confirmation dialog. +3. Click **Restart replication**. + +### Restart multiple tables + +1. Click **Restart all tables**, or open the arrow menu beside it and select **Restart failed tables only**. +2. Review the affected tables, data replacement, and cost details. The failed-tables action affects every table that is failed when the request runs. +3. Confirm **Restart all tables** or **Restart failed tables**, matching the scope you chose. + +If a table restart fails, refresh the pipeline and table states before retrying. Start the pipeline explicitly if it remains **Stopped**. diff --git a/apps/docs/content/guides/database/replication/pipelines.mdx b/apps/docs/content/guides/database/replication/pipelines.mdx index 4717a020e4a..71e7d113682 100644 --- a/apps/docs/content/guides/database/replication/pipelines.mdx +++ b/apps/docs/content/guides/database/replication/pipelines.mdx @@ -1,210 +1,69 @@ --- id: 'pipelines' title: 'Set up Pipelines' -description: 'Create a Supabase Pipeline using Postgres logical replication.' +description: 'Create a managed pipeline using Postgres logical replication.' subtitle: 'Configure publications, destinations, and Supabase Pipelines.' sidebar_label: 'Setting up' --- <$Partial path="pipelines-public-alpha.mdx" /> -Supabase Pipelines is a managed CDC product for moving data from Supabase Postgres to supported destination systems. It uses **Postgres logical replication** with the open-source [Supabase ETL engine](https://github.com/supabase/etl). You choose a destination in the Dashboard, and Supabase runs the pipeline that sends database changes to that destination. +Supabase Pipelines replicates Postgres data to a destination you configure in the Dashboard. Supabase runs the pipeline using the open-source [Supabase ETL](https://github.com/supabase/etl). -Pipelines has two replication phases: +**Initial sync** copies existing rows from selected tables. **Ongoing replication** uses change data capture (CDC) to apply subsequent inserts, updates, deletes, and truncates. -- **Initial sync**: A one-time copy of the existing rows in the published tables. -- **Ongoing replication (CDC)**: Continuously captures and applies subsequent inserts, updates, deletes, and truncates. +[Create a pipeline](#setup-overview), then refer to [settings](#pipeline-settings), [management](#managing-your-pipeline), or [advanced publication options](#advanced-publication-options) as needed. -Managed Pipelines run in **AWS `eu-central-1` (Frankfurt)**. Choose a destination region as close as possible to Frankfurt to reduce network latency and replication lag. +## Before you start -## Pricing +Check the [plan and access requirements](/docs/guides/database/replication/pipelines-faq#which-plans-support-pipelines) and [check supported destinations](/docs/guides/database/replication#supported-destinations). -<$Partial path="billing/pricing/pricing_pipelines.mdx" /> +### Region -For billing examples and optimization guidance, see [Manage Pipelines usage](/docs/guides/platform/manage-your-usage/pipelines). +Managed Pipelines run in AWS `eu-central-1` (Frankfurt), independently of your source and destination regions. This region is fixed. Choose nearby destination resources to reduce latency and replication lag. + +### Pricing + +Pipelines charges for configured pipeline hours, initial sync data, and ongoing replication data. Destination-provider charges are separate. See [rates, estimates, and billing examples](/docs/guides/platform/manage-your-usage/pipelines). ## Setup overview -Pipelines requires two main components: a **Postgres publication** (defines what to replicate) and a **destination** (where data is sent). Supabase runs the managed pipeline that reads from the publication and writes to the destination. Follow these steps to set up your replication pipeline. +Prepare your destination, then enable and configure Pipelines in the Dashboard. A Postgres publication selects the data to replicate; you can create one during setup. - +### Step 1: Prepare your destination [#step-1-create-a-postgres-publication] -If you already have a Postgres publication set up, you can skip to [Step 2: Enable Pipelines](#step-2-enable-pipelines). +Follow your destination guide to prepare its resources and credentials, and check its source table requirements: - - -### Step 1: Create a Postgres publication - -A Postgres publication defines which tables and change types will be replicated from your database. You can create a basic publication in the Dashboard while configuring the destination, or use SQL when you need column lists, row filters, schema-wide publications, or other advanced options. - -- **Dashboard**: Continue to [Step 2](#step-2-enable-pipelines). In Step 3, open the **Publication** selector, click **New publication**, enter a name, and select at least one table. -- **SQL**: Create the publication now using one of the examples below, then select it when you configure the destination. - -#### Creating a publication with SQL - -The following SQL examples assume you have `users` and `orders` tables in your database. - -##### Publication for specific tables - -```sql --- Create publication for both tables -create publication pub_users_orders -for table users, orders; -``` - -This publication tracks all changes (INSERT, UPDATE, DELETE, TRUNCATE) for both the `users` and `orders` tables. - -##### Publication for all tables in a schema - -```sql --- Create a publication for all tables in the public schema -create publication pub_all_public for tables in schema public; -``` - -This tracks changes for all existing and future tables in the `public` schema. - -##### Publication for all tables - -```sql --- Create a publication for all tables -create publication pub_all_tables for all tables; -``` - -This tracks changes for all tables in your database. - - - -`FOR ALL TABLES` includes tables in Supabase-managed schemas, including the internal `etl` tables that Pipelines creates. Prefer `FOR TABLES IN SCHEMA public` or list the application tables explicitly unless you intend to replicate every eligible table in the database. - - - -#### Advanced publication options - -##### Selecting specific columns - -You can replicate only a subset of columns from a table: - -```sql --- Replicate only specific columns from the users table -create publication pub_users_subset -for table users (id, email, created_at); -``` - -This only replicates the `id`, `email`, and `created_at` columns from the `users` table. - -##### Filtering rows with a predicate - -You can filter which rows to replicate using a `WHERE` clause: - -```sql --- Only replicate active users -create publication pub_active_users -for table users where (status = 'active'); - --- Only replicate recent orders -create publication pub_recent_orders -for table orders where (created_at > '2024-01-01'); -``` - -##### Partitioned tables - -Pipelines follows Postgres publication semantics for partitioned tables. The `publish_via_partition_root` publication setting controls whether changes from partitions are emitted as the partition root or as the leaf partitions. - -| Publication setting | What gets replicated | Destination shape | -| ------------------------------------------ | ---------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------- | -| `publish_via_partition_root = true` | Rows from the published partition root, including rows stored in its leaf partitions | One table matching the published partition root | -| `publish_via_partition_root = false` | Rows from the leaf partitions under the published partition root | One table per replicated leaf partition | -| Not set in SQL | Same as `false`, because Postgres defaults `publish_via_partition_root` to `false` | One table per replicated leaf partition | -| Publishing an individual leaf partition | The leaf partition itself, regardless of `publish_via_partition_root` | One table for that leaf partition | -| `FOR ALL TABLES` or `FOR TABLES IN SCHEMA` | Partition roots plus regular tables when `true`; leaf partitions plus regular tables when `false` or unset | Destination tables follow the effective Postgres publication table list | - -For example, if `orders` is partitioned by month: - -```sql --- Replicate the whole partition hierarchy as the parent table. -create publication pub_orders_root -for table orders -with (publish_via_partition_root = true); - --- Replicate each leaf partition as its own table. -create publication pub_orders_leaves -for table orders -with (publish_via_partition_root = false); -``` - -Use `publish_via_partition_root = true` when you want analytics queries to read from a single destination table that has the parent table's schema. Use `false` when each partition should remain a separate destination table. - -Publications created from the Dashboard replication flow use `publish_via_partition_root = true`. If you create or alter a publication manually with SQL, set this option explicitly so the destination shape matches what you expect. - -On Postgres 15 and newer, row filters on partition publications apply during both the initial sync and ongoing replication. Pipelines uses the row filter attached to the effective publication table entry: the published partition root when `publish_via_partition_root = true`, and the published leaf relation when `publish_via_partition_root = false`. - -The publication setting controls which Postgres relation becomes a destination table. It does not copy the source table's physical partitioning configuration, partition key, or partition bounds to the destination. - - - -With `publish_via_partition_root = true`, truncating an individual leaf partition is not replicated as a truncate event for the published parent. This is useful for append-only data such as events: you can truncate old leaf partitions to keep Postgres storage bounded while retaining the rows already copied to the destination. If you want the destination to be truncated too, run `TRUNCATE` on the published partition root. - -Rows retained only in the destination aren't a permanent archive. A table reset or full pipeline initial sync rebuilds the destination from the rows that still exist in Postgres. - - - -#### Viewing publications in the Dashboard - -After creating a publication via SQL, you can view it in the Dashboard: - -1. Navigate to the [**Database > Publications**](/dashboard/project/_/database/publications) section of the Dashboard -2. You'll see all your publications listed with their tables +- [BigQuery](/docs/guides/database/replication/bigquery) +- [ClickHouse](/docs/guides/database/replication/clickhouse) +- [DuckLake](/docs/guides/database/replication/ducklake) +- [Snowflake](/docs/guides/database/replication/snowflake) {/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} ### Step 2: Enable Pipelines -Before creating a managed replication pipeline, enable Pipelines for your project: + + +Pipelines installs an `etl` schema and a database event trigger to track replication and schema changes. If your application already has an `etl` schema, rename it before enabling Pipelines. See [what Pipelines installs and how to remove it](/docs/guides/database/replication/pipelines-faq#what-does-pipelines-install-in-the-database). + + 1. Navigate to the [**Database > Replication**](/dashboard/project/_/database/replication) section of the Dashboard -2. Click **Add destination** to show the replication side panel -3. Select a Pipelines destination, such as **BigQuery**, or **ClickHouse**, **DuckLake**, or **Snowflake** if your organization has Early Access -4. Click **Enable Pipelines** +2. Click **Add pipeline** to show the replication side panel +3. Select a destination available to your organization +4. If Pipelines is not yet enabled, click **Enable Pipelines**, review the dialog, and confirm **Enable Pipelines**. ### Step 3: Configure a destination -Once Pipelines is enabled and you have a Postgres publication, configure a destination. The destination is where your replicated data will be stored, while the pipeline is the active Postgres replication process that continuously streams changes from your database to that destination. +1. Enter a **Name** and select a **Publication**. Use an existing publication, or click **New publication**, name it, and select at least one table. For schema-wide replication or advanced options, [create the publication with SQL](#creating-a-publication-with-sql). +2. Under **Initial sync**, keep **All tables** to copy existing rows, or [choose which tables to copy](#choosing-which-tables-to-copy) if you only need future changes for some tables. +3. Enter the credentials and destination-specific settings prepared in Step 1. +4. Optionally expand **Advanced settings** to change [batching, initial sync concurrency, or slot recovery](#advanced-settings). +5. Click **Create and start pipeline**. Fix any **Required** issues. For **Warnings**, review the risk, click **Create and start pipeline anyway**, and confirm that you want to continue. +6. Review the estimated initial sync cost and ongoing charges, then confirm creation. -#### Choose and configure your destination - -{/* supa-mdx-lint-disable-next-line Rule003Spelling */} -Follow these steps to configure your destination. Each destination has its own setup requirements and data model. **BigQuery** is currently available. **ClickHouse**, **DuckLake**, and **Snowflake** are in Early Access. [Request access](/go/supabase-pipelines-new-destinations) to these destinations. - -1. Navigate to the [**Database > Replication**](/dashboard/project/_/database/replication) section of the Dashboard -2. Click **Add destination** if the destination side panel isn't already open - -3. Select the destination type - -4. Configure the destination details: - - **Destination name**: A name to identify this destination - - **Publication**: Select an existing publication, or click **New publication** to create one by choosing a name and at least one table - - **Region**: Managed Pipelines run in the fixed **AWS `eu-central-1` (Frankfurt)** region. This can't be changed. In your destination provider, choose nearby destination resources when possible. - -5. Configure the destination-specific settings. See the destination guide for required credentials, permissions, and limitations: - - [BigQuery](/docs/guides/database/replication/bigquery) - - [ClickHouse (Early Access)](/docs/guides/database/replication/clickhouse) - - [DuckLake (Early Access)](/docs/guides/database/replication/ducklake) - - [Snowflake (Early Access)](/docs/guides/database/replication/snowflake) - -6. Optionally expand **Advanced settings** to tune pipeline behavior: - - | Setting | Default | Allowed values | Description | - | ------------------------------ | -------------------- | ---------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | - | **Batch wait time** | `10000` milliseconds | Whole milliseconds, `0` or greater | Maximum time after the first buffered initial-sync row or ongoing change before the pipeline flushes a partially filled batch. Internal size and memory limits can flush it earlier. Lower values can reduce batching delay; higher values can improve destination write efficiency. | - | **Table sync workers** | `4` workers | Whole number greater than `0` | Maximum number of tables synced in parallel during the initial sync. Each active table sync temporarily uses one additional replication slot, up to `N + 1` slots including the pipeline's main slot. | - | **Copy connections per table** | `4` connections | Whole number greater than `0` | Maximum source database connections used to copy one table in parallel. With multiple table sync workers, source connection usage can scale with both settings. More connections can speed up large tables until the source database, network, or destination becomes the bottleneck. | - | **Invalidated slot behavior** | `Error` | `Error` or `Recreate` | What happens when the main replication slot can no longer continue from retained WAL. **Error** blocks startup for manual recovery. **Recreate** resets table sync state, rebuilds the slot on the next start, and runs the initial sync again for every replicated table. | - - Leave these settings at their defaults unless you need to tune initial sync speed, latency, or recovery behavior. - - Use **Invalidated slot behavior** carefully. If **Recreate** is selected and the pipeline starts after Postgres has invalidated the main replication slot, the pipeline resets its saved table-sync state, creates a new slot, and replaces each destination table through a new initial sync. This destructive restart is required for consistency because the old slot can no longer provide every change the pipeline missed, and the data processed during the new initial sync is billed again. - -7. Click **Create and start pipeline** to begin replication +The pipeline copies the selected tables' existing rows and begins ongoing replication for every published table. -The pipeline begins the initial sync from your database to your destination. - ### Step 4: Monitor your pipeline -After you create and start the pipeline, its destination appears in the destinations list. You can monitor the pipeline's status and performance from the Dashboard. +Open the pipeline and check that tables progress to **Live** and replication lag catches up. See [pipeline states, metrics, and logs](/docs/guides/database/replication/pipelines-monitoring) to assess progress and diagnose problems. -For comprehensive monitoring instructions including pipeline states, metrics, and logs, see [Monitor pipeline status](/docs/guides/database/replication/pipelines-monitoring). +## Pipeline settings -### Managing your pipeline +Review initial sync choices and creation checks before tuning advanced settings. -You can manage your pipeline from the destinations list using the actions menu. +### Choosing which tables to copy - +**Initial sync** controls which publication tables copy their existing rows. Ongoing replication includes new changes from every published table. -Available actions: +| Selection | Existing rows copied | +| ------------------------------ | ---------------------------------------------- | +| **All tables** | Every published table; the default | +| **All except selected tables** | Every published table except those you exclude | +| **Selected tables only** | Only the tables you include | +| **No tables** | None; replicate new changes only | + +Selection changes affect unfinished initial syncs. To copy a completed table again, include it here, apply the settings, and [restart its replication](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). + +Skipping initial sync does not preserve or attach to previously loaded destination data. A table restart erases destination data even when initial sync is skipped. + +### Creation checks + +The Dashboard checks source access, logical replication settings and capacity, the publication, destination connectivity, and destination table requirements. Publications must contain at least one table and exclude the internal `etl` schema. `FOR ALL TABLES` is rejected. + +A warning can recommend increasing `max_slot_wal_keep_size` so Postgres retains changes during initial sync. Higher retention uses more source storage; check available disk before applying the recommendation. See [WAL configuration](/docs/guides/database/replication#logical-replication-configuration) and [monitoring](/docs/guides/database/replication/pipelines-monitoring#replication-lag-metrics). + +### Advanced settings + +Leave these settings at their defaults unless you need to tune latency, initial sync speed, or recovery behavior. + +| Setting | Behavior | +| -------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| **Batch wait time** | Default: `10000` milliseconds. Maximum wait after the first buffered row or change before flushing a partial batch. Size and memory limits can flush earlier. Accepts whole milliseconds from `0`. | +| **Table sync workers** | Default: `4`. Maximum tables copied concurrently. Each active worker uses an additional replication slot. Accepts whole numbers greater than `0`. | +| **Initial sync connections per table** | Default: `4`. Maximum source connections copying one table. Total connection use increases with both concurrency settings. Accepts whole numbers greater than `0`. | +| **Invalidated slot behavior** | Default: **Block startup**, which requires manual recovery. **Recreate slot** rebuilds an invalid main slot and restarts replication for all tables from scratch on the next start. | + +Lower batch wait times reduce batching delay; higher values can improve write efficiency. For concurrency tradeoffs, see [Initial sync and table-sync slots](/docs/guides/database/replication/pipelines-monitoring#initial-sync-and-table-sync-slots). + +**Recreate slot replaces every destination table**, including tables excluded from initial sync. Review [lost-slot recovery](/docs/guides/database/replication/pipelines-monitoring#respond-based-on-the-slot-status) before enabling it. + +## Managing your pipeline + +Use the three-dot actions menu on a pipeline row: - **Start pipeline**: Begin replication for a stopped pipeline - **Update available**: Review and apply the latest managed pipeline version when an update is available -- **Stop pipeline**: Request a graceful stop. The pipeline can remain **Stopping** for up to five minutes while in-flight work finishes. New changes queue in the WAL, and configured pipeline-hour billing continues while stopped. -- **Restart pipeline**: Stop and start the pipeline (required after publication changes) -- **Edit destination**: Modify destination settings like credentials or advanced options -- **Delete destination**: Remove the destination and permanently stop replication +- **Stop pipeline**: Finish in-flight work and stop. Shutdown can take several minutes. WAL accumulates and pipeline-hour billing continues while stopped. +- **Restart pipeline**: Restart with the saved settings and replication progress. Required after adding or removing publication tables. Review the [recovery behavior](#pipeline-restarts-and-recovery) below before restarting during initial sync. +- **Edit pipeline**: Modify settings like credentials, initial sync selection, or advanced options. Click **Apply and restart pipeline** for an active pipeline or **Apply and start pipeline** for a stopped pipeline. +- **Delete pipeline**: Delete the pipeline and stop replication. Already replicated data remains at the destination. -{/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} +### Viewing publications in the Dashboard -### Disabling Pipelines - -To turn off Pipelines for a project, delete all Pipelines destinations first. After all destinations are removed, open the three-dot actions menu on the Replication page and click **Disable Pipelines**. - -For cleanup details, see [What happens when you disable Pipelines?](/docs/guides/database/replication/pipelines-faq#what-happens-when-you-disable-pipelines). +View publications and their tables in [**Database > Publications**](/dashboard/project/_/database/publications). ### Adding or removing tables -If you need to modify which tables are replicated after your replication pipeline is already running, follow these steps: - -If your Postgres publication uses `FOR ALL TABLES` or `FOR TABLES IN SCHEMA`, new tables in that scope are automatically included in the publication. However, you still **must restart the replication pipeline** for the changes to take effect. +`FOR TABLES IN SCHEMA` includes new tables automatically, but Pipelines discovers them only after a pipeline restart. To exclude an implicitly included table, change the publication scope or use an explicit table list; `ALTER PUBLICATION ... DROP TABLE` cannot exclude it. +These examples use the `pub_users_orders` publication from the [SQL example](#publication-for-specific-tables). Replace it and the table names with your own; tables must already exist. + #### Adding tables to replication -1. Add the table to your publication using SQL: +1. Add existing source tables to your explicit table-list publication using SQL: ```sql - -- Add a single table to an existing publication - alter publication pub_users_orders add table products; - - -- Or add multiple tables at once - alter publication pub_users_orders add table products, categories; + alter publication pub_users_orders + add table products, categories; ``` -2. **Restart the replication pipeline** using the actions menu (see [Managing your pipeline](#managing-your-pipeline)) for the changes to take effect. +2. Review the [initial sync selection](#choosing-which-tables-to-copy) for the added tables, then select **Restart pipeline** from the pipeline's actions menu. #### Removing tables from replication 1. Remove the table from your Postgres publication using SQL: ```sql - -- Remove a single table from a publication - alter publication pub_users_orders drop table orders; - - -- Or remove multiple tables at once - alter publication pub_users_orders drop table orders, products; + alter publication pub_users_orders + drop table orders, products; ``` -2. **Restart the replication pipeline** using the actions menu (see [Managing your pipeline](#managing-your-pipeline)) for the changes to take effect. +2. Select **Restart pipeline** from the pipeline's actions menu. + +After the restart, Pipelines removes its replication state for those tables. Source tables and destination tables, including data already replicated, remain unchanged. Other pipelines using publications that still include the tables are unaffected. If multiple pipelines use the changed publication, restart each one. + +You can delete the destination tables yourself after the restart. **Don't modify or delete tables still managed by Pipelines**; doing so can stop replication and require a [table restart](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). + +### Pipeline restarts and recovery + +A pipeline restart does not request a fresh copy of every table. Tables that completed initial sync resume from saved progress. An interrupted initial sync can start again from scratch: Pipelines deletes the partial destination data and copies the table again according to the **Initial sync** selection. This can also happen to an unfinished table when another table's restart temporarily stops the pipeline. + +If the main replication slot is lost and **Recreate slot** is enabled, startup rebuilds all replicated tables. Review [lost-slot recovery](/docs/guides/database/replication/pipelines-monitoring#respond-based-on-the-slot-status) for its data-loss and billing effects. To deliberately rebuild specific tables, use [Restarting tables](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). + +{/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} + +### Disabling Pipelines + +Delete all pipelines first. Then open the three-dot actions menu on the Replication page and click **Disable Pipelines**. + +For cleanup details, see [What happens when you disable Pipelines?](/docs/guides/database/replication/pipelines-faq#what-happens-when-you-disable-pipelines). + +## Creating a publication with SQL + +The following SQL examples assume you have `users` and `orders` tables with the referenced columns in your database. Schema-wide publications, column lists, and row filters require Postgres 15 or later. Choose one publication scope; the examples are alternatives. + +### Publication for specific tables + +```sql +create publication pub_users_orders +for table users, orders; +``` + +This publication includes inserts, updates, deletes, and truncates for `users` and `orders`. + +### Publication for all tables in a schema + +```sql +create publication pub_all_public +for tables in schema public; +``` + +This tracks changes for all existing and future tables in the `public` schema. + +To include multiple application schemas, pass a comma-separated list. For example, if your tables are in `public` and `analytics`: + +```sql +create publication pub_application_schemas +for tables in schema public, analytics; +``` + +Both schemas must already exist. This includes their existing and future tables without including the internal `etl` schema. Restart the pipeline after adding new tables so it discovers them. See [Postgres schema publications](https://www.postgresql.org/docs/15/sql-createpublication.html) for syntax and requirements. + +### Publication for all tables + +Pipelines rejects `FOR ALL TABLES` because it includes the [internal `etl` tables](/docs/guides/database/replication/pipelines-faq#what-does-pipelines-install-in-the-database). Replicating them can interfere with replication state. Use `FOR TABLES IN SCHEMA` for application schemas or list tables explicitly. Replace any existing `FOR ALL TABLES` publication; Postgres cannot narrow it in place. + +## Advanced publication options + +Use these options to control which data is published and how partitioned tables appear at the destination. + +### Selecting specific columns + +You can replicate only a subset of columns from a table: + +```sql +create publication pub_users_subset +for table users (id, email, created_at); +``` + +For updates and deletes, the column list must include all replica-identity columns. With `REPLICA IDENTITY FULL`, that means all columns, so you cannot publish a subset. Also check your destination's source table requirements. + +### Filtering rows with a predicate + +You can filter which rows to replicate using a `WHERE` clause. These examples copy matching existing rows during initial sync and then replicate matching inserts only: + +```sql +-- Only replicate active users +create publication pub_active_users +for table users +where (status = 'active') +with (publish = 'insert'); + +-- Only replicate recent orders +create publication pub_recent_orders +for table orders +where (created_at > '2024-01-01') +with (publish = 'insert'); +``` + +To also publish updates and deletes, every column used in the row filter must be covered by the table's replica identity. For example, `REPLICA IDENTITY FULL` covers all columns, at the cost of more WAL. Review [Postgres row-filter restrictions](https://www.postgresql.org/docs/15/logical-replication-row-filter.html) and your destination's source table requirements before enabling those operations. + +Postgres evaluates the filter before sending changes. A row matches only when the expression is `true`; `false` and `NULL` do not match. For an update, it checks both the old and new row: + +| Old row matches | New row matches | Change sent to Pipelines | +| --------------- | --------------- | ------------------------ | +| Yes | Yes | Update | +| No | Yes | Insert | +| Yes | No | Delete | +| No | No | No change | + +This applies when updates are published. Row filters do not limit `TRUNCATE`: a published truncate affects the whole destination table. See [Other publication changes](#other-publication-changes) before changing a filter on an existing pipeline. + +### Partitioned tables + +`publish_via_partition_root` controls whether partition changes arrive as one parent table or separate leaf tables: + +| Setting | Destination shape | +| ----------------------------- | ----------------------------------------------------------------------- | +| `true` | One table matching the published parent, including rows from its leaves | +| `false` or unset in SQL | One table per replicated leaf | +| Publishing an individual leaf | One table for that leaf, regardless of this setting | + +Schema-wide publications follow the same rule for partitioned tables; regular tables remain separate. + +For example, if `orders` is partitioned by month: + +```sql +-- Publish all partitions as the parent table. +create publication pub_orders_root +for table orders +with (publish_via_partition_root = true); + +-- Publish each leaf as a separate table. +create publication pub_orders_leaves +for table orders +with (publish_via_partition_root = false); +``` + +Publications created from the Dashboard replication flow default to `publish_via_partition_root = true`. Clear **Publish partitions as the parent table** to use `false`. If you create or alter a publication manually with SQL, set this option explicitly so the destination shape matches what you expect. + +On Postgres 15 and later, row filters apply during both initial sync and ongoing replication: + +- With `publish_via_partition_root = true`, the published parent's filter and column list apply, even if a leaf has its own filter or column list. +- With `false`, each leaf's filter and column list apply. Define row filters on the leaves; Postgres rejects them on the partitioned parent in this mode. + +See [Postgres partition row filters](https://www.postgresql.org/docs/15/logical-replication-row-filter.html#LOGICAL-REPLICATION-ROW-FILTER-PARTITIONED-TABLE) for examples. + +The publication setting controls which Postgres relation becomes a destination table. It does not copy the source table's physical partitioning configuration, partition key, or partition bounds to the destination. + +Choose the mode before starting replication: + +- With `true`, new changes in a newly created or attached partition flow through the already tracked parent without a restart. Attaching a partition does not copy rows that were already in it. +- With `false`, a new leaf partition has its own table identity. Restart the pipeline to discover it; its existing rows are copied only if it is selected for initial sync. + +To change modes, stop the pipeline, change the option, review **Initial sync** for the new table identities, then restart. Pipelines removes state for untracked tables and discovers new ones, copying existing rows only if selected for initial sync. Old destination tables remain; their data is not merged or split automatically. -Don't manually delete or modify a destination table managed by Pipelines. This can stop replication and require a new initial sync. Supported source schema changes are documented below. To permanently remove a destination table, first remove its source table from the publication and restart the pipeline, then delete the destination table. Alternatively, delete the whole destination. See the [Pipelines FAQ](/docs/guides/database/replication/pipelines-faq#what-happens-if-a-table-is-deleted-at-the-destination) for details. +With `publish_via_partition_root = true`, truncating a leaf partition keeps destination rows; truncating the published parent clears them. + +A [table restart](/docs/guides/database/replication/pipelines-monitoring#restarting-tables) discards retained rows. It copies current source data again only if the table is selected for initial sync. -### Schema change support +### Other publication changes -Schema change support is destination-specific and limited. See the [BigQuery](/docs/guides/database/replication/bigquery#schema-change-support), [ClickHouse](/docs/guides/database/replication/clickhouse#schema-change-support), [DuckLake](/docs/guides/database/replication/ducklake#schema-change-support), or [Snowflake](/docs/guides/database/replication/snowflake#schema-change-support) guide for the exact behavior. +Publication changes do not all require a pipeline restart: -### How it works +| Change | When it takes effect | +| ---------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------- | +| Add or remove published columns on an already tracked table | Applied while the pipeline runs, subject to the destination's [schema-change restrictions](#schema-change-support). | +| Change a row filter | Postgres applies the new filter to ongoing changes without a pipeline restart. | +| Change published operations (`insert`, `update`, `delete`, `truncate`) | Postgres changes which operations it sends without a pipeline restart. Previously omitted operations are not replayed. | +| Change `publish_via_partition_root` | Stop the pipeline before changing this option, then restart it to discover the new table identities. See [Partitioned tables](#partitioned-tables). | -Once configured, a replication pipeline: +Changing a row filter affects future changes only: it neither copies newly included historical rows nor removes rows that no longer match. A pipeline restart does not rebuild history. To rebuild it, include affected tables in initial sync and [restart their replication](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). -1. **Captures** changes from your Postgres database using Postgres publications and logical replication -2. **Sends** the changes through the managed pipeline -3. **Loads** the data to your destination +Wait for initial sync to finish before changing its filter; changing a filter does not update a copy already in progress. -Pipelines automatically optimizes how changes are delivered to the destination. It maps published source columns and values to destination-compatible names and types, but doesn't provide user-defined transformations. +`ALTER PUBLICATION ... SET TABLE` replaces the table list. When changing column lists or filters, include every table you want to keep; omitted tables are removed. -### Troubleshooting +## Schema change support -If you encounter issues during setup: +Pipelines applies supported changes to replicated columns as replication progresses. It does not mirror every Postgres DDL operation. Check the supported changes and destination behavior for [BigQuery](/docs/guides/database/replication/bigquery#schema-change-support), [ClickHouse](/docs/guides/database/replication/clickhouse#schema-change-support), [DuckLake](/docs/guides/database/replication/ducklake#schema-change-support), or [Snowflake](/docs/guides/database/replication/snowflake#schema-change-support) before altering replicated tables. -- **Publication not appearing**: Ensure you created the Postgres publication via SQL and refresh the dashboard -- **Tables not showing in publication**: Verify your tables meet the requirements of the selected destination. BigQuery and ClickHouse `ReplacingMergeTree` require a source primary key. ClickHouse updates require `REPLICA IDENTITY FULL`; deletes require primary-key or full identity. DuckLake updates and deletes require a primary-key identity, replica-identity index, or full identity. With a primary-key identity or replica-identity index, include every identity column in the publication. Snowflake requires `REPLICA IDENTITY FULL` when updates are published. -- **Pipeline failed to start**: Check the error message in the status view for specific details -- **No data being replicated**: Verify your Postgres publication includes the correct tables and event types +Unsupported changes have different outcomes: -For more troubleshooting help, see the [Pipelines FAQ](/docs/guides/database/replication/pipelines-faq). +- **Data type changes are skipped with a warning** in every destination, including changes to precision or scale. The destination keeps its existing type. Compatible values may continue to replicate, but later writes or schema changes can fail. +- **Unsupported defaults and tightening `NOT NULL` are skipped with a warning.** Replication can continue with a different default or a more permissive destination column. Each destination guide describes these differences; skipping a default does not skip the values Postgres supplies in row changes. +- **Incompatible schema changes are rejected with an error**, such as changing the primary-key definition used by BigQuery or ClickHouse's `ReplacingMergeTree`. Failed or interrupted schema changes can require manual recovery. -### Limitations +Check [replication logs](/docs/guides/database/replication/pipelines-monitoring#viewing-logs) for warnings and errors after changing a replicated table. + +After a type change, or when a schema error cannot resume safely, resolve any source incompatibility and [restart replication for the affected tables](/docs/guides/database/replication/pipelines-monitoring#restarting-tables). This rebuilds their destination schema and **deletes their existing destination data**. Select the tables for **Initial sync** to copy their current source rows again. Restarting the pipeline alone does not rebuild tables. Do not repair managed destination objects manually. + +## Limitations Pipelines has the following limitations: -- **Row identity**: Requirements are destination-specific. BigQuery and ClickHouse `ReplacingMergeTree` require a source primary key and its published columns. ClickHouse updates require `REPLICA IDENTITY FULL`; deletes require primary-key or full identity. DuckLake insert-only tables don't require a key, but updates and deletes require a primary-key identity, replica-identity index, or full identity. With a primary-key identity or replica-identity index, include every identity column in the publication. Snowflake insert-only tables don't require a key. Snowflake deletes require a published primary-key or replica-identity index unless full identity is used. Snowflake updates require `REPLICA IDENTITY FULL`. +- **Source tables**: Primary-key, replica-identity, and publication-column requirements depend on the destination. See its destination guide before creating the pipeline. +- **Transformations**: Pipelines maps names and types for the destination but does not run user-defined transformations. - **Custom data types**: Custom values replicate as strings. Check that your destination can interpret those string values correctly. +- **Arrays**: Only one-dimensional arrays are supported. Non-default lower bounds are not preserved. Check the destination guide for additional array restrictions. - **Generated columns**: Generated columns are skipped. Use triggers to store derived values in regular columns if you need them in the destination. -- **Replica identity**: Updates and deletes need the mode required by the destination. See the [BigQuery](/docs/guides/database/replication/bigquery#source-table-requirements), [ClickHouse](/docs/guides/database/replication/clickhouse#source-table-requirements), [DuckLake](/docs/guides/database/replication/ducklake#source-table-requirements), and [Snowflake](/docs/guides/database/replication/snowflake#source-table-requirements) requirements. -- **Schema changes**: Support is destination-specific and limited. -- **No user-defined transformations**: Pipelines performs destination-compatible type and name mapping, but doesn't run custom transformations -- **At-least-once processing**: In rare recovery cases, an acknowledged batch can be processed and counted again. BigQuery, DuckLake, and the default ClickHouse `ReplacingMergeTree` layout maintain current-state tables. ClickHouse `MergeTree` and Snowflake store append-only histories, so consumers must tolerate repeated events. See [Can data be processed more than once?](/docs/guides/database/replication/pipelines-faq#can-data-be-processed-more-than-once) for details. Destination-specific limitations, such as row size and type mappings, are documented in each destination guide. - -### Next steps - -- [Set up BigQuery](/docs/guides/database/replication/bigquery) -- [Set up ClickHouse](/docs/guides/database/replication/clickhouse) -- [Set up DuckLake](/docs/guides/database/replication/ducklake) -- [Set up Snowflake](/docs/guides/database/replication/snowflake) -- [Monitor pipeline status](/docs/guides/database/replication/pipelines-monitoring) -- [View the Pipelines FAQ](/docs/guides/database/replication/pipelines-faq) diff --git a/apps/docs/content/guides/database/replication/snowflake.mdx b/apps/docs/content/guides/database/replication/snowflake.mdx index db0af2240fd..dbade3c3a6f 100644 --- a/apps/docs/content/guides/database/replication/snowflake.mdx +++ b/apps/docs/content/guides/database/replication/snowflake.mdx @@ -10,14 +10,30 @@ sidebar_label: 'Snowflake' <$Partial path="pipelines-public-alpha.mdx" /> - - -The Snowflake destination is in Early Access and available only to approved organizations. [Request access](/go/supabase-pipelines-new-destinations) before following this guide. - - +The Snowflake destination is in private alpha and available only to approved organizations. [Request access](/go/supabase-pipelines-new-destinations) before following this guide. [Snowflake](https://www.snowflake.com/) is a managed data platform. Supabase Pipelines writes an append-only change history for each replicated Postgres table to Snowflake. +[Prepare resources](#prepare-snowflake-resources), [configure the destination](#configure-snowflake-as-a-destination), then [query replicated data](#query-and-materialize-current-state). + +## Source table requirements + +Required `REPLICA IDENTITY` depends on the operations enabled in the Postgres publication: + +| Published operations | Required replica identity | +| -------------------- | ---------------------------------------------------------------------------------------------------------- | +| `INSERT` only | No row identity is required. | +| `DELETE` | A primary key, an identity index (`USING INDEX`), or full identity (`FULL`). Publish all identity columns. | +| `UPDATE` | `REPLICA IDENTITY FULL`. | + +Set full replica identity before publishing updates: + +```sql +alter table public.your_table replica identity full; +``` + +`REPLICA IDENTITY FULL` increases WAL volume, but lets Pipelines construct complete new rows when Postgres omits unchanged out-of-line TOAST values. The setting applies only to new WAL records. If retained WAL already contains an incompatible update, restart replication for the affected table after changing the setting. + {/* supa-mdx-lint-disable-next-line Rule001HeadingCase */} ## Prepare Snowflake resources @@ -41,14 +57,13 @@ create schema if not exists PIPELINES_DB.REPLICATED; grant usage on database PIPELINES_DB to role PIPELINES_ROLE; grant usage on schema PIPELINES_DB.REPLICATED to role PIPELINES_ROLE; grant create table on schema PIPELINES_DB.REPLICATED to role PIPELINES_ROLE; -grant create pipe on schema PIPELINES_DB.REPLICATED to role PIPELINES_ROLE; ``` The pipeline role must own destination tables so it can alter, truncate, or drop them. Don't pre-create destination tables under another role. -The role also needs `CREATE PIPE` so Snowflake can create each table's managed default Snowpipe Streaming pipe, named `
-STREAMING`, when Pipelines opens a channel. Pipelines does not need a virtual warehouse, stage, or manually created pipe. +Snowflake creates each table's managed default pipe, `
-STREAMING`, automatically. No virtual warehouse, stage, or manually created pipe is required. See [Snowpipe Streaming access privileges](https://docs.snowflake.com/en/user-guide/snowpipe-streaming/snowpipe-streaming-access-control) for the ingestion requirements. -Use a separate role and warehouse for downstream queries and transformations. The pipeline service role does not need query or transformation privileges. +Use a separate role and warehouse for downstream queries and transformations. ### Keep the SQL and streaming roles aligned @@ -57,20 +72,24 @@ Pipelines uses two Snowflake interfaces: - SQL requests use the optional **Role** configured in the Dashboard. When **Role** is empty, they use the user's default role. - Snowpipe Streaming uses the user's `DEFAULT_ROLE`. It does not use the optional **Role** setting. -Use one dedicated role for both interfaces. Set it as the service user's `DEFAULT_ROLE`. Leave **Role** empty in the Dashboard or set it to the same role. If the roles differ, SQL validation and table creation can succeed while streaming fails. +Set the pipeline role as the service user's `DEFAULT_ROLE`. Leave **Role** empty or set it to that same role. Otherwise, SQL validation can succeed while streaming fails. ### Generate a key pair -Pipelines authenticates with an RSA key pair. Snowflake requires a key of at least 2048 bits and recommends PKCS #8. Generate an unencrypted private key: +Pipelines authenticates with an RSA key pair. Snowflake requires a key of at least 2048 bits and recommends PKCS #8. Choose one of the following commands to generate `rsa_key.p8`. + +For an unencrypted private key: ```bash -openssl genrsa 2048 | openssl pkcs8 -topk8 -inform PEM -out rsa_key.p8 -nocrypt +openssl genrsa 2048 | openssl pkcs8 -topk8 \ + -inform PEM -out rsa_key.p8 -nocrypt ``` -To use a passphrase, generate an encrypted PKCS #8 private key: +Or, for a passphrase-protected private key: ```bash -openssl genrsa 2048 | openssl pkcs8 -topk8 -v2 des3 -inform PEM -out rsa_key.p8 +openssl genrsa 2048 | openssl pkcs8 -topk8 -v2 des3 \ + -inform PEM -out rsa_key.p8 ``` Derive the public key: @@ -103,27 +122,27 @@ Enter the result as **Account ID**, for example `MYORG-MYACCOUNT`. Do not enter ## Configure Snowflake as a destination -1. Navigate to the [**Database > Replication**](/dashboard/project/_/database/replication) section of the Dashboard. -2. Click **Add destination**. -3. Select **Snowflake**. If it isn't available, [request Early Access](/go/supabase-pipelines-new-destinations). -4. Select a Postgres publication and enter a destination name. -5. Enter the Snowflake settings: - - **Account ID**: The organization and account identifier, such as `MYORG-MYACCOUNT`. - - **User**: The dedicated unquoted service user, such as `PIPELINES_USER`. - - **Database**: The destination database, such as `PIPELINES_DB`. - - **Schema**: The dedicated destination schema, such as `REPLICATED`. - - **Role**: Leave empty to use the user's `DEFAULT_ROLE`. If set, enter the same role. - - **Private key**: The complete PEM-encoded private key, including its begin and end lines. - - **Private key passphrase**: Required only for an encrypted PKCS #8 key. -6. Review the [source table requirements](#source-table-requirements) and click **Create and start pipeline**. +Follow [Set up Pipelines](/docs/guides/database/replication/pipelines#setup-overview) and select **Snowflake**. Enter these settings: -Enter the database and schema identifiers exactly as stored in Snowflake. Unquoted identifiers are stored in uppercase. Managed Pipelines run in **AWS `eu-central-1` (Frankfurt)**. When possible, use a Snowflake account near Frankfurt. +| Field | Value | +| -------------------------- | --------------------------------------------------------------------------------------------------------------- | +| **Account ID** | `MYORG-MYACCOUNT`, for example; use an organization-account identifier | +| **User** | `PIPELINES_USER`, or your unquoted service user | +| **Database** | `PIPELINES_DB`, or your destination database | +| **Schema** | `REPLICATED`, or your dedicated destination schema | +| **Role** | The service user's default role name, or empty; see [role alignment](#keep-the-sql-and-streaming-roles-aligned) | +| **Private key** | Complete PEM, including begin and end lines | +| **Private key passphrase** | Only for an encrypted PKCS #8 key | + +Click **Create and start pipeline** and complete the validation and cost confirmations. + +Enter the database and schema identifiers exactly as stored in Snowflake. Unquoted identifiers are stored in uppercase. Choose an account near the [managed pipeline region](/docs/guides/database/replication/pipelines#region). ## How it works -Pipelines uses Snowflake's SQL REST API to validate the database and schema, create, evolve, and reset destination tables, and apply source `TRUNCATE` operations. It sends initial and ongoing row data through Snowpipe Streaming. These operations do not use a virtual warehouse. +Pipelines uses Snowflake's SQL REST API to validate the database and schema, create, evolve, and recreate destination tables, and apply source `TRUNCATE` operations. It sends initial and ongoing row data through Snowpipe Streaming. -Validation checks authentication, database and schema visibility, and that `QUOTED_IDENTIFIERS_IGNORE_CASE` is `FALSE`. It does not verify that the role can create or own tables, create pipes, or write through Snowpipe Streaming. +Validation checks authentication, database and schema visibility, and that `QUOTED_IDENTIFIERS_IGNORE_CASE` is `FALSE`. It does not verify that the role can create or own tables or write through Snowpipe Streaming. ### Destination table names @@ -134,7 +153,7 @@ Pipelines maps each Postgres schema and table pair to one Snowflake table name. | `public.orders` | `PUBLIC_ORDERS` | | `sales_eu.order_items` | `SALES__EU_ORDER__ITEMS` | -Postgres schema and table names cannot start or end with `_` or contain `"` or `;`. Names that differ only in case map to the same Snowflake name. Use lowercase Postgres names to avoid collisions. Source column names are preserved as quoted identifiers, except for the reserved metadata names below. +Postgres schema and table names cannot start or end with `_` or contain `"` or `;`. Names that differ only in case map to the same Snowflake name. Use lowercase Postgres names to avoid collisions. Source column names are preserved as quoted identifiers, except for the [reserved metadata names](#append-only-change-history). ### Append-only change history @@ -151,16 +170,14 @@ Snowflake tables are an event history, not a current-state replica: - An insert appends the new row. - An update appends the complete new row. It does not append a before image. -- A delete appends the complete old row for `REPLICA IDENTITY FULL`. For a primary-key or `USING INDEX` identity, it appends only the identity columns and sets all other source columns to `NULL`. +- A delete appends the complete old row for `REPLICA IDENTITY FULL`. For a primary-key or `USING INDEX` identity, it sends only the identity columns. Other columns can contain destination defaults or `NULL`; do not treat them as the deleted row's original values. - A source `TRUNCATE` truncates the Snowflake table, resets its streaming state, and does not append a truncate event. -To derive current state, group by a stable source identity and select the row with the latest `_cdc_sequence_number`. Exclude identities whose latest operation is `delete`. See [Query and materialize current state](#query-and-materialize-current-state) for SQL examples. +The sequence number orders changes but is not a globally unique event ID. Snowpipe committed offsets suppress routine replay; consumers must still tolerate [duplicate processing](/docs/guides/database/replication/pipelines-faq#can-data-be-processed-more-than-once). -The sequence number is used for ordering and checkpointing. It is not a globally unique event ID. Pipelines provides at-least-once delivery, so consumers must tolerate duplicates. Snowpipe committed offsets suppress routine replay but do not change this guarantee. +A [table restart](/docs/guides/database/replication/pipelines-monitoring#restarting-tables) drops the Snowflake table and managed streaming state, erasing its history. It cannot recover past events. [Removing a table from the publication](/docs/guides/database/replication/pipelines#removing-tables-from-replication) leaves its destination history in place. -Resetting a table drops and recreates its Snowflake table and managed streaming state. This erases its history. Removing a table from the Postgres publication stops new changes after the pipeline restarts. The existing Snowflake table remains. - -## Query and materialize current state +## Query replicated data [#query-and-materialize-current-state] Use the replicated change history to build a current-state dataset for reports and analytics. Pipelines maintains the history table. You create and maintain the queries, views, or dynamic tables that read it. @@ -275,6 +292,12 @@ The five-minute `target_lag` is an example freshness target relative to the hist Dynamic-table refreshes consume warehouse compute, and the materialized results consume storage. These costs are additional to ingestion and querying. Start with a freshness target that meets your reporting needs and measure a representative workload. A dedicated warehouse helps isolate refresh costs. See Snowflake's [dynamic table cost guide](https://docs.snowflake.com/en/user-guide/dynamic-tables/cost). +### Use streams and tasks + +Snowflake [streams and tasks](https://docs.snowflake.com/en/user-guide/data-pipelines-intro) can maintain a separate table with scheduled `MERGE` statements. Use this option when you need control over the update procedure or schedule. Snowflake's [SCD Type 1 examples](https://docs.snowflake.com/en/user-guide/dynamic-tables/migrate-streams-tasks#scd-type-1-upsert) compare this approach with dynamic tables. + +Adapt the merge to Pipelines' `"_cdc_operation"` and `"_cdc_sequence_number"` columns. A stream on the history table sees appended rows, including rows representing source updates and deletes. Your job must interpret those operations, load existing history, tolerate replay, and rebuild current state after a source truncate or pipeline table reset. + ### Maintain derived objects Pipelines maintains the replicated history table, but does not update your view or dynamic-table definitions. @@ -287,30 +310,6 @@ Pipelines maintains the replicated history table, but does not update your view Recreating a dynamic table initializes its contents again and uses compute. See Snowflake's [dynamic table modification guide](https://docs.snowflake.com/en/user-guide/dynamic-tables/modify) for changes that require reinitialization. -### Use streams and tasks - -Snowflake [streams and tasks](https://docs.snowflake.com/en/user-guide/data-pipelines-intro) can maintain a separate table with scheduled `MERGE` statements. Use this option when you need control over the update procedure or schedule. Snowflake's [SCD Type 1 examples](https://docs.snowflake.com/en/user-guide/dynamic-tables/migrate-streams-tasks#scd-type-1-upsert) compare this approach with dynamic tables. - -Adapt the merge to Pipelines' `"_cdc_operation"` and `"_cdc_sequence_number"` columns. A stream on the history table sees appended rows, including rows representing source updates and deletes. Your job must interpret those operations, load existing history, tolerate replay, and rebuild current state after a source truncate or pipeline table reset. - -## Source table requirements - -Required `REPLICA IDENTITY` depends on the operations enabled in the Postgres publication: - -| Published operations | Required replica identity | -| -------------------- | -------------------------------------------------------------------------------------------------------------- | -| `INSERT` only | No row identity is required. | -| `DELETE` | A primary key, `REPLICA IDENTITY USING INDEX`, or `REPLICA IDENTITY FULL`. Identity columns must be published. | -| `UPDATE` | `REPLICA IDENTITY FULL`. | - -Set full replica identity before publishing updates: - -```sql -alter table public.your_table replica identity full; -``` - -`REPLICA IDENTITY FULL` increases WAL volume, but lets Pipelines construct complete new rows when Postgres omits unchanged out-of-line TOAST values. The setting applies only to new WAL records. If retained WAL already contains an incompatible update, reset the affected table after changing the setting. - ## Type mapping Pipelines creates Snowflake columns with these mappings: @@ -332,46 +331,42 @@ Pipelines uses `VARCHAR` for character and text types, `numeric`, `time with tim Additional limits apply: - Multi-dimensional arrays aren't supported. Non-default lower bounds on one-dimensional arrays aren't preserved. -- Non-finite floating-point and `numeric` values are rejected. +- Non-finite `real` and `double precision` values are rejected. Non-finite `numeric` values are preserved as strings in `VARCHAR` columns. - An uncompressed serialized row larger than 2 MiB is rejected. - Source primary-key, unique, check, length, precision, and nullability constraints aren't copied. Only the two CDC metadata columns are `NOT NULL`. ## Schema change support -Snowflake schema change support is limited during Early Access. +Pipelines supports: -Supported changes: +- Adding, renaming, or dropping columns +- Adding or removing published columns on tracked tables -- Add a column. -- Rename a column. -- Drop a column. +Replicated columns remain nullable in Snowflake, and changes to existing column defaults are not propagated. Initial table creation can copy compatible literal defaults. Columns added in Postgres can copy string, numeric, or boolean literal defaults; other defaults are omitted. Postgres still supplies the source values through replication. -Unsupported or limited changes: +Schema changes also affect stored history: renaming a column changes its name in old events, dropping it removes its historical values, and adding one with a default can populate older rows. -- Changing a column type isn't supported. -- Source table and schema renames aren't supported. -- Changes to nullability or existing column defaults are ignored. -- Initial table creation can copy compatible literal defaults. Added columns can copy string, numeric, or boolean literal defaults. Other defaults are omitted. +Previously excluded columns are added without defaults, leaving historical events `NULL` for those columns. Removing a published column drops its destination values; adding it again does not restore them. -Snowflake DDL changes existing history. Adding a column with a default can populate older rows. Renaming a column changes the historical schema. Dropping a column removes it from old events. Snowflake DDL is not transactional, so an interrupted multi-column change can leave a partially applied schema. Apart from [enabling change tracking](#materialize-with-a-dynamic-table), do not alter managed destination objects manually. If the pipeline remains failed after a restart, [contact support](/dashboard/support/new). +For type changes, unsupported changes, and interrupted schema changes, see the shared [schema-change behavior and recovery](/docs/guides/database/replication/pipelines#schema-change-support). Apart from [enabling change tracking](#materialize-with-a-dynamic-table), do not alter managed destination objects manually. ## Troubleshooting -| Symptom | What to check | -| ----------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| Authentication fails | Confirm the account identifier and user, the registered public-key fingerprint, the complete private-key PEM, and the passphrase. Don't provide a passphrase for an unencrypted key. | -| Database or schema isn't found | Match the database and schema names exactly, including case. Confirm that the role has `USAGE` on both. | -| Validation succeeds but table initialization fails | Confirm that the role has `CREATE TABLE` and `CREATE PIPE` on the schema. Check that another role does not own a table with the same name. Snowflake manages the default pipe. Use a dedicated empty schema. | -| Validation or table creation succeeds but writes fail | Confirm that the service user's `DEFAULT_ROLE` is the pipeline role. Leave **Role** empty or set it to the same role. Confirm that the role has `CREATE PIPE`. Check that Snowflake network policies allow the account control endpoint and discovered Snowpipe ingest host. | -| Updates or deletes fail | Check the publication's operations, replica identity, and included identity columns. Updates require `REPLICA IDENTITY FULL`. | -| A row is rejected | Check for multi-dimensional arrays, non-finite numbers, serialized rows larger than 2 MiB, or source columns named `_cdc_operation` or `_cdc_sequence_number`. | -| A schema change fails | Check the supported changes above. Snowflake DDL can be partially applied, so do not repair managed tables manually. [Contact support](/dashboard/support/new) with the pipeline ID and error details. | +| Symptom | What to check | +| ------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| Authentication fails | [Account identifier](#find-the-account-identifier), user, [key fingerprint and PEM](#generate-a-key-pair), and passphrase. Omit the passphrase for an unencrypted key. | +| Database or schema isn't found | Exact identifier case and `USAGE` permissions on both objects | +| Validation passes but table creation fails | [Grants and ownership](#prepare-snowflake-resources), including `CREATE TABLE` | +| Tables are created but writes fail | [SQL and streaming role alignment](#keep-the-sql-and-streaming-roles-aligned), table permissions, and network access to the account control and discovered Snowpipe ingest endpoints | +| Updates or deletes fail | [Source replica identity and published columns](#source-table-requirements) | +| A row is rejected | [Type and row-size limits](#type-mapping) and [reserved metadata columns](#append-only-change-history) | +| A schema change fails | [Supported changes](#schema-change-support); do not repair managed tables manually | -Use [pipeline monitoring](/docs/guides/database/replication/pipelines-monitoring) and [replication logs](/dashboard/project/_/logs/replication-logs) to inspect table state, lag, and errors. +Use [pipeline monitoring](/docs/guides/database/replication/pipelines-monitoring) to inspect errors. For unresolved failures, [contact support](/dashboard/support/new) with the pipeline ID and error details. ## Additional resources - [Snowflake key-pair authentication](https://docs.snowflake.com/en/user-guide/key-pair-auth) - [Snowflake account identifiers](https://docs.snowflake.com/en/user-guide/admin-account-identifier) - [Snowpipe Streaming default pipe](https://docs.snowflake.com/en/user-guide/snowpipe-streaming/snowpipe-streaming-pipe-object) -- [Snowpipe Streaming operations and privileges](https://docs.snowflake.com/en/user-guide/snowpipe-streaming/snowpipe-streaming-operations) +- [Snowpipe Streaming access control](https://docs.snowflake.com/en/user-guide/snowpipe-streaming/snowpipe-streaming-access-control) diff --git a/apps/docs/content/guides/platform/manage-your-usage/pipelines.mdx b/apps/docs/content/guides/platform/manage-your-usage/pipelines.mdx index fb3585f34f9..2325101bf3c 100644 --- a/apps/docs/content/guides/platform/manage-your-usage/pipelines.mdx +++ b/apps/docs/content/guides/platform/manage-your-usage/pipelines.mdx @@ -3,51 +3,30 @@ id: 'manage-usage-pipelines' title: 'Manage Pipelines usage' --- +Understand [Pipelines charges](#what-you-are-charged-for), compare [billing examples](#billing-examples), and [reduce unnecessary usage](#optimize-usage). + ## What you are charged for -You are charged for configured pipelines and pipeline data processed. Data processed is billed at different rates during initial sync and ongoing replication. Pipelines are charged by the hour for as long as they are configured, including while they are stopped. +Pipelines has three charges: -- **Pipeline hours** measure how long each pipeline remains configured. Delete a pipeline to end this charge. +- **Pipeline hours** measure how long each pipeline remains configured, including while stopped. Delete the pipeline to end this charge. - **Initial sync data processed** is the Postgres row data accepted by the destination when a table is first synchronized or synchronized again. - **Ongoing replication data processed** is the Postgres row data accepted by the destination for subsequent database changes. It depends on how much your published data changes, not on the source table size, WAL size, or destination's compressed storage size. +Tables excluded from **Initial sync** do not incur an initial sync data charge. Pipeline-hour charges and charges for ongoing changes still apply. See [Choosing which tables to copy](/docs/guides/database/replication/pipelines#choosing-which-tables-to-copy). + Destination-provider charges are separate. For example, Google Cloud can charge for BigQuery ingestion, storage, and CDC compute. -## How data processed is measured - -Pipeline data processed is the amount of logical row data emitted by Postgres for replication, successfully processed by a pipeline, and accepted by its destination. It is not based on physical table storage or destination-specific encoding, making usage consistent across destinations. - -The measurement includes: - -- **Initial sync and resynchronization**: Row data emitted by Postgres COPY. -- **Ongoing replication**: Row values Postgres emits for inserts, updates, and deletes. Updates include new row values and any previous identity values Postgres emits. Deletes include the emitted identity values. - -Failed destination write attempts that Pipelines retries are not counted. Data is counted only after the destination acknowledges successful processing. In rare cases, Pipelines can count an acknowledged batch but crash or be interrupted before its replication checkpoint is persisted. Recovery can then process and count the same data again. - -Data successfully processed again as part of a user-requested resynchronization, table restart, or pipeline reset is also counted again. - -### Cost estimates - -The Dashboard provides a quick planning estimate of initial sync volume and cost using information already available about your source tables. It is designed to give you a useful indication before initial sync begins without first scanning and encoding all the data that the sync will process. - -If an estimate is unavailable, you can still create the pipeline or restart tables. - - - -Use this estimate as a planning guide rather than an exact quote. The final volume is measured from the data successfully processed during initial sync and can vary based on your published data and filters. - - - -Actual charges use the logical Postgres row data copied after publication column and row filters and accepted by the destination. - -### Usage on your invoice - -Usage is shown as "ETL Pipeline Hours", "ETL Copy Backfill Data GB", and "ETL Replicated Data GB" on your invoice. - ## Pricing <$Partial path="billing/pricing/pricing_pipelines.mdx" /> +## Cost estimates + +The Dashboard estimates initial sync volume and cost from source table metadata without scanning all rows. You can create a pipeline or restart table replication even if an estimate is unavailable. + +Estimates are planning guidance, not exact quotes. Actual charges reflect row data copied after publication column and row filters and accepted by the destination. + ## Billing examples ### Billing period without an initial sync @@ -88,7 +67,7 @@ Multiple projects had a configured pipeline for the entire month and processed d ### After deleting a pipeline after one day -Pipeline hours are billed in arrears for as long as a pipeline is configured, including while it is stopped. After you delete the pipeline, pipeline-hour billing ends. +Pipeline hours are billed in arrears. Deleting the pipeline after 24 hours ends that charge; other project charges continue. | Line Item | Hours | Costs | | ----------------------------- | ----- | --------------------------- | @@ -106,3 +85,20 @@ Pipeline hours are billed in arrears for as long as a pipeline is configured, in - Include only the tables and columns that you need at the destination. - Keep high-churn tables out of the publication when their changes are not needed for analytics. - If you no longer require replication, delete the pipeline through your [project's replication settings](/dashboard/project/_/database/replication) to stop pipeline-hour charges. + +## How data processed is measured + +Data processed measures logical Postgres row data accepted by the destination, independent of physical storage size or destination encoding. + +The measurement includes: + +- **Initial sync and resynchronization**: Row data emitted by Postgres COPY. +- **Ongoing replication**: Row values Postgres emits for inserts, updates, and deletes. Updates include new row values and any previous identity values Postgres emits. Deletes include the emitted identity values. + +Failed destination write attempts that Pipelines retries are not counted. Data is counted only after the destination acknowledges successful processing. In rare cases, Pipelines can count an acknowledged batch but crash or be interrupted before its replication checkpoint is persisted. Recovery can then process and count the same data again. + +Data successfully processed again after [restarting table replication](/docs/guides/database/replication/pipelines-monitoring#restarting-tables), repeating an interrupted initial sync, or recovering a lost main slot is also counted again. **Restart pipeline** normally resumes saved progress, but [startup recovery](/docs/guides/database/replication/pipelines#pipeline-restarts-and-recovery) can require copying data again. + +## Usage on your invoice + +Usage is shown as "ETL Pipeline Hours", "ETL Copy Backfill Data GB", and "ETL Replicated Data GB" on your invoice. diff --git a/apps/studio/components/interfaces/Database/Replication/DestinationPanel/DestinationForm/AdvancedSettings.tsx b/apps/studio/components/interfaces/Database/Replication/DestinationPanel/DestinationForm/AdvancedSettings.tsx index 472e5b427ef..49efc3e609a 100644 --- a/apps/studio/components/interfaces/Database/Replication/DestinationPanel/DestinationForm/AdvancedSettings.tsx +++ b/apps/studio/components/interfaces/Database/Replication/DestinationPanel/DestinationForm/AdvancedSettings.tsx @@ -218,7 +218,7 @@ export const AdvancedSettings = ({ diff --git a/supa-mdx-lint/Rule003Spelling.toml b/supa-mdx-lint/Rule003Spelling.toml index c4525a61846..f04d5c79a4e 100644 --- a/supa-mdx-lint/Rule003Spelling.toml +++ b/supa-mdx-lint/Rule003Spelling.toml @@ -23,6 +23,7 @@ allow_list = [ "[A-Za-z0-9_-]+(\\.[A-Z-a-z0-9_-]+)+(\\/[A-Za-z0-9_-]+)*", "[Aa]dd-ons?", "AAGUID", + "[Aa]fterward", "[Aa]llowlists?", "[Aa]uditability", "[Aa]utomations?", @@ -36,6 +37,7 @@ allow_list = [ "[Bb]reakpoints?", "[Bb]uilt-ins?", "[Bb]undlers?", + "[Cc]anceled", "[Cc]atalogs?", "[Cc]hangelogs?", "CircleCI", @@ -75,6 +77,7 @@ allow_list = [ "[Ee]xfiltrat(e|ed|es|ing)?", "[Ff]astpath", "[Ff]atals", + "[Ff]avors?", "[Ff]ootguns?", "[Ff]rontend", "[Gg]apless", @@ -141,6 +144,7 @@ allow_list = [ "[Rr]eplayability", "[Rr]epos?", "[Rr]esultingly", + "[Rr]esyncs?", "[Rr]untimes?", "[Ss]anitization", "[Ss]erverless",