mirror of
https://github.com/supabase/supabase.git
synced 2026-10-05 09:25:06 +03:00
* fix(studio): check job_run_details size in cron display When cron.job_run_details grows too large (200k+ rows), loading the cron jobs overview can timeout and affect other queries by pulling excessive data into shared buffers. This change: - Estimates table size using pg_stat before fetching cron jobs data - Shows a cleanup notice when the table exceeds the threshold - Provides batched deletion using ctid ranges to avoid buffer pollution - Allows scheduling an automated daily cleanup cron job - Handles timeout errors gracefully with a "suspected overflow" state The useCronJobsData hook now returns a discriminated union status that tracks loading, estimate-error, overflow-confirmed, overflow-suspected, and ready states, allowing the UI to respond appropriately to each case. * fix(studio): use index when querying cron.job_run_details cron.job_run_details is only indexed by runid, not by start_time. Change the query to use the runid index (which gives the same result, since runid is auto-incrementing).
206 lines
6.7 KiB
TypeScript
206 lines
6.7 KiB
TypeScript
import { useCallback, useRef, useState } from 'react'
|
|
import { toast } from 'sonner'
|
|
|
|
import type { ConnectionVars } from '@/data/common.types'
|
|
import { useExecuteSqlMutation } from 'data/sql/execute-sql-mutation'
|
|
import {
|
|
CTID_BATCH_PAGE_SIZE,
|
|
getDeleteOldCronJobRunDetailsByCtidKey,
|
|
getDeleteOldCronJobRunDetailsByCtidSql,
|
|
getJobRunDetailsPageCountKey,
|
|
getJobRunDetailsPageCountSql,
|
|
getScheduleDeleteCronJobRunDetailsKey,
|
|
getScheduleDeleteCronJobRunDetailsSql,
|
|
} from 'data/sql/queries/delete-cron-job-run-details'
|
|
import { CLEANUP_INTERVALS } from './CronJobsTab.constants'
|
|
|
|
// Delay between batches to allow other queries to proceed (in milliseconds)
|
|
const BATCH_DELAY_MS = 100
|
|
|
|
type UseCronJobsCleanupActionsOptions = ConnectionVars
|
|
|
|
export interface BatchDeletionProgress {
|
|
currentBatch: number
|
|
totalBatches: number
|
|
totalRowsDeleted: number
|
|
}
|
|
|
|
export type CleanupState =
|
|
| { status: 'idle' }
|
|
| { status: 'deleting'; progress: BatchDeletionProgress }
|
|
| { status: 'delete-success'; totalRowsDeleted: number }
|
|
| { status: 'delete-error'; error: string }
|
|
| { status: 'scheduling' }
|
|
| { status: 'schedule-success' }
|
|
| { status: 'schedule-error'; error: string }
|
|
|
|
export const useCronJobsCleanupActions = ({
|
|
projectRef,
|
|
connectionString,
|
|
}: UseCronJobsCleanupActionsOptions) => {
|
|
const [cleanupInterval, setCleanupInterval] = useState(CLEANUP_INTERVALS[0].value)
|
|
const [cleanupState, setCleanupState] = useState<CleanupState>({ status: 'idle' })
|
|
|
|
// Ref to track cancellation
|
|
const cancelledRef = useRef(false)
|
|
|
|
const { mutateAsync: executeSql } = useExecuteSqlMutation({
|
|
onError: () => {}, // Error handled inline
|
|
})
|
|
|
|
/**
|
|
* Run batched deletion using ctid ranges.
|
|
* This approach scans the table in page chunks to avoid:
|
|
* - Buffer cache pollution from full table scans
|
|
* - Long-running transactions that block vacuum
|
|
* - Lock accumulation from deleting millions of rows at once
|
|
*/
|
|
const runBatchedDeletion = useCallback(
|
|
async (interval: string) => {
|
|
if (!projectRef) {
|
|
console.error('[CronJobsTab > batch deletion] Project reference is required')
|
|
toast.error('There was an error running the cleanup. Please try again.')
|
|
return
|
|
}
|
|
|
|
cancelledRef.current = false
|
|
|
|
try {
|
|
// Step 1: Get the total number of pages in the table
|
|
setCleanupState({
|
|
status: 'deleting',
|
|
progress: { currentBatch: 0, totalBatches: 0, totalRowsDeleted: 0 },
|
|
})
|
|
|
|
const pageCountResult = await executeSql({
|
|
projectRef,
|
|
connectionString,
|
|
sql: getJobRunDetailsPageCountSql(),
|
|
queryKey: getJobRunDetailsPageCountKey(projectRef),
|
|
})
|
|
|
|
const rawTotalPages = pageCountResult.result?.[0]?.num_pages ?? 0
|
|
const totalPages = Number(rawTotalPages)
|
|
if (!Number.isFinite(totalPages) || totalPages < 0) {
|
|
throw new Error(
|
|
`[CronJobs > cleanup actions] Invalid page count returned: ${rawTotalPages}`
|
|
)
|
|
}
|
|
|
|
if (totalPages === 0) {
|
|
setCleanupState({ status: 'delete-success', totalRowsDeleted: 0 })
|
|
toast.success('The job_run_details table is empty.')
|
|
return
|
|
}
|
|
|
|
const totalBatches = Math.ceil(totalPages / CTID_BATCH_PAGE_SIZE)
|
|
let totalRowsDeleted = 0
|
|
|
|
// Step 2: Iterate through pages in batches
|
|
for (let batch = 0; batch < totalBatches; batch++) {
|
|
// Check for cancellation
|
|
if (cancelledRef.current) {
|
|
setCleanupState({ status: 'idle' })
|
|
toast.info('Deletion cancelled.')
|
|
return
|
|
}
|
|
|
|
const startPage = batch * CTID_BATCH_PAGE_SIZE
|
|
const endPage = Math.min((batch + 1) * CTID_BATCH_PAGE_SIZE, totalPages + 1)
|
|
|
|
setCleanupState({
|
|
status: 'deleting',
|
|
progress: {
|
|
currentBatch: batch + 1,
|
|
totalBatches,
|
|
totalRowsDeleted,
|
|
},
|
|
})
|
|
|
|
const deleteResult = await executeSql({
|
|
projectRef,
|
|
connectionString,
|
|
sql: getDeleteOldCronJobRunDetailsByCtidSql(interval, startPage, endPage),
|
|
queryKey: getDeleteOldCronJobRunDetailsByCtidKey(projectRef, interval, startPage),
|
|
})
|
|
|
|
const deletedCount = deleteResult.result?.[0]?.deleted_count ?? 0
|
|
totalRowsDeleted += deletedCount
|
|
|
|
if (cancelledRef.current) {
|
|
setCleanupState({ status: 'idle' })
|
|
toast.info('Deletion cancelled.')
|
|
return
|
|
}
|
|
|
|
if (batch < totalBatches - 1) {
|
|
await new Promise((resolve) => setTimeout(resolve, BATCH_DELAY_MS))
|
|
}
|
|
}
|
|
|
|
setCleanupState({ status: 'delete-success', totalRowsDeleted })
|
|
toast.success(
|
|
`Deleted ${totalRowsDeleted.toLocaleString()} cron job runs older than ${interval}.`
|
|
)
|
|
} catch (error) {
|
|
console.error('[CronJobs] Batch deletion failed with error: %O', error)
|
|
const errorMessage = error instanceof Error ? error.message : 'Unknown error'
|
|
setCleanupState({ status: 'delete-error', error: errorMessage })
|
|
toast.error('Running the cleanup failed. Please try again.')
|
|
}
|
|
},
|
|
[projectRef, connectionString, executeSql]
|
|
)
|
|
|
|
/**
|
|
* Schedule a daily cleanup job.
|
|
* This should only be called after a successful initial deletion.
|
|
*/
|
|
const scheduleCleanup = useCallback(
|
|
async (interval: string) => {
|
|
if (!projectRef) {
|
|
console.error('[CronJobsTab > schedule cleanup] Project reference is required')
|
|
toast.error('There was an error scheduling the cleanup. Please try again.')
|
|
return
|
|
}
|
|
|
|
try {
|
|
setCleanupState({ status: 'scheduling' })
|
|
|
|
await executeSql({
|
|
projectRef,
|
|
connectionString,
|
|
sql: getScheduleDeleteCronJobRunDetailsSql(interval),
|
|
queryKey: getScheduleDeleteCronJobRunDetailsKey(projectRef, interval),
|
|
})
|
|
|
|
setCleanupState({ status: 'schedule-success' })
|
|
toast.success('Scheduled daily cleanup job.')
|
|
} catch (error) {
|
|
console.error('[CronJobs] Failed to schedule cleanup with error: %O', error)
|
|
const errorMessage = error instanceof Error ? error.message : 'Unknown error'
|
|
setCleanupState({ status: 'schedule-error', error: errorMessage })
|
|
toast.error('Scheduling the cleanup job failed. Please try again.')
|
|
}
|
|
},
|
|
[projectRef, connectionString, executeSql]
|
|
)
|
|
|
|
/**
|
|
* Cancel an in-progress deletion.
|
|
*/
|
|
const cancelDeletion = useCallback(() => {
|
|
cancelledRef.current = true
|
|
setCleanupState({ status: 'idle' })
|
|
}, [])
|
|
|
|
return {
|
|
cleanupInterval,
|
|
setCleanupInterval,
|
|
cleanupState,
|
|
runBatchedDeletion,
|
|
scheduleCleanup,
|
|
cancelDeletion,
|
|
}
|
|
}
|