From c39e284641dc61de253c927bb223af471195c98c Mon Sep 17 00:00:00 2001 From: Jordi Enric <37541088+jordienr@users.noreply.github.com> Date: Wed, 15 Apr 2026 16:33:29 +0200 Subject: [PATCH] Observability: remove healthy/unhealthy badge from Service Health (#44771) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Problem The "Healthy / Unhealthy" badge on the Observability overview was alarming — showing **UNHEALTHY** even when every bar in the chart looked fine. Two root causes: 1. **The threshold is aggressive.** Any period where the aggregate error rate is ≥ 1% flips the badge to "Unhealthy", even if that 1% came from a short burst that is visually indistinguishable in the chart. 2. **Period-wide aggregation hides spikes.** The badge status is computed over the entire selected time window (e.g. 24 h). A 5-minute spike at 20% errors diluted across 24 h of mostly-clean traffic can push the aggregate just over 1%, triggering "Unhealthy" while all chart bars look green. The badge wording ("Unhealthy") also implies a current service problem, whereas the underlying metric is a historical aggregate — making it easy to misread. ## Change Remove the badge entirely. The per-row error/warning rate indicator (e.g. `● 1.34% errors`) already surfaces the key signal without the alarming label, and the bar chart lets users see the actual shape of traffic over time. ## On spike visibility in charts The charts already use **COUNT per time bucket** (not averages), so individual bars faithfully represent event volume. The bucket granularity does compress spikes for longer windows (hourly buckets for 1–3 day views, daily for 7-day), but that's a separate concern from the badge. If we want to surface burst detection in the future, a better approach would be per-bucket threshold highlighting rather than a single period-wide badge. https://claude.ai/code/session_01E1ejWyuR9BV4qcTyiGGVVY ## Summary by CodeRabbit * **Bug Fixes** * Removed the service health status indicator from the Service Health Table. * **New Features** * Replaced per-row bar charts with a line chart showing error/warning rates alongside OK series. * Added a centered "No data" placeholder when chart data is empty and preserved click interactions on chart points. * Y-axis values now display as percentages. --------- Co-authored-by: Claude --- .../Observability/ObservabilityOverview.tsx | 23 +++----- .../Observability/ServiceHealthTable.tsx | 59 ++----------------- 2 files changed, 13 insertions(+), 69 deletions(-) diff --git a/apps/studio/components/interfaces/Observability/ObservabilityOverview.tsx b/apps/studio/components/interfaces/Observability/ObservabilityOverview.tsx index 32871b08ff7..b0c7d614ac0 100644 --- a/apps/studio/components/interfaces/Observability/ObservabilityOverview.tsx +++ b/apps/studio/components/interfaces/Observability/ObservabilityOverview.tsx @@ -112,25 +112,21 @@ export const ObservabilityOverview = () => { const dbServiceData = overviewData.services.db - // Creates a 1-hour time window for the clicked bar for log filtering + // Navigate to the log view scoped to the clicked bar's bucket window const handleBarClick = useCallback( - (serviceKey: string, logsUrl: string) => (datum: any) => { + (logsUrl: string) => (datum: any) => { if (!datum?.timestamp) return - const datumTimestamp = dayjs(datum.timestamp) - // Round down to the start of the hour - const start = datumTimestamp.startOf('hour').toISOString() - // Add 1 hour to get the end of the hour - const end = datumTimestamp.startOf('hour').add(1, 'hour').toISOString() - - const queryParams = new URLSearchParams({ - its: start, - ite: end, - }) + // datum.timestamp is already the UTC-truncated bucket boundary from timestamp_trunc(), + // so use it directly to avoid local-timezone startOf() misalignment (e.g. UTC+5:30). + const unit = interval === '1hr' ? 'minute' : 'hour' + const start = datum.timestamp + const end = dayjs.utc(datum.timestamp).add(1, unit).toISOString() + const queryParams = new URLSearchParams({ its: start, ite: end }) router.push(`${logsUrl}?${queryParams.toString()}`) }, - [router] + [router, interval] ) return ( @@ -181,7 +177,6 @@ export const ObservabilityOverview = () => { }))} serviceData={overviewData.services} onBarClick={handleBarClick} - interval={interval} datetimeFormat={datetimeFormat} /> diff --git a/apps/studio/components/interfaces/Observability/ServiceHealthTable.tsx b/apps/studio/components/interfaces/Observability/ServiceHealthTable.tsx index 0a13b6feaed..ab092f3ffab 100644 --- a/apps/studio/components/interfaces/Observability/ServiceHealthTable.tsx +++ b/apps/studio/components/interfaces/Observability/ServiceHealthTable.tsx @@ -1,11 +1,11 @@ import { ChevronRight, HelpCircle } from 'lucide-react' import Link from 'next/link' -import { Badge, Card, CardContent, Loading, Tooltip, TooltipContent, TooltipTrigger } from 'ui' +import { Card, CardContent, Loading, Tooltip, TooltipContent, TooltipTrigger } from 'ui' import { LogsBarChart } from 'ui-patterns/LogsBarChart' import { ButtonTooltip } from '../../ui/ButtonTooltip' import type { LogsBarChartDatum } from '../ProjectHome/ProjectUsage.metrics' -import { getHealthStatus, type ServiceKey } from './ObservabilityOverview.utils' +import type { ServiceKey } from './ObservabilityOverview.utils' type ServiceConfig = { key: ServiceKey @@ -24,13 +24,10 @@ type ServiceData = { isLoading: boolean } -type IntervalKey = '1hr' | '1day' | '7day' - export type ServiceHealthTableProps = { services: ServiceConfig[] serviceData: Record - onBarClick: (serviceKey: string, logsUrl: string) => (datum: LogsBarChartDatum) => void - interval: IntervalKey + onBarClick: (logsUrl: string) => (datum: LogsBarChartDatum) => void datetimeFormat: string } @@ -43,41 +40,6 @@ const SERVICE_DESCRIPTIONS: Record = { postgrest: 'Auto-generated REST API for your database', } -const getStatusLabel = (status: 'healthy' | 'error' | 'unknown'): string => { - switch (status) { - case 'healthy': - return 'Healthy' - case 'error': - return 'Unhealthy' - case 'unknown': - return 'Unknown' - } -} - -const getStatusTooltip = (status: 'healthy' | 'error' | 'unknown'): string => { - switch (status) { - case 'healthy': - return 'Error rate is below 1%' - case 'error': - return 'Error rate is 1% or higher' - case 'unknown': - return 'Insufficient data (fewer than 100 requests)' - } -} - -const getStatusVariant = ( - status: 'healthy' | 'error' | 'unknown' -): 'success' | 'warning' | 'destructive' | 'default' => { - switch (status) { - case 'healthy': - return 'success' - case 'error': - return 'destructive' - case 'unknown': - return 'default' - } -} - type ServiceRowProps = { service: ServiceConfig data: ServiceData @@ -86,11 +48,6 @@ type ServiceRowProps = { } const ServiceRow = ({ service, data, onBarClick, datetimeFormat }: ServiceRowProps) => { - const { status } = getHealthStatus(data.errorRate, data.total) - const statusLabel = getStatusLabel(status) - const statusVariant = getStatusVariant(status) - const statusTooltip = getStatusTooltip(status) - const errorRate = data.total > 0 ? data.errorRate : 0 const warningRate = data.total > 0 ? (data.warningCount / data.total) * 100 : 0 @@ -119,14 +76,6 @@ const ServiceRow = ({ service, data, onBarClick, datetimeFormat }: ServiceRowPro
- - - {statusLabel} - - -

{statusTooltip}

-
-
)