* fix(supabase): stop server clients leaking a 30s refresh ticker per request
`autoRefreshToken` defaults to true in supabase-js, and off-browser
@supabase/auth-js starts the refresh ticker unconditionally:
// in non-browser environments the refresh token ticker runs always
this.startAutoRefresh()
That is a setInterval firing every 30 s. It calls unref(), so the process
still exits, tests pass, and Vercel never notices because the process is
torn down long before the tickers accumulate. But unref() does not make a
timer collectable: it stays registered in the event loop and remains a GC
root for its callback, which closes over the GoTrueClient, the
SupabaseClient, and the whole request scope around it.
A long-running self-hosted instance therefore leaks one timer plus one
entire request graph (socket, IncomingMessage, ServerResponse, headers,
route context: ~100 kB) per client constructed. One died of "JavaScript
heap out of memory" after 42 h, the last 24 of them completely idle. The
heap snapshot showed 445 retained request graphs and ~1050 Timeouts in
the 30 000 ms bucket, retained via `autoRefreshTicker`, and the rate
matched the traffic exactly: the Docker healthcheck polls /api/health
every 30 s and the webhook dispatch cron runs every minute, so
3 clients/min x 148 min = 444.
- new lib/supabase/service-client.ts: createServiceRoleClient() applies
SERVER_AUTH_OPTIONS, spread LAST so a caller passing its own auth block
cannot re-enable the ticker
- 22 call sites migrated; only booking-templates/sync/cron had ever
passed the options itself
- guard 9 in no-new-antipatterns.mjs fails CI on any new value import of
supabase-js's createClient outside the wrapper; type-only imports are
fine. Verified to fail on a deliberate regression and pass once fixed
- browser clients untouched: a signed-in tab genuinely needs the refresh,
and lib/supabase/client.ts is built on createBrowserClient anyway
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
* fix(checks): catch namespace imports in the leaky-supabase-client guard
The guard only matched named imports, so
import * as sb from '@supabase/supabase-js'
sb.createClient(url, key)
reached createClient through member access without ever naming it, and
passed. Verified against the real script before and after: the shape is
flagged now, and `import type * as sb` still passes.
Namespace value imports are treated as leaky outright rather than tracking
member access, which keeps the check a regex over source text with no new
dependency.
Review also suggested excluding *.test.tsx alongside *.test.ts. Skipped: the
repo has no .test.tsx files, and all four sibling checks in this file use
`.test.ts`. Diverging in one of them would read as an accident; if such files
appear, all four should change together.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
---------
Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
122 lines
4.1 KiB
TypeScript
122 lines
4.1 KiB
TypeScript
import { createServiceRoleClient } from '@/lib/supabase/service-client'
|
|
import { NextResponse } from 'next/server'
|
|
import { createLogger } from '@/lib/logger'
|
|
|
|
const log = createLogger('health')
|
|
|
|
const CACHE_TTL_MS = 5_000
|
|
|
|
type HealthBody = {
|
|
status: 'healthy' | 'unhealthy'
|
|
timestamp: string
|
|
version: string
|
|
}
|
|
|
|
type CheckResult = {
|
|
body: HealthBody
|
|
status: number
|
|
}
|
|
|
|
type CachedResult = CheckResult & { expires: number }
|
|
|
|
// In-memory cache shared across requests in the same process. Docker's
|
|
// healthcheck polls every 30 s, so the cache always returns fresh data to it,
|
|
// but a public flood (multiple requests/second) is served from RAM and never
|
|
// reaches Postgres. The cache is intentionally tiny (one entry) because the
|
|
// endpoint takes no parameters.
|
|
let cached: CachedResult | null = null
|
|
|
|
// Holds the pending check when one is in flight so concurrent cache misses
|
|
// share a single Postgres round-trip. Cleared as soon as the promise settles.
|
|
// Bounds the worst case to one DB query per CACHE_TTL_MS window regardless of
|
|
// burst arrival rate (e.g. a load-balancer replaying queued probes).
|
|
let pending: Promise<CheckResult> | null = null
|
|
|
|
/**
|
|
* GET /api/health
|
|
* Public health check endpoint (no auth required).
|
|
*
|
|
* Error details are logged server-side only: never echoed to the response
|
|
* body, which would expose Postgres error text on a public endpoint. The
|
|
* logger receives only error.code/error.message; raw Supabase error objects
|
|
* may include schema names, table names, or query fragments that should
|
|
* never reach application logs.
|
|
*
|
|
* Results are cached for {@link CACHE_TTL_MS} so flood traffic does not
|
|
* hammer Postgres with a service-role query per request.
|
|
*/
|
|
export async function GET() {
|
|
const now = Date.now()
|
|
if (cached && cached.expires > now) {
|
|
return NextResponse.json(cached.body, { status: cached.status })
|
|
}
|
|
|
|
const inFlight = pending ?? (pending = runAndCache(now))
|
|
try {
|
|
const result = await inFlight
|
|
return NextResponse.json(result.body, { status: result.status })
|
|
} finally {
|
|
if (pending === inFlight) pending = null
|
|
}
|
|
}
|
|
|
|
async function runAndCache(now: number): Promise<CheckResult> {
|
|
const result = await runHealthCheck()
|
|
cached = { ...result, expires: now + CACHE_TTL_MS }
|
|
return result
|
|
}
|
|
|
|
async function runHealthCheck(): Promise<CheckResult> {
|
|
const supabaseUrl = process.env.NEXT_PUBLIC_SUPABASE_URL
|
|
const supabaseServiceKey = process.env.SUPABASE_SERVICE_ROLE_KEY
|
|
|
|
if (!supabaseUrl || !supabaseServiceKey) {
|
|
log.error('Missing Supabase configuration for health check')
|
|
return {
|
|
body: { status: 'unhealthy', timestamp: new Date().toISOString(), version: '1.0.0' },
|
|
status: 503,
|
|
}
|
|
}
|
|
|
|
try {
|
|
const supabase = createServiceRoleClient(supabaseUrl, supabaseServiceKey)
|
|
const { error } = await supabase
|
|
.from('fiscal_periods')
|
|
.select('id', { count: 'exact', head: true })
|
|
.limit(1)
|
|
|
|
if (error) {
|
|
// PostgrestError is a plain object: { code, message, details, hint }.
|
|
// details/hint can contain table or column names; log only the
|
|
// operationally useful fields.
|
|
log.error('Database health check failed', {
|
|
errCode: error.code ?? null,
|
|
errMessage: error.message ?? null,
|
|
})
|
|
return {
|
|
body: { status: 'unhealthy', timestamp: new Date().toISOString(), version: '1.0.0' },
|
|
status: 503,
|
|
}
|
|
}
|
|
|
|
return {
|
|
body: { status: 'healthy', timestamp: new Date().toISOString(), version: '1.0.0' },
|
|
status: 200,
|
|
}
|
|
} catch (err) {
|
|
// Caught Error instances are reduced to {name, message, code} by the
|
|
// logger's redactor; never pass the raw value lest a deep stack containing
|
|
// query strings ends up in production logs.
|
|
const e = err as { name?: unknown; message?: unknown; code?: unknown }
|
|
log.error('Health check unexpected error', {
|
|
errName: typeof e?.name === 'string' ? e.name : null,
|
|
errMessage: typeof e?.message === 'string' ? e.message : null,
|
|
errCode: typeof e?.code === 'string' ? e.code : null,
|
|
})
|
|
return {
|
|
body: { status: 'unhealthy', timestamp: new Date().toISOString(), version: '1.0.0' },
|
|
status: 503,
|
|
}
|
|
}
|
|
}
|