diff --git a/app/api/documents/verify/cron/__tests__/route.test.ts b/app/api/documents/verify/cron/__tests__/route.test.ts new file mode 100644 index 00000000..5b3cb3f4 --- /dev/null +++ b/app/api/documents/verify/cron/__tests__/route.test.ts @@ -0,0 +1,253 @@ +/** + * Tests for the nightly document integrity-verify cron. + * + * Covers the two production defects fixed in this route: + * - the run must fit its budget (maxDuration 300 + batch default 200), and + * - a document whose storage object cannot be downloaded must surface as an + * audit incident (INTEGRITY_FAILURE / DOCUMENT_OBJECT_MISSING) AND get its + * last_integrity_check_at stamped so it stops head-blocking the + * nulls-first queue every night. + */ +import { describe, it, expect, vi, beforeEach } from 'vitest' +import { createHash } from 'node:crypto' +import { NextResponse } from 'next/server' +import { verifyCronSecret } from '@/lib/auth/cron' + +vi.mock('@/lib/auth/cron', () => ({ + verifyCronSecret: vi.fn(() => null), +})) + +interface MockDoc { + id: string + user_id: string + company_id: string + storage_path: string + sha256_hash: string + file_name: string +} + +const state = { + documents: [] as MockDoc[], + fetchError: null as { message: string } | null, + downloadResults: new Map(), + updates: [] as Array<{ values: Record; id: string }>, + auditInserts: [] as Array>, + auditInsertError: null as { message: string } | null, + limitCalls: [] as number[], +} + +function makeMockClient() { + return { + from: (table: string) => { + if (table === 'document_attachments') { + return { + select: () => ({ + eq: () => ({ + order: () => ({ + limit: (n: number) => { + state.limitCalls.push(n) + return Promise.resolve({ data: state.documents, error: state.fetchError }) + }, + }), + }), + }), + update: (values: Record) => ({ + eq: (_column: string, id: string) => { + state.updates.push({ values, id }) + return Promise.resolve({ error: null }) + }, + }), + } + } + if (table === 'audit_log') { + return { + insert: (row: Record) => { + state.auditInserts.push(row) + return Promise.resolve({ error: state.auditInsertError }) + }, + } + } + throw new Error(`unexpected table: ${table}`) + }, + storage: { + from: () => ({ + download: (path: string) => + Promise.resolve( + state.downloadResults.get(path) ?? { + data: null, + error: { message: 'Object not found' }, + } + ), + }), + }, + } +} + +vi.mock('@supabase/supabase-js', () => ({ + createClient: vi.fn(() => makeMockClient()), +})) + +import { GET, maxDuration } from '../route' + +function cronRequest(): Request { + return new Request('http://localhost:3000/api/documents/verify/cron') +} + +function makeDoc(overrides: Partial = {}): MockDoc { + return { + id: 'doc-1', + user_id: 'user-1', + company_id: 'company-1', + storage_path: 'user-1/company-1/inbox/file.pdf', + sha256_hash: 'deadbeef', + file_name: 'file.pdf', + ...overrides, + } +} + +/** Register a downloadable object whose bytes hash to the returned sha256. */ +function registerObject(path: string, content: string): string { + const buf = Buffer.from(content, 'utf8') + const arrayBuffer = buf.buffer.slice(buf.byteOffset, buf.byteOffset + buf.byteLength) + state.downloadResults.set(path, { + data: { arrayBuffer: async () => arrayBuffer }, + error: null, + }) + return createHash('sha256').update(buf).digest('hex') +} + +beforeEach(() => { + vi.clearAllMocks() + process.env.NEXT_PUBLIC_SUPABASE_URL = 'https://example.supabase.co' + process.env.SUPABASE_SERVICE_ROLE_KEY = 'service-role-key' + delete process.env.DOCUMENT_VERIFY_BATCH_SIZE + state.documents = [] + state.fetchError = null + state.downloadResults.clear() + state.updates = [] + state.auditInserts = [] + state.auditInsertError = null + state.limitCalls = [] +}) + +describe('GET /api/documents/verify/cron', () => { + it('returns 401 when the cron secret is invalid', async () => { + vi.mocked(verifyCronSecret).mockReturnValueOnce( + NextResponse.json({ error: 'Unauthorized' }, { status: 401 }) + ) + + const response = await GET(cronRequest()) + + expect(response.status).toBe(401) + expect(state.limitCalls).toHaveLength(0) + }) + + it('declares a 300s function budget', () => { + expect(maxDuration).toBe(300) + }) + + it('requests a batch of 200 by default and honors the env override', async () => { + await GET(cronRequest()) + expect(state.limitCalls).toEqual([200]) + + process.env.DOCUMENT_VERIFY_BATCH_SIZE = '50' + await GET(cronRequest()) + expect(state.limitCalls).toEqual([200, 50]) + }) + + it('stamps last_integrity_check_at on a successful verification', async () => { + const doc = makeDoc() + const hash = registerObject(doc.storage_path, '%PDF-1.4 demo content') + state.documents = [{ ...doc, sha256_hash: hash }] + + const response = await GET(cronRequest()) + const json = await response.json() + + expect(json).toEqual({ + processed: 1, + verified: 1, + failures: 0, + missingObjects: 0, + errors: 0, + }) + expect(state.updates).toHaveLength(1) + expect(state.updates[0].id).toBe(doc.id) + expect(state.updates[0].values.last_integrity_check_at).toEqual(expect.any(String)) + expect(state.auditInserts).toHaveLength(0) + }) + + it('writes an INTEGRITY_FAILURE audit row and still stamps on hash mismatch', async () => { + const doc = makeDoc({ sha256_hash: 'not-the-real-hash' }) + registerObject(doc.storage_path, 'tampered content') + state.documents = [doc] + + const response = await GET(cronRequest()) + const json = await response.json() + + expect(json.failures).toBe(1) + expect(json.missingObjects).toBe(0) + expect(state.updates).toHaveLength(1) + expect(state.auditInserts).toHaveLength(1) + expect(state.auditInserts[0].action).toBe('INTEGRITY_FAILURE') + expect(String(state.auditInserts[0].description)).not.toContain('DOCUMENT_OBJECT_MISSING') + }) + + it('surfaces a missing storage object as an audit incident AND stamps the check', async () => { + const missing = makeDoc({ id: 'doc-missing', storage_path: 'user-1/company-1/gone.pdf' }) + const healthy = makeDoc({ id: 'doc-healthy', storage_path: 'user-1/company-1/ok.pdf' }) + const healthyHash = registerObject(healthy.storage_path, 'healthy content') + state.documents = [missing, { ...healthy, sha256_hash: healthyHash }] + + const response = await GET(cronRequest()) + const json = await response.json() + + // The audit row is the incident surface for the missing object. + expect(state.auditInserts).toHaveLength(1) + const audit = state.auditInserts[0] + expect(audit.action).toBe('INTEGRITY_FAILURE') + expect(audit.record_id).toBe('doc-missing') + expect(String(audit.description)).toContain('DOCUMENT_OBJECT_MISSING') + expect(audit.new_state).toMatchObject({ reason: 'DOCUMENT_OBJECT_MISSING' }) + + // Both documents are stamped: the failing one must stop head-blocking + // the nulls-first queue, and the healthy one was verified. + expect(state.updates.map((u) => u.id).sort()).toEqual(['doc-healthy', 'doc-missing']) + + expect(json).toEqual({ + processed: 2, + verified: 1, + failures: 0, + missingObjects: 1, + errors: 1, + }) + }) + + it('does not stamp a missing object when the audit insert fails, so it retries next run', async () => { + const missing = makeDoc({ id: 'doc-missing', storage_path: 'user-1/company-1/gone.pdf' }) + state.documents = [missing] + state.auditInsertError = { message: 'insert blocked' } + + const response = await GET(cronRequest()) + const json = await response.json() + + expect(state.auditInserts).toHaveLength(1) + expect(state.updates).toHaveLength(0) + expect(json.missingObjects).toBe(0) + expect(json.errors).toBe(1) + }) + + it('returns an error envelope when the document fetch fails', async () => { + state.fetchError = { message: 'db down' } + + const response = await GET(cronRequest()) + + expect(response.status).toBeGreaterThanOrEqual(500) + }) + + it('reports zero processed when there is nothing to verify', async () => { + const response = await GET(cronRequest()) + const json = await response.json() + + expect(json).toEqual({ message: 'No documents to verify', processed: 0 }) + }) +}) diff --git a/app/api/documents/verify/cron/route.ts b/app/api/documents/verify/cron/route.ts index f72cde26..36f4aee1 100644 --- a/app/api/documents/verify/cron/route.ts +++ b/app/api/documents/verify/cron/route.ts @@ -4,11 +4,23 @@ import { withCronContext } from '@/lib/api/with-cron-context' import { errorResponse, errorResponseFromCode } from '@/lib/errors/get-structured-error' /** - * GET /api/documents/verify/cron: weekly Sunday 03:00 UTC. + * GET /api/documents/verify/cron: nightly 03:00 UTC (schedule in vercel.json). * Spot-checks WORM archive integrity by recomputing SHA-256 for the next * batch of documents and writing INTEGRITY_FAILURE rows to the audit log - * for any mismatches. + * for any mismatches. Documents whose storage object cannot be downloaded + * get an INTEGRITY_FAILURE row marked DOCUMENT_OBJECT_MISSING and are still + * stamped as checked so they stop head-blocking the nulls-first queue. */ + +// Vercel function budget; verification is sequential, see batch size below. +export const maxDuration = 300 + +// Measured ~0.8s per document (download + hash + stamp), so 200 documents +// finish in ~160s with headroom inside the 300s budget. The previous default +// of 500 hit the platform timeout around item ~250 every night, so the tail +// of the queue was never reached. +const DEFAULT_VERIFY_BATCH_SIZE = 200 + export const GET = withCronContext('cron.documents_verify', async (_request, ctx) => { const supabaseUrl = process.env.NEXT_PUBLIC_SUPABASE_URL const supabaseServiceKey = process.env.SUPABASE_SERVICE_ROLE_KEY @@ -27,7 +39,7 @@ export const GET = withCronContext('cron.documents_verify', async (_request, ctx .select('id, user_id, company_id, storage_path, sha256_hash, file_name') .eq('is_current_version', true) .order('last_integrity_check_at', { ascending: true, nullsFirst: true }) - .limit(parseInt(process.env.DOCUMENT_VERIFY_BATCH_SIZE || '500', 10)) + .limit(parseInt(process.env.DOCUMENT_VERIFY_BATCH_SIZE || '', 10) || DEFAULT_VERIFY_BATCH_SIZE) if (fetchError) { ctx.log.error('failed to fetch documents for verify', fetchError) @@ -40,6 +52,7 @@ export const GET = withCronContext('cron.documents_verify', async (_request, ctx let verified = 0 let failures = 0 + let missingObjects = 0 const summary = await ctx.forEach('document', documents, async (doc, itemCtx) => { const { data: fileData, error: downloadError } = await supabase.storage @@ -47,7 +60,44 @@ export const GET = withCronContext('cron.documents_verify', async (_request, ctx .download(doc.storage_path) if (downloadError || !fileData) { - throw new Error(downloadError?.message || 'download_failed') + // The storage object is unreadable: surface it as an incident in the + // audit log. The action stays INTEGRITY_FAILURE because the DB check + // constraint audit_log_action_check allows a fixed set of actions; + // the DOCUMENT_OBJECT_MISSING marker in description and new_state + // distinguishes a missing object from a hash mismatch. + const reason = downloadError?.message || 'download_failed' + const { error: auditError } = await supabase.from('audit_log').insert({ + user_id: doc.user_id, + company_id: doc.company_id, + action: 'INTEGRITY_FAILURE', + table_name: 'document_attachments', + record_id: doc.id, + description: `DOCUMENT_OBJECT_MISSING: storage object for document "${doc.file_name}" at "${doc.storage_path}" could not be downloaded: ${reason}`, + old_state: { sha256_hash: doc.sha256_hash }, + new_state: { reason: 'DOCUMENT_OBJECT_MISSING', download_error: reason }, + }) + + if (auditError) { + // Leave last_integrity_check_at untouched so the document is retried + // (and the incident write re-attempted) on the next run. + throw new Error(`audit insert failed for missing object: ${auditError.message}`) + } + + // Stamp the check so the row stops sorting to the head of the + // nulls-first queue every night; the audit row above is the durable + // incident surface. + await supabase + .from('document_attachments') + .update({ last_integrity_check_at: new Date().toISOString() }) + .eq('id', doc.id) + + missingObjects++ + itemCtx.log.error('document object missing', new Error(reason), { + documentId: doc.id, + fileName: doc.file_name, + storagePath: doc.storage_path, + }) + throw new Error(`DOCUMENT_OBJECT_MISSING: ${reason}`) } const buffer = await fileData.arrayBuffer() @@ -90,6 +140,7 @@ export const GET = withCronContext('cron.documents_verify', async (_request, ctx processed: summary.total, verified, failures, + missingObjects, downloadErrors: summary.failed, }) @@ -97,6 +148,7 @@ export const GET = withCronContext('cron.documents_verify', async (_request, ctx processed: summary.total, verified, failures, + missingObjects, errors: summary.failed, }) }) diff --git a/scripts/seed-demo-account.ts b/scripts/seed-demo-account.ts index 88f3544a..1af84d25 100644 --- a/scripts/seed-demo-account.ts +++ b/scripts/seed-demo-account.ts @@ -22,6 +22,7 @@ import { createClient } from '@supabase/supabase-js' import { config as dotenv } from 'dotenv' +import { createHash } from 'node:crypto' import { resolve } from 'node:path' import { encryptPersonnummer } from '@/lib/salary/personnummer' @@ -1890,19 +1891,36 @@ async function seedInboxAndUncategorized( ): Promise { console.log('[6] inbox AWS PDF + 5 uncategorized + voucher gaps') - // Synthetic AWS PDF storage row (no actual file upload: storage path - // exists for demo, file content can be uploaded later via UI) - const fakeHash = 'demo' + Math.random().toString(36).slice(2, 18).padEnd(60, '0') + // Synthetic AWS invoice PDF: upload a tiny but valid PDF so the nightly + // integrity-verify cron can download the object and match its real + // SHA-256, instead of failing forever on a fabricated hash with no file. + const awsPdfBuffer = Buffer.from( + [ + '%PDF-1.4', + '1 0 obj << /Type /Catalog /Pages 2 0 R >> endobj', + '2 0 obj << /Type /Pages /Kids [3 0 R] /Count 1 >> endobj', + '3 0 obj << /Type /Page /Parent 2 0 R /MediaBox [0 0 595 842] >> endobj', + 'trailer << /Root 1 0 R >>', + '%%EOF', + ].join('\n'), + 'utf8' + ) + const awsPdfPath = `${ctx.userId}/${ctx.companyId}/inbox/aws-2026-05-05.pdf` + const { error: uploadErr } = await sb.storage + .from('documents') + .upload(awsPdfPath, awsPdfBuffer, { contentType: 'application/pdf', upsert: true }) + if (uploadErr) throw new Error(`storage upload AWS PDF: ${uploadErr.message}`) + const awsPdfHash = createHash('sha256').update(awsPdfBuffer).digest('hex') const { data: doc, error: docErr } = await sb .from('document_attachments') .insert({ user_id: ctx.userId, company_id: ctx.companyId, - storage_path: `${ctx.userId}/${ctx.companyId}/inbox/aws-2026-05-05.pdf`, + storage_path: awsPdfPath, file_name: 'aws-2026-05-05.pdf', - file_size_bytes: 124567, + file_size_bytes: awsPdfBuffer.length, mime_type: 'application/pdf', - sha256_hash: fakeHash, + sha256_hash: awsPdfHash, version: 1, is_current_version: true, uploaded_by: ctx.userId,