* fix(invoice-inbox): read whole PDFs (last-page slice + truncation retry) PDF extraction read only part of well-structured PDFs, two confirmed mechanisms (21-day prod window: 49 sliced docs, 29 silent empties): - The auto-extract page budget was 3 (Bedrock-latency legacy, issue #553) and the slice kept only the first pages, so multi-page invoices lost the final page where totals, OCR and 'Att betala' sit. The budget is now 8 on pdf-native backends (Claude reads PDFs directly); the slice always keeps the last page. Rasterizing self-host backends keep the old budget of 3. - A max_tokens-truncated model answer was parsed as-is, failed, and became an all-null extraction with no trace. extractFromDocument now reports stop_reason max_tokens / finish_reason length as truncated; the extractor retries once at double AI_EXTRACTION_MAX_TOKENS and logs ai_extraction_truncated either way. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_012xosyW53HUa9JoFiDayhSk * fix(invoice-inbox): sweep cutoff covers the slower two-call extraction Skeptic finding on #2014: the crash-recovery sweep flipped 'processing' rows to an empty skeleton after 2 minutes, but a deferred extraction can now legitimately run 3-5 minutes (8 native pages plus one truncation retry at a doubled token cap), so the sweep stole the row and the CAS discarded the worker's real result. Cutoff raised to 10 minutes. Also: pages_partial_note made period-agnostic (old rows were extracted from first-pages-only slices, so naming the last page was retroactively wrong for them), and two stale first-pages-only comments updated. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_012xosyW53HUa9JoFiDayhSk * fix(invoice-inbox): keep the first extraction response when the retry throws CodeRabbit finding on #2014: a throttled/failed retry call bubbled to the outer catch before rawText was assigned, discarding a first response whose text may parse fine despite the truncation flag. The retry is now caught locally (logged as ai_extraction_retry_failed) and the first result flows on. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_012xosyW53HUa9JoFiDayhSk --------- Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
99 lines
3.4 KiB
TypeScript
99 lines
3.4 KiB
TypeScript
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
|
import { PDFDocument } from 'pdf-lib'
|
|
import {
|
|
slicePdfForExtraction,
|
|
maxPagesForAutoExtract,
|
|
MAX_PAGES_FOR_AUTO_EXTRACT,
|
|
MAX_PAGES_FOR_AUTO_EXTRACT_NATIVE,
|
|
} from '@/extensions/general/invoice-inbox/lib/upload-and-extract'
|
|
|
|
// Pages get index-encoded widths (500 + i) so tests can assert exactly WHICH
|
|
// source pages survived the slice, not just how many.
|
|
async function makePdf(pageCount: number): Promise<ArrayBuffer> {
|
|
const pdf = await PDFDocument.create()
|
|
for (let i = 0; i < pageCount; i++) pdf.addPage([500 + i, 700])
|
|
const bytes = await pdf.save()
|
|
const out = new ArrayBuffer(bytes.byteLength)
|
|
new Uint8Array(out).set(bytes)
|
|
return out
|
|
}
|
|
|
|
async function pageWidths(buffer: ArrayBuffer): Promise<number[]> {
|
|
const pdf = await PDFDocument.load(buffer)
|
|
return Array.from({ length: pdf.getPageCount() }, (_, i) =>
|
|
Math.round(pdf.getPage(i).getSize().width)
|
|
)
|
|
}
|
|
|
|
describe('slicePdfForExtraction', () => {
|
|
it('keeps the first maxPages-1 pages plus the LAST page (where totals sit)', async () => {
|
|
const sliced = await slicePdfForExtraction(await makePdf(10), 8)
|
|
expect(sliced).not.toBeNull()
|
|
// Pages 0..6 plus page 9: widths 500..506 and 509.
|
|
expect(await pageWidths(sliced!)).toEqual([500, 501, 502, 503, 504, 505, 506, 509])
|
|
})
|
|
|
|
it('keeps the last page also on the old 3-page budget', async () => {
|
|
const sliced = await slicePdfForExtraction(await makePdf(5), 3)
|
|
expect(await pageWidths(sliced!)).toEqual([500, 501, 504])
|
|
})
|
|
|
|
it('copies the document unchanged when it fits the budget', async () => {
|
|
const sliced = await slicePdfForExtraction(await makePdf(3), 8)
|
|
expect(await pageWidths(sliced!)).toEqual([500, 501, 502])
|
|
})
|
|
|
|
it('returns null on an unparseable buffer', async () => {
|
|
const garbage = new TextEncoder().encode('not a pdf').buffer as ArrayBuffer
|
|
expect(await slicePdfForExtraction(garbage, 8)).toBeNull()
|
|
})
|
|
})
|
|
|
|
describe('maxPagesForAutoExtract', () => {
|
|
const ENV = [
|
|
'AWS_ACCESS_KEY_ID',
|
|
'AWS_SECRET_ACCESS_KEY',
|
|
'ANTHROPIC_API_KEY',
|
|
'AI_PROVIDER',
|
|
'AI_BASE_URL',
|
|
'AI_API_KEY',
|
|
'AI_MODEL',
|
|
'AI_PDF_MODE',
|
|
] as const
|
|
let saved: Record<string, string | undefined> = {}
|
|
beforeEach(() => {
|
|
saved = {}
|
|
for (const k of ENV) {
|
|
saved[k] = process.env[k]
|
|
delete process.env[k]
|
|
}
|
|
})
|
|
afterEach(() => {
|
|
for (const k of ENV) {
|
|
if (saved[k] === undefined) delete process.env[k]
|
|
else process.env[k] = saved[k]
|
|
}
|
|
})
|
|
|
|
it('uses the higher budget when the backend reads PDFs natively (Claude)', () => {
|
|
process.env.AWS_ACCESS_KEY_ID = 'AKIAEXAMPLE'
|
|
process.env.AWS_SECRET_ACCESS_KEY = 'secret'
|
|
expect(maxPagesForAutoExtract()).toBe(MAX_PAGES_FOR_AUTO_EXTRACT_NATIVE)
|
|
})
|
|
|
|
it('keeps the conservative budget on a rasterizing OpenAI-compatible backend', () => {
|
|
process.env.AI_PROVIDER = 'openai-compatible'
|
|
process.env.AI_BASE_URL = 'http://localhost:8000/v1'
|
|
process.env.AI_MODEL = 'some-model'
|
|
expect(maxPagesForAutoExtract()).toBe(MAX_PAGES_FOR_AUTO_EXTRACT)
|
|
})
|
|
|
|
it('follows AI_PDF_MODE=native on an OpenAI-compatible backend', () => {
|
|
process.env.AI_PROVIDER = 'openai-compatible'
|
|
process.env.AI_BASE_URL = 'http://localhost:8000/v1'
|
|
process.env.AI_MODEL = 'some-model'
|
|
process.env.AI_PDF_MODE = 'native'
|
|
expect(maxPagesForAutoExtract()).toBe(MAX_PAGES_FOR_AUTO_EXTRACT_NATIVE)
|
|
})
|
|
})
|