* fix(import): handle BOMs at the byte level in decodeFileContent Inspect leading bytes before decoding: EF BB BF strips the UTF-8 BOM and decodes the remainder (falling back to Windows-1252 for the remainder only, so the fallback can no longer produce a literal mojibake prefix), and FF FE / FE FF decode as UTF-16LE/BE. stripBOM additionally strips a literal mojibake BOM prefix for string paths pre-decoded elsewhere. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(import): make an explicit SEB choice at least as good as auto-detect Three changes for the SEB bank CSV report: - parseBankFile: when an explicit format parses 0 transactions, fall back to auto-detection; a different format that parses rows is returned with a prepended info issue naming both formats. A working explicit parse is never overridden, and explicit generic_csv (the manual mapping escape hatch) is exempt. - SEB profile: sniff the header delimiter (';' vs ',') and split with the quote-aware parseCSVLine; accept a bare Datum date column as a lowest priority tier in parse only, never in detect. Its user-reachable issue strings are now Swedish. - Import page: when a parse yields 0 transactions, show the parser's real issues instead of only the generic no-transactions hint. The v1 agent route now decodes through the shared decodeFileContent and stamps external ids, import_source, and the stored file format from the format the parse result actually carries, so fallback imports dedup identically to auto-detected ones. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * docs(api-v1): the bank import route also decodes UTF-16 CodeRabbit on #1565: decodeFileContent gained UTF-16LE/BE BOM support but the route overview and the registered endpoint description still listed only UTF-8 / Windows-1252. Skill regenerated (apiskill:generate). Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: Jakob Wennberg <311770904+jakobwennberg-oss@users.noreply.github.com> Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
204 lines
6.7 KiB
TypeScript
204 lines
6.7 KiB
TypeScript
/**
|
|
* Bank file parser: main entry point
|
|
*
|
|
* Auto-detects Swedish bank file formats and parses to normalized transactions.
|
|
* Supports Nordea, SEB, Swedbank, Handelsbanken CSV and ISO 20022 camt.053 XML.
|
|
*/
|
|
|
|
import * as crypto from 'crypto'
|
|
import type { BankFileFormat, BankFileFormatId, BankFileParseResult, ParsedBankTransaction } from './types'
|
|
import { nordeaFormat } from './formats/nordea'
|
|
import { nordeaBusinessFormat } from './formats/nordea-business'
|
|
import { sebFormat } from './formats/seb'
|
|
import { swedbankFormat } from './formats/swedbank'
|
|
import { handelsbankenFormat } from './formats/handelsbanken'
|
|
import { lansforsakringarFormat } from './formats/lansforsakringar'
|
|
import { icaBankenFormat } from './formats/ica-banken'
|
|
import { skandiaFormat } from './formats/skandia'
|
|
import { lunarFormat } from './formats/lunar'
|
|
import { northmillFormat } from './formats/northmill'
|
|
import { wiseFormat } from './formats/wise'
|
|
import { wiseStatementFormat } from './formats/wise-statement'
|
|
import { camt053Format } from './formats/camt053'
|
|
import { genericCSVFormat } from './formats/generic-csv'
|
|
|
|
/**
|
|
* Ordered list of format detectors.
|
|
* camt.053 first (XML detection is unambiguous), then bank-specific CSV formats.
|
|
* New bank formats go after existing ones but before generic_csv.
|
|
* Generic CSV is last: it never auto-detects (manual fallback only).
|
|
*/
|
|
const FORMATS: BankFileFormat[] = [
|
|
camt053Format,
|
|
nordeaFormat,
|
|
nordeaBusinessFormat,
|
|
sebFormat,
|
|
swedbankFormat,
|
|
handelsbankenFormat,
|
|
lansforsakringarFormat,
|
|
icaBankenFormat,
|
|
skandiaFormat,
|
|
lunarFormat,
|
|
northmillFormat,
|
|
wiseFormat,
|
|
wiseStatementFormat,
|
|
genericCSVFormat,
|
|
]
|
|
|
|
/**
|
|
* Get a format by its ID
|
|
*/
|
|
export function getFormat(id: BankFileFormatId): BankFileFormat | undefined {
|
|
return FORMATS.find((f) => f.id === id)
|
|
}
|
|
|
|
/**
|
|
* Get all available formats
|
|
*/
|
|
export function getAllFormats(): BankFileFormat[] {
|
|
return FORMATS
|
|
}
|
|
|
|
/**
|
|
* Auto-detect the bank file format from content and filename
|
|
*
|
|
* Returns the first matching format, or null if no format matches.
|
|
* Uses filename extension as a hint (e.g. .xml for camt.053).
|
|
*/
|
|
export function detectFileFormat(content: string, filename: string): BankFileFormat | null {
|
|
for (const format of FORMATS) {
|
|
if (format.detect(content, filename)) {
|
|
return format
|
|
}
|
|
}
|
|
return null
|
|
}
|
|
|
|
/**
|
|
* Parse a bank file with auto-detection or explicit format
|
|
*
|
|
* @param content - File content as string (already decoded)
|
|
* @param filename - Original filename (used for format detection hints)
|
|
* @param formatId - Optional explicit format to use (skips auto-detection)
|
|
*/
|
|
export function parseBankFile(
|
|
content: string,
|
|
filename: string,
|
|
formatId?: BankFileFormatId
|
|
): BankFileParseResult {
|
|
let format: BankFileFormat | undefined
|
|
|
|
if (formatId) {
|
|
format = getFormat(formatId)
|
|
if (!format) {
|
|
return {
|
|
format: formatId,
|
|
format_name: 'Unknown',
|
|
transactions: [],
|
|
date_from: null,
|
|
date_to: null,
|
|
issues: [{ row: 0, message: `Okänt format: ${formatId}`, severity: 'error' }],
|
|
stats: { total_rows: 0, parsed_rows: 0, skipped_rows: 0, total_income: 0, total_expenses: 0 },
|
|
}
|
|
}
|
|
|
|
const explicitResult = format.parse(content)
|
|
if (explicitResult.transactions.length > 0 || formatId === 'generic_csv') {
|
|
// A working explicit parse is never overridden. generic_csv is also
|
|
// exempt: it is the manual column-mapping escape hatch and its default
|
|
// mapping legitimately parses 0 rows before the user maps columns.
|
|
return explicitResult
|
|
}
|
|
|
|
// The explicit choice parsed nothing: fall back to auto-detection so an
|
|
// explicitly selected bank is never WORSE than "Automatisk identifiering".
|
|
// detectFileFormat can never return generic_csv (its detect() is always
|
|
// false), so this cannot reroute the UI into the mapping flow.
|
|
const detected = detectFileFormat(content, filename)
|
|
if (detected && detected.id !== formatId) {
|
|
const detectedResult = detected.parse(content)
|
|
if (detectedResult.transactions.length > 0) {
|
|
return {
|
|
...detectedResult,
|
|
issues: [
|
|
{
|
|
row: 0,
|
|
message: `Filen matchade inte det valda formatet (${format.name}) och tolkades istället som ${detected.name}.`,
|
|
severity: 'info',
|
|
},
|
|
...detectedResult.issues,
|
|
],
|
|
}
|
|
}
|
|
}
|
|
|
|
return explicitResult
|
|
} else {
|
|
format = detectFileFormat(content, filename) || undefined
|
|
if (!format) {
|
|
// Build diagnostic message listing which formats were tried
|
|
const tried = FORMATS
|
|
.filter(f => f.id !== 'generic_csv')
|
|
.map(f => f.name)
|
|
const firstLine = content.split('\n')[0]?.substring(0, 80) || ''
|
|
return {
|
|
format: 'generic_csv',
|
|
format_name: 'Unknown',
|
|
transactions: [],
|
|
date_from: null,
|
|
date_to: null,
|
|
issues: [{
|
|
row: 0,
|
|
message: `Kunde inte identifiera bankformat. Testade: ${tried.join(', ')}. Första raden: "${firstLine}". Välj bank manuellt eller använd "Annan CSV".`,
|
|
severity: 'error',
|
|
}],
|
|
stats: { total_rows: 0, parsed_rows: 0, skipped_rows: 0, total_income: 0, total_expenses: 0 },
|
|
}
|
|
}
|
|
}
|
|
|
|
return format.parse(content)
|
|
}
|
|
|
|
/**
|
|
* Generate a stable external_id for a parsed bank transaction.
|
|
*
|
|
* For CSV files: SHA-256 of (format + date + description + amount + row_index)
|
|
* For camt.053: Uses the entry reference from the XML if available
|
|
*
|
|
* Two identical transactions on the same day will get different IDs due to row_index.
|
|
*/
|
|
export function generateExternalId(
|
|
tx: ParsedBankTransaction,
|
|
formatId: BankFileFormatId,
|
|
rowIndex: number
|
|
): string {
|
|
// For camt.053, prefer the raw_line which contains the entry reference
|
|
if (formatId === 'camt053' && tx.raw_line && !tx.raw_line.startsWith('camt053_entry_')) {
|
|
return `camt053_${tx.raw_line}`
|
|
}
|
|
|
|
// Wise carries the stable transfer ID (TRANSFER-…, PLAN_ORDER-…, plus a
|
|
// `-fee` suffix for fee rows) in raw_line: use it so re-importing the same
|
|
// statement dedups exactly instead of relying on the row hash.
|
|
if (formatId === 'wise' && tx.raw_line) {
|
|
return `wise_${tx.raw_line}`
|
|
}
|
|
|
|
if (formatId === 'wise_statement' && tx.raw_line) {
|
|
return `wise_${tx.raw_line}`
|
|
}
|
|
|
|
// For CSV formats, create a composite hash
|
|
const composite = `${formatId}|${tx.date}|${tx.description}|${tx.amount}|${rowIndex}`
|
|
const hash = crypto.createHash('sha256').update(composite).digest('hex').substring(0, 16)
|
|
return `${formatId}_${hash}`
|
|
}
|
|
|
|
/**
|
|
* Generate a file hash for dedup of the same file being uploaded twice
|
|
*/
|
|
export function generateFileHash(content: string): string {
|
|
return crypto.createHash('sha256').update(content).digest('hex')
|
|
}
|