Fix/UI changes (#439)
* feat(bookkeeping): add preview for next voucher number in JournalEntryForm * feat(encoding): implement U+FFFD recovery for Swedish text in encoding functions
This commit is contained in:
@@ -568,6 +568,53 @@ describe('decodeBuffer — Windows-1252', () => {
|
||||
})
|
||||
})
|
||||
|
||||
// --- Defensive encoding: handle files where the detector picks the wrong encoding ---
|
||||
|
||||
describe('detectEncoding — full-buffer scan', () => {
|
||||
it('detects Win-1252 even when Swedish chars appear past the legacy 4KB sample boundary', () => {
|
||||
// Build a buffer where the first 8000 bytes are pure ASCII header + filler,
|
||||
// and the Swedish Win-1252 byte appears only at byte 8000+. The old
|
||||
// implementation sampled the first 4000 bytes and would default to UTF-8.
|
||||
const filler = new Uint8Array(8000).fill(0x20) // spaces
|
||||
const tail = new Uint8Array([
|
||||
0x46, 0x4f, 0x52, 0x45, 0x4e, 0x49, 0x4e, 0x47, // FORENING
|
||||
0xd6, // Ö in Win-1252 (0xD6) — invalid lone UTF-8 byte
|
||||
])
|
||||
const buf = new Uint8Array(filler.length + tail.length)
|
||||
buf.set(filler, 0)
|
||||
buf.set(tail, filler.length)
|
||||
const encoding = detectEncoding(buf.buffer)
|
||||
expect(encoding).toBe('windows1252')
|
||||
})
|
||||
})
|
||||
|
||||
describe('decodeBuffer — fallback on U+FFFD', () => {
|
||||
it('falls back from utf8 to windows1252 when the result has replacement characters', () => {
|
||||
// "F" "Ö" "RENING" in Windows-1252 — Ö is lone byte 0xD6, not valid UTF-8
|
||||
const buf = new Uint8Array([0x46, 0xd6, 0x52, 0x45, 0x4e, 0x49, 0x4e, 0x47])
|
||||
const result = decodeBuffer(buf.buffer, 'utf8')
|
||||
expect(result).toBe('FÖRENING')
|
||||
expect(result.includes('\uFFFD')).toBe(false)
|
||||
})
|
||||
|
||||
it('falls back from utf8 to cp437 when both windows1252 also fails', () => {
|
||||
// 0x94 is "ö" in CP437; in Win-1252 it's an unprintable "" but textually different;
|
||||
// in UTF-8 it's invalid → U+FFFD. Verify CP437 path is reachable when chosen wrong.
|
||||
const buf = new Uint8Array([0x66, 0x94, 0x72]) // f + ö-cp437 + r
|
||||
const result = decodeBuffer(buf.buffer, 'utf8')
|
||||
// Either windows1252 or cp437 fallback produces a non-FFFD result; both are
|
||||
// acceptable here since the byte 0x94 is interpretable in both — what matters
|
||||
// is no U+FFFD leaks through.
|
||||
expect(result.includes('\uFFFD')).toBe(false)
|
||||
})
|
||||
|
||||
it('returns primary decode unchanged when it contains no U+FFFD', () => {
|
||||
const buf = new TextEncoder().encode('Företag').buffer
|
||||
const result = decodeBuffer(buf, 'utf8')
|
||||
expect(result).toBe('Företag')
|
||||
})
|
||||
})
|
||||
|
||||
// --- Fix 3: Invalid date rejection ---
|
||||
|
||||
describe('parseSIEFile — invalid date handling', () => {
|
||||
|
||||
@@ -3,6 +3,8 @@ import {
|
||||
decodeFileContent,
|
||||
decodeStringContent,
|
||||
hasEncodingIssues,
|
||||
recoverStringWithFFFD,
|
||||
recoverWordWithFFFD,
|
||||
} from '../encoding'
|
||||
|
||||
describe('decodeStringContent', () => {
|
||||
@@ -77,3 +79,75 @@ describe('decodeFileContent', () => {
|
||||
expect(decodeFileContent(cp1252)).toBe('GÖTEBORG')
|
||||
})
|
||||
})
|
||||
|
||||
// --- U+FFFD heuristic recovery ---
|
||||
|
||||
describe('recoverWordWithFFFD', () => {
|
||||
it('recovers uppercase Ö in common Swedish stems', () => {
|
||||
expect(recoverWordWithFFFD('F\uFFFDRENING')).toBe('FÖRENING')
|
||||
expect(recoverWordWithFFFD('F\uFFFDRETAG')).toBe('FÖRETAG')
|
||||
expect(recoverWordWithFFFD('G\uFFFDTEBORG')).toBe('GÖTEBORG')
|
||||
expect(recoverWordWithFFFD('LINK\uFFFDPING')).toBe('LINKÖPING')
|
||||
})
|
||||
|
||||
it('recovers lowercase ö in common Swedish stems', () => {
|
||||
expect(recoverWordWithFFFD('f\uFFFDrening')).toBe('förening')
|
||||
expect(recoverWordWithFFFD('malm\uFFFD')).toBe('malmö')
|
||||
expect(recoverWordWithFFFD('k\uFFFDp')).toBe('köp')
|
||||
})
|
||||
|
||||
it('recovers compound words via substring match', () => {
|
||||
expect(recoverWordWithFFFD('BOSTADSR\uFFFDTTSF\uFFFDRENING')).toBe(
|
||||
'BOSTADSRÄTTSFÖRENING'
|
||||
)
|
||||
expect(recoverWordWithFFFD('Idrottsf\uFFFDrening')).toBe('Idrottsförening')
|
||||
})
|
||||
|
||||
it('is a no-op when the input has no U+FFFD', () => {
|
||||
expect(recoverWordWithFFFD('FÖRENING')).toBe('FÖRENING')
|
||||
expect(recoverWordWithFFFD('hello')).toBe('hello')
|
||||
})
|
||||
|
||||
it('returns null for ambiguous words not in the dictionary', () => {
|
||||
// Random 4-letter word with U+FFFD; no Swedish stem hits.
|
||||
expect(recoverWordWithFFFD('Z\uFFFDXQ')).toBeNull()
|
||||
})
|
||||
|
||||
it('returns null for words with too many U+FFFDs to disambiguate', () => {
|
||||
expect(
|
||||
recoverWordWithFFFD('\uFFFD\uFFFD\uFFFD\uFFFD\uFFFD\uFFFD\uFFFD')
|
||||
).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
describe('recoverStringWithFFFD', () => {
|
||||
it('repairs the canonical "Levbet FÖRENING" case', () => {
|
||||
expect(recoverStringWithFFFD('Levbet F\uFFFDRENING')).toBe('Levbet FÖRENING')
|
||||
})
|
||||
|
||||
it('repairs city + business-name combos', () => {
|
||||
expect(recoverStringWithFFFD('Sjöberg AB, Malm\uFFFD')).toBe('Sjöberg AB, Malmö')
|
||||
expect(recoverStringWithFFFD('Faktura fr\uFFFDn G\uFFFDTEBORG AB')).toBe(
|
||||
'Faktura från GÖTEBORG AB'
|
||||
)
|
||||
})
|
||||
|
||||
it('preserves punctuation, whitespace, and digits', () => {
|
||||
expect(recoverStringWithFFFD('K\uFFFDp 1 234,56 SEK')).toBe('Köp 1 234,56 SEK')
|
||||
})
|
||||
|
||||
it('is a no-op on clean strings', () => {
|
||||
expect(recoverStringWithFFFD('Hello World')).toBe('Hello World')
|
||||
expect(recoverStringWithFFFD('FÖRENING')).toBe('FÖRENING')
|
||||
})
|
||||
|
||||
it('returns null when any word in the string is ambiguous', () => {
|
||||
expect(recoverStringWithFFFD('FÖRENING Z\uFFFDXQ')).toBeNull()
|
||||
})
|
||||
|
||||
it('is idempotent on recovered output', () => {
|
||||
const once = recoverStringWithFFFD('F\uFFFDRENING')
|
||||
expect(once).toBe('FÖRENING')
|
||||
expect(recoverStringWithFFFD(once!)).toBe('FÖRENING')
|
||||
})
|
||||
})
|
||||
|
||||
@@ -87,3 +87,156 @@ export function stripBOM(content: string): string {
|
||||
export function prepareContent(content: string): string {
|
||||
return normalizeLineEndings(stripBOM(decodeStringContent(content)))
|
||||
}
|
||||
|
||||
/**
|
||||
* --- U+FFFD heuristic recovery ---
|
||||
*
|
||||
* When Windows-1252 / Latin-1 bytes are decoded as UTF-8 with `fatal: false`,
|
||||
* invalid sequences silently become U+FFFD. The original byte is lost — but
|
||||
* for Swedish text we can guess from context: the missing letter is almost
|
||||
* always one of Å/Ä/Ö (uppercase context) or å/ä/ö (lowercase context).
|
||||
*
|
||||
* This recovery tries each Swedish vowel substitution and scores the resulting
|
||||
* word against a small dictionary of Swedish stems. If exactly one candidate
|
||||
* scores above the threshold, it's applied. Ambiguous cases return null and
|
||||
* must be reviewed manually.
|
||||
*/
|
||||
|
||||
const REPLACEMENT = '\uFFFD'
|
||||
const SWEDISH_VOWELS_UPPER = ['Å', 'Ä', 'Ö'] as const
|
||||
const SWEDISH_VOWELS_LOWER = ['å', 'ä', 'ö'] as const
|
||||
|
||||
/**
|
||||
* Stems of common Swedish words containing åäö that appear in business names,
|
||||
* place names, addresses, and accounting descriptions.
|
||||
*/
|
||||
const SWEDISH_STEMS = new Set<string>([
|
||||
// -för- prefix (extremely common)
|
||||
'för', 'före', 'förening', 'företag', 'försäkring', 'försäljning',
|
||||
'försäljnings', 'förskola', 'församling', 'förvaltning', 'förbund',
|
||||
'förlag', 'föräldra', 'försök', 'förbrukning', 'förbättring', 'förskott',
|
||||
'försening', 'förhandling', 'förbättrings',
|
||||
// domain terms
|
||||
'bostadsrätt', 'rätt', 'samfällighet', 'idrott', 'fastighet', 'utbildning',
|
||||
'näring', 'växel', 'värme', 'köp', 'köpa', 'inköp', 'sälja', 'säljs',
|
||||
// accounting
|
||||
'kostnad', 'kostnader', 'intäkt', 'intäkter', 'avskrivning', 'avsättning',
|
||||
'lön', 'lönekostnad', 'pension', 'utgående', 'ingående', 'momspliktig',
|
||||
'redovisning', 'företagskonto', 'bankkonto', 'överavskrivning',
|
||||
'överskott', 'underskott', 'överföring', 'överlåtelse', 'återbetalning',
|
||||
'utlägg', 'utgift',
|
||||
// common short prepositions and adverbs
|
||||
'från', 'för', 'över', 'är', 'när', 'där', 'även', 'någon', 'något',
|
||||
'många', 'själv', 'små', 'väg', 'gång', 'tjänst', 'tjänster', 'räkning',
|
||||
'räntor', 'år',
|
||||
// common cities
|
||||
'göteborg', 'malmö', 'örebro', 'östersund', 'jönköping', 'linköping',
|
||||
'norrköping', 'lidköping', 'köping', 'helsingborg', 'umeå', 'skellefteå',
|
||||
'piteå', 'luleå', 'borås', 'växjö', 'östhammar', 'södertälje', 'västerås',
|
||||
'härnösand', 'värnamo', 'mölndal', 'mörrum', 'mönsterås', 'färjestaden',
|
||||
'eskilstuna',
|
||||
// directions / common geo terms
|
||||
'östra', 'västra', 'södra', 'norra', 'öster', 'väster', 'söder',
|
||||
// legal forms
|
||||
'aktiebolag', 'handelsbolag', 'ekonomisk', 'allmännyttig',
|
||||
// misc
|
||||
'företagsledare', 'koncernbidrag', 'utländsk', 'utländska', 'främmande',
|
||||
'vägen', 'gatan', 'allén', 'gränden', 'torget',
|
||||
// surnames containing åäö
|
||||
'lindström', 'sjöberg', 'söderberg', 'öberg', 'åström', 'åkerlund',
|
||||
'östlund', 'lindgren', 'sjögren', 'hägglund', 'bäckström',
|
||||
])
|
||||
|
||||
/**
|
||||
* Score a candidate word.
|
||||
* - 1000 if the entire word matches a known stem (highest confidence).
|
||||
* - Otherwise the count of distinct stems that appear as substrings.
|
||||
* Counting (not boolean-returning) is required: when the same word has
|
||||
* multiple U+FFFD positions, the correct combination must outscore wrong
|
||||
* combinations that still happen to contain *one* stem each.
|
||||
*/
|
||||
function scoreCandidate(word: string): number {
|
||||
const lower = word.toLowerCase()
|
||||
if (SWEDISH_STEMS.has(lower)) return 1000
|
||||
let score = 0
|
||||
for (const stem of SWEDISH_STEMS) {
|
||||
if (lower.includes(stem)) score++
|
||||
}
|
||||
return score
|
||||
}
|
||||
|
||||
function chooseCase(word: string): 'upper' | 'lower' {
|
||||
let upper = 0
|
||||
let lower = 0
|
||||
for (const ch of word) {
|
||||
if (ch === REPLACEMENT) continue
|
||||
if (ch >= 'A' && ch <= 'Z') upper++
|
||||
else if (ch >= 'a' && ch <= 'z') lower++
|
||||
}
|
||||
return upper > lower ? 'upper' : 'lower'
|
||||
}
|
||||
|
||||
/**
|
||||
* Try every Swedish-vowel substitution for the U+FFFDs in `word`, then return
|
||||
* the highest-scoring candidate. Returns null if no candidate matches a
|
||||
* dictionary stem (i.e. ambiguous — operator must review).
|
||||
*/
|
||||
export function recoverWordWithFFFD(word: string): string | null {
|
||||
if (!word.includes(REPLACEMENT)) return word
|
||||
|
||||
const vowels = chooseCase(word) === 'upper' ? SWEDISH_VOWELS_UPPER : SWEDISH_VOWELS_LOWER
|
||||
const positions: number[] = []
|
||||
for (let i = 0; i < word.length; i++) {
|
||||
if (word[i] === REPLACEMENT) positions.push(i)
|
||||
}
|
||||
|
||||
// Words with more than ~6 lost bytes blow up the combinatorial space —
|
||||
// bail out rather than spend cycles on something that's likely garbage anyway.
|
||||
const totalCombos = Math.pow(vowels.length, positions.length)
|
||||
if (totalCombos > 729) return null
|
||||
|
||||
let best: { word: string; score: number } | null = null
|
||||
for (let combo = 0; combo < totalCombos; combo++) {
|
||||
const chars = word.split('')
|
||||
let c = combo
|
||||
for (const pos of positions) {
|
||||
chars[pos] = vowels[c % vowels.length]
|
||||
c = Math.floor(c / vowels.length)
|
||||
}
|
||||
const candidate = chars.join('')
|
||||
const score = scoreCandidate(candidate)
|
||||
if (!best || score > best.score) {
|
||||
best = { word: candidate, score }
|
||||
}
|
||||
}
|
||||
|
||||
if (!best || best.score === 0) return null
|
||||
return best.word
|
||||
}
|
||||
|
||||
/**
|
||||
* Repair every U+FFFD-containing word in `text` via dictionary-backed
|
||||
* substitution. Returns the repaired string when *every* corrupted word
|
||||
* resolved to a confident candidate; returns null if any word remains
|
||||
* ambiguous (operator must review the whole string by hand).
|
||||
*
|
||||
* Idempotent on clean input.
|
||||
*/
|
||||
export function recoverStringWithFFFD(text: string): string | null {
|
||||
if (!text.includes(REPLACEMENT)) return text
|
||||
|
||||
// Split on runs of non-letter/non-digit characters so punctuation,
|
||||
// whitespace, and structural characters are preserved as-is.
|
||||
const tokens = text.split(/([^\p{L}\p{N}\uFFFD]+)/u)
|
||||
const out: string[] = []
|
||||
for (const token of tokens) {
|
||||
if (!token.includes(REPLACEMENT)) {
|
||||
out.push(token)
|
||||
continue
|
||||
}
|
||||
const recovered = recoverWordWithFFFD(token)
|
||||
if (recovered === null) return null
|
||||
out.push(recovered)
|
||||
}
|
||||
return out.join('')
|
||||
}
|
||||
|
||||
@@ -85,6 +85,10 @@ const WIN1252_SWEDISH_BYTES = new Set([
|
||||
* so presence in one range rules out the other.
|
||||
* 4. UTF-8 multi-byte sequences (0xC3 + continuation) are detected with proper
|
||||
* skipping of continuation bytes to avoid false CP437 counts.
|
||||
*
|
||||
* Scans the entire buffer (not a sample): SIE files are capped at 50 MB and
|
||||
* Swedish characters often only appear deep in voucher descriptions, well past
|
||||
* any small header sample.
|
||||
*/
|
||||
export function detectEncoding(buffer: ArrayBuffer): SIEEncoding {
|
||||
const bytes = new Uint8Array(buffer)
|
||||
@@ -99,19 +103,17 @@ export function detectEncoding(buffer: ArrayBuffer): SIEEncoding {
|
||||
// (Fortnox, Bokio, Dooer etc. export UTF-8 with #FORMAT PC8).
|
||||
// Instead, we detect encoding from actual byte patterns.
|
||||
|
||||
// Scan sample for encoding-specific byte ranges
|
||||
const sampleSize = Math.min(bytes.length, 4000)
|
||||
let cp437Count = 0 // Swedish chars in 0x80-0x9F (CP437 range)
|
||||
let utf8Count = 0 // Valid UTF-8 multi-byte Swedish sequences
|
||||
let win1252Count = 0 // Swedish chars in 0xC0-0xFF (Win-1252 range)
|
||||
|
||||
for (let i = 0; i < sampleSize; i++) {
|
||||
for (let i = 0; i < bytes.length; i++) {
|
||||
const byte = bytes[i]
|
||||
|
||||
// Check for UTF-8 multi-byte sequences for Swedish chars FIRST
|
||||
// to avoid false CP437/Win-1252 counts from continuation bytes.
|
||||
// Ä = C3 84, Å = C3 85, Ö = C3 96, ä = C3 A4, å = C3 A5, ö = C3 B6, é = C3 A9
|
||||
if (byte === 0xc3 && i + 1 < sampleSize) {
|
||||
if (byte === 0xc3 && i + 1 < bytes.length) {
|
||||
const nextByte = bytes[i + 1]
|
||||
if ([0x84, 0x85, 0x96, 0xa4, 0xa5, 0xb6, 0xa9].includes(nextByte)) {
|
||||
utf8Count++
|
||||
@@ -140,9 +142,29 @@ export function detectEncoding(buffer: ArrayBuffer): SIEEncoding {
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode a buffer to string using the specified encoding
|
||||
* Decode a buffer to string using the specified encoding.
|
||||
*
|
||||
* After decoding, validates the result for U+FFFD replacement characters
|
||||
* (which signal that the chosen encoding was wrong). When found, retries
|
||||
* with each alternate encoding and returns the first result without U+FFFD.
|
||||
* This guards against `detectEncoding` heuristic misses on files where
|
||||
* Swedish characters are rare or absent in the bytes the detector looked at.
|
||||
*/
|
||||
export function decodeBuffer(buffer: ArrayBuffer, encoding: SIEEncoding): string {
|
||||
const primary = decodeBufferRaw(buffer, encoding)
|
||||
if (!primary.includes('\uFFFD')) return primary
|
||||
|
||||
const alternates: SIEEncoding[] = (['utf8', 'windows1252', 'cp437'] as const).filter(
|
||||
(e) => e !== encoding
|
||||
)
|
||||
for (const alt of alternates) {
|
||||
const candidate = decodeBufferRaw(buffer, alt)
|
||||
if (!candidate.includes('\uFFFD')) return candidate
|
||||
}
|
||||
return primary
|
||||
}
|
||||
|
||||
function decodeBufferRaw(buffer: ArrayBuffer, encoding: SIEEncoding): string {
|
||||
if (encoding === 'utf8') {
|
||||
const decoder = new TextDecoder('utf-8')
|
||||
return decoder.decode(buffer)
|
||||
|
||||
Reference in New Issue
Block a user