Files
accounted/lib/parties/name-extract.ts
T
82859d01db feat(parties): name the company inside a voucher text, and stop asking SCB about foreign ones (#2265)
* feat(parties): name the company inside a voucher text, and stop asking SCB about foreign ones

The registry picker searched SCB on the whole display name, which for an
assistant-written voucher is a sentence, so "1511768101 · Visma Spcs AB,
faktura ..." never matched and foreign suppliers produced an empty list
with no explanation.

- lib/parties/name-extract.ts: name candidates read out of the text,
  anchored on legal-form words (AB, AB (publ), Inc., Ltd, B.V., GmbH, Oy,
  ...) and on country words, plus EU VAT numbers. Every candidate is a
  substring of the text; foreign forms and countries mark the candidate
  as one SCB cannot hold.
- Suggestions: the display name prefers the legal person named in the
  text ("TIC identity" becomes "The Intelligence Company AB (publ)"),
  the voucher texts are stored as a ledger fact for the picker, the
  country is stored when the text says, and a single foreign VAT number
  in the text becomes the party's VAT number.
- GET .../enrich/candidates plans the search: Swedish legal person first,
  cleaned head last, at most three queries, stopping at the first hit;
  no SCB call when the best reading is foreign, the response says which
  company it read and where.
- Picker: "X ser ut att vara ett utländskt bolag (Irland). SCB:s register
  täcker bara svenska företag." with a hint to save by name and VAT
  number; alternate readings offered as one-click searches when the
  first found nothing.
- nameQuery strips stacked legal-form suffixes ("AB (publ)").
- The queue builds itself whenever the books hold counterparts it has
  not seen, not only on a first visit; the toast only appears when
  something was created.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

* fix(parties): take a text-derived VAT number only on the expense side

A customer's VAT number steers reverse charge on outgoing invoices, so it
must come from a document or a person, never from a text heuristic. A
supplier's is informational and may still be read from the voucher text.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

---------

Co-authored-by: Jakob Wennberg <311770904+jakobwennberg-oss@users.noreply.github.com>
Co-authored-by: Claude Fable 5.1 <noreply@anthropic.com>
2026-09-04 13:43:41 +02:00

280 lines
13 KiB
TypeScript

/**
* Parties: which company a voucher text is talking about.
*
* Descriptions written by people and by the assistant carry the counterpart
* somewhere inside a sentence: "1511768101 · Visma Spcs AB, faktura
* 2025-10-02", "TIC identity BG 0000005786439 Bg-bet. via internet · Faktura
* 20250746, The Intelligence Company AB (publ)", "Utlägg Framer · Framer B.V.
* (NL), webbdesignverktyg". The ledger key groups such vouchers; this module
* names them. It anchors on legal-form words (AB, Inc., B.V., GmbH, Oy, ...)
* and on country words, and reads EU VAT numbers out of the text. Nothing is
* generated: every candidate is a substring of the text, returned in the
* order worth trying against a register. A foreign legal form or country
* means SCB's register cannot hold the company, so the caller can say so
* instead of searching in vain.
*/
import { displayNameFromVoucherText } from './ledger-key'
export interface NameCandidate {
/** The name as written, with its legal form when there is one. */
name: string
/** Canonical legal form, e.g. 'AB', 'B.V.', 'Inc.'. */
legalForm?: string
/** ISO 3166-1 alpha-2 when the form or the text says so. */
country?: string
/** Not a Swedish legal person: SCB's register cannot hold it. */
foreign: boolean
source: 'legal_form' | 'country' | 'head'
}
export interface VatNumberHit {
vat: string
country: string
}
interface FormSpec {
pattern: string
canonical: string
foreign: boolean
country?: string
/** Short uppercase tokens are matched as written; words are not. */
caseSensitive: boolean
}
// Order matters where one form contains another (Pte. Ltd. before Ltd,
// Oyj before Oy, ASA before AS, AB (publ) before AB).
const FORMS: FormSpec[] = [
{ pattern: 'AB\\s*\\(publ\\)', canonical: 'AB (publ)', foreign: false, caseSensitive: true },
{ pattern: 'Aktiebolag(?:et)?', canonical: 'AB', foreign: false, caseSensitive: false },
{ pattern: 'AB', canonical: 'AB', foreign: false, caseSensitive: true },
{ pattern: 'HB', canonical: 'HB', foreign: false, caseSensitive: true },
{ pattern: 'KB', canonical: 'KB', foreign: false, caseSensitive: true },
{ pattern: 'ekonomisk förening', canonical: 'ek. för.', foreign: false, caseSensitive: false },
{ pattern: 'ek\\.?\\s*för\\.?', canonical: 'ek. för.', foreign: false, caseSensitive: false },
{ pattern: 'Pte\\.?\\s*Ltd\\.?', canonical: 'Pte. Ltd.', foreign: true, country: 'SG', caseSensitive: false },
{ pattern: 'Pty\\.?\\s*Ltd\\.?', canonical: 'Pty Ltd', foreign: true, country: 'AU', caseSensitive: false },
{ pattern: 'Inc\\.?', canonical: 'Inc.', foreign: true, country: 'US', caseSensitive: true },
{ pattern: 'Incorporated', canonical: 'Inc.', foreign: true, country: 'US', caseSensitive: false },
{ pattern: 'Corp\\.?', canonical: 'Corp.', foreign: true, country: 'US', caseSensitive: true },
{ pattern: 'Corporation', canonical: 'Corp.', foreign: true, country: 'US', caseSensitive: false },
{ pattern: 'LLC|L\\.L\\.C\\.', canonical: 'LLC', foreign: true, country: 'US', caseSensitive: true },
{ pattern: 'PBC', canonical: 'PBC', foreign: true, country: 'US', caseSensitive: true },
{ pattern: 'Ltd\\.?', canonical: 'Ltd', foreign: true, caseSensitive: true },
{ pattern: 'Limited', canonical: 'Ltd', foreign: true, caseSensitive: false },
{ pattern: 'PLC|plc', canonical: 'PLC', foreign: true, country: 'GB', caseSensitive: true },
{ pattern: 'LLP', canonical: 'LLP', foreign: true, caseSensitive: true },
{ pattern: 'GmbH(?:\\s*&\\s*Co\\.?\\s*KG)?', canonical: 'GmbH', foreign: true, country: 'DE', caseSensitive: true },
{ pattern: 'e\\.V\\.', canonical: 'e.V.', foreign: true, country: 'DE', caseSensitive: true },
{ pattern: 'AG', canonical: 'AG', foreign: true, caseSensitive: true },
{ pattern: 'B\\.V\\.|BV', canonical: 'B.V.', foreign: true, country: 'NL', caseSensitive: true },
{ pattern: 'N\\.V\\.|NV', canonical: 'N.V.', foreign: true, country: 'NL', caseSensitive: true },
{ pattern: 'Oyj', canonical: 'Oyj', foreign: true, country: 'FI', caseSensitive: true },
{ pattern: 'Oy', canonical: 'Oy', foreign: true, country: 'FI', caseSensitive: true },
{ pattern: 'ApS', canonical: 'ApS', foreign: true, country: 'DK', caseSensitive: true },
{ pattern: 'A/S', canonical: 'A/S', foreign: true, country: 'DK', caseSensitive: true },
{ pattern: 'ASA', canonical: 'ASA', foreign: true, country: 'NO', caseSensitive: true },
{ pattern: 'AS', canonical: 'AS', foreign: true, caseSensitive: true },
{ pattern: 'S\\.A\\.S\\.|SAS', canonical: 'SAS', foreign: true, country: 'FR', caseSensitive: true },
{ pattern: 'SARL|S\\.à\\.?\\s?r\\.l\\.|Sàrl|Sarl', canonical: 'SARL', foreign: true, caseSensitive: true },
{ pattern: 'S\\.A\\.', canonical: 'S.A.', foreign: true, caseSensitive: true },
{ pattern: 'S\\.L\\.', canonical: 'S.L.', foreign: true, country: 'ES', caseSensitive: true },
{ pattern: 'S\\.r\\.l\\.|Srl', canonical: 'S.r.l.', foreign: true, country: 'IT', caseSensitive: true },
{ pattern: 'S\\.p\\.A\\.|SpA', canonical: 'S.p.A.', foreign: true, country: 'IT', caseSensitive: true },
{ pattern: 'Kft\\.?', canonical: 'Kft.', foreign: true, country: 'HU', caseSensitive: true },
{ pattern: 'Zrt\\.?', canonical: 'Zrt.', foreign: true, country: 'HU', caseSensitive: true },
{ pattern: 'Sp\\.?\\s*z\\s*o\\.?\\s*o\\.?', canonical: 'Sp. z o.o.', foreign: true, country: 'PL', caseSensitive: false },
{ pattern: 'UAB', canonical: 'UAB', foreign: true, country: 'LT', caseSensitive: true },
{ pattern: 'OÜ', canonical: 'OÜ', foreign: true, country: 'EE', caseSensitive: true },
{ pattern: 'SIA', canonical: 'SIA', foreign: true, country: 'LV', caseSensitive: true },
{ pattern: 's\\.r\\.o\\.', canonical: 's.r.o.', foreign: true, caseSensitive: false },
{ pattern: 'd\\.o\\.o\\.', canonical: 'd.o.o.', foreign: true, caseSensitive: false },
{ pattern: 'Lda\\.?', canonical: 'Lda', foreign: true, country: 'PT', caseSensitive: true },
]
const FORM_REGEXES = FORMS.map((f) => ({
spec: f,
re: new RegExp(`(?:^|[\\s(])(${f.pattern})(?=$|[\\s.,;:)])`, f.caseSensitive ? 'u' : 'iu'),
}))
// Country words as they appear in voucher text, Swedish and English.
const COUNTRY_WORDS: Array<[RegExp, string]> = [
[/\b(?:Sverige|Sweden)\b/iu, 'SE'],
[/\b(?:Ireland|Irland)\b/iu, 'IE'],
[/\b(?:USA|U\.S\.A\.|United States)\b/iu, 'US'],
[/\b(?:UK|U\.K\.|United Kingdom|Storbritannien|England)\b/u, 'GB'],
[/\b(?:Nederländerna|Netherlands|Holland)\b/iu, 'NL'],
[/\b(?:Cypern|Cyprus)\b/iu, 'CY'],
[/\b(?:Tyskland|Germany|Deutschland)\b/iu, 'DE'],
[/\b(?:Finland)\b/iu, 'FI'],
[/\b(?:Danmark|Denmark)\b/iu, 'DK'],
[/\b(?:Norge|Norway)\b/iu, 'NO'],
[/\b(?:Frankrike|France)\b/iu, 'FR'],
[/\b(?:Spanien|Spain)\b/iu, 'ES'],
[/\b(?:Italien|Italy)\b/iu, 'IT'],
[/\b(?:Singapore)\b/iu, 'SG'],
[/\b(?:Estland|Estonia)\b/iu, 'EE'],
[/\b(?:Lettland|Latvia)\b/iu, 'LV'],
[/\b(?:Litauen|Lithuania)\b/iu, 'LT'],
[/\b(?:Polen|Poland)\b/iu, 'PL'],
[/\b(?:Schweiz|Switzerland)\b/iu, 'CH'],
[/\b(?:Österrike|Austria)\b/iu, 'AT'],
[/\b(?:Belgien|Belgium)\b/iu, 'BE'],
[/\b(?:Luxemburg|Luxembourg)\b/iu, 'LU'],
[/\b(?:Portugal)\b/iu, 'PT'],
[/\b(?:Tjeckien|Czechia|Czech Republic)\b/iu, 'CZ'],
[/\b(?:Ungern|Hungary)\b/iu, 'HU'],
[/\b(?:Kanada|Canada)\b/iu, 'CA'],
[/\b(?:Australien|Australia)\b/iu, 'AU'],
[/\b(?:Indien|India)\b/iu, 'IN'],
[/\b(?:Kina|China)\b/iu, 'CN'],
[/\b(?:Japan)\b/iu, 'JP'],
]
// Two- or three-letter codes only inside parentheses: "(NL)", "(USA)".
const CODE_IN_PARENS = /\((?:[^()]*?,\s*)?([A-Z]{2,3})\)/u
const CODE_MAP: Record<string, string> = {
USA: 'US', US: 'US', UK: 'GB', GB: 'GB', IE: 'IE', NL: 'NL', CY: 'CY', DE: 'DE', FI: 'FI', DK: 'DK', NO: 'NO',
FR: 'FR', ES: 'ES', IT: 'IT', SG: 'SG', EE: 'EE', LV: 'LV', LT: 'LT', PL: 'PL', CH: 'CH', AT: 'AT', BE: 'BE',
LU: 'LU', PT: 'PT', CZ: 'CZ', HU: 'HU', CA: 'CA', AU: 'AU', IN: 'IN', CN: 'CN', JP: 'JP', SE: 'SE',
}
// Words that precede a name without being part of it.
const LEAD_WORDS = new Set([
'utlägg', 'faktura', 'fakturor', 'leverantörsfaktura', 'levfakt', 'levfkt', 'kundfaktura', 'kvitto', 'betalning',
'kundbet', 'kundbetalning', 'kundinbetalning', 'levbet', 'leverantörsbetalning', 'utbet', 'inbet', 'betalt', 'betald',
'delbetalning', 'delbet', 'inbetalning', 'utbetalning', 'till', 'från', 'av', 'för', 'hos', 'via', 'och', 'rättelse',
'ankomst', 'ref', 'inköp', 'köp', 'abonnemang', 'prenumeration', 'månadsavgift', 'avgift', 'konsult', 'tjänst',
'kortköp/uttag', 'kortköp', 'uttag', 'överföring', 'internet', 'bg-bet.', 'bg-bet', 'pg-bet.', 'autogiro',
])
const VAT_RE = /\b(AT|BE|BG|HR|CY|CZ|DK|EE|FI|FR|DE|EL|HU|IE|IT|LV|LT|LU|MT|NL|PL|PT|RO|SK|SI|ES|SE|GB|XI)\s?([0-9A-Z]{8,12})\b/gu
function stripVatSweden(text: string): string {
// "VAT-Sweden 25%" is a tax line on foreign invoices, not a country.
return text.replace(/VAT\s?-?\s?Sweden/giu, ' ')
}
function countryHint(text: string): string | undefined {
const cleaned = stripVatSweden(text)
const code = CODE_IN_PARENS.exec(cleaned)?.[1]
if (code && CODE_MAP[code]) return CODE_MAP[code]
for (const [re, country] of COUNTRY_WORDS) if (re.test(cleaned)) return country
return undefined
}
/** EU-style VAT numbers written in the text, deduplicated, SE included. */
export function extractVatNumbers(text: string): VatNumberHit[] {
const out = new Map<string, VatNumberHit>()
for (const m of text.matchAll(VAT_RE)) {
const body = m[2]!
if ((body.match(/\d/g) ?? []).length < 7) continue
const vat = `${m[1]}${body}`
if (!out.has(vat)) out.set(vat, { vat, country: m[1]! })
}
return [...out.values()]
}
function splitSegments(text: string): string[] {
return text
.split(/\s*(?:·|,|;|:|\||\n|\s[-–—]\s)\s*/u)
.map((s) => s.trim())
.filter(Boolean)
}
function nameTokensBefore(before: string): string[] {
let tokens = before.trim().split(/\s+/u).filter(Boolean)
// A parenthesis closes whatever came before it: "(ankomst 2) Acme".
const lastParen = tokens.map((t) => t.includes(')')).lastIndexOf(true)
if (lastParen >= 0) tokens = tokens.slice(lastParen + 1)
while (tokens.length) {
const t = tokens[0]!
const bare = t.replace(/^[("'`]+|[)"'`]+$/gu, '')
if (!/\p{L}/u.test(bare) || LEAD_WORDS.has(bare.toLowerCase()) || /^\d{4}-\d{2}(-\d{2})?$/u.test(bare)) {
tokens.shift()
continue
}
break
}
return tokens.slice(-6).map((t) => t.replace(/^[("'`]+|[)"'`.]+$/gu, ''))
}
function legalFormCandidate(segment: string, next: string | undefined, whole: string): NameCandidate | null {
let best: { index: number; length: number; spec: FormSpec } | null = null
for (const { spec, re } of FORM_REGEXES) {
const m = re.exec(segment)
if (!m) continue
const index = m.index + m[0].length - m[1]!.length
if (!best || index < best.index) best = { index, length: m[1]!.length, spec }
}
if (!best) return null
const tokens = nameTokensBefore(segment.slice(0, best.index))
if (tokens.length === 0) return null
const after = `${segment.slice(best.index + best.length)} ${next ?? ''}`
const country = best.spec.country ?? countryHint(after) ?? countryHint(whole)
const foreign = best.spec.foreign
return {
name: `${tokens.join(' ')} ${best.spec.canonical}`,
legalForm: best.spec.canonical,
...(country ? { country } : {}),
foreign,
source: 'legal_form',
}
}
function countryCandidate(segment: string): NameCandidate | null {
for (const [re, country] of COUNTRY_WORDS) {
if (country === 'SE') continue
const m = re.exec(stripVatSweden(segment))
if (!m) continue
const tokens = nameTokensBefore(segment.slice(0, m.index))
if (tokens.length === 0 || tokens.length > 4) return null
return { name: `${tokens.join(' ')} ${m[0]}`, country, foreign: true, source: 'country' }
}
return null
}
/**
* Name candidates in the order worth trying: Swedish legal persons first,
* then names anchored on a country word, then the cleaned head of the text.
* The head is marked foreign when the text as a whole points abroad.
*/
export function extractNameCandidates(text: string): NameCandidate[] {
const out: NameCandidate[] = []
const seen = new Set<string>()
const push = (c: NameCandidate | null) => {
if (!c) return
const k = c.name.toLowerCase()
if (seen.has(k)) return
seen.add(k)
out.push(c)
}
const segments = splitSegments(text)
const forms: NameCandidate[] = []
const countries: NameCandidate[] = []
segments.forEach((seg, i) => {
const f = legalFormCandidate(seg, segments[i + 1], text)
if (f) forms.push(f)
else {
const c = countryCandidate(seg)
if (c) countries.push(c)
}
})
forms.filter((c) => !c.foreign).forEach(push)
forms.filter((c) => c.foreign).forEach(push)
countries.forEach(push)
const head = displayNameFromVoucherText(text)
if (head.length >= 2) {
const textCountry = countryHint(text)
const foreignVat = extractVatNumbers(text).some((v) => v.country !== 'SE')
const foreign = out.some((c) => c.foreign) || (textCountry !== undefined && textCountry !== 'SE') || foreignVat
push({
name: head,
...(textCountry && textCountry !== 'SE' ? { country: textCountry } : {}),
foreign,
source: 'head',
})
}
return out
}