Files
bin/finance/app/parsers/csv_formats.py
2026-07-19 21:34:48 +02:00

246 lines
9.7 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Parser für Kontoumsatz-CSV-Exporte (DKB, VR, HVB).
Siehe „CSV-Formatreferenz" in docs/superpowers/plans/2026-07-19-ausbaustufe-3.md
für die verbindliche Spalten-/Formatspezifikation je Bank.
"""
import csv
import io
import re
from dataclasses import dataclass
from datetime import date
from decimal import Decimal
from pathlib import Path
from app.parsers.base import (ParsedStatement, ParsedTransaction, ParserError,
parse_german_amount, parse_german_date)
RE_KONTOSTAND = re.compile(r"Kontostand vom (\d{2}\.\d{2}\.\d{4}):")
RE_ZEITRAUM = re.compile(r"(\d{2}\.\d{2}\.\d{4})\s*-\s*(\d{2}\.\d{2}\.\d{4})")
@dataclass
class ParsedCsv:
statement: ParsedStatement
anchor: tuple[date, Decimal] | None
balance_checkable: bool
def _detect_encoding(raw: bytes) -> str:
if raw.startswith(b"\xef\xbb\xbf"):
return "utf-8-sig"
if raw.startswith(b"\xff\xfe"):
return "utf-16-le"
if raw.startswith(b"\xfe\xff"):
return "utf-16-be"
sample = raw[:64]
if b"\x00" in sample:
even_zeros = sum(1 for i in range(0, len(sample), 2) if sample[i] == 0)
odd_zeros = sum(1 for i in range(1, len(sample), 2) if i < len(sample) and sample[i] == 0)
return "utf-16-be" if even_zeros > odd_zeros else "utf-16-le"
return "utf-8"
def _decode(raw: bytes) -> str:
enc = _detect_encoding(raw)
text = raw.decode(enc, errors="ignore")
if text.startswith(""):
text = text[1:]
return text
def _strip_currency(s: str) -> str:
return s.replace("", "").replace("EUR", "").replace("\xa0", " ").strip()
def _parse_amount(s: str) -> Decimal:
return parse_german_amount(_strip_currency(s))
def _parse_two_digit_year_date(s: str) -> date:
m = re.fullmatch(r"(\d{2})\.(\d{2})\.(\d{2})", s.strip())
if not m:
raise ParserError(f"DKB-CSV: Kein Datum im Format TT.MM.JJ: {s!r}")
day, month, yy = int(m.group(1)), int(m.group(2)), int(m.group(3))
year = 2000 + yy if yy <= 69 else 1900 + yy
try:
return date(year, month, day)
except ValueError:
raise ParserError(f"DKB-CSV: Ungueltiges Datum: {s!r}")
def detect_csv_format(head_bytes: bytes) -> str | None:
try:
text = _decode(head_bytes)
except (LookupError, UnicodeError):
return None
lines = text.splitlines()
if not lines:
return None
first = lines[0]
if first.startswith("Kontonummer;") and "Buchungsdatum" in first and "Valuta" in first \
and "Betrag" in first:
return "hvb_csv"
if first.startswith("Bezeichnung Auftragskonto;") and "IBAN Auftragskonto" in first:
return "vr_csv"
head = "\n".join(lines[:6])
if "Kontostand vom" in head and "Buchungsdatum" in head and "Zahlungsempf" in head:
return "dkb_csv"
return None
def parse_csv(path: Path) -> ParsedCsv:
raw = path.read_bytes()
fmt = detect_csv_format(raw)
if fmt is None:
raise ParserError("CSV-Format nicht erkannt")
text = _decode(raw)
if fmt == "dkb_csv":
return _parse_dkb(text)
if fmt == "vr_csv":
return _parse_vr(text)
return _parse_hvb(text)
def _rows(text: str) -> list[list[str]]:
return list(csv.reader(io.StringIO(text), delimiter=";", quotechar='"'))
def _parse_dkb(text: str) -> ParsedCsv:
rows = _rows(text)
if len(rows) < 2:
raise ParserError("DKB-CSV: Datei zu kurz")
iban = rows[0][1].strip() if len(rows[0]) > 1 else None
period_start = period_end = None
kontostand_date: date | None = None
kontostand_amount: Decimal | None = None
header_idx = None
for i, row in enumerate(rows):
if row and row[0].strip() == "Zeitraum:" and len(row) > 1:
m = RE_ZEITRAUM.search(row[1])
if m:
period_start = parse_german_date(m.group(1))
period_end = parse_german_date(m.group(2))
if row and RE_KONTOSTAND.search(row[0] or "") and len(row) > 1:
m = RE_KONTOSTAND.search(row[0])
kontostand_date = parse_german_date(m.group(1))
kontostand_amount = _parse_amount(row[1])
if row and row[0].strip() == "Buchungsdatum":
header_idx = i
break
if header_idx is None:
raise ParserError("DKB-CSV: Spaltenkopf nicht gefunden")
if kontostand_date is None or kontostand_amount is None:
raise ParserError("DKB-CSV: Kontostand-Zeile nicht gefunden")
header = [h.strip() for h in rows[header_idx]]
idx = {name: i for i, name in enumerate(header)}
required = ["Buchungsdatum", "Wertstellung", "Status", "Zahlungspflichtige*r",
"Zahlungsempfänger*in", "Verwendungszweck", "Umsatztyp", "Betrag (€)"]
for col in required:
if col not in idx:
raise ParserError(f"DKB-CSV: Spalte fehlt: {col}")
txs: list[ParsedTransaction] = []
for row in rows[header_idx + 1:]:
if not row or len(row) < len(header):
continue
if row[idx["Status"]].strip() != "Gebucht":
continue
amount = _parse_amount(row[idx["Betrag (€)"]])
counterparty = (row[idx["Zahlungsempfänger*in"]] if amount < 0
else row[idx["Zahlungspflichtige*r"]]).strip()
purpose = f"{row[idx['Umsatztyp']].strip()} {row[idx['Verwendungszweck']].strip()}".strip()
booking = _parse_two_digit_year_date(row[idx["Buchungsdatum"]])
value = _parse_two_digit_year_date(row[idx["Wertstellung"]])
txs.append(ParsedTransaction(booking_date=booking, value_date=value, amount=amount,
purpose=purpose, counterparty=counterparty))
txs.reverse() # file is sorted descending -> chronologically ascending
stmt = ParsedStatement(bank="dkb_csv", iban=iban, period_start=period_start,
period_end=period_end, opening_balance=None,
closing_balance=None, transactions=txs)
return ParsedCsv(statement=stmt, anchor=(kontostand_date, kontostand_amount),
balance_checkable=False)
def _parse_vr(text: str) -> ParsedCsv:
rows = _rows(text)
if not rows:
raise ParserError("VR-CSV: Datei leer")
header = [h.strip() for h in rows[0]]
idx = {name: i for i, name in enumerate(header)}
required = ["IBAN Auftragskonto", "Buchungstag", "Valutadatum", "Name Zahlungsbeteiligter",
"Buchungstext", "Verwendungszweck", "Betrag", "Saldo nach Buchung"]
for col in required:
if col not in idx:
raise ParserError(f"VR-CSV: Spalte fehlt: {col}")
data_rows = [(i, row) for i, row in enumerate(rows[1:], start=2) if row and len(row) >= len(header)]
if not data_rows:
raise ParserError("VR-CSV: keine Buchungszeilen gefunden")
iban = rows[1][idx["IBAN Auftragskonto"]].strip() if len(rows) > 1 else None
# File is sorted descending (newest first) -> reverse for chronological ascending.
asc = list(reversed(data_rows))
txs: list[ParsedTransaction] = []
saldi: list[Decimal] = []
prev_saldo: Decimal | None = None
for line_no, row in asc:
amount = _parse_amount(row[idx["Betrag"]])
saldo = _parse_amount(row[idx["Saldo nach Buchung"]])
if prev_saldo is not None and prev_saldo + amount != saldo:
raise ParserError(f"VR-CSV: Saldo-Kette inkonsistent bei Zeile {line_no}")
prev_saldo = saldo
booking = parse_german_date(row[idx["Buchungstag"]])
value = parse_german_date(row[idx["Valutadatum"]])
purpose = f"{row[idx['Buchungstext']].strip()} {row[idx['Verwendungszweck']].strip()}".strip()
counterparty = row[idx["Name Zahlungsbeteiligter"]].strip()
txs.append(ParsedTransaction(booking_date=booking, value_date=value, amount=amount,
purpose=purpose, counterparty=counterparty))
saldi.append(saldo)
opening_balance = saldi[0] - txs[0].amount
closing_balance = saldi[-1]
anchor = (txs[-1].booking_date, closing_balance)
stmt = ParsedStatement(bank="vr_csv", iban=iban,
period_start=txs[0].booking_date, period_end=txs[-1].booking_date,
opening_balance=opening_balance, closing_balance=closing_balance,
transactions=txs)
return ParsedCsv(statement=stmt, anchor=anchor, balance_checkable=True)
def _parse_hvb(text: str) -> ParsedCsv:
rows = _rows(text)
if not rows:
raise ParserError("HVB-CSV: Datei leer")
header = [h.strip() for h in rows[0]]
idx = {name: i for i, name in enumerate(header)}
required = ["Kontonummer", "Buchungsdatum", "Valuta", "Verwendungszweck", "Betrag"]
for col in required:
if col not in idx:
raise ParserError(f"HVB-CSV: Spalte fehlt: {col}")
data_rows = [row for row in rows[1:] if row and len(row) >= len(header)]
if not data_rows:
raise ParserError("HVB-CSV: keine Buchungszeilen gefunden")
kontonummer = data_rows[0][idx["Kontonummer"]].strip()
txs: list[ParsedTransaction] = []
for row in data_rows: # already chronologically ascending
booking = parse_german_date(row[idx["Buchungsdatum"]])
value = parse_german_date(row[idx["Valuta"]])
amount = _parse_amount(row[idx["Betrag"]])
purpose = row[idx["Verwendungszweck"]].strip()
txs.append(ParsedTransaction(booking_date=booking, value_date=value, amount=amount,
purpose=purpose, counterparty=""))
stmt = ParsedStatement(bank="hvb_csv", iban=kontonummer,
period_start=txs[0].booking_date, period_end=txs[-1].booking_date,
opening_balance=None, closing_balance=None, transactions=txs)
return ParsedCsv(statement=stmt, anchor=None, balance_checkable=False)