Files
lifetrack/apps/api/app/modules/finance/pipeline.py
T
MeeJayandClaude Opus 5 93f0689c1e Initial import: LifeTrack v1 (santé, vape, finances)
Tracker de vie auto-hébergé : suivi poids/calories/sport avec planning de
pesées, sevrage tabac (vape) avec modèle de coût DIY et économies, et
finances personnelles avec import de relevés bancaires.

Architecture : FastAPI + SQLAlchemy 2.0 + PostgreSQL 16, React 18 + TS +
Vite + Tailwind + ECharts, déploiement Docker Compose. Modules
auto-découverts des deux côtés (pkgutil / import.meta.glob) et framework
de connecteurs à deux voies (importeurs de fichiers + ingestion JSON)
pour brancher de nouvelles sources sans toucher au noyau.

Validé : 292 tests pytest, tsc + vite build, contrat API/web vérifié
contre le schéma OpenAPI, et déploiement Docker réel sur PostgreSQL 16
(28 tables, SPA servie par nginx, wizard de premier démarrage).

Documentation : README.md, docs/GUIDE.md, CONVENTIONS.md, et les
documents de conception et de recherche dans docs/.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-14 10:48:57 +02:00

457 lines
14 KiB
Python

"""Import pipeline: decode -> parse -> normalize -> dedup -> rules -> insert.
Reference: datamodel-finance.md §4. The whole run is one SQL transaction
(atomicity, §4.4) and is idempotent thanks to the occurrence-aware dedup hash.
Every imported row references the central `import_runs` row (FK ON DELETE
CASCADE), which is what makes `DELETE /api/imports/{id}` a real rollback.
"""
import hashlib
import uuid
from collections import defaultdict
from dataclasses import dataclass, field
from datetime import date, timedelta
from decimal import Decimal
from typing import Any
from sqlalchemy import select
from sqlalchemy.orm import Session
from app.core.errors import DomainValidationError
from app.core.importing.base import ImporterParseError
from app.core.timeutils import utcnow
from app.modules.finance.categorize import (
apply_rules_to_tx,
detect_transfers,
get_transfer_category_id,
load_enabled_rules,
)
from app.modules.finance.enums import ImportStatus, SourceKind
from app.modules.finance.models import (
FinAccount,
FinImportRun,
FinSourceProfile,
FinTransaction,
)
from app.modules.finance.normalize import (
compute_dedup_hash,
light_clean,
occurrence_key,
)
from app.modules.finance.parsers import (
NormalizedRow,
ParseResult,
decode_bytes,
parse_csv,
parse_ofx,
parse_paypal_csv,
)
from app.modules.imports.models import ImportRun
TRANSFER_WINDOW_DAYS = 3
MAX_STATS_ERRORS = 50
# Importer ids registered in importers.py, per source kind.
IMPORTER_ID_BY_KIND: dict[str, str] = {
SourceKind.CSV: "bank_generic_csv",
SourceKind.OFX: "bank_ofx",
SourceKind.PAYPAL_CSV: "paypal_csv",
}
def file_sha256(raw: bytes) -> str:
return hashlib.sha256(raw).hexdigest()
def parse_file(
kind: str, config: dict[str, Any], raw: bytes, default_currency: str = "EUR"
) -> ParseResult:
"""Dispatch to the parser of the profile kind (§4.2)."""
if kind == SourceKind.OFX:
return parse_ofx(config or {}, raw, default_currency)
text = decode_bytes(raw, (config or {}).get("encoding", "auto"))
if kind == SourceKind.PAYPAL_CSV:
return parse_paypal_csv(config or {}, text, default_currency)
return parse_csv(config or {}, text, default_currency)
def prepare_rows(
account_id: uuid.UUID, rows: list[NormalizedRow]
) -> list[tuple[NormalizedRow, str]]:
"""Number identical tuples (occurrence) then hash each row (§4.3)."""
counter: dict[tuple, int] = defaultdict(int)
prepared: list[tuple[NormalizedRow, str]] = []
for row in rows:
key = occurrence_key(row.booked_date, row.amount, row.label_raw)
occurrence = counter[key]
counter[key] += 1
prepared.append(
(
row,
compute_dedup_hash(
account_id, row.booked_date, row.amount, row.label_raw, occurrence
),
)
)
return prepared
def _existing_keys(
db: Session,
account_id: uuid.UUID,
date_min: date | None,
date_max: date | None,
) -> tuple[set[str], set[str]]:
"""Set-based duplicate lookup: one query for hashes, one for external ids."""
hash_stmt = select(FinTransaction.dedup_hash).where(
FinTransaction.account_id == account_id
)
if date_min is not None and date_max is not None:
hash_stmt = hash_stmt.where(
FinTransaction.booked_date >= date_min,
FinTransaction.booked_date <= date_max,
)
hashes = set(db.scalars(hash_stmt).all())
ext_stmt = select(FinTransaction.external_id).where(
FinTransaction.account_id == account_id,
FinTransaction.external_id.is_not(None),
)
externals = {e for e in db.scalars(ext_stmt).all() if e}
return hashes, externals
def _date_bounds(rows: list[NormalizedRow]) -> tuple[date | None, date | None]:
if not rows:
return None, None
dates = [row.booked_date for row in rows]
return min(dates), max(dates)
def build_transaction(
user_id: int,
account: FinAccount,
row: NormalizedRow,
dedup_hash: str,
import_run_id: int | None,
) -> FinTransaction:
return FinTransaction(
id=uuid.uuid4(),
user_id=user_id,
account_id=account.id,
booked_date=row.booked_date,
value_date=row.value_date,
amount=row.amount,
currency=(row.currency or account.currency or "EUR").upper()[:3],
label_raw=row.label_raw,
label_clean=light_clean(row.label_raw),
counterparty=row.counterparty,
external_id=row.external_id,
dedup_hash=dedup_hash,
import_run_id=import_run_id,
)
@dataclass
class PreviewResult:
rows_preview: list[NormalizedRow] = field(default_factory=list)
rows_total: int = 0
rows_error: int = 0
rows_skipped_filtered: int = 0
would_skip_duplicates: int = 0
date_min: date | None = None
date_max: date | None = None
errors: list[dict[str, Any]] = field(default_factory=list)
duplicate_file_of: int | None = None
def preview_import(
db: Session,
user_id: int,
account: FinAccount,
profile: FinSourceProfile,
raw: bytes,
preview_size: int = 20,
) -> PreviewResult:
"""Parse without writing anything (§9.6 `POST /imports/preview`)."""
if not raw:
raise DomainValidationError("Le fichier envoyé est vide.")
try:
parsed = parse_file(profile.kind, profile.config or {}, raw, account.currency)
except ImporterParseError as exc:
raise DomainValidationError(str(exc)) from exc
prepared = prepare_rows(account.id, parsed.rows)
date_min, date_max = _date_bounds(parsed.rows)
hashes, externals = _existing_keys(db, account.id, date_min, date_max)
duplicates = 0
seen: set[str] = set()
for row, dedup_hash in prepared:
if (row.external_id and row.external_id in externals) or dedup_hash in hashes:
duplicates += 1
continue
if dedup_hash in seen:
duplicates += 1
continue
seen.add(dedup_hash)
previous = db.scalars(
select(FinImportRun.import_run_id)
.where(
FinImportRun.user_id == user_id,
FinImportRun.file_sha256 == file_sha256(raw),
FinImportRun.status == ImportStatus.COMPLETED,
)
.limit(1)
).first()
return PreviewResult(
rows_preview=parsed.rows[:preview_size],
rows_total=parsed.rows_total,
rows_error=parsed.rows_error,
rows_skipped_filtered=parsed.rows_filtered,
would_skip_duplicates=duplicates,
date_min=date_min,
date_max=date_max,
errors=parsed.errors,
duplicate_file_of=previous,
)
def _failed_run(
db: Session,
user_id: int,
account: FinAccount,
profile: FinSourceProfile,
filename: str,
raw: bytes,
message: str,
) -> FinImportRun:
"""Persist a failed run after a rollback (no transaction inserted)."""
db.rollback()
started = utcnow()
central = ImportRun(
user_id=user_id,
importer_id=IMPORTER_ID_BY_KIND.get(profile.kind, "bank_generic_csv"),
domain="finance",
filename=filename,
file_size=len(raw),
status="failed",
error_details=[{"row": 0, "message": message}],
started_at=started,
finished_at=utcnow(),
)
db.add(central)
db.flush()
run = FinImportRun(
id=uuid.uuid4(),
user_id=user_id,
import_run_id=central.id,
source_profile_id=profile.id,
account_id=account.id,
filename=filename,
file_sha256=file_sha256(raw),
status=ImportStatus.FAILED,
started_at=started,
finished_at=utcnow(),
stats={},
error_message=message,
)
db.add(run)
db.commit()
return run
def run_import(
db: Session,
user_id: int,
account: FinAccount,
profile: FinSourceProfile,
filename: str,
raw: bytes,
) -> FinImportRun:
"""Execute one synchronous import run (§4.4)."""
if not raw:
raise DomainValidationError("Le fichier envoyé est vide.")
started = utcnow()
checksum = file_sha256(raw)
duplicate_of = db.scalars(
select(FinImportRun.import_run_id)
.where(
FinImportRun.user_id == user_id,
FinImportRun.file_sha256 == checksum,
FinImportRun.status == ImportStatus.COMPLETED,
)
.limit(1)
).first()
central = ImportRun(
user_id=user_id,
importer_id=IMPORTER_ID_BY_KIND.get(profile.kind, "bank_generic_csv"),
domain="finance",
filename=filename,
file_size=len(raw),
status="completed",
error_details=[],
started_at=started,
)
db.add(central)
db.flush()
run = FinImportRun(
id=uuid.uuid4(),
user_id=user_id,
import_run_id=central.id,
source_profile_id=profile.id,
account_id=account.id,
filename=filename,
file_sha256=checksum,
status=ImportStatus.RUNNING,
started_at=started,
stats={},
)
db.add(run)
try:
parsed = parse_file(profile.kind, profile.config or {}, raw, account.currency)
except ImporterParseError as exc:
return _failed_run(db, user_id, account, profile, filename, raw, str(exc))
except Exception: # noqa: BLE001 — a broken file must not 500 the API
return _failed_run(
db,
user_id,
account,
profile,
filename,
raw,
"Erreur inattendue pendant la lecture du fichier.",
)
# §4.2: the run only fails when EVERY data row errored. A file whose rows
# were all filtered out on purpose (PayPal authorisations, non-EUR lines…)
# is a legitimate empty import, not a failure.
if parsed.rows_total and parsed.rows_error == parsed.rows_total:
return _failed_run(
db,
user_id,
account,
profile,
filename,
raw,
"Aucune ligne exploitable — vérifiez le profil de source.",
)
prepared = prepare_rows(account.id, parsed.rows)
date_min, date_max = _date_bounds(parsed.rows)
hashes, externals = _existing_keys(db, account.id, date_min, date_max)
rules = load_enabled_rules(db, user_id)
transfer_cat = get_transfer_category_id(db, user_id)
to_insert: list[FinTransaction] = []
skipped_duplicate = 0
rules_applied = 0
for row, dedup_hash in prepared:
if (row.external_id and row.external_id in externals) or dedup_hash in hashes:
skipped_duplicate += 1
continue
hashes.add(dedup_hash)
if row.external_id:
externals.add(row.external_id)
tx = build_transaction(user_id, account, row, dedup_hash, central.id)
if apply_rules_to_tx(tx, rules, transfer_cat):
rules_applied += 1
to_insert.append(tx)
# Atomicity (§4.4): the whole run is one SQL transaction; anything going
# wrong rolls every row back and the run is stored as failed.
transfers = 0
try:
db.add_all(to_insert)
db.flush()
if date_min is not None and date_max is not None and to_insert:
transfers = detect_transfers(
db,
user_id,
date_min - timedelta(days=TRANSFER_WINDOW_DAYS),
date_max + timedelta(days=TRANSFER_WINDOW_DAYS),
)
except Exception: # noqa: BLE001 — never leave a half-written run behind
return _failed_run(
db,
user_id,
account,
profile,
filename,
raw,
"L'enregistrement des transactions a échoué : import annulé.",
)
stats: dict[str, Any] = {
"rows_total": parsed.rows_total,
"rows_imported": len(to_insert),
"rows_skipped_duplicate": skipped_duplicate,
"rows_skipped_filtered": parsed.rows_filtered,
"rows_error": parsed.rows_error,
"rules_applied": rules_applied,
"transfers_detected": transfers,
"date_min": date_min.isoformat() if date_min else None,
"date_max": date_max.isoformat() if date_max else None,
"errors": parsed.errors[:MAX_STATS_ERRORS],
}
if parsed.errors_truncated or len(parsed.errors) > MAX_STATS_ERRORS:
stats["errors_truncated"] = True
if duplicate_of is not None:
stats["duplicate_file_of"] = duplicate_of
run.status = ImportStatus.COMPLETED
run.finished_at = utcnow()
run.stats = stats
central.rows_total = parsed.rows_total
central.rows_inserted = len(to_insert)
central.rows_duplicates = skipped_duplicate
central.rows_errors = parsed.rows_error
central.error_details = parsed.errors[:MAX_STATS_ERRORS]
central.finished_at = run.finished_at
try:
db.commit()
except Exception: # noqa: BLE001 — never leave a half-written run behind
return _failed_run(
db,
user_id,
account,
profile,
filename,
raw,
"L'enregistrement des transactions a échoué : import annulé.",
)
return run
def transaction_dedup_hash(
db: Session,
account_id: uuid.UUID,
booked_date: date,
amount: Decimal,
label_raw: str,
) -> str:
"""Hash for a manually entered transaction: occurrence = count of twins."""
occurrence = 0
stmt = select(FinTransaction.dedup_hash).where(
FinTransaction.account_id == account_id,
FinTransaction.booked_date == booked_date,
FinTransaction.amount == amount,
)
existing = set(db.scalars(stmt).all())
candidate = compute_dedup_hash(
account_id, booked_date, amount, label_raw, occurrence
)
while candidate in existing:
occurrence += 1
candidate = compute_dedup_hash(
account_id, booked_date, amount, label_raw, occurrence
)
return candidate