Files
Personal-Finance-Management/app/services/bank_import_service.py
T
2026-06-05 18:14:16 -04:00

916 lines
35 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
Bank Statement Import Service
Supported formats:
PDF — Digital bank statement PDFs; text extracted via pdfplumber then
parsed by Groq LLM (requires GROQ_API_KEY).
OFX / QFX — Open Financial Exchange (exported by most US banks & Quicken)
CSV — Auto-detects Chase, Bank of America, Citi, Capital One, Discover,
USAA, Wells Fargo; falls back to a generic heuristic; asks user
to map columns when detection fails.
Public API:
parse_file(file_bytes, filename, col_map=None) → dict
do_import(rows, default_account_id, skip_dupes=True) → (imported, skipped)
auto_categorize(description) → category_name_str
"""
import csv
import io
import re
import logging
from datetime import datetime
log = logging.getLogger(__name__)
# ── Keyword → PFM category name map ──────────────────────────────────────────
KEYWORD_CATEGORIES = {
'Food & Dining': [
'restaurant', 'dining', 'cafe', 'coffee', 'starbucks', 'mcdonald',
'burger', 'pizza', 'sushi', 'taco', 'chipotle', 'subway', 'doordash',
'ubereats', 'grubhub', 'seamless', 'instacart', 'grocery', 'safeway',
'whole foods', 'trader joe', 'kroger', 'wegmans', 'publix', 'aldi',
'panera', 'dunkin', 'chick-fil-a', 'wendy', 'taco bell', 'kfc',
'popeyes', 'five guys', 'domino', 'papa john', 'little caesar',
'denny', 'ihop', 'olive garden', 'applebee', 'cheesecake factory',
],
'Transport': [
'uber', 'lyft', 'taxi', 'gas station', 'shell', 'chevron', 'exxon',
'mobil', 'citgo', 'speedway', 'sunoco', 'circle k', 'wawa fuel',
'parking', 'toll', 'transit', 'metro', 'bart', 'mta ', 'wmata',
'amtrak', 'greyhound', 'zipcar', 'enterprise rent', 'hertz', 'avis',
'budget rent', 'national car', 'jiffy lube', 'valvoline', 'firestone',
'oil change', 'autozone', 'advance auto', 'napa auto', 'pep boys',
'airline', 'delta ', 'united air', 'southwest', 'american air',
'jetblue', 'spirit air',
],
'Shopping': [
'amazon', 'ebay', 'walmart', 'target', 'costco', 'best buy',
'apple store', 'nike', 'adidas', 'gap', 'old navy', 'h&m', 'zara',
'nordstrom', 'macy', 'tj maxx', 'marshalls', 'ross store',
'dollar tree', 'dollar general', 'home depot', 'lowes', 'ikea',
'bed bath', 'wayfair', 'etsy', 'paypal ', 'venmo ',
],
'Entertainment': [
'netflix', 'spotify', 'hulu', 'disney+', 'hbo max', 'amazon prime',
'apple tv', 'youtube premium', 'twitch', 'steam ', 'playstation',
'xbox', 'nintendo', 'movie', 'cinema', 'theater', 'concert',
'ticketmaster', 'stubhub', 'eventbrite', 'live nation', 'amc ',
'regal cinema', 'fandango', 'bowling', 'arcade', 'museum', 'zoo',
],
'Utilities': [
'electric', 'gas utility', 'natural gas', 'water bill', 'sewage',
'trash', 'waste mgmt', 'internet', 'comcast', 'xfinity', 'verizon',
'at&t', 'att bill', 't-mobile', 'sprint', 'charter', 'spectrum',
'cox comm', 'centurylink', 'utility bill', 'pge ', 'con edison',
],
'Health': [
'pharmacy', 'cvs ', 'walgreens', 'rite aid', 'hospital', 'clinic',
'doctor', 'dental', 'dentist', 'vision center', 'optometrist',
'labcorp', 'quest diag', 'urgent care', 'medical', 'health care',
'gym', 'planet fitness', 'la fitness', 'equinox', 'peloton',
],
'Housing': [
'rent payment', 'mortgage', 'hoa ', 'property tax', 'maintenance',
'plumber', 'electrician', 'hvac', 'pest control', 'lawn care',
'landscap', 'cleaning service', 'airbnb', 'vrbo', 'hotel', 'motel',
'hilton', 'marriott', 'hyatt', 'sheraton', 'holiday inn',
],
'Subscriptions': [
'subscription', 'membership fee', 'annual fee', 'monthly fee',
'adobe ', 'microsoft 365', 'office 365', 'dropbox', 'icloud',
'google one', '1password', 'nordvpn', 'expressvpn', 'duolingo',
'headspace', 'calm app', 'audible', 'kindle unlimited',
'new york times', 'wall street journal', 'nytimes.com',
],
'Education': [
'tuition', 'university', 'college', 'coursera', 'udemy',
'linkedin learning', 'pluralsight', 'skillshare', 'textbook', 'chegg',
'grammarly', 'turnitin', 'canvas ', 'blackboard',
],
'Insurance': [
'insurance', 'geico', 'state farm', 'allstate', 'progressive ins',
'liberty mutual', 'farmers ins', 'aaa ins', 'aflac', 'metlife',
'cigna', 'aetna', 'humana', 'blue cross', 'united health',
],
'Gifts': [
'gift card', 'donation', 'charity', 'church tithe', 'gofundme',
'red cross', 'goodwill', 'salvation army', 'habitat humanity',
],
'Investment': [
'robinhood', 'fidelity inv', 'schwab', 'vanguard', 'e*trade',
'td ameritrade', 'coinbase', 'binance', 'crypto purchase',
],
'Salary': [
'payroll', 'direct deposit', 'salary', 'wages deposit', 'paycheck',
'adp llc', 'paychex', 'gusto payroll', 'paylocity', 'intuit payroll',
],
'Freelance': [
'freelance', 'consulting fee', 'contract payment', 'upwork',
'fiverr', 'toptal',
],
'Other Income': [
'interest paid', 'dividend', 'refund', 'cashback', 'reward',
'reimburs', 'tax refund', 'irs treas',
],
}
# ── Known bank CSV format profiles ───────────────────────────────────────────
BANK_FORMATS = [
{
'name': 'Chase',
'required': {'transaction date', 'description', 'amount'},
'map': {
'date': 'transaction date', 'description': 'description',
'amount': 'amount', 'positive_is_expense': False,
},
},
{
'name': 'Bank of America',
'required': {'posted date', 'payee', 'amount'},
'map': {
'date': 'posted date', 'description': 'payee',
'amount': 'amount', 'positive_is_expense': False,
},
},
{
'name': 'Citi',
'required': {'date', 'description', 'debit', 'credit'},
'map': {
'date': 'date', 'description': 'description',
'debit': 'debit', 'credit': 'credit',
},
},
{
'name': 'Capital One',
'required': {'transaction date', 'posted date', 'card no.', 'description', 'debit', 'credit'},
'map': {
'date': 'transaction date', 'description': 'description',
'debit': 'debit', 'credit': 'credit',
},
},
{
'name': 'Discover',
'required': {'trans. date', 'post date', 'description', 'amount', 'category'},
'map': {
'date': 'trans. date', 'description': 'description',
'amount': 'amount', 'positive_is_expense': True,
},
},
{
'name': 'American Express',
'required': {'date', 'description', 'amount', 'extended details', 'appears on your statement as'},
'map': {
'date': 'date', 'description': 'description',
'amount': 'amount', 'positive_is_expense': False,
},
},
{
'name': 'USAA',
'required': {'date', 'description', 'original description', 'category', 'amount', 'status'},
'map': {
'date': 'date', 'description': 'description',
'amount': 'amount', 'positive_is_expense': False,
},
},
{
'name': 'Wells Fargo',
'required': {'date', 'amount', 'description'},
'map': {
'date': 'date', 'description': 'description',
'amount': 'amount', 'positive_is_expense': False,
},
},
]
_DATE_FMTS = [
'%Y-%m-%d', '%m/%d/%Y', '%d/%m/%Y', '%Y/%m/%d',
'%d-%m-%Y', '%m-%d-%Y', '%b %d, %Y', '%d %b %Y',
'%B %d, %Y', '%d %B %Y', '%m/%d/%y',
]
# ── Internal helpers ──────────────────────────────────────────────────────────
def _parse_date(s):
s = str(s).strip()
# OFX: 20250115120000[-5:EST] — take first 8 digits
if re.match(r'^\d{8}', s):
return datetime.strptime(s[:8], '%Y%m%d').date()
for fmt in _DATE_FMTS:
try:
return datetime.strptime(s, fmt).date()
except ValueError:
pass
raise ValueError(f'Unrecognised date: {s!r}')
def _clean_amount(s):
if s is None:
return 0.0
cleaned = re.sub(r'[^\d.\-+]', '', str(s).strip())
try:
return float(cleaned) if cleaned else 0.0
except ValueError:
return 0.0
def _extract_tag(text, tag):
"""Extract value of an OFX SGML/XML tag (stops at newline or next tag)."""
m = re.search(rf'<{tag}>\s*([^\r\n<]+)', text, re.IGNORECASE)
return m.group(1).strip() if m else ''
# ── OFX / QFX parser ─────────────────────────────────────────────────────────
def _parse_ofx(content):
"""Parse OFX/QFX (handles both SGML and XML variants)."""
# Strip header block above <OFX>
ofx_start = re.search(r'<OFX>', content, re.IGNORECASE)
if ofx_start:
content = content[ofx_start.start():]
# Try XML-style first (closing tags present)
xml_blocks = re.findall(r'<STMTTRN>(.*?)</STMTTRN>', content,
re.IGNORECASE | re.DOTALL)
if not xml_blocks:
# SGML-style: split on opening tags
xml_blocks = re.split(r'<STMTTRN>', content, flags=re.IGNORECASE)[1:]
raw = []
for block in xml_blocks:
dtposted = _extract_tag(block, 'DTPOSTED')
trnamt = _extract_tag(block, 'TRNAMT')
trntype = _extract_tag(block, 'TRNTYPE').upper()
fitid = _extract_tag(block, 'FITID')
name = _extract_tag(block, 'NAME')
memo = _extract_tag(block, 'MEMO')
if not dtposted or not trnamt:
continue
try:
txn_date = _parse_date(dtposted)
except ValueError:
continue
amount = _clean_amount(trnamt)
if amount == 0.0:
continue
# OFX sign: negative = debit (expense), positive = credit (income)
if amount < 0:
txn_type = 'expense'
else:
txn_type = 'income'
amount = abs(amount)
# TRNTYPE tag can override when sign is ambiguous
if trntype in ('DEBIT', 'CHECK', 'PAYMENT', 'REPEATPMT', 'ATM', 'POS', 'XFER'):
txn_type = 'expense'
elif trntype in ('CREDIT', 'INT', 'DIV', 'DIRECTDEP', 'DEPOSIT', 'REFUND'):
txn_type = 'income'
description = name or memo or f'OFX {trntype}'
raw.append({
'date': txn_date,
'transaction_type': txn_type,
'amount': amount,
'description': description,
'notes': memo if memo and memo != name else '',
'source_id': fitid,
})
log.info('[bank_import] OFX: parsed %d transactions', len(raw))
return raw
# ── CSV helpers ───────────────────────────────────────────────────────────────
def _detect_csv_format(headers_set):
"""Return BANK_FORMATS entry whose required columns are all present, or None."""
for fmt in BANK_FORMATS:
if fmt['required'].issubset(headers_set):
return fmt
return None
def _build_generic_map(headers_set):
"""
Attempt to build a column map from an unknown bank CSV.
Looks for common column name patterns.
Returns col_map dict or None if key columns cannot be found.
"""
date_names = ['date', 'posted date', 'transaction date', 'trans. date',
'post date', 'settlement date', 'value date', 'effective date']
desc_names = ['description', 'payee', 'name', 'merchant', 'memo',
'transaction description', 'details', 'narrative', 'remarks']
amt_names = ['amount', 'transaction amount', 'net amount']
debit_names = ['debit', 'withdrawal', 'withdrawals', 'charge', 'charges',
'debit amount', 'out', 'outflow']
cred_names = ['credit', 'deposit', 'deposits', 'payment', 'in',
'credit amount', 'inflow']
date_col = next((c for c in date_names if c in headers_set), None)
desc_col = next((c for c in desc_names if c in headers_set), None)
amt_col = next((c for c in amt_names if c in headers_set), None)
debit_col = next((c for c in debit_names if c in headers_set), None)
cred_col = next((c for c in cred_names if c in headers_set), None)
if not date_col or not desc_col:
return None
if not amt_col and not debit_col and not cred_col:
return None
return {
'date': date_col,
'description': desc_col,
'amount': amt_col,
'debit': debit_col,
'credit': cred_col,
'positive_is_expense': False,
}
def _parse_csv_with_map(content, col_map):
"""
Parse CSV content using the provided column map.
Returns (raw_rows, errors).
raw_rows: list of dicts with date (date obj), transaction_type, amount, description, notes, source_id.
"""
reader = csv.DictReader(io.StringIO(content))
# Build {lowercase: original_case} lookup for header matching
header_lookup = {h.strip().lower(): h.strip() for h in (reader.fieldnames or [])}
def get_col(key):
mapped = col_map.get(key)
if not mapped:
return None
return header_lookup.get(mapped.lower())
date_col = get_col('date')
desc_col = get_col('description')
amt_col = get_col('amount')
debit_col = get_col('debit')
cred_col = get_col('credit')
positive_is_expense = col_map.get('positive_is_expense', False)
raw = []
errors = []
for i, row in enumerate(reader, start=2):
raw_date = row.get(date_col, '').strip() if date_col else ''
if not raw_date:
continue # skip blank rows silently
try:
txn_date = _parse_date(raw_date)
except ValueError as e:
errors.append(f'Row {i}: {e}')
continue
description = (row.get(desc_col, '') if desc_col else '').strip()
if not description:
description = f'Transaction on {raw_date}'
# Amount resolution: debit/credit columns take priority
if debit_col or cred_col:
debit = abs(_clean_amount(row.get(debit_col, '') if debit_col else ''))
credit = abs(_clean_amount(row.get(cred_col, '') if cred_col else ''))
if debit > 0 and credit > 0:
# Both filled — treat as expense (unusual, but be safe)
amount = debit
txn_type = 'expense'
elif debit > 0:
amount = debit
txn_type = 'expense'
elif credit > 0:
amount = credit
txn_type = 'income'
else:
continue # skip zero / blank rows
elif amt_col:
amount_raw = _clean_amount(row.get(amt_col, ''))
if amount_raw == 0.0:
continue
if positive_is_expense:
txn_type = 'expense' if amount_raw > 0 else 'income'
else:
txn_type = 'expense' if amount_raw < 0 else 'income'
amount = abs(amount_raw)
else:
errors.append(f'Row {i}: no amount column')
continue
raw.append({
'date': txn_date,
'transaction_type': txn_type,
'amount': amount,
'description': description,
'notes': '',
'source_id': None,
})
return raw, errors
# ── Auto-categorizer ──────────────────────────────────────────────────────────
def auto_categorize(description):
"""
Return a PFM category name based on keywords in the description.
Returns empty string when no match is found.
"""
lower = description.lower()
for cat_name, keywords in KEYWORD_CATEGORIES.items():
if any(kw in lower for kw in keywords):
return cat_name
return ''
def _build_cat_id_map():
from app.models.category import Category
return {c.name: c.id for c in Category.query.filter_by(is_active=True).all()}
def build_category_map():
"""Public alias — shared by teller, plaid, and schwab services."""
return _build_cat_id_map()
# ── Enrichment ────────────────────────────────────────────────────────────────
def _enrich(raw_rows):
"""Add category_id + category_name to raw rows; convert date to ISO string."""
cat_map = _build_cat_id_map()
enriched = []
for r in raw_rows:
d = r['date']
cat_name = auto_categorize(r['description'])
enriched.append({
**r,
'date': d.isoformat() if hasattr(d, 'isoformat') else str(d),
'category_name': cat_name,
'category_id': cat_map.get(cat_name),
})
return enriched
# ── PDF parser ────────────────────────────────────────────────────────────────
# Maximum characters sent to Groq (~8 K tokens; covers ~6 months of transactions)
_PDF_MAX_CHARS = 30_000
def _groq_parse_statement(text):
"""
Send extracted PDF text to Groq and return a list of raw transaction dicts.
Raises RuntimeError on API / parse failure.
"""
import json
import requests as _requests
from flask import current_app
api_key = current_app.config.get('GROQ_API_KEY', '')
if not api_key:
raise RuntimeError('GROQ_API_KEY is not configured — cannot AI-parse PDFs.')
model = current_app.config.get('GROQ_MODEL', 'llama-3.3-70b-versatile')
prompt = (
'Extract ALL transactions from the bank statement text below.\n'
'Return ONLY a JSON array — no explanation, no markdown fences.\n\n'
'Each element must have exactly these keys:\n'
' "date" YYYY-MM-DD\n'
' "description" clean merchant or payee name\n'
' "amount" positive number (never negative)\n'
' "transaction_type" "expense" for debits/withdrawals/charges;\n'
' "income" for credits/deposits/refunds/interest\n\n'
'Rules:\n'
'- Include every transaction; do not omit any.\n'
'- Skip opening balance, closing balance, and summary lines.\n'
'- If the year is missing, infer it from the statement period header.\n'
'- amount is always a positive float.\n\n'
f'Bank statement text:\n{text}'
)
try:
resp = _requests.post(
'https://api.groq.com/openai/v1/chat/completions',
headers={
'Authorization': f'Bearer {api_key}',
'Content-Type': 'application/json',
},
json={
'model': model,
'messages': [
{'role': 'system',
'content': 'You are a precise financial data extractor. '
'Respond with valid JSON only — nothing else.'},
{'role': 'user', 'content': prompt},
],
'max_tokens': 4096,
'temperature': 0.1,
'stream': False,
},
timeout=60,
)
resp.raise_for_status()
raw_content = resp.json()['choices'][0]['message']['content']
log.info('[bank_import] Groq response length=%d chars', len(raw_content))
except Exception as exc:
log.error('[bank_import] Groq API call failed: %s', exc, exc_info=True)
raise RuntimeError(f'Groq API error: {exc}')
# Strip markdown code fences if the model wrapped its output
content = raw_content.strip()
if '```' in content:
m = re.search(r'```(?:json)?\s*([\s\S]+?)\s*```', content)
content = m.group(1).strip() if m else content
# Ensure we start at the JSON array
if not content.startswith('['):
m = re.search(r'\[[\s\S]+\]', content)
if m:
content = m.group(0)
try:
transactions = json.loads(content)
except (ValueError, json.JSONDecodeError) as exc:
log.error('[bank_import] Groq JSON parse failed: %s | raw=%r', exc, raw_content[:500])
raise RuntimeError(
f'Groq returned a response that could not be parsed as JSON: {exc}. '
'Try again or use the CSV export from your bank instead.'
)
if not isinstance(transactions, list):
raise RuntimeError('Groq returned unexpected structure (expected a JSON array).')
raw_rows = []
for txn in transactions:
try:
txn_date = _parse_date(str(txn.get('date', '')))
except ValueError:
continue
amount = float(txn.get('amount', 0) or 0)
if amount <= 0:
continue
txn_type = str(txn.get('transaction_type', 'expense')).lower().strip()
if txn_type not in ('income', 'expense'):
txn_type = 'expense'
raw_rows.append({
'date': txn_date,
'transaction_type': txn_type,
'amount': amount,
'description': str(txn.get('description', '')).strip() or 'Bank transaction',
'notes': '',
'source_id': None,
})
log.info('[bank_import] Groq extracted %d transactions from PDF', len(raw_rows))
return raw_rows
_TABLE_DATE_HDRS = {'date', 'posted', 'transaction date', 'trans date',
'posting date', 'value date', 'effective date', 'settled'}
_TABLE_DESC_HDRS = {'description', 'payee', 'merchant', 'memo', 'transaction',
'details', 'name', 'narrative', 'particulars', 'reference'}
_TABLE_DEBIT_HDRS = {'debit', 'withdrawal', 'withdrawals', 'charge', 'charges',
'amount debited', 'payment', 'dr'}
_TABLE_CRED_HDRS = {'credit', 'deposit', 'deposits', 'amount credited',
'cr', 'inflow'}
_TABLE_AMT_HDRS = {'amount', 'transaction amount', 'net amount'}
def _pdfplumber_table_parse(pdf_handle):
"""
Try to extract transactions directly from pdfplumber table structures.
Iterates every page, finds tables whose headers match bank-statement
patterns, and converts rows to raw transaction dicts.
Returns a (possibly empty) list of raw dicts; never raises.
"""
all_rows = []
for page in pdf_handle.pages:
try:
tables = page.extract_tables()
except Exception:
continue
for table in tables:
if not table or len(table) < 2:
continue
# Normalise headers (lower-case, strip)
raw_headers = [str(h).strip().lower() if h else '' for h in table[0]]
date_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_DATE_HDRS), None)
desc_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_DESC_HDRS), None)
amt_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_AMT_HDRS), None)
debit_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_DEBIT_HDRS), None)
cred_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_CRED_HDRS), None)
# Need at least date + description + one amount column
if date_col is None or desc_col is None:
continue
if amt_col is None and debit_col is None and cred_col is None:
continue
col_max = max(c for c in [date_col, desc_col, amt_col, debit_col, cred_col]
if c is not None)
for row in table[1:]:
if not row or len(row) <= col_max:
continue
date_str = str(row[date_col]).strip() if row[date_col] else ''
if not date_str or date_str.lower() in ('', 'none', '-', '--', 'n/a'):
continue
try:
txn_date = _parse_date(date_str)
except ValueError:
continue
description = str(row[desc_col]).strip() if row[desc_col] else ''
if not description or description.lower() in ('', 'none'):
continue
if debit_col is not None or cred_col is not None:
debit = abs(_clean_amount(row[debit_col] if debit_col is not None else ''))
credit = abs(_clean_amount(row[cred_col] if cred_col is not None else ''))
if debit > 0:
amount, txn_type = debit, 'expense'
elif credit > 0:
amount, txn_type = credit, 'income'
else:
continue
else:
raw_amt = _clean_amount(row[amt_col] if row[amt_col] else '')
if raw_amt == 0.0:
continue
txn_type = 'expense' if raw_amt < 0 else 'income'
amount = abs(raw_amt)
all_rows.append({
'date': txn_date,
'transaction_type': txn_type,
'amount': amount,
'description': description,
'notes': '',
'source_id': None,
})
return all_rows
def _parse_pdf(file_bytes):
"""
Extract transactions from a digital bank-statement PDF.
Strategy (in order):
1. pdfplumber table extraction — fast, free, no API call needed.
Used when structured tables with recognisable headers are found and
yield at least 3 rows.
2. pdfplumber text extraction → Groq LLM — handles unstructured
or narrative-style statements.
Returns (raw_rows, warnings_list).
Raises RuntimeError for unrecoverable problems (scanned PDF, bad file, etc.).
"""
try:
import pdfplumber
except ImportError:
raise RuntimeError(
'pdfplumber is not installed. Run: pip install pdfplumber==0.11.4'
)
warnings = []
text_parts = []
try:
with pdfplumber.open(io.BytesIO(file_bytes)) as pdf:
num_pages = len(pdf.pages)
log.info('[bank_import] PDF has %d page(s)', num_pages)
# ── Strategy 1: structured table extraction ──────────────────────
table_rows = _pdfplumber_table_parse(pdf)
if len(table_rows) >= 3:
log.info(
'[bank_import] PDF table extraction: %d rows (skipping Groq)',
len(table_rows),
)
return table_rows, warnings
log.info(
'[bank_import] PDF table extraction yielded %d row(s) — falling back to Groq',
len(table_rows),
)
# ── Strategy 2: text extraction → Groq ──────────────────────────
for page in pdf.pages:
text = page.extract_text(x_tolerance=2, y_tolerance=2)
if text:
text_parts.append(text)
except Exception as exc:
log.error('[bank_import] pdfplumber failed: %s', exc, exc_info=True)
raise RuntimeError(
f'Could not read the PDF ({exc}). '
'Make sure the file is not password-protected and is a valid PDF.'
)
full_text = '\n'.join(text_parts).strip()
log.info('[bank_import] PDF text extracted: %d characters', len(full_text))
if len(full_text) < 150:
raise RuntimeError(
'This PDF appears to be a scanned image (no text layer found). '
'Please download the electronic/digital statement from your bank portal, '
'or export your transactions as a CSV or OFX file instead.'
)
truncated = len(full_text) > _PDF_MAX_CHARS
if truncated:
full_text = full_text[:_PDF_MAX_CHARS]
warnings.append(
f'The PDF is large — only the first {_PDF_MAX_CHARS:,} characters were '
'sent to the AI parser. If transactions are missing, split the statement '
'into shorter date ranges and import each part separately.'
)
raw_rows = _groq_parse_statement(full_text)
if not raw_rows:
raise RuntimeError(
'The AI could not find any transactions in this PDF. '
'Check that the file is a bank statement and not a summary or letter, '
'then try again.'
)
return raw_rows, warnings
# ── Public parse entry point ──────────────────────────────────────────────────
def parse_file(file_bytes, filename, col_map=None):
"""
Detect format and parse a bank statement file.
Returns dict:
rows — list of enriched transaction dicts (empty on needs_mapping)
errors — list of non-fatal parse error strings
format_name — human-readable detected format
needs_mapping — True when CSV columns could not be auto-detected
headers — list of CSV column headers (populated when needs_mapping)
"""
fname_lower = filename.lower()
# ── PDF — must be handled before decode attempt ────────────────────────
if fname_lower.endswith('.pdf') or (
isinstance(file_bytes, bytes) and file_bytes[:4] == b'%PDF'
):
raw, warnings = _parse_pdf(file_bytes)
return {
'rows': _enrich(raw), 'errors': warnings,
'format_name': 'PDF (AI-parsed)',
'needs_mapping': False, 'headers': [],
}
if isinstance(file_bytes, bytes):
try:
content = file_bytes.decode('utf-8-sig')
except UnicodeDecodeError:
content = file_bytes.decode('latin-1')
else:
content = file_bytes
# ── OFX / QFX ─────────────────────────────────────────────────────────
if fname_lower.endswith(('.ofx', '.qfx')) or '<OFX>' in content.upper()[:1000]:
raw = _parse_ofx(content)
return {
'rows': _enrich(raw), 'errors': [],
'format_name': 'OFX / QFX',
'needs_mapping': False, 'headers': [],
}
# ── CSV ────────────────────────────────────────────────────────────────
reader = csv.DictReader(io.StringIO(content))
headers = [h.strip() for h in (reader.fieldnames or [])]
headers_set = {h.lower() for h in headers}
# User-supplied mapping (from the mapping UI)
if col_map:
effective_map = col_map
format_name = 'Custom mapping'
else:
fmt = _detect_csv_format(headers_set)
if fmt:
effective_map = fmt['map']
format_name = fmt['name']
else:
effective_map = _build_generic_map(headers_set)
if effective_map:
format_name = 'Generic CSV'
else:
log.info('[bank_import] CSV column auto-detect failed; headers=%s', headers)
return {
'rows': [], 'errors': [],
'format_name': 'Unknown',
'needs_mapping': True, 'headers': headers,
}
raw, errors = _parse_csv_with_map(content, effective_map)
log.info('[bank_import] CSV parsed %d rows (format=%s, errors=%d)',
len(raw), format_name, len(errors))
return {
'rows': _enrich(raw), 'errors': errors,
'format_name': format_name,
'needs_mapping': False, 'headers': [],
}
# ── Import ─────────────────────────────────────────────────────────────────────
def do_import(rows, default_account_id, skip_dupes=True):
"""
Persist a list of row dicts to the DB.
Each row must have: date (ISO str), transaction_type, amount, description.
Optional: category_id, notes, source_id (bank ref for dupe detection).
Returns (imported_count, skipped_count).
"""
from datetime import date as date_cls
from app.extensions import db
from app.models.transaction import Transaction
from app.services.account_service import calc_balance
imported = 0
skipped = 0
affected = set()
for r in rows:
txn_date = date_cls.fromisoformat(r['date']) if isinstance(r['date'], str) else r['date']
txn_type = r['transaction_type']
amount = float(r['amount'])
description = (r.get('description') or '').strip() or 'Imported'
category_id = r.get('category_id') or None
account_id = int(r.get('account_id') or default_account_id)
notes = (r.get('notes') or '').strip()
source_id = r.get('source_id') or None
if skip_dupes:
if source_id:
dup = Transaction.query.filter(
Transaction.notes.like(f'%import:{source_id}%')
).first()
else:
dup = Transaction.query.filter_by(
date=txn_date,
amount=amount,
transaction_type=txn_type,
description=description,
).first()
if dup:
skipped += 1
continue
note_parts = []
if source_id:
note_parts.append(f'import:{source_id}')
if notes:
note_parts.append(notes)
txn = Transaction(
date=txn_date,
transaction_type=txn_type,
amount=amount,
description=description,
category_id=category_id,
account_id=account_id,
notes=' | '.join(note_parts),
)
db.session.add(txn)
affected.add(account_id)
imported += 1
db.session.commit()
for aid in affected:
calc_balance(aid)
log.info('[bank_import] do_import: imported=%d skipped=%d', imported, skipped)
return imported, skipped