06/05 Optimize app

This commit is contained in:
2026-06-05 15:51:57 -04:00
parent 458044201e
commit 025f3f8823
14 changed files with 425 additions and 64 deletions
+124 -1
View File
@@ -572,9 +572,115 @@ def _groq_parse_statement(text):
return raw_rows
_TABLE_DATE_HDRS = {'date', 'posted', 'transaction date', 'trans date',
'posting date', 'value date', 'effective date', 'settled'}
_TABLE_DESC_HDRS = {'description', 'payee', 'merchant', 'memo', 'transaction',
'details', 'name', 'narrative', 'particulars', 'reference'}
_TABLE_DEBIT_HDRS = {'debit', 'withdrawal', 'withdrawals', 'charge', 'charges',
'amount debited', 'payment', 'dr'}
_TABLE_CRED_HDRS = {'credit', 'deposit', 'deposits', 'amount credited',
'cr', 'inflow'}
_TABLE_AMT_HDRS = {'amount', 'transaction amount', 'net amount'}
def _pdfplumber_table_parse(pdf_handle):
"""
Try to extract transactions directly from pdfplumber table structures.
Iterates every page, finds tables whose headers match bank-statement
patterns, and converts rows to raw transaction dicts.
Returns a (possibly empty) list of raw dicts; never raises.
"""
all_rows = []
for page in pdf_handle.pages:
try:
tables = page.extract_tables()
except Exception:
continue
for table in tables:
if not table or len(table) < 2:
continue
# Normalise headers (lower-case, strip)
raw_headers = [str(h).strip().lower() if h else '' for h in table[0]]
date_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_DATE_HDRS), None)
desc_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_DESC_HDRS), None)
amt_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_AMT_HDRS), None)
debit_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_DEBIT_HDRS), None)
cred_col = next((i for i, h in enumerate(raw_headers)
if h in _TABLE_CRED_HDRS), None)
# Need at least date + description + one amount column
if date_col is None or desc_col is None:
continue
if amt_col is None and debit_col is None and cred_col is None:
continue
col_max = max(c for c in [date_col, desc_col, amt_col, debit_col, cred_col]
if c is not None)
for row in table[1:]:
if not row or len(row) <= col_max:
continue
date_str = str(row[date_col]).strip() if row[date_col] else ''
if not date_str or date_str.lower() in ('', 'none', '-', '--', 'n/a'):
continue
try:
txn_date = _parse_date(date_str)
except ValueError:
continue
description = str(row[desc_col]).strip() if row[desc_col] else ''
if not description or description.lower() in ('', 'none'):
continue
if debit_col is not None or cred_col is not None:
debit = abs(_clean_amount(row[debit_col] if debit_col is not None else ''))
credit = abs(_clean_amount(row[cred_col] if cred_col is not None else ''))
if debit > 0:
amount, txn_type = debit, 'expense'
elif credit > 0:
amount, txn_type = credit, 'income'
else:
continue
else:
raw_amt = _clean_amount(row[amt_col] if row[amt_col] else '')
if raw_amt == 0.0:
continue
txn_type = 'expense' if raw_amt < 0 else 'income'
amount = abs(raw_amt)
all_rows.append({
'date': txn_date,
'transaction_type': txn_type,
'amount': amount,
'description': description,
'notes': '',
'source_id': None,
})
return all_rows
def _parse_pdf(file_bytes):
"""
Extract text from a digital PDF using pdfplumber, then parse with Groq.
Extract transactions from a digital bank-statement PDF.
Strategy (in order):
1. pdfplumber table extraction — fast, free, no API call needed.
Used when structured tables with recognisable headers are found and
yield at least 3 rows.
2. pdfplumber text extraction → Groq LLM — handles unstructured
or narrative-style statements.
Returns (raw_rows, warnings_list).
Raises RuntimeError for unrecoverable problems (scanned PDF, bad file, etc.).
@@ -593,10 +699,27 @@ def _parse_pdf(file_bytes):
with pdfplumber.open(io.BytesIO(file_bytes)) as pdf:
num_pages = len(pdf.pages)
log.info('[bank_import] PDF has %d page(s)', num_pages)
# ── Strategy 1: structured table extraction ──────────────────────
table_rows = _pdfplumber_table_parse(pdf)
if len(table_rows) >= 3:
log.info(
'[bank_import] PDF table extraction: %d rows (skipping Groq)',
len(table_rows),
)
return table_rows, warnings
log.info(
'[bank_import] PDF table extraction yielded %d row(s) — falling back to Groq',
len(table_rows),
)
# ── Strategy 2: text extraction → Groq ──────────────────────────
for page in pdf.pages:
text = page.extract_text(x_tolerance=2, y_tolerance=2)
if text:
text_parts.append(text)
except Exception as exc:
log.error('[bank_import] pdfplumber failed: %s', exc, exc_info=True)
raise RuntimeError(
+70 -51
View File
@@ -1,5 +1,10 @@
"""
Export Service — CSV, Excel, and PDF generation for transactions and reports.
Memory-efficient exports:
- CSV: streaming generator (rows written one at a time, never all in memory)
- Excel: openpyxl write-only mode + DB yield_per(500) avoids loading the full
result set into Python at once
"""
import io
@@ -11,13 +16,20 @@ from app.models.transaction import Transaction
# ── CSV export ────────────────────────────────────────────────────────────────
def transactions_to_csv(transactions):
"""Return a CSV string of transactions."""
output = io.StringIO()
writer = csv.writer(output)
def transactions_csv_stream(query):
"""
Generator that yields CSV text one row at a time.
Pass the SQLAlchemy *query* (not a list) — rows are fetched in 500-row batches.
Use with Flask's stream_with_context() for a true streaming response.
"""
buf = io.StringIO()
writer = csv.writer(buf)
writer.writerow(['Date', 'Type', 'Description', 'Category', 'Account', 'Amount', 'Notes'])
for txn in transactions:
yield buf.getvalue()
buf.seek(0); buf.truncate()
for txn in query.yield_per(500):
writer.writerow([
txn.date.strftime('%Y-%m-%d'),
txn.transaction_type,
@@ -27,80 +39,87 @@ def transactions_to_csv(transactions):
float(txn.amount),
txn.notes or '',
])
output.seek(0)
return output.getvalue()
yield buf.getvalue()
buf.seek(0); buf.truncate()
# ── Excel export ──────────────────────────────────────────────────────────────
def transactions_to_excel(transactions, period_label='Transactions'):
"""Return Excel bytes for a list of transactions."""
def transactions_to_excel(query, period_label='Transactions'):
"""
Return Excel bytes built from a SQLAlchemy *query* using openpyxl write-only
mode. Rows are fetched 500 at a time so the full result set is never held in
Python memory simultaneously.
"""
from openpyxl import Workbook
from openpyxl.styles import Font, PatternFill, Alignment, Border, Side
from openpyxl.styles import Font, PatternFill, Alignment
from openpyxl.cell import WriteOnlyCell
from openpyxl.utils import get_column_letter
wb = Workbook()
ws = wb.active
ws.title = period_label[:31] # max 31 chars
symbol = current_app.config.get('APP_CURRENCY_SYMBOL', '$')
# Header style
wb = Workbook(write_only=True)
ws = wb.create_sheet(title=period_label[:31])
col_widths = [12, 10, 40, 18, 18, 16, 30]
for i, w in enumerate(col_widths, 1):
ws.column_dimensions[get_column_letter(i)].width = w
header_fill = PatternFill(start_color='0F172A', end_color='0F172A', fill_type='solid')
header_font = Font(color='F1F5F9', bold=True, size=10)
thin = Side(style='thin', color='E2E8F0')
border = Border(bottom=Side(style='thin', color='E2E8F0'))
headers = ['Date', 'Type', 'Description', 'Category', 'Account',
f'Amount ({symbol})', 'Notes']
header_row = []
for h in headers:
c = WriteOnlyCell(ws, value=h)
c.font = header_font
c.fill = header_fill
header_row.append(c)
ws.append(header_row)
headers = ['Date', 'Type', 'Description', 'Category', 'Account', f'Amount ({symbol})', 'Notes']
col_widths = [12, 10, 40, 18, 18, 16, 30]
for col, (header, width) in enumerate(zip(headers, col_widths), 1):
cell = ws.cell(row=1, column=col, value=header)
cell.font = header_font
cell.fill = header_fill
cell.alignment = Alignment(horizontal='left', vertical='center')
ws.column_dimensions[get_column_letter(col)].width = width
ws.row_dimensions[1].height = 22
# Data rows
income_fill = PatternFill(start_color='F0FDF4', end_color='F0FDF4', fill_type='solid')
expense_fill = PatternFill(start_color='FFF7F7', end_color='FFF7F7', fill_type='solid')
row_font = Font(size=10)
amt_fmt = '#,##0.00'
for row_num, txn in enumerate(transactions, 2):
running_total = 0.0
for txn in query.yield_per(500):
fill = income_fill if txn.transaction_type == 'income' else expense_fill
data = [
amt = float(txn.amount)
running_total += amt
row_vals = [
txn.date.strftime('%Y-%m-%d'),
txn.transaction_type.title(),
txn.description,
txn.category.name if txn.category else '',
txn.account.name if txn.account else '',
float(txn.amount),
amt,
txn.notes or '',
]
for col, value in enumerate(data, 1):
cell = ws.cell(row=row_num, column=col, value=value)
cell.fill = fill
cell.border = border
cell.font = Font(size=10)
if col == 6:
cell.number_format = f'#,##0.00'
cell.alignment = Alignment(horizontal='right')
row = []
for col_idx, value in enumerate(row_vals, 1):
c = WriteOnlyCell(ws, value=value)
c.font = row_font
c.fill = fill
if col_idx == 6:
c.number_format = amt_fmt
c.alignment = Alignment(horizontal='right')
row.append(c)
ws.append(row)
# Totals row
total_row = len(transactions) + 2
ws.cell(row=total_row, column=5, value='TOTAL').font = Font(bold=True, size=10)
total_cell = ws.cell(row=total_row, column=6,
value=sum(float(t.amount) for t in transactions))
total_cell.font = Font(bold=True, size=10)
total_cell.number_format = f'#,##0.00'
total_cell.alignment = Alignment(horizontal='right')
# Totals row — plain cells (write-only, no random access)
total_lbl = WriteOnlyCell(ws, value='TOTAL')
total_lbl.font = Font(bold=True, size=10)
total_val = WriteOnlyCell(ws, value=running_total)
total_val.font = Font(bold=True, size=10)
total_val.number_format = amt_fmt
total_val.alignment = Alignment(horizontal='right')
ws.append(['', '', '', '', total_lbl, total_val, ''])
output = io.BytesIO()
wb.save(output)
output.seek(0)
return output.getvalue()
return output.read()
# ── PDF export ────────────────────────────────────────────────────────────────
+93
View File
@@ -323,6 +323,99 @@ def update_prices(investment_ids=None):
return updated
def check_and_save_price_alerts(threshold: float = 5.0) -> int:
"""
Fetch today's day-change for every unique ticker that has an active holding.
For any ticker where |day_change_pct| >= threshold, write an AiInsight row
with insight_type='alert' so the investments page can surface a banner.
Deduplicates by ticker so each ticker's Groq/Yahoo call happens only once.
Returns the number of alerts saved.
"""
import json
from app.models.ai_insight import AiInsight
today = datetime.utcnow().date()
investments = Investment.query.filter(
Investment.ticker != None,
Investment.ticker != '',
Investment.is_active == True,
).all()
if not investments:
return 0
# Collect unique tickers and their holding names
ticker_map = {} # ticker → asset_name (first one found)
for inv in investments:
t = inv.ticker.upper()
if t not in ticker_map:
ticker_map[t] = inv.asset_name
alerts = []
for ticker, asset_name in ticker_map.items():
try:
change = fetch_day_change(ticker)
except Exception:
continue
if not change:
continue
pct = change.get('day_change_pct') or 0
if abs(pct) >= threshold:
alerts.append({
'ticker': ticker,
'asset_name': asset_name,
'day_change_pct': round(pct, 2),
'current_price': change.get('current'),
})
if not alerts:
return 0
# Upsert: overwrite any earlier alert from today
existing = AiInsight.query.filter_by(
insight_date=today, insight_type='alert'
).first()
content_json = json.dumps(alerts)
if existing:
existing.content = content_json
else:
db.session.add(AiInsight(
insight_date=today,
insight_type='alert',
content=content_json,
prompt_summary=f'price_alert threshold={threshold}%',
))
try:
db.session.commit()
log.info('[investment] saved %d price alert(s) (threshold=%.1f%%)', len(alerts), threshold)
except Exception as e:
db.session.rollback()
log.error('[investment] failed to save price alerts: %s', e)
return len(alerts)
def get_price_alerts():
"""
Return today's price alert list (from ai_insights) or [] if none exist.
Each item: {ticker, asset_name, day_change_pct, current_price}
"""
import json
from app.models.ai_insight import AiInsight
today = datetime.utcnow().date()
row = AiInsight.query.filter_by(insight_date=today, insight_type='alert').first()
if not row:
return []
try:
return json.loads(row.content)
except Exception:
return []
def get_portfolio_summary():
"""
Return portfolio-level aggregates across all active investments.