04/30 Web Checker web app ver. 1.0
This commit is contained in:
@@ -0,0 +1,341 @@
|
||||
"""
|
||||
routes/ai_summary.py — AI document analysis routes.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import json
|
||||
import io
|
||||
import os
|
||||
import re
|
||||
import requests as http_requests
|
||||
from flask import (Blueprint, render_template, request, redirect, url_for,
|
||||
flash, session, jsonify)
|
||||
from models import (
|
||||
get_all_criteria, get_active_criteria, create_criterion, update_criterion,
|
||||
delete_criterion, save_ai_analysis, get_ai_analysis_history,
|
||||
get_ai_analysis_detail, log_action,
|
||||
)
|
||||
from config import get_setting
|
||||
from utils.decorators import login_required, admin_required
|
||||
|
||||
logger = logging.getLogger("routes.ai_summary")
|
||||
ai_summary_bp = Blueprint("ai_summary", __name__, url_prefix="/ai-summary")
|
||||
|
||||
GROQ_MODELS = [
|
||||
"llama-3.3-70b-versatile",
|
||||
"llama-3.1-8b-instant",
|
||||
"gemma2-9b-it",
|
||||
]
|
||||
|
||||
SUPPORTED_EXT = {".txt", ".md", ".csv", ".pdf", ".doc", ".docx", ".xlsx", ".xls"}
|
||||
|
||||
|
||||
@ai_summary_bp.route("/")
|
||||
@login_required
|
||||
def ai_summary():
|
||||
user = session["user"]
|
||||
criteria = get_all_criteria()
|
||||
history = get_ai_analysis_history(
|
||||
user_id=(None if user["role"] == "admin" else user["id"])
|
||||
)
|
||||
groq_key = get_setting("groq.api_key", "")
|
||||
groq_model = get_setting("groq.model", GROQ_MODELS[0])
|
||||
return render_template("ai_summary.html",
|
||||
criteria=criteria, history=history,
|
||||
groq_key_set=bool(groq_key),
|
||||
groq_model=groq_model,
|
||||
groq_models=GROQ_MODELS,
|
||||
is_admin=user["role"] == "admin")
|
||||
|
||||
|
||||
@ai_summary_bp.route("/analyze", methods=["POST"])
|
||||
@login_required
|
||||
def analyze():
|
||||
user = session["user"]
|
||||
files = request.files.getlist("documents[]")
|
||||
model = request.form.get("model", get_setting("groq.model", GROQ_MODELS[0]))
|
||||
# Read API key: env var takes priority, then DB setting
|
||||
api_key = (os.environ.get("GROQ_API_KEY") or "").strip() or get_setting("groq.api_key", "").strip()
|
||||
|
||||
if not api_key:
|
||||
return jsonify({"error": "Groq API key is not configured. Set GROQ_API_KEY in .env or via Admin → Settings."}), 400
|
||||
if not files or all(f.filename == "" for f in files):
|
||||
return jsonify({"error": "No files uploaded."}), 400
|
||||
|
||||
texts = []
|
||||
file_names = []
|
||||
errors = []
|
||||
|
||||
for f in files:
|
||||
if not f.filename:
|
||||
continue
|
||||
ext = ("." + f.filename.rsplit(".", 1)[-1].lower()) if "." in f.filename else ""
|
||||
if ext not in SUPPORTED_EXT:
|
||||
errors.append(f"{f.filename}: unsupported file type")
|
||||
continue
|
||||
file_names.append(f.filename)
|
||||
try:
|
||||
content = _extract_text(f, ext)
|
||||
if content and content.strip():
|
||||
texts.append(f"=== {f.filename} ===\n{content.strip()}")
|
||||
else:
|
||||
errors.append(f"{f.filename}: no text could be extracted")
|
||||
except Exception as e:
|
||||
logger.warning(f"Could not extract text from {f.filename}: {e}")
|
||||
errors.append(f"{f.filename}: {e}")
|
||||
|
||||
if not texts:
|
||||
detail = "; ".join(errors) if errors else "Check the file format and try again."
|
||||
return jsonify({"error": f"Could not extract text from any uploaded file. {detail}"}), 400
|
||||
|
||||
criteria = get_active_criteria()
|
||||
combined = "\n\n".join(texts)
|
||||
|
||||
try:
|
||||
result = _call_groq(api_key, model, combined, criteria)
|
||||
except Exception as e:
|
||||
logger.error(f"Groq API error: {e}")
|
||||
return jsonify({"error": f"AI analysis failed: {e}"}), 500
|
||||
|
||||
criteria_snap = json.dumps([
|
||||
{"title": c["title"], "description": c["description"]} for c in criteria
|
||||
])
|
||||
analysis_id = save_ai_analysis(
|
||||
user["id"], ", ".join(file_names), model,
|
||||
result.get("verdict"), criteria_snap, result.get("summary", "")
|
||||
)
|
||||
log_action(user["id"], "AI_ANALYSIS", "ai_analysis_log", analysis_id,
|
||||
f"AI analysis on {len(file_names)} file(s). Verdict: {result.get('verdict')}.")
|
||||
|
||||
return jsonify({"analysis_id": analysis_id, "model": model,
|
||||
"file_count": len(file_names), **result})
|
||||
|
||||
|
||||
def _extract_text(file_obj, ext: str) -> str:
|
||||
"""Extract plain text from an uploaded file object."""
|
||||
data = file_obj.read()
|
||||
|
||||
if ext in (".txt", ".md", ".csv"):
|
||||
return data.decode("utf-8", errors="replace")
|
||||
|
||||
if ext == ".pdf":
|
||||
import pypdf
|
||||
reader = pypdf.PdfReader(io.BytesIO(data))
|
||||
pages = []
|
||||
for page in reader.pages:
|
||||
t = page.extract_text()
|
||||
if t:
|
||||
pages.append(t)
|
||||
return "\n".join(pages)
|
||||
|
||||
if ext in (".docx", ".doc"):
|
||||
# Use python-docx (installed as 'docx') — do NOT use docx2txt
|
||||
from docx import Document
|
||||
doc = Document(io.BytesIO(data))
|
||||
lines = [para.text for para in doc.paragraphs if para.text.strip()]
|
||||
return "\n".join(lines)
|
||||
|
||||
if ext in (".xlsx", ".xls"):
|
||||
import openpyxl
|
||||
wb = openpyxl.load_workbook(io.BytesIO(data), read_only=True, data_only=True)
|
||||
lines = []
|
||||
for ws in wb.worksheets:
|
||||
for row in ws.iter_rows(values_only=True):
|
||||
line = "\t".join(str(c) if c is not None else "" for c in row)
|
||||
if line.strip():
|
||||
lines.append(line)
|
||||
return "\n".join(lines)
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
# Office address used as origin for distance/travel-time estimates.
|
||||
_OFFICE_ADDRESS = "2815 Hartland Road, Falls Church, VA 22043, USA"
|
||||
|
||||
# Stage 1 — extraction prompt (always sent)
|
||||
_EXTRACTION_PROMPT = """You are an expert government procurement analyst.
|
||||
The user has provided {n} document(s). Your job is to extract specific information from each document and present it in a clean, structured format.
|
||||
|
||||
IMPORTANT — Our office is located at:
|
||||
{office}
|
||||
Use this as the ORIGIN address for all driving distance and travel time calculations in field #9 below.
|
||||
|
||||
For EACH document, extract and clearly label the following fields (write "N/A" if a field is not found):
|
||||
|
||||
1. Solicitation Number
|
||||
2. Solicitation Type (e.g. RFP, RFQ, IFB, etc.)
|
||||
3. Set-Aside (e.g. Small Business, 8(a), N/A)
|
||||
4. Description / Scope of Work
|
||||
5. Work Site / Location(s)
|
||||
6. Pre-Proposal Conference / Site-Visit (date, time, full address)
|
||||
7. Point of Contact (POC) (name, phone, email)
|
||||
8. Total Square Footage (if applicable)
|
||||
9. Driving Distance & Travel Time
|
||||
- Origin: {office}
|
||||
- Destination: Pre-Proposal Conference or primary Work Site address
|
||||
- Provide your best estimate of driving distance (miles) and typical driving time using major highways
|
||||
- Note that these are AI estimates; actual times may vary with traffic
|
||||
10. Last Day to Submit Questions
|
||||
11. Due Date & Time
|
||||
12. Any other notable requirements or deadlines
|
||||
|
||||
After the per-document breakdown, provide a detailed OVERALL SUMMARY covering:
|
||||
|
||||
A. Scope of Work
|
||||
B. Contract Period
|
||||
C. Proposal Submission Requirements
|
||||
D. Key Deadlines & Action Items
|
||||
|
||||
Be precise, detailed, and use bullet points throughout.
|
||||
If information is not explicitly stated in the documents, note it as "Not specified in the document."
|
||||
|
||||
DOCUMENTS:
|
||||
{documents}"""
|
||||
|
||||
# Stage 2 — criteria evaluation suffix (appended only when active criteria exist)
|
||||
_CRITERIA_PROMPT_SUFFIX = """
|
||||
|
||||
================================================================================
|
||||
OPPORTUNITY ALIGNMENT EVALUATION
|
||||
================================================================================
|
||||
|
||||
After completing the extraction and summary above, evaluate whether this
|
||||
opportunity aligns with our company's interests based on the following criteria.
|
||||
|
||||
OUR EVALUATION CRITERIA:
|
||||
{criteria_list}
|
||||
|
||||
For EACH criterion above:
|
||||
- State whether the opportunity MEETS, DOES NOT MEET, or PARTIALLY MEETS it.
|
||||
- Provide a brief, specific explanation citing details from the document(s).
|
||||
|
||||
Then provide an OVERALL RECOMMENDATION using EXACTLY one of these three labels
|
||||
on its own line (this label is machine-read — do not alter it):
|
||||
|
||||
RECOMMENDATION: PURSUE
|
||||
RECOMMENDATION: PASS
|
||||
RECOMMENDATION: UNCLEAR
|
||||
|
||||
Use PURSUE if the opportunity clearly meets most criteria and presents strong
|
||||
alignment. Use PASS if it clearly fails key criteria. Use UNCLEAR if the
|
||||
documents lack sufficient information to make a confident determination.
|
||||
|
||||
End with a 2-3 sentence EXECUTIVE SUMMARY explaining your recommendation
|
||||
in plain business language."""
|
||||
|
||||
|
||||
def _call_groq(api_key: str, model: str, text: str, criteria: list) -> dict:
|
||||
"""Call the Groq chat completions REST API directly (no SDK required)."""
|
||||
import re
|
||||
|
||||
n = text.count("=== ") or 1 # count file separators for the prompt header
|
||||
|
||||
# Build the two-stage prompt matching the desktop app exactly
|
||||
prompt = _EXTRACTION_PROMPT.format(
|
||||
n=n, office=_OFFICE_ADDRESS, documents=text[:14000]
|
||||
)
|
||||
|
||||
if criteria:
|
||||
criteria_list = "\n".join(
|
||||
f" {i+1}. {c['title']}: {c['description']}"
|
||||
for i, c in enumerate(criteria)
|
||||
)
|
||||
prompt += _CRITERIA_PROMPT_SUFFIX.format(criteria_list=criteria_list)
|
||||
|
||||
response = http_requests.post(
|
||||
"https://api.groq.com/openai/v1/chat/completions",
|
||||
headers={
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
json={
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": prompt}],
|
||||
"max_tokens": 4096,
|
||||
"temperature": 0.2,
|
||||
},
|
||||
timeout=90,
|
||||
)
|
||||
if not response.ok:
|
||||
logger.error(f"Groq API {response.status_code}: {response.text[:300]}")
|
||||
response.raise_for_status()
|
||||
|
||||
content = response.json()["choices"][0]["message"]["content"] or ""
|
||||
|
||||
# Parse the machine-readable RECOMMENDATION label (only present when criteria used)
|
||||
verdict = None
|
||||
if criteria:
|
||||
match = re.search(
|
||||
r"RECOMMENDATION\s*:\s*(PURSUE|PASS|UNCLEAR)",
|
||||
content, re.IGNORECASE,
|
||||
)
|
||||
if match:
|
||||
verdict = match.group(1).upper()
|
||||
|
||||
return {"verdict": verdict, "summary": content}
|
||||
|
||||
|
||||
# ─── Criteria Management (admin only) ─────────────────────────────────────────
|
||||
|
||||
@ai_summary_bp.route("/criteria/create", methods=["POST"])
|
||||
@admin_required
|
||||
def create_criterion_view():
|
||||
admin = session["user"]
|
||||
title = request.form.get("title", "").strip()
|
||||
desc = request.form.get("description", "").strip()
|
||||
is_active = request.form.get("is_active", "1") == "1"
|
||||
sort_order = int(request.form.get("sort_order", 0))
|
||||
try:
|
||||
create_criterion(admin["id"], title, desc, is_active, sort_order)
|
||||
flash(f"Criterion '{title}' created.", "success")
|
||||
except Exception as e:
|
||||
flash(f"Error: {e}", "danger")
|
||||
return redirect(url_for("ai_summary.ai_summary"))
|
||||
|
||||
|
||||
@ai_summary_bp.route("/criteria/<int:criterion_id>/edit", methods=["POST"])
|
||||
@admin_required
|
||||
def edit_criterion_view(criterion_id):
|
||||
admin = session["user"]
|
||||
title = request.form.get("title", "").strip()
|
||||
desc = request.form.get("description", "").strip()
|
||||
is_active = request.form.get("is_active", "1") == "1"
|
||||
sort_order = int(request.form.get("sort_order", 0))
|
||||
try:
|
||||
update_criterion(admin["id"], criterion_id, title, desc, is_active, sort_order)
|
||||
flash(f"Criterion '{title}' updated.", "success")
|
||||
except Exception as e:
|
||||
flash(f"Error: {e}", "danger")
|
||||
return redirect(url_for("ai_summary.ai_summary"))
|
||||
|
||||
|
||||
@ai_summary_bp.route("/criteria/<int:criterion_id>/delete", methods=["POST"])
|
||||
@admin_required
|
||||
def delete_criterion_view(criterion_id):
|
||||
admin = session["user"]
|
||||
try:
|
||||
delete_criterion(admin["id"], criterion_id)
|
||||
flash("Criterion deleted.", "success")
|
||||
except Exception as e:
|
||||
flash(f"Error: {e}", "danger")
|
||||
return redirect(url_for("ai_summary.ai_summary"))
|
||||
|
||||
|
||||
@ai_summary_bp.route("/history/<int:analysis_id>")
|
||||
@login_required
|
||||
def analysis_detail(analysis_id):
|
||||
record = get_ai_analysis_detail(analysis_id)
|
||||
if not record:
|
||||
return jsonify({"error": "Not found"}), 404
|
||||
user = session["user"]
|
||||
if user["role"] != "admin" and record["user_id"] != user["id"]:
|
||||
return jsonify({"error": "Access denied"}), 403
|
||||
return jsonify({
|
||||
"id": record["id"],
|
||||
"username": record.get("username"),
|
||||
"file_names": record["file_names"],
|
||||
"model": record["model"],
|
||||
"verdict": record["verdict"],
|
||||
"analyzed_at": str(record["analyzed_at"]),
|
||||
"summary_text": record["summary_text"],
|
||||
})
|
||||
Reference in New Issue
Block a user