04/22 Add support old document *.doc

This commit is contained in:
2026-04-22 15:10:08 -04:00
parent 87cf54124a
commit b7e2a164c7
2 changed files with 95 additions and 6 deletions
+2
View File
@@ -8,3 +8,5 @@ reportlab>=4.0.0
groq>=1.0.0 groq>=1.0.0
pypdf>=3.0.0 pypdf>=3.0.0
python-docx>=1.0.0 python-docx>=1.0.0
pywin32>=306
docx2txt>=0.8
+93 -6
View File
@@ -37,16 +37,16 @@ logger = logging.getLogger("ai_summary_view")
SUPPORTED_EXT = { SUPPORTED_EXT = {
".txt", ".md", ".csv", ".txt", ".md", ".csv",
".pdf", ".pdf",
".docx", ".doc", ".docx",
".xlsx", ".xls", ".xlsx", ".xls",
} }
FILE_DIALOG_TYPES = [ FILE_DIALOG_TYPES = [
("Supported documents", ("Supported documents",
"*.txt *.md *.csv *.pdf *.docx *.xlsx *.xls"), "*.txt *.md *.csv *.pdf *.docx *.doc *.xlsx *.xls"),
("Text files", "*.txt *.md *.csv"), ("Text files", "*.txt *.md *.csv"),
("PDF files", "*.pdf"), ("PDF files", "*.pdf"),
("Word documents", "*.docx"), ("Word documents", "*.docx *.doc"),
("Excel files", "*.xlsx *.xls"), ("Excel files", "*.xlsx *.xls"),
("All files", "*.*"), ("All files", "*.*"),
] ]
@@ -65,11 +65,18 @@ _CFG_KEY_KEY = "api_key"
_CFG_KEY_MODEL = "model" _CFG_KEY_MODEL = "model"
# -- Focused extraction prompt (procurement / solicitation) -------------------- # -- Focused extraction prompt (procurement / solicitation) --------------------
_OFFICE_ADDRESS = "2815 Hartland Road, Falls Church, VA 22043, USA"
_EXTRACTION_PROMPT = """\ _EXTRACTION_PROMPT = """\
You are an expert government procurement analyst. You are an expert government procurement analyst.
The user has provided {n} document(s). Your job is to extract specific \ The user has provided {n} document(s). Your job is to extract specific \
information from each document and present it in a clean, structured format. information from each document and present it in a clean, structured format.
IMPORTANT — Our office is located at:
{office}
Use this as the ORIGIN address for all driving distance and travel time \
calculations in field #9 below.
For EACH document, extract and clearly label the following fields \ For EACH document, extract and clearly label the following fields \
(write "N/A" if a field is not found): (write "N/A" if a field is not found):
@@ -78,10 +85,16 @@ For EACH document, extract and clearly label the following fields \
3. Set-Aside (e.g. Small Business, 8(a), N/A) 3. Set-Aside (e.g. Small Business, 8(a), N/A)
4. Description / Scope of Work 4. Description / Scope of Work
5. Work Site / Location(s) 5. Work Site / Location(s)
6. Pre-Proposal Conference / Site-Visit (date, time, location — mandatory or optional) 6. Pre-Proposal Conference / Site-Visit (date, time, full address — mandatory or optional)
7. Point of Contact (POC) (name, phone, email) 7. Point of Contact (POC) (name, phone, email)
8. Total Square Footage (if applicable) 8. Total Square Footage (if applicable)
9. Driving Distance / Travel Time to Pre-Proposal Conference 9. Driving Distance & Travel Time
- Origin: {office}
- Destination: the Pre-Proposal Conference or Site-Visit address from field #6 \
(if no conference, use the primary Work Site address from field #5)
- Provide your best estimate of driving distance (miles) and typical \
driving time using major highways
- Note that these are AI estimates; actual times may vary with traffic
10. Last Day to Submit Questions 10. Last Day to Submit Questions
11. Due Date & Time 11. Due Date & Time
12. Any other notable requirements or deadlines 12. Any other notable requirements or deadlines
@@ -626,7 +639,7 @@ class AiSummaryView(ttk.Frame):
# 2. Build prompt # 2. Build prompt
n = len(file_paths) n = len(file_paths)
prompt = _EXTRACTION_PROMPT.format( prompt = _EXTRACTION_PROMPT.format(
n=n, documents=combined_text) n=n, office=_OFFICE_ADDRESS, documents=combined_text)
# 3. Call Groq API # 3. Call Groq API
self.after(0, lambda: self._set_status( self.after(0, lambda: self._set_status(
@@ -715,6 +728,8 @@ def _extract_text(file_path: str) -> str:
return _read_pdf(file_path) return _read_pdf(file_path)
if ext == ".docx": if ext == ".docx":
return _read_docx(file_path) return _read_docx(file_path)
if ext == ".doc":
return _read_doc(file_path)
if ext in (".xlsx", ".xls"): if ext in (".xlsx", ".xls"):
return _read_excel(file_path) return _read_excel(file_path)
raise ValueError(f"Unsupported file type: {ext}") raise ValueError(f"Unsupported file type: {ext}")
@@ -750,6 +765,7 @@ def _read_pdf(path: str) -> str:
def _read_docx(path: str) -> str: def _read_docx(path: str) -> str:
"""Read modern .docx (Office Open XML) via python-docx."""
try: try:
import docx import docx
doc = docx.Document(path) doc = docx.Document(path)
@@ -761,6 +777,76 @@ def _read_docx(path: str) -> str:
) )
def _read_doc(path: str) -> str:
"""Read legacy .doc (binary OLE) files.
Strategy (Windows):
1. win32com — automates Word to save as temp .docx, then reads it.
Requires Microsoft Word to be installed; most reliable.
2. docx2txt — pure-Python fallback that can sometimes handle simple .doc
files by treating them as ZIP-like structures (works for some .doc).
3. Raw text scrape — last resort ASCII extraction from the binary.
"""
# --- Strategy 1: Word COM automation (Windows + Word installed) -----------
try:
import win32com.client
import tempfile
import pythoncom
pythoncom.CoInitialize()
word = win32com.client.Dispatch("Word.Application")
word.Visible = False
try:
abs_path = os.path.abspath(path)
doc = word.Documents.Open(abs_path)
# Save as a temp .docx so python-docx can read it cleanly
with tempfile.NamedTemporaryFile(suffix=".docx", delete=False) as tf:
tmp_path = tf.name
doc.SaveAs2(tmp_path, FileFormat=16) # 16 = wdFormatXMLDocument
doc.Close(False)
text = _read_docx(tmp_path)
return text
finally:
try:
word.Quit()
except Exception:
pass
try:
os.unlink(tmp_path)
except Exception:
pass
pythoncom.CoUninitialize()
except Exception:
pass # Word not installed or COM error — try next strategy
# --- Strategy 2: docx2txt (pure Python, works on some .doc files) --------
try:
import docx2txt
text = docx2txt.process(path)
if text and text.strip():
return text
except Exception:
pass
# --- Strategy 3: raw ASCII scrape from binary ----------------------------
try:
with open(path, "rb") as f:
raw = f.read()
# Extract printable ASCII runs of 4+ chars
import re
chunks = re.findall(rb"[ -~]{4,}", raw)
text = "\n".join(c.decode("ascii", errors="ignore") for c in chunks)
if text.strip():
return text
except Exception:
pass
raise ValueError(
f"Could not extract text from '{os.path.basename(path)}'.\n"
"For best results with .doc files, install Microsoft Word "
"or convert the file to .docx before uploading."
)
def _read_excel(path: str) -> str: def _read_excel(path: str) -> str:
try: try:
import openpyxl import openpyxl
@@ -787,6 +873,7 @@ def _file_icon(path: str) -> str:
ext = os.path.splitext(path)[1].lower() ext = os.path.splitext(path)[1].lower()
return { return {
".pdf": "📄", ".pdf": "📄",
".doc": "📝",
".docx": "📝", ".docx": "📝",
".xlsx": "📊", ".xlsx": "📊",
".xls": "📊", ".xls": "📊",