From df3ba262215b15a1a6f2841fa2945e187db447ce Mon Sep 17 00:00:00 2001 From: NguyenND Date: Mon, 22 Dec 2025 15:23:29 -0500 Subject: [PATCH] Updated address_normalization_fix file --- address_normalization_fix.py | 220 +++++++++++++++++++++++++++++++++-- 1 file changed, 212 insertions(+), 8 deletions(-) diff --git a/address_normalization_fix.py b/address_normalization_fix.py index be2882c..fda4532 100644 --- a/address_normalization_fix.py +++ b/address_normalization_fix.py @@ -17,11 +17,77 @@ Solution: - Normalize addresses before geocoding - Use fuzzy matching to detect identical locations - Prevent re-geocoding of essentially the same address +- Extract and compare street number + street name as primary identifier """ import re from difflib import SequenceMatcher +# Try to import logger_handler for logging (optional - won't break if not available) +try: + from logger_handler import AppLogger + logger_handler = AppLogger() + LOGGING_ENABLED = True +except ImportError: + logger_handler = None + LOGGING_ENABLED = False + + +def _log_activity(action, message): + """Helper function to log activity if logger is available""" + if LOGGING_ENABLED and logger_handler: + try: + logger_handler.log_user_activity(action, message) + except Exception: + pass # Ignore logging errors + + +def extract_street_address(address): + """ + Extract the core street address (number + street name) from an address string. + This is the most reliable identifier for location matching. + + Args: + address: Normalized address string + + Returns: + Core street address string (e.g., "3402 s glebe rd") + """ + if not address: + return "" + + # Pattern to match: street number + optional directional + street name + street type + # Examples: "3402 south glebe road", "7100 gordon rd", "123 n main st" + street_pattern = r'^(\d+[-\w]*)\s+([nsew]?\s*[\w\s]+?\s*(?:rd|st|ave|dr|ln|ct|blvd|pkwy|cir|pl|ter|hwy|way|trail|pike|run|walk|path|loop))' + + match = re.search(street_pattern, address.lower()) + if match: + street_num = match.group(1).strip() + street_name = match.group(2).strip() + # Clean up extra spaces + street_name = re.sub(r'\s+', ' ', street_name) + return f"{street_num} {street_name}" + + # Fallback: try to extract just number + next few words + simple_pattern = r'^(\d+[-\w]*)\s+([\w\s]+)' + match = re.search(simple_pattern, address.lower()) + if match: + street_num = match.group(1).strip() + # Take words until we hit something that looks like a city/state + words = match.group(2).split() + street_words = [] + for word in words: + # Stop at state abbreviations or zip codes + if re.match(r'^[a-z]{2}$', word) and word in ['va', 'md', 'dc', 'ca', 'ny', 'tx', 'fl', 'pa', 'il', 'oh', 'ga', 'nc', 'nj']: + break + if re.match(r'^\d{5}', word): + break + street_words.append(word) + if street_words: + return f"{street_num} {' '.join(street_words[:4])}" # Limit to 4 words + + return address + def normalize_address(address): """ @@ -98,7 +164,11 @@ def normalize_address(address): # Known neighborhood keywords to remove (these don't affect geocoding) neighborhood_keywords = ['hills', 'heights', 'park', 'village', 'estates', - 'manor', 'gardens', 'terrace', 'commons', 'plaza'] + 'manor', 'gardens', 'terrace', 'commons', 'plaza', + 'downtown', 'midtown', 'uptown', 'district', 'center', + 'crossing', 'corner', 'square', 'point', 'landing', + 'aurora', 'crystal', 'forest', 'lake', 'river', 'creek', + 'meadow', 'valley', 'ridge', 'grove', 'glen', 'woods'] for i, part in enumerate(parts): part_clean = part.strip() @@ -157,15 +227,93 @@ def normalize_address(address): return normalized -def addresses_are_similar(addr1, addr2, threshold=0.90): +def extract_address_components(address): + """ + Extract key components from an address for comparison. + + Args: + address: Address string (raw or normalized) + + Returns: + Dictionary with extracted components: + - street_number: The street number (e.g., "3402") + - street_name: The street name with type (e.g., "s glebe rd") + - city: City name if found + - state: State abbreviation if found + - zip_code: ZIP code if found + """ + if not address: + return {} + + addr_lower = address.lower().strip() + + components = { + 'street_number': None, + 'street_name': None, + 'city': None, + 'state': None, + 'zip_code': None + } + + # Extract street number (at the beginning) + street_num_match = re.match(r'^(\d+[-\w]*)', addr_lower) + if street_num_match: + components['street_number'] = street_num_match.group(1) + + # Extract ZIP code + zip_match = re.search(r'\b(\d{5})(?:-\d{4})?\b', addr_lower) + if zip_match: + components['zip_code'] = zip_match.group(1) + + # Extract state (2-letter abbreviation before or after zip) + state_match = re.search(r'\b([a-z]{2})\s*(?:\d{5}|$)', addr_lower) + if state_match: + potential_state = state_match.group(1) + # Validate it's a real state abbreviation + valid_states = ['al', 'ak', 'az', 'ar', 'ca', 'co', 'ct', 'de', 'fl', 'ga', + 'hi', 'id', 'il', 'in', 'ia', 'ks', 'ky', 'la', 'me', 'md', + 'ma', 'mi', 'mn', 'ms', 'mo', 'mt', 'ne', 'nv', 'nh', 'nj', + 'nm', 'ny', 'nc', 'nd', 'oh', 'ok', 'or', 'pa', 'ri', 'sc', + 'sd', 'tn', 'tx', 'ut', 'vt', 'va', 'wa', 'wv', 'wi', 'wy', 'dc'] + if potential_state in valid_states: + components['state'] = potential_state + + # Extract street name (between number and city/state/zip) + # This is the trickiest part + if components['street_number']: + # Remove street number from beginning + remainder = addr_lower[len(components['street_number']):].strip() + remainder = remainder.lstrip(',').strip() + + # Look for street type keywords + street_types = ['rd', 'st', 'ave', 'dr', 'ln', 'ct', 'blvd', 'pkwy', 'cir', + 'pl', 'ter', 'hwy', 'way', 'trail', 'pike', 'run', 'walk', + 'path', 'loop', 'road', 'street', 'avenue', 'drive', 'lane', + 'court', 'boulevard', 'parkway', 'circle', 'place', 'terrace', + 'highway'] + + for st_type in street_types: + pattern = rf'^([\w\s]+?\s*{st_type})\b' + match = re.search(pattern, remainder) + if match: + components['street_name'] = match.group(1).strip() + break + + return components + + +def addresses_are_similar(addr1, addr2, threshold=0.85): """ Check if two addresses are similar enough to be considered the same location - Uses fuzzy string matching to handle minor variations + Uses multiple comparison strategies for robust matching: + 1. Direct street address comparison (highest priority) + 2. Component-based comparison + 3. Fuzzy string matching on normalized addresses Args: addr1: First address string addr2: Second address string - threshold: Similarity threshold (0-1), default 0.90 (90% similar) + threshold: Similarity threshold (0-1), default 0.85 (85% similar) Returns: Boolean indicating if addresses are similar @@ -184,6 +332,10 @@ def addresses_are_similar(addr1, addr2, threshold=0.90): if not addr1 or not addr2: return False + print(f"\nšŸ” ADDRESS SIMILARITY CHECK:") + print(f" Address 1: {addr1}") + print(f" Address 2: {addr2}") + # Normalize both addresses norm1 = normalize_address(addr1) norm2 = normalize_address(addr2) @@ -193,16 +345,68 @@ def addresses_are_similar(addr1, addr2, threshold=0.90): print(f"āœ… Addresses match exactly after normalization") return True - # Calculate similarity using difflib SequenceMatcher - similarity = SequenceMatcher(None, norm1, norm2).ratio() + # STRATEGY 1: Extract and compare core street addresses + # This is the most reliable method for catching cases like: + # "3402 South Glebe Road Arlington VA 22202" vs + # "3402, South Glebe Road, Aurora Hills, Arlington VA 22202" + street1 = extract_street_address(norm1) + street2 = extract_street_address(norm2) + print(f" Street Address 1: '{street1}'") + print(f" Street Address 2: '{street2}'") + + if street1 and street2: + street_similarity = SequenceMatcher(None, street1, street2).ratio() + print(f" Street similarity: {street_similarity:.2%}") + + # If street addresses are very similar (>92%), addresses are the same + if street_similarity >= 0.92: + print(f"āœ… SIMILAR - Street addresses match ({street_similarity:.2%})") + return True + + # STRATEGY 2: Component-based comparison + comp1 = extract_address_components(addr1) + comp2 = extract_address_components(addr2) + + print(f" Components 1: {comp1}") + print(f" Components 2: {comp2}") + + # If street numbers match exactly and street names are similar + if comp1.get('street_number') and comp2.get('street_number'): + if comp1['street_number'] == comp2['street_number']: + # Same street number - check street name similarity + if comp1.get('street_name') and comp2.get('street_name'): + name_sim = SequenceMatcher(None, + comp1['street_name'], + comp2['street_name']).ratio() + print(f" Street name similarity: {name_sim:.2%}") + + if name_sim >= 0.85: + # Also check if zip codes match (if both have them) + if comp1.get('zip_code') and comp2.get('zip_code'): + if comp1['zip_code'] == comp2['zip_code']: + print(f"āœ… SIMILAR - Same street number, similar name, same ZIP") + return True + else: + # No zip to compare, but street info matches + print(f"āœ… SIMILAR - Same street number, similar street name") + return True + + # STRATEGY 3: Full normalized address fuzzy matching + similarity = SequenceMatcher(None, norm1, norm2).ratio() is_similar = similarity >= threshold - print(f"šŸ“Š Address similarity check:") + print(f"šŸ“Š Full address similarity:") print(f" Address 1 (normalized): {norm1}") print(f" Address 2 (normalized): {norm2}") print(f" Similarity score: {similarity:.2%}") print(f" Threshold: {threshold:.2%}") print(f" Result: {'āœ… SIMILAR (same location)' if is_similar else 'āŒ DIFFERENT (different locations)'}") - return is_similar + # Log the address similarity check result + _log_activity( + 'address_similarity_check', + f"Compared addresses: similarity={similarity:.2%}, result={'SIMILAR' if is_similar else 'DIFFERENT'}" + ) + + return is_similar \ No newline at end of file