Updated address_normalization_fix file
This commit is contained in:
@@ -17,11 +17,77 @@ Solution:
|
|||||||
- Normalize addresses before geocoding
|
- Normalize addresses before geocoding
|
||||||
- Use fuzzy matching to detect identical locations
|
- Use fuzzy matching to detect identical locations
|
||||||
- Prevent re-geocoding of essentially the same address
|
- Prevent re-geocoding of essentially the same address
|
||||||
|
- Extract and compare street number + street name as primary identifier
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import re
|
import re
|
||||||
from difflib import SequenceMatcher
|
from difflib import SequenceMatcher
|
||||||
|
|
||||||
|
# Try to import logger_handler for logging (optional - won't break if not available)
|
||||||
|
try:
|
||||||
|
from logger_handler import AppLogger
|
||||||
|
logger_handler = AppLogger()
|
||||||
|
LOGGING_ENABLED = True
|
||||||
|
except ImportError:
|
||||||
|
logger_handler = None
|
||||||
|
LOGGING_ENABLED = False
|
||||||
|
|
||||||
|
|
||||||
|
def _log_activity(action, message):
|
||||||
|
"""Helper function to log activity if logger is available"""
|
||||||
|
if LOGGING_ENABLED and logger_handler:
|
||||||
|
try:
|
||||||
|
logger_handler.log_user_activity(action, message)
|
||||||
|
except Exception:
|
||||||
|
pass # Ignore logging errors
|
||||||
|
|
||||||
|
|
||||||
|
def extract_street_address(address):
|
||||||
|
"""
|
||||||
|
Extract the core street address (number + street name) from an address string.
|
||||||
|
This is the most reliable identifier for location matching.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
address: Normalized address string
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Core street address string (e.g., "3402 s glebe rd")
|
||||||
|
"""
|
||||||
|
if not address:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
# Pattern to match: street number + optional directional + street name + street type
|
||||||
|
# Examples: "3402 south glebe road", "7100 gordon rd", "123 n main st"
|
||||||
|
street_pattern = r'^(\d+[-\w]*)\s+([nsew]?\s*[\w\s]+?\s*(?:rd|st|ave|dr|ln|ct|blvd|pkwy|cir|pl|ter|hwy|way|trail|pike|run|walk|path|loop))'
|
||||||
|
|
||||||
|
match = re.search(street_pattern, address.lower())
|
||||||
|
if match:
|
||||||
|
street_num = match.group(1).strip()
|
||||||
|
street_name = match.group(2).strip()
|
||||||
|
# Clean up extra spaces
|
||||||
|
street_name = re.sub(r'\s+', ' ', street_name)
|
||||||
|
return f"{street_num} {street_name}"
|
||||||
|
|
||||||
|
# Fallback: try to extract just number + next few words
|
||||||
|
simple_pattern = r'^(\d+[-\w]*)\s+([\w\s]+)'
|
||||||
|
match = re.search(simple_pattern, address.lower())
|
||||||
|
if match:
|
||||||
|
street_num = match.group(1).strip()
|
||||||
|
# Take words until we hit something that looks like a city/state
|
||||||
|
words = match.group(2).split()
|
||||||
|
street_words = []
|
||||||
|
for word in words:
|
||||||
|
# Stop at state abbreviations or zip codes
|
||||||
|
if re.match(r'^[a-z]{2}$', word) and word in ['va', 'md', 'dc', 'ca', 'ny', 'tx', 'fl', 'pa', 'il', 'oh', 'ga', 'nc', 'nj']:
|
||||||
|
break
|
||||||
|
if re.match(r'^\d{5}', word):
|
||||||
|
break
|
||||||
|
street_words.append(word)
|
||||||
|
if street_words:
|
||||||
|
return f"{street_num} {' '.join(street_words[:4])}" # Limit to 4 words
|
||||||
|
|
||||||
|
return address
|
||||||
|
|
||||||
|
|
||||||
def normalize_address(address):
|
def normalize_address(address):
|
||||||
"""
|
"""
|
||||||
@@ -98,7 +164,11 @@ def normalize_address(address):
|
|||||||
|
|
||||||
# Known neighborhood keywords to remove (these don't affect geocoding)
|
# Known neighborhood keywords to remove (these don't affect geocoding)
|
||||||
neighborhood_keywords = ['hills', 'heights', 'park', 'village', 'estates',
|
neighborhood_keywords = ['hills', 'heights', 'park', 'village', 'estates',
|
||||||
'manor', 'gardens', 'terrace', 'commons', 'plaza']
|
'manor', 'gardens', 'terrace', 'commons', 'plaza',
|
||||||
|
'downtown', 'midtown', 'uptown', 'district', 'center',
|
||||||
|
'crossing', 'corner', 'square', 'point', 'landing',
|
||||||
|
'aurora', 'crystal', 'forest', 'lake', 'river', 'creek',
|
||||||
|
'meadow', 'valley', 'ridge', 'grove', 'glen', 'woods']
|
||||||
|
|
||||||
for i, part in enumerate(parts):
|
for i, part in enumerate(parts):
|
||||||
part_clean = part.strip()
|
part_clean = part.strip()
|
||||||
@@ -157,15 +227,93 @@ def normalize_address(address):
|
|||||||
return normalized
|
return normalized
|
||||||
|
|
||||||
|
|
||||||
def addresses_are_similar(addr1, addr2, threshold=0.90):
|
def extract_address_components(address):
|
||||||
|
"""
|
||||||
|
Extract key components from an address for comparison.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
address: Address string (raw or normalized)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Dictionary with extracted components:
|
||||||
|
- street_number: The street number (e.g., "3402")
|
||||||
|
- street_name: The street name with type (e.g., "s glebe rd")
|
||||||
|
- city: City name if found
|
||||||
|
- state: State abbreviation if found
|
||||||
|
- zip_code: ZIP code if found
|
||||||
|
"""
|
||||||
|
if not address:
|
||||||
|
return {}
|
||||||
|
|
||||||
|
addr_lower = address.lower().strip()
|
||||||
|
|
||||||
|
components = {
|
||||||
|
'street_number': None,
|
||||||
|
'street_name': None,
|
||||||
|
'city': None,
|
||||||
|
'state': None,
|
||||||
|
'zip_code': None
|
||||||
|
}
|
||||||
|
|
||||||
|
# Extract street number (at the beginning)
|
||||||
|
street_num_match = re.match(r'^(\d+[-\w]*)', addr_lower)
|
||||||
|
if street_num_match:
|
||||||
|
components['street_number'] = street_num_match.group(1)
|
||||||
|
|
||||||
|
# Extract ZIP code
|
||||||
|
zip_match = re.search(r'\b(\d{5})(?:-\d{4})?\b', addr_lower)
|
||||||
|
if zip_match:
|
||||||
|
components['zip_code'] = zip_match.group(1)
|
||||||
|
|
||||||
|
# Extract state (2-letter abbreviation before or after zip)
|
||||||
|
state_match = re.search(r'\b([a-z]{2})\s*(?:\d{5}|$)', addr_lower)
|
||||||
|
if state_match:
|
||||||
|
potential_state = state_match.group(1)
|
||||||
|
# Validate it's a real state abbreviation
|
||||||
|
valid_states = ['al', 'ak', 'az', 'ar', 'ca', 'co', 'ct', 'de', 'fl', 'ga',
|
||||||
|
'hi', 'id', 'il', 'in', 'ia', 'ks', 'ky', 'la', 'me', 'md',
|
||||||
|
'ma', 'mi', 'mn', 'ms', 'mo', 'mt', 'ne', 'nv', 'nh', 'nj',
|
||||||
|
'nm', 'ny', 'nc', 'nd', 'oh', 'ok', 'or', 'pa', 'ri', 'sc',
|
||||||
|
'sd', 'tn', 'tx', 'ut', 'vt', 'va', 'wa', 'wv', 'wi', 'wy', 'dc']
|
||||||
|
if potential_state in valid_states:
|
||||||
|
components['state'] = potential_state
|
||||||
|
|
||||||
|
# Extract street name (between number and city/state/zip)
|
||||||
|
# This is the trickiest part
|
||||||
|
if components['street_number']:
|
||||||
|
# Remove street number from beginning
|
||||||
|
remainder = addr_lower[len(components['street_number']):].strip()
|
||||||
|
remainder = remainder.lstrip(',').strip()
|
||||||
|
|
||||||
|
# Look for street type keywords
|
||||||
|
street_types = ['rd', 'st', 'ave', 'dr', 'ln', 'ct', 'blvd', 'pkwy', 'cir',
|
||||||
|
'pl', 'ter', 'hwy', 'way', 'trail', 'pike', 'run', 'walk',
|
||||||
|
'path', 'loop', 'road', 'street', 'avenue', 'drive', 'lane',
|
||||||
|
'court', 'boulevard', 'parkway', 'circle', 'place', 'terrace',
|
||||||
|
'highway']
|
||||||
|
|
||||||
|
for st_type in street_types:
|
||||||
|
pattern = rf'^([\w\s]+?\s*{st_type})\b'
|
||||||
|
match = re.search(pattern, remainder)
|
||||||
|
if match:
|
||||||
|
components['street_name'] = match.group(1).strip()
|
||||||
|
break
|
||||||
|
|
||||||
|
return components
|
||||||
|
|
||||||
|
|
||||||
|
def addresses_are_similar(addr1, addr2, threshold=0.85):
|
||||||
"""
|
"""
|
||||||
Check if two addresses are similar enough to be considered the same location
|
Check if two addresses are similar enough to be considered the same location
|
||||||
Uses fuzzy string matching to handle minor variations
|
Uses multiple comparison strategies for robust matching:
|
||||||
|
1. Direct street address comparison (highest priority)
|
||||||
|
2. Component-based comparison
|
||||||
|
3. Fuzzy string matching on normalized addresses
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
addr1: First address string
|
addr1: First address string
|
||||||
addr2: Second address string
|
addr2: Second address string
|
||||||
threshold: Similarity threshold (0-1), default 0.90 (90% similar)
|
threshold: Similarity threshold (0-1), default 0.85 (85% similar)
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Boolean indicating if addresses are similar
|
Boolean indicating if addresses are similar
|
||||||
@@ -184,6 +332,10 @@ def addresses_are_similar(addr1, addr2, threshold=0.90):
|
|||||||
if not addr1 or not addr2:
|
if not addr1 or not addr2:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
print(f"\n🔍 ADDRESS SIMILARITY CHECK:")
|
||||||
|
print(f" Address 1: {addr1}")
|
||||||
|
print(f" Address 2: {addr2}")
|
||||||
|
|
||||||
# Normalize both addresses
|
# Normalize both addresses
|
||||||
norm1 = normalize_address(addr1)
|
norm1 = normalize_address(addr1)
|
||||||
norm2 = normalize_address(addr2)
|
norm2 = normalize_address(addr2)
|
||||||
@@ -193,16 +345,68 @@ def addresses_are_similar(addr1, addr2, threshold=0.90):
|
|||||||
print(f"✅ Addresses match exactly after normalization")
|
print(f"✅ Addresses match exactly after normalization")
|
||||||
return True
|
return True
|
||||||
|
|
||||||
# Calculate similarity using difflib SequenceMatcher
|
# STRATEGY 1: Extract and compare core street addresses
|
||||||
similarity = SequenceMatcher(None, norm1, norm2).ratio()
|
# This is the most reliable method for catching cases like:
|
||||||
|
# "3402 South Glebe Road Arlington VA 22202" vs
|
||||||
|
# "3402, South Glebe Road, Aurora Hills, Arlington VA 22202"
|
||||||
|
street1 = extract_street_address(norm1)
|
||||||
|
street2 = extract_street_address(norm2)
|
||||||
|
|
||||||
|
print(f" Street Address 1: '{street1}'")
|
||||||
|
print(f" Street Address 2: '{street2}'")
|
||||||
|
|
||||||
|
if street1 and street2:
|
||||||
|
street_similarity = SequenceMatcher(None, street1, street2).ratio()
|
||||||
|
print(f" Street similarity: {street_similarity:.2%}")
|
||||||
|
|
||||||
|
# If street addresses are very similar (>92%), addresses are the same
|
||||||
|
if street_similarity >= 0.92:
|
||||||
|
print(f"✅ SIMILAR - Street addresses match ({street_similarity:.2%})")
|
||||||
|
return True
|
||||||
|
|
||||||
|
# STRATEGY 2: Component-based comparison
|
||||||
|
comp1 = extract_address_components(addr1)
|
||||||
|
comp2 = extract_address_components(addr2)
|
||||||
|
|
||||||
|
print(f" Components 1: {comp1}")
|
||||||
|
print(f" Components 2: {comp2}")
|
||||||
|
|
||||||
|
# If street numbers match exactly and street names are similar
|
||||||
|
if comp1.get('street_number') and comp2.get('street_number'):
|
||||||
|
if comp1['street_number'] == comp2['street_number']:
|
||||||
|
# Same street number - check street name similarity
|
||||||
|
if comp1.get('street_name') and comp2.get('street_name'):
|
||||||
|
name_sim = SequenceMatcher(None,
|
||||||
|
comp1['street_name'],
|
||||||
|
comp2['street_name']).ratio()
|
||||||
|
print(f" Street name similarity: {name_sim:.2%}")
|
||||||
|
|
||||||
|
if name_sim >= 0.85:
|
||||||
|
# Also check if zip codes match (if both have them)
|
||||||
|
if comp1.get('zip_code') and comp2.get('zip_code'):
|
||||||
|
if comp1['zip_code'] == comp2['zip_code']:
|
||||||
|
print(f"✅ SIMILAR - Same street number, similar name, same ZIP")
|
||||||
|
return True
|
||||||
|
else:
|
||||||
|
# No zip to compare, but street info matches
|
||||||
|
print(f"✅ SIMILAR - Same street number, similar street name")
|
||||||
|
return True
|
||||||
|
|
||||||
|
# STRATEGY 3: Full normalized address fuzzy matching
|
||||||
|
similarity = SequenceMatcher(None, norm1, norm2).ratio()
|
||||||
is_similar = similarity >= threshold
|
is_similar = similarity >= threshold
|
||||||
|
|
||||||
print(f"📊 Address similarity check:")
|
print(f"📊 Full address similarity:")
|
||||||
print(f" Address 1 (normalized): {norm1}")
|
print(f" Address 1 (normalized): {norm1}")
|
||||||
print(f" Address 2 (normalized): {norm2}")
|
print(f" Address 2 (normalized): {norm2}")
|
||||||
print(f" Similarity score: {similarity:.2%}")
|
print(f" Similarity score: {similarity:.2%}")
|
||||||
print(f" Threshold: {threshold:.2%}")
|
print(f" Threshold: {threshold:.2%}")
|
||||||
print(f" Result: {'✅ SIMILAR (same location)' if is_similar else '❌ DIFFERENT (different locations)'}")
|
print(f" Result: {'✅ SIMILAR (same location)' if is_similar else '❌ DIFFERENT (different locations)'}")
|
||||||
|
|
||||||
return is_similar
|
# Log the address similarity check result
|
||||||
|
_log_activity(
|
||||||
|
'address_similarity_check',
|
||||||
|
f"Compared addresses: similarity={similarity:.2%}, result={'SIMILAR' if is_similar else 'DIFFERENT'}"
|
||||||
|
)
|
||||||
|
|
||||||
|
return is_similar
|
||||||
Reference in New Issue
Block a user