359 lines
14 KiB
Python
359 lines
14 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
Unified Multi-Surface Scraper for VeriPath Audits.
|
||
Extracts GBP + Apple Maps + Website, merges into verified JSON contract.
|
||
|
||
Working surfaces:
|
||
- Google Business Profile (via noworneverev/google-maps-scraper)
|
||
- Apple Maps (headless Firefox → /data/search JSON)
|
||
- Website (direct HTTP fetch)
|
||
|
||
Bot-walled (headless detection, no free bypass):
|
||
- Bing Places, Yelp
|
||
"""
|
||
import asyncio
|
||
import json
|
||
import sys
|
||
import re
|
||
import os
|
||
from datetime import datetime, timezone
|
||
|
||
# Ensure scraper is importable
|
||
sys.path.insert(0, os.path.expanduser("~/deps/google-maps-scraper/src"))
|
||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||
|
||
from gmaps_scraper.scraper import GoogleMapsScraper
|
||
from apple_scraper import scrape_apple
|
||
from bing_provider import scrape_bing
|
||
|
||
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
|
||
|
||
|
||
async def geocode(query):
|
||
"""Get coordinates from Google Maps search."""
|
||
import urllib.request
|
||
import urllib.parse
|
||
|
||
search_url = f"https://www.google.com/maps/search/{urllib.parse.quote(query)}/"
|
||
req = urllib.request.Request(search_url, headers={"User-Agent": "Mozilla/5.0"})
|
||
try:
|
||
with urllib.request.urlopen(req, timeout=10) as resp:
|
||
html = resp.read().decode("utf-8", errors="replace")
|
||
except Exception:
|
||
return None
|
||
|
||
m = re.search(r'"center":\{"lat":(-?\d+\.\d+),"lng":(-?\d+\.\d+)}', html)
|
||
if m:
|
||
return float(m.group(1)), float(m.group(2))
|
||
|
||
m = re.search(r'@(-?\d+\.\d+),(-?\d+\.\d+),\d+z', html)
|
||
if m:
|
||
return float(m.group(1)), float(m.group(2))
|
||
|
||
return None
|
||
|
||
|
||
def build_gmaps_url(query, coords):
|
||
"""Build Google Maps search URL."""
|
||
import urllib.parse
|
||
encoded = urllib.parse.quote(query)
|
||
if coords:
|
||
lat, lon = coords
|
||
return f"https://www.google.com/maps/search/{encoded}/@{lat},{lon},14z/data=!3m1!4b1"
|
||
return f"https://www.google.com/maps/search/{encoded}/data=!3m1!4b1"
|
||
|
||
|
||
# Known search-page titles (not real places) — same set the lib rejects
|
||
_SEARCH_PAGE_TITLES = {"results", "Results", "検索結果", "搜尋結果", "搜索结果"}
|
||
|
||
|
||
def _looks_like_match(place_name, business):
|
||
"""Reject search-page titles and names that share no significant token with the query."""
|
||
if not place_name:
|
||
return False
|
||
name = place_name.strip()
|
||
if name.lower() in _SEARCH_PAGE_TITLES:
|
||
return False
|
||
# Significant tokens = words >=4 chars from the business query
|
||
biz_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', business)}
|
||
name_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', name)}
|
||
if not biz_tokens:
|
||
return True
|
||
# Require at least one shared significant token (e.g. "Crystal", "Plumbing")
|
||
return bool(biz_tokens & name_tokens)
|
||
|
||
|
||
async def scrape_google(query, coords, business=""):
|
||
"""Scrape Google Business Profile data, validating the matched result."""
|
||
url = build_gmaps_url(query, coords)
|
||
try:
|
||
async with GoogleMapsScraper() as scraper:
|
||
result = await scraper.scrape(url)
|
||
if result and result.place:
|
||
p = result.place
|
||
# Validation gate: reject wrong-entity matches before trusting
|
||
if not _looks_like_match(p.name, business or query):
|
||
print(f" [REJECT] GBP matched wrong entity: '{p.name}' (expected ~'{business}')")
|
||
return None
|
||
# Hours: ["Thursday9 AM–6 PM", "SundayClosed", ...]
|
||
hours_dict = {}
|
||
if p.hours:
|
||
for h in p.hours:
|
||
m = re.match(r'(Monday|Tuesday|Wednesday|Thursday|Friday|Saturday|Sunday)(.+)', h, re.IGNORECASE)
|
||
if m:
|
||
day = m.group(1).lower()
|
||
time_str = m.group(2).strip()
|
||
time_str = time_str.replace('\u202f', ' ').replace('\u2013', '-').replace('\ue14d', '').strip()
|
||
hours_dict[day] = time_str
|
||
|
||
return {
|
||
# Strip PUA glyphs (map-pin \ue0b0 etc.) from address + phone — BMP PUA (U+E000–U+F8FF) breaks downstream normalize
|
||
"name": p.name,
|
||
"address": re.sub(r'[\ue000-\uf8ff]', '', p.address or '').strip(),
|
||
"phone": re.sub(r'[\ue000-\uf8ff]', '', p.phone or '').strip() if p.phone else p.phone,
|
||
"website": p.website,
|
||
"rating": p.rating,
|
||
"reviews": p.review_count,
|
||
"hours": hours_dict,
|
||
"category": p.category,
|
||
"price_level": p.price_level,
|
||
"description": p.description,
|
||
"photos_count": p.photos_count,
|
||
"latitude": p.latitude,
|
||
"longitude": p.longitude,
|
||
"url": p.google_maps_url,
|
||
"permanently_closed": p.permanently_closed,
|
||
"temporarily_closed": p.temporarily_closed,
|
||
}
|
||
except Exception as e:
|
||
print(f" [FAIL] Google: {e}")
|
||
return None
|
||
|
||
|
||
def fetch_website_data(website_url):
|
||
"""Extract basic info from business website."""
|
||
if not website_url:
|
||
return None
|
||
import urllib.request
|
||
try:
|
||
req = urllib.request.Request(website_url, headers={"User-Agent": "Mozilla/5.0"})
|
||
with urllib.request.urlopen(req, timeout=10) as resp:
|
||
html = resp.read().decode("utf-8", errors="replace")
|
||
|
||
data = {"source_url": website_url}
|
||
|
||
# Extract title
|
||
m = re.search(r'<title>([^<]+)</title>', html, re.IGNORECASE)
|
||
if m:
|
||
data["title"] = m.group(1).strip()
|
||
|
||
# Extract meta description
|
||
m = re.search(r'<meta[^>]*name=["\']description["\'][^>]*content=["\']([^"\']+)["\']', html, re.IGNORECASE)
|
||
if not m:
|
||
m = re.search(r'<meta[^>]*content=["\']([^"\']+)["\'][^>]*name=["\']description["\']', html, re.IGNORECASE)
|
||
if m:
|
||
data["description"] = m.group(1).strip()
|
||
|
||
# Extract phone from page
|
||
phone_match = re.search(r'(\(?\d{3}\)?[\s-]?\d{3}[\s-]?\d{4})', html)
|
||
if phone_match:
|
||
data["phone_on_page"] = phone_match.group(1)
|
||
|
||
return data
|
||
|
||
except Exception as e:
|
||
print(f" [FAIL] Website: {e}")
|
||
return None
|
||
|
||
|
||
def merge_and_validate(gbp_data=None, apple_data=None, website_data=None, bing_data=None):
|
||
"""Merge data from all surfaces and flag mismatches."""
|
||
contract = {
|
||
"audit": {
|
||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||
"tool": "multi_scraper_v1",
|
||
"surfaces_checked": {
|
||
"google_business_profile": bool(gbp_data),
|
||
"apple_maps": bool(apple_data),
|
||
"bing_places": bool(bing_data),
|
||
"website": bool(website_data),
|
||
},
|
||
"verification": {},
|
||
},
|
||
"primary_source": "google_business_profile" if gbp_data else None,
|
||
"name": None,
|
||
"address": None,
|
||
"phone": None,
|
||
"website": None,
|
||
"rating": None,
|
||
"reviews": None,
|
||
"reviews_sample": None, # Yelp-sourced via Apple Maps proxy
|
||
"hours": None,
|
||
"category": None,
|
||
"price_level": None,
|
||
"photos_count": None,
|
||
"description": None,
|
||
"coordinates": None,
|
||
"closed_status": None,
|
||
"sources": {
|
||
"google_business_profile": gbp_data,
|
||
"apple_maps": apple_data,
|
||
"bing_places": bing_data,
|
||
"website": website_data,
|
||
},
|
||
}
|
||
|
||
# Cross-source NAP verification
|
||
sources = [("google", gbp_data), ("apple", apple_data)]
|
||
for field in ["name", "phone", "website"]:
|
||
values = {}
|
||
for src_name, src_data in sources:
|
||
if src_data and src_data.get(field):
|
||
values[src_name] = src_data[field]
|
||
if len(values) > 1:
|
||
first_val = list(values.values())[0]
|
||
mismatches = {k: v for k, v in values.items() if v != first_val}
|
||
if mismatches:
|
||
contract["audit"]["verification"][f"{field}_mismatch"] = {
|
||
"primary": first_val,
|
||
"conflicts": mismatches,
|
||
}
|
||
|
||
# Hours cross-reference
|
||
if gbp_data and apple_data:
|
||
gbp_hours = gbp_data.get("hours") or {}
|
||
apple_hours = apple_data.get("hours") or {}
|
||
if gbp_hours and apple_hours:
|
||
hour_mismatches = {}
|
||
for day in gbp_hours:
|
||
if day in apple_hours and gbp_hours[day] != apple_hours[day]:
|
||
hour_mismatches[day] = {
|
||
"google": gbp_hours[day],
|
||
"apple": apple_hours[day],
|
||
}
|
||
if hour_mismatches:
|
||
contract["audit"]["verification"]["hours_mismatch"] = hour_mismatches
|
||
|
||
# Use GBP as primary source — but only if it has real data
|
||
gbp_usable = gbp_data and gbp_data.get("name") and gbp_data.get("name") not in ("Hours", "Open", "Closed", "") and gbp_data.get("address")
|
||
if gbp_usable:
|
||
contract["name"] = gbp_data.get("name")
|
||
contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("address") or "").strip()
|
||
contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("phone") or "").strip()
|
||
contract["website"] = gbp_data.get("website")
|
||
contract["rating"] = gbp_data.get("rating")
|
||
contract["reviews"] = gbp_data.get("reviews")
|
||
contract["reviews_sample"] = apple_data.get("reviews_sample") if apple_data else None
|
||
contract["hours"] = gbp_data.get("hours")
|
||
contract["category"] = gbp_data.get("category")
|
||
contract["price_level"] = gbp_data.get("price_level")
|
||
contract["photos_count"] = gbp_data.get("photos_count")
|
||
contract["description"] = gbp_data.get("description")
|
||
contract["coordinates"] = {
|
||
"lat": gbp_data.get("latitude"),
|
||
"lon": gbp_data.get("longitude"),
|
||
}
|
||
contract["closed_status"] = {
|
||
"permanently_closed": gbp_data.get("permanently_closed"),
|
||
"temporarily_closed": gbp_data.get("temporarily_closed"),
|
||
}
|
||
elif apple_data:
|
||
# ponytail: fallback to Apple if GBP returned None fields
|
||
contract["name"] = apple_data.get("name") or "Unknown"
|
||
contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("address") or "").strip()
|
||
contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("phone") or "").strip()
|
||
contract["website"] = apple_data.get("website") or ""
|
||
contract["rating"] = apple_data.get("rating")
|
||
contract["reviews"] = apple_data.get("reviews")
|
||
contract["hours"] = apple_data.get("hours")
|
||
contract["primary_source"] = "apple_maps"
|
||
|
||
return contract
|
||
|
||
|
||
async def main():
|
||
if len(sys.argv) < 3:
|
||
print("Usage: multi_scraper.py <business_name> <city, state>")
|
||
sys.exit(1)
|
||
|
||
business = sys.argv[1]
|
||
location = sys.argv[2]
|
||
query = f"{business} {location}"
|
||
|
||
print(f"Scraping: {query}")
|
||
print("=" * 60)
|
||
|
||
# Step 1: Geocode
|
||
print("Geocoding location...")
|
||
coords = await geocode(query)
|
||
if coords:
|
||
print(f" Coords: {coords[0]}, {coords[1]}")
|
||
|
||
# Step 2: Scrape GBP (primary)
|
||
print("\nScraping Google Business Profile...")
|
||
gbp_data = await scrape_google(query, coords, business=business)
|
||
if gbp_data:
|
||
print(f" [OK] Google: {gbp_data['name']} ({gbp_data['rating']} ★, {gbp_data['reviews']} reviews)")
|
||
print(f" Hours: {gbp_data.get('hours', {})}")
|
||
|
||
# Step 3: Scrape Apple Maps
|
||
print("\nScraping Apple Maps...")
|
||
apple_data = await scrape_apple(query)
|
||
if apple_data:
|
||
print(f" [OK] Apple: {apple_data.get('name')} ({apple_data.get('rating')} ★, {apple_data.get('reviews')} reviews)")
|
||
if apple_data.get('hours'):
|
||
print(f" Hours: {apple_data.get('hours')}")
|
||
else:
|
||
print(" [FAIL] Apple Maps returned no data")
|
||
|
||
# Step 3b: Scrape Bing Places (via web search)
|
||
print("\nScraping Bing Places...")
|
||
await asyncio.sleep(5) # ponytail: rate limit, separate from Apple Maps browser launch
|
||
bing_data = await scrape_bing(query)
|
||
if bing_data:
|
||
print(f" [OK] Bing: {bing_data.get('name')} ({bing_data.get('rating')} ★, {bing_data.get('reviews')} reviews)")
|
||
if bing_data.get('hours'):
|
||
print(f" Hours: {bing_data.get('hours')}")
|
||
else:
|
||
print(" [FAIL] Bing returned no data")
|
||
|
||
# Step 4: Fetch website
|
||
print("\nFetching website...")
|
||
website_url = gbp_data.get("website") if gbp_data else None
|
||
if not website_url and apple_data:
|
||
website_url = apple_data.get("website")
|
||
website_data = fetch_website_data(website_url)
|
||
if website_data:
|
||
print(f" [OK] Website: {website_data.get('source_url')}")
|
||
else:
|
||
print(" [SKIP] No website URL found")
|
||
|
||
# Step 5: Merge and validate
|
||
print("\nMerging sources...")
|
||
contract = merge_and_validate(gbp_data=gbp_data, apple_data=apple_data, website_data=website_data, bing_data=bing_data)
|
||
|
||
# Step 6: Save
|
||
safe_name = re.sub(r"[^\w\s-]", "", business).strip().replace(" ", "_").lower()
|
||
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||
filename = f"{safe_name}_multi_surface_{date_str}.json"
|
||
# Accept optional output path as 3rd argument; default to script dir
|
||
output_dir = sys.argv[3] if len(sys.argv) > 3 else BASE_DIR
|
||
os.makedirs(output_dir, exist_ok=True)
|
||
filepath = os.path.join(output_dir, filename)
|
||
|
||
with open(filepath, "w") as f:
|
||
json.dump(contract, f, indent=2, ensure_ascii=False)
|
||
|
||
print(f"\nSaved: {filepath}")
|
||
print(f"Primary source: {contract['primary_source']}")
|
||
|
||
# Print verification summary
|
||
if contract["audit"]["verification"]:
|
||
print("\n⚠️ Mismatches detected:")
|
||
for key, val in contract["audit"]["verification"].items():
|
||
print(f" {key}: {json.dumps(val)}")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
asyncio.run(main())
|