#!/usr/bin/env python3 """ Unified Multi-Surface Scraper for VeriPath Audits. Extracts GBP + Apple Maps + Website, merges into verified JSON contract. Working surfaces: - Google Business Profile (via noworneverev/google-maps-scraper) - Apple Maps (headless Firefox → /data/search JSON) - Website (direct HTTP fetch) Bot-walled (headless detection, no free bypass): - Bing Places, Yelp """ import asyncio import json import sys import re import os from datetime import datetime, timezone # Ensure scraper is importable sys.path.insert(0, "/tmp/google-maps-scraper/src") sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from gmaps_scraper.scraper import GoogleMapsScraper from apple_scraper import scrape_apple from bing_provider import scrape_bing BASE_DIR = os.path.dirname(os.path.abspath(__file__)) async def geocode(query): """Get coordinates from Google Maps search.""" import urllib.request import urllib.parse search_url = f"https://www.google.com/maps/search/{urllib.parse.quote(query)}/" req = urllib.request.Request(search_url, headers={"User-Agent": "Mozilla/5.0"}) try: with urllib.request.urlopen(req, timeout=10) as resp: html = resp.read().decode("utf-8", errors="replace") except Exception: return None m = re.search(r'"center":\{"lat":(-?\d+\.\d+),"lng":(-?\d+\.\d+)}', html) if m: return float(m.group(1)), float(m.group(2)) m = re.search(r'@(-?\d+\.\d+),(-?\d+\.\d+),\d+z', html) if m: return float(m.group(1)), float(m.group(2)) return None def build_gmaps_url(query, coords): """Build Google Maps search URL.""" import urllib.parse encoded = urllib.parse.quote(query) if coords: lat, lon = coords return f"https://www.google.com/maps/search/{encoded}/@{lat},{lon},14z/data=!3m1!4b1" return f"https://www.google.com/maps/search/{encoded}/data=!3m1!4b1" # Known search-page titles (not real places) — same set the lib rejects _SEARCH_PAGE_TITLES = {"results", "Results", "検索結果", "搜尋結果", "搜索结果"} def _looks_like_match(place_name, business): """Reject search-page titles and names that share no significant token with the query.""" if not place_name: return False name = place_name.strip() if name.lower() in _SEARCH_PAGE_TITLES: return False # Significant tokens = words >=4 chars from the business query biz_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', business)} name_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', name)} if not biz_tokens: return True # Require at least one shared significant token (e.g. "Crystal", "Plumbing") return bool(biz_tokens & name_tokens) async def scrape_google(query, coords, business=""): """Scrape Google Business Profile data, validating the matched result.""" url = build_gmaps_url(query, coords) try: async with GoogleMapsScraper() as scraper: result = await scraper.scrape(url) if result and result.place: p = result.place # Validation gate: reject wrong-entity matches before trusting if not _looks_like_match(p.name, business or query): print(f" [REJECT] GBP matched wrong entity: '{p.name}' (expected ~'{business}')") return None # Hours: ["Thursday9 AM–6 PM", "SundayClosed", ...] hours_dict = {} if p.hours: for h in p.hours: m = re.match(r'(Monday|Tuesday|Wednesday|Thursday|Friday|Saturday|Sunday)(.+)', h, re.IGNORECASE) if m: day = m.group(1).lower() time_str = m.group(2).strip() time_str = time_str.replace('\u202f', ' ').replace('\u2013', '-').replace('\ue14d', '').strip() hours_dict[day] = time_str return { # Strip PUA glyphs (map-pin \ue0b0 etc.) from address + phone — BMP PUA (U+E000–U+F8FF) breaks downstream normalize "name": p.name, "address": re.sub(r'[\ue000-\uf8ff]', '', p.address or '').strip(), "phone": re.sub(r'[\ue000-\uf8ff]', '', p.phone or '').strip() if p.phone else p.phone, "website": p.website, "rating": p.rating, "reviews": p.review_count, "hours": hours_dict, "category": p.category, "price_level": p.price_level, "description": p.description, "photos_count": p.photos_count, "latitude": p.latitude, "longitude": p.longitude, "url": p.google_maps_url, "permanently_closed": p.permanently_closed, "temporarily_closed": p.temporarily_closed, } except Exception as e: print(f" [FAIL] Google: {e}") return None def fetch_website_data(website_url): """Extract basic info from business website.""" if not website_url: return None import urllib.request try: req = urllib.request.Request(website_url, headers={"User-Agent": "Mozilla/5.0"}) with urllib.request.urlopen(req, timeout=10) as resp: html = resp.read().decode("utf-8", errors="replace") data = {"source_url": website_url} # Extract title m = re.search(r'([^<]+)', html, re.IGNORECASE) if m: data["title"] = m.group(1).strip() # Extract meta description m = re.search(r']*name=["\']description["\'][^>]*content=["\']([^"\']+)["\']', html, re.IGNORECASE) if not m: m = re.search(r']*content=["\']([^"\']+)["\'][^>]*name=["\']description["\']', html, re.IGNORECASE) if m: data["description"] = m.group(1).strip() # Extract phone from page phone_match = re.search(r'(\(?\d{3}\)?[\s-]?\d{3}[\s-]?\d{4})', html) if phone_match: data["phone_on_page"] = phone_match.group(1) return data except Exception as e: print(f" [FAIL] Website: {e}") return None def merge_and_validate(gbp_data=None, apple_data=None, website_data=None, bing_data=None): """Merge data from all surfaces and flag mismatches.""" contract = { "audit": { "timestamp": datetime.now(timezone.utc).isoformat(), "tool": "multi_scraper_v1", "surfaces_checked": { "google_business_profile": bool(gbp_data), "apple_maps": bool(apple_data), "bing_places": bool(bing_data), "website": bool(website_data), }, "verification": {}, }, "primary_source": "google_business_profile" if gbp_data else None, "name": None, "address": None, "phone": None, "website": None, "rating": None, "reviews": None, "reviews_sample": None, # Yelp-sourced via Apple Maps proxy "hours": None, "category": None, "price_level": None, "photos_count": None, "description": None, "coordinates": None, "closed_status": None, "sources": { "google_business_profile": gbp_data, "apple_maps": apple_data, "bing_places": bing_data, "website": website_data, }, } # Cross-source NAP verification sources = [("google", gbp_data), ("apple", apple_data)] for field in ["name", "phone", "website"]: values = {} for src_name, src_data in sources: if src_data and src_data.get(field): values[src_name] = src_data[field] if len(values) > 1: first_val = list(values.values())[0] mismatches = {k: v for k, v in values.items() if v != first_val} if mismatches: contract["audit"]["verification"][f"{field}_mismatch"] = { "primary": first_val, "conflicts": mismatches, } # Hours cross-reference if gbp_data and apple_data: gbp_hours = gbp_data.get("hours") or {} apple_hours = apple_data.get("hours") or {} if gbp_hours and apple_hours: hour_mismatches = {} for day in gbp_hours: if day in apple_hours and gbp_hours[day] != apple_hours[day]: hour_mismatches[day] = { "google": gbp_hours[day], "apple": apple_hours[day], } if hour_mismatches: contract["audit"]["verification"]["hours_mismatch"] = hour_mismatches # Use GBP as primary source — but only if it has real data gbp_usable = gbp_data and gbp_data.get("name") and gbp_data.get("name") not in ("Hours", "Open", "Closed", "") and gbp_data.get("address") if gbp_usable: contract["name"] = gbp_data.get("name") contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("address") or "").strip() contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("phone") or "").strip() contract["website"] = gbp_data.get("website") contract["rating"] = gbp_data.get("rating") contract["reviews"] = gbp_data.get("reviews") contract["reviews_sample"] = apple_data.get("reviews_sample") if apple_data else None contract["hours"] = gbp_data.get("hours") contract["category"] = gbp_data.get("category") contract["price_level"] = gbp_data.get("price_level") contract["photos_count"] = gbp_data.get("photos_count") contract["description"] = gbp_data.get("description") contract["coordinates"] = { "lat": gbp_data.get("latitude"), "lon": gbp_data.get("longitude"), } contract["closed_status"] = { "permanently_closed": gbp_data.get("permanently_closed"), "temporarily_closed": gbp_data.get("temporarily_closed"), } elif apple_data: # ponytail: fallback to Apple if GBP returned None fields contract["name"] = apple_data.get("name") or "Unknown" contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("address") or "").strip() contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("phone") or "").strip() contract["website"] = apple_data.get("website") or "" contract["rating"] = apple_data.get("rating") contract["reviews"] = apple_data.get("reviews") contract["hours"] = apple_data.get("hours") contract["primary_source"] = "apple_maps" return contract async def main(): if len(sys.argv) < 3: print("Usage: multi_scraper.py ") sys.exit(1) business = sys.argv[1] location = sys.argv[2] query = f"{business} {location}" print(f"Scraping: {query}") print("=" * 60) # Step 1: Geocode print("Geocoding location...") coords = await geocode(query) if coords: print(f" Coords: {coords[0]}, {coords[1]}") # Step 2: Scrape GBP (primary) print("\nScraping Google Business Profile...") gbp_data = await scrape_google(query, coords, business=business) if gbp_data: print(f" [OK] Google: {gbp_data['name']} ({gbp_data['rating']} ★, {gbp_data['reviews']} reviews)") print(f" Hours: {gbp_data.get('hours', {})}") # Step 3: Scrape Apple Maps print("\nScraping Apple Maps...") apple_data = await scrape_apple(query) if apple_data: print(f" [OK] Apple: {apple_data.get('name')} ({apple_data.get('rating')} ★, {apple_data.get('reviews')} reviews)") if apple_data.get('hours'): print(f" Hours: {apple_data.get('hours')}") else: print(" [FAIL] Apple Maps returned no data") # Step 3b: Scrape Bing Places (via web search) print("\nScraping Bing Places...") await asyncio.sleep(5) # ponytail: rate limit, separate from Apple Maps browser launch bing_data = await scrape_bing(query) if bing_data: print(f" [OK] Bing: {bing_data.get('name')} ({bing_data.get('rating')} ★, {bing_data.get('reviews')} reviews)") if bing_data.get('hours'): print(f" Hours: {bing_data.get('hours')}") else: print(" [FAIL] Bing returned no data") # Step 4: Fetch website print("\nFetching website...") website_url = gbp_data.get("website") if gbp_data else None if not website_url and apple_data: website_url = apple_data.get("website") website_data = fetch_website_data(website_url) if website_data: print(f" [OK] Website: {website_data.get('source_url')}") else: print(" [SKIP] No website URL found") # Step 5: Merge and validate print("\nMerging sources...") contract = merge_and_validate(gbp_data=gbp_data, apple_data=apple_data, website_data=website_data, bing_data=bing_data) # Step 6: Save safe_name = re.sub(r"[^\w\s-]", "", business).strip().replace(" ", "_").lower() date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d") filename = f"{safe_name}_multi_surface_{date_str}.json" # Accept optional output path as 3rd argument; default to script dir output_dir = sys.argv[3] if len(sys.argv) > 3 else BASE_DIR os.makedirs(output_dir, exist_ok=True) filepath = os.path.join(output_dir, filename) with open(filepath, "w") as f: json.dump(contract, f, indent=2, ensure_ascii=False) print(f"\nSaved: {filepath}") print(f"Primary source: {contract['primary_source']}") # Print verification summary if contract["audit"]["verification"]: print("\n⚠️ Mismatches detected:") for key, val in contract["audit"]["verification"].items(): print(f" {key}: {json.dumps(val)}") if __name__ == "__main__": asyncio.run(main())