diff --git a/implementation/auditing/multi_scraper.py b/implementation/auditing/multi_scraper.py
new file mode 100644
index 0000000..f38314b
--- /dev/null
+++ b/implementation/auditing/multi_scraper.py
@@ -0,0 +1,319 @@
+#!/usr/bin/env python3
+"""
+Unified Multi-Surface Scraper for VeriPath Audits.
+Extracts GBP + Apple Maps + Website, merges into verified JSON contract.
+
+Working surfaces:
+ - Google Business Profile (via noworneverev/google-maps-scraper)
+ - Apple Maps (headless Firefox → /data/search JSON)
+ - Website (direct HTTP fetch)
+
+Bot-walled (headless detection, no free bypass):
+ - Bing Places, Yelp
+"""
+import asyncio
+import json
+import sys
+import re
+import os
+from datetime import datetime, timezone
+
+# Ensure scraper is importable
+sys.path.insert(0, "/tmp/google-maps-scraper/src")
+sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
+
+from gmaps_scraper.scraper import GoogleMapsScraper
+from apple_scraper import scrape_apple
+from bing_provider import scrape_bing
+
+BASE_DIR = os.path.dirname(os.path.abspath(__file__))
+
+
+async def geocode(query):
+ """Get coordinates from Google Maps search."""
+ import urllib.request
+ import urllib.parse
+
+ search_url = f"https://www.google.com/maps/search/{urllib.parse.quote(query)}/"
+ req = urllib.request.Request(search_url, headers={"User-Agent": "Mozilla/5.0"})
+ try:
+ with urllib.request.urlopen(req, timeout=10) as resp:
+ html = resp.read().decode("utf-8", errors="replace")
+ except Exception:
+ return None
+
+ m = re.search(r'"center":\{"lat":(-?\d+\.\d+),"lng":(-?\d+\.\d+)}', html)
+ if m:
+ return float(m.group(1)), float(m.group(2))
+
+ m = re.search(r'@(-?\d+\.\d+),(-?\d+\.\d+),\d+z', html)
+ if m:
+ return float(m.group(1)), float(m.group(2))
+
+ return None
+
+
+def build_gmaps_url(query, coords):
+ """Build Google Maps search URL."""
+ import urllib.parse
+ encoded = urllib.parse.quote(query)
+ if coords:
+ lat, lon = coords
+ return f"https://www.google.com/maps/search/{encoded}/@{lat},{lon},14z/data=!3m1!4b1"
+ return f"https://www.google.com/maps/search/{encoded}/data=!3m1!4b1"
+
+
+async def scrape_google(query, coords):
+ """Scrape Google Business Profile data."""
+ url = build_gmaps_url(query, coords)
+ try:
+ async with GoogleMapsScraper() as scraper:
+ result = await scraper.scrape(url)
+ if result and result.place:
+ p = result.place
+ # Hours: ["Thursday9 AM–6 PM", "SundayClosed", ...]
+ hours_dict = {}
+ if p.hours:
+ for h in p.hours:
+ m = re.match(r'(Monday|Tuesday|Wednesday|Thursday|Friday|Saturday|Sunday)(.+)', h, re.IGNORECASE)
+ if m:
+ day = m.group(1).lower()
+ time_str = m.group(2).strip()
+ time_str = time_str.replace('\u202f', ' ').replace('\u2013', '-').replace('\ue14d', '').strip()
+ hours_dict[day] = time_str
+
+ return {
+ "name": p.name,
+ "address": p.address,
+ "phone": p.phone,
+ "website": p.website,
+ "rating": p.rating,
+ "reviews": p.review_count,
+ "hours": hours_dict,
+ "category": p.category,
+ "price_level": p.price_level,
+ "description": p.description,
+ "photos_count": p.photos_count,
+ "latitude": p.latitude,
+ "longitude": p.longitude,
+ "url": p.google_maps_url,
+ "permanently_closed": p.permanently_closed,
+ "temporarily_closed": p.temporarily_closed,
+ }
+ except Exception as e:
+ print(f" [FAIL] Google: {e}")
+ return None
+
+
+def fetch_website_data(website_url):
+ """Extract basic info from business website."""
+ if not website_url:
+ return None
+ import urllib.request
+ try:
+ req = urllib.request.Request(website_url, headers={"User-Agent": "Mozilla/5.0"})
+ with urllib.request.urlopen(req, timeout=10) as resp:
+ html = resp.read().decode("utf-8", errors="replace")
+
+ data = {"source_url": website_url}
+
+ # Extract title
+ m = re.search(r'
([^<]+)', html, re.IGNORECASE)
+ if m:
+ data["title"] = m.group(1).strip()
+
+ # Extract meta description
+ m = re.search(r']*name=["\']description["\'][^>]*content=["\']([^"\']+)["\']', html, re.IGNORECASE)
+ if not m:
+ m = re.search(r']*content=["\']([^"\']+)["\'][^>]*name=["\']description["\']', html, re.IGNORECASE)
+ if m:
+ data["description"] = m.group(1).strip()
+
+ # Extract phone from page
+ phone_match = re.search(r'(\(?\d{3}\)?[\s-]?\d{3}[\s-]?\d{4})', html)
+ if phone_match:
+ data["phone_on_page"] = phone_match.group(1)
+
+ return data
+
+ except Exception as e:
+ print(f" [FAIL] Website: {e}")
+ return None
+
+
+def merge_and_validate(gbp_data=None, apple_data=None, website_data=None, bing_data=None):
+ """Merge data from all surfaces and flag mismatches."""
+ contract = {
+ "audit": {
+ "timestamp": datetime.now(timezone.utc).isoformat(),
+ "tool": "multi_scraper_v1",
+ "surfaces_checked": {
+ "google_business_profile": bool(gbp_data),
+ "apple_maps": bool(apple_data),
+ "bing_places": bool(bing_data),
+ "website": bool(website_data),
+ },
+ "verification": {},
+ },
+ "primary_source": "google_business_profile" if gbp_data else None,
+ "name": None,
+ "address": None,
+ "phone": None,
+ "website": None,
+ "rating": None,
+ "reviews": None,
+ "reviews_sample": None, # Yelp-sourced via Apple Maps proxy
+ "hours": None,
+ "category": None,
+ "price_level": None,
+ "photos_count": None,
+ "description": None,
+ "coordinates": None,
+ "closed_status": None,
+ "sources": {
+ "google_business_profile": gbp_data,
+ "apple_maps": apple_data,
+ "bing_places": bing_data,
+ "website": website_data,
+ },
+ }
+
+ # Cross-source NAP verification
+ sources = [("google", gbp_data), ("apple", apple_data)]
+ for field in ["name", "phone", "website"]:
+ values = {}
+ for src_name, src_data in sources:
+ if src_data and src_data.get(field):
+ values[src_name] = src_data[field]
+ if len(values) > 1:
+ first_val = list(values.values())[0]
+ mismatches = {k: v for k, v in values.items() if v != first_val}
+ if mismatches:
+ contract["audit"]["verification"][f"{field}_mismatch"] = {
+ "primary": first_val,
+ "conflicts": mismatches,
+ }
+
+ # Hours cross-reference
+ if gbp_data and apple_data:
+ gbp_hours = gbp_data.get("hours") or {}
+ apple_hours = apple_data.get("hours") or {}
+ if gbp_hours and apple_hours:
+ hour_mismatches = {}
+ for day in gbp_hours:
+ if day in apple_hours and gbp_hours[day] != apple_hours[day]:
+ hour_mismatches[day] = {
+ "google": gbp_hours[day],
+ "apple": apple_hours[day],
+ }
+ if hour_mismatches:
+ contract["audit"]["verification"]["hours_mismatch"] = hour_mismatches
+
+ # Use GBP as primary source
+ if gbp_data:
+ contract["name"] = gbp_data.get("name")
+ contract["address"] = re.sub(r'[\U000E0000-\U000EFFFF]', '', gbp_data.get("address", "")).strip()
+ contract["phone"] = re.sub(r'[\U000E0000-\U000EFFFF]', '', gbp_data.get("phone", "")).strip()
+ contract["website"] = gbp_data.get("website")
+ contract["rating"] = gbp_data.get("rating")
+ contract["reviews"] = gbp_data.get("reviews")
+ contract["reviews_sample"] = apple_data.get("reviews_sample") if apple_data else None
+ contract["hours"] = gbp_data.get("hours")
+ contract["category"] = gbp_data.get("category")
+ contract["price_level"] = gbp_data.get("price_level")
+ contract["photos_count"] = gbp_data.get("photos_count")
+ contract["description"] = gbp_data.get("description")
+ contract["coordinates"] = {
+ "lat": gbp_data.get("latitude"),
+ "lon": gbp_data.get("longitude"),
+ }
+ contract["closed_status"] = {
+ "permanently_closed": gbp_data.get("permanently_closed"),
+ "temporarily_closed": gbp_data.get("temporarily_closed"),
+ }
+
+ return contract
+
+
+async def main():
+ if len(sys.argv) < 3:
+ print("Usage: multi_scraper.py ")
+ sys.exit(1)
+
+ business = sys.argv[1]
+ location = sys.argv[2]
+ query = f"{business} {location}"
+
+ print(f"Scraping: {query}")
+ print("=" * 60)
+
+ # Step 1: Geocode
+ print("Geocoding location...")
+ coords = await geocode(query)
+ if coords:
+ print(f" Coords: {coords[0]}, {coords[1]}")
+
+ # Step 2: Scrape GBP (primary)
+ print("\nScraping Google Business Profile...")
+ gbp_data = await scrape_google(query, coords)
+ if gbp_data:
+ print(f" [OK] Google: {gbp_data['name']} ({gbp_data['rating']} ★, {gbp_data['reviews']} reviews)")
+ print(f" Hours: {gbp_data.get('hours', {})}")
+
+ # Step 3: Scrape Apple Maps
+ print("\nScraping Apple Maps...")
+ apple_data = await scrape_apple(query)
+ if apple_data:
+ print(f" [OK] Apple: {apple_data.get('name')} ({apple_data.get('rating')} ★, {apple_data.get('reviews')} reviews)")
+ if apple_data.get('hours'):
+ print(f" Hours: {apple_data.get('hours')}")
+ else:
+ print(" [FAIL] Apple Maps returned no data")
+
+ # Step 3b: Scrape Bing Places (via web search)
+ print("\nScraping Bing Places...")
+ await asyncio.sleep(5) # ponytail: rate limit, separate from Apple Maps browser launch
+ bing_data = await scrape_bing(query)
+ if bing_data:
+ print(f" [OK] Bing: {bing_data.get('name')} ({bing_data.get('rating')} ★, {bing_data.get('reviews')} reviews)")
+ if bing_data.get('hours'):
+ print(f" Hours: {bing_data.get('hours')}")
+ else:
+ print(" [FAIL] Bing returned no data")
+
+ # Step 4: Fetch website
+ print("\nFetching website...")
+ website_url = gbp_data.get("website") if gbp_data else None
+ if not website_url and apple_data:
+ website_url = apple_data.get("website")
+ website_data = fetch_website_data(website_url)
+ if website_data:
+ print(f" [OK] Website: {website_data.get('source_url')}")
+ else:
+ print(" [SKIP] No website URL found")
+
+ # Step 5: Merge and validate
+ print("\nMerging sources...")
+ contract = merge_and_validate(gbp_data=gbp_data, apple_data=apple_data, website_data=website_data, bing_data=bing_data)
+
+ # Step 6: Save
+ safe_name = re.sub(r"[^\w\s-]", "", business).strip().replace(" ", "_").lower()
+ date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
+ filename = f"{safe_name}_multi_surface_{date_str}.json"
+ filepath = os.path.join(BASE_DIR, filename)
+
+ with open(filepath, "w") as f:
+ json.dump(contract, f, indent=2, ensure_ascii=False)
+
+ print(f"\nSaved: {filepath}")
+ print(f"Primary source: {contract['primary_source']}")
+
+ # Print verification summary
+ if contract["audit"]["verification"]:
+ print("\n⚠️ Mismatches detected:")
+ for key, val in contract["audit"]["verification"].items():
+ print(f" {key}: {json.dumps(val)}")
+
+
+if __name__ == "__main__":
+ asyncio.run(main())