Files
veripath/implementation/auditing/multi_scraper.py
T

359 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
Unified Multi-Surface Scraper for VeriPath Audits.
Extracts GBP + Apple Maps + Website, merges into verified JSON contract.
Working surfaces:
- Google Business Profile (via noworneverev/google-maps-scraper)
- Apple Maps (headless Firefox → /data/search JSON)
- Website (direct HTTP fetch)
Bot-walled (headless detection, no free bypass):
- Bing Places, Yelp
"""
import asyncio
import json
import sys
import re
import os
from datetime import datetime, timezone
# Ensure scraper is importable
sys.path.insert(0, os.path.expanduser("~/deps/google-maps-scraper/src"))
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from gmaps_scraper.scraper import GoogleMapsScraper
from apple_scraper import scrape_apple
from bing_provider import scrape_bing
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
async def geocode(query):
"""Get coordinates from Google Maps search."""
import urllib.request
import urllib.parse
search_url = f"https://www.google.com/maps/search/{urllib.parse.quote(query)}/"
req = urllib.request.Request(search_url, headers={"User-Agent": "Mozilla/5.0"})
try:
with urllib.request.urlopen(req, timeout=10) as resp:
html = resp.read().decode("utf-8", errors="replace")
except Exception:
return None
m = re.search(r'"center":\{"lat":(-?\d+\.\d+),"lng":(-?\d+\.\d+)}', html)
if m:
return float(m.group(1)), float(m.group(2))
m = re.search(r'@(-?\d+\.\d+),(-?\d+\.\d+),\d+z', html)
if m:
return float(m.group(1)), float(m.group(2))
return None
def build_gmaps_url(query, coords):
"""Build Google Maps search URL."""
import urllib.parse
encoded = urllib.parse.quote(query)
if coords:
lat, lon = coords
return f"https://www.google.com/maps/search/{encoded}/@{lat},{lon},14z/data=!3m1!4b1"
return f"https://www.google.com/maps/search/{encoded}/data=!3m1!4b1"
# Known search-page titles (not real places) — same set the lib rejects
_SEARCH_PAGE_TITLES = {"results", "Results", "検索結果", "搜尋結果", "搜索结果"}
def _looks_like_match(place_name, business):
"""Reject search-page titles and names that share no significant token with the query."""
if not place_name:
return False
name = place_name.strip()
if name.lower() in _SEARCH_PAGE_TITLES:
return False
# Significant tokens = words >=4 chars from the business query
biz_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', business)}
name_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', name)}
if not biz_tokens:
return True
# Require at least one shared significant token (e.g. "Crystal", "Plumbing")
return bool(biz_tokens & name_tokens)
async def scrape_google(query, coords, business=""):
"""Scrape Google Business Profile data, validating the matched result."""
url = build_gmaps_url(query, coords)
try:
async with GoogleMapsScraper() as scraper:
result = await scraper.scrape(url)
if result and result.place:
p = result.place
# Validation gate: reject wrong-entity matches before trusting
if not _looks_like_match(p.name, business or query):
print(f" [REJECT] GBP matched wrong entity: '{p.name}' (expected ~'{business}')")
return None
# Hours: ["Thursday9 AM6 PM", "SundayClosed", ...]
hours_dict = {}
if p.hours:
for h in p.hours:
m = re.match(r'(Monday|Tuesday|Wednesday|Thursday|Friday|Saturday|Sunday)(.+)', h, re.IGNORECASE)
if m:
day = m.group(1).lower()
time_str = m.group(2).strip()
time_str = time_str.replace('\u202f', ' ').replace('\u2013', '-').replace('\ue14d', '').strip()
hours_dict[day] = time_str
return {
# Strip PUA glyphs (map-pin \ue0b0 etc.) from address + phone — BMP PUA (U+E000U+F8FF) breaks downstream normalize
"name": p.name,
"address": re.sub(r'[\ue000-\uf8ff]', '', p.address or '').strip(),
"phone": re.sub(r'[\ue000-\uf8ff]', '', p.phone or '').strip() if p.phone else p.phone,
"website": p.website,
"rating": p.rating,
"reviews": p.review_count,
"hours": hours_dict,
"category": p.category,
"price_level": p.price_level,
"description": p.description,
"photos_count": p.photos_count,
"latitude": p.latitude,
"longitude": p.longitude,
"url": p.google_maps_url,
"permanently_closed": p.permanently_closed,
"temporarily_closed": p.temporarily_closed,
}
except Exception as e:
print(f" [FAIL] Google: {e}")
return None
def fetch_website_data(website_url):
"""Extract basic info from business website."""
if not website_url:
return None
import urllib.request
try:
req = urllib.request.Request(website_url, headers={"User-Agent": "Mozilla/5.0"})
with urllib.request.urlopen(req, timeout=10) as resp:
html = resp.read().decode("utf-8", errors="replace")
data = {"source_url": website_url}
# Extract title
m = re.search(r'<title>([^<]+)</title>', html, re.IGNORECASE)
if m:
data["title"] = m.group(1).strip()
# Extract meta description
m = re.search(r'<meta[^>]*name=["\']description["\'][^>]*content=["\']([^"\']+)["\']', html, re.IGNORECASE)
if not m:
m = re.search(r'<meta[^>]*content=["\']([^"\']+)["\'][^>]*name=["\']description["\']', html, re.IGNORECASE)
if m:
data["description"] = m.group(1).strip()
# Extract phone from page
phone_match = re.search(r'(\(?\d{3}\)?[\s-]?\d{3}[\s-]?\d{4})', html)
if phone_match:
data["phone_on_page"] = phone_match.group(1)
return data
except Exception as e:
print(f" [FAIL] Website: {e}")
return None
def merge_and_validate(gbp_data=None, apple_data=None, website_data=None, bing_data=None):
"""Merge data from all surfaces and flag mismatches."""
contract = {
"audit": {
"timestamp": datetime.now(timezone.utc).isoformat(),
"tool": "multi_scraper_v1",
"surfaces_checked": {
"google_business_profile": bool(gbp_data),
"apple_maps": bool(apple_data),
"bing_places": bool(bing_data),
"website": bool(website_data),
},
"verification": {},
},
"primary_source": "google_business_profile" if gbp_data else None,
"name": None,
"address": None,
"phone": None,
"website": None,
"rating": None,
"reviews": None,
"reviews_sample": None, # Yelp-sourced via Apple Maps proxy
"hours": None,
"category": None,
"price_level": None,
"photos_count": None,
"description": None,
"coordinates": None,
"closed_status": None,
"sources": {
"google_business_profile": gbp_data,
"apple_maps": apple_data,
"bing_places": bing_data,
"website": website_data,
},
}
# Cross-source NAP verification
sources = [("google", gbp_data), ("apple", apple_data)]
for field in ["name", "phone", "website"]:
values = {}
for src_name, src_data in sources:
if src_data and src_data.get(field):
values[src_name] = src_data[field]
if len(values) > 1:
first_val = list(values.values())[0]
mismatches = {k: v for k, v in values.items() if v != first_val}
if mismatches:
contract["audit"]["verification"][f"{field}_mismatch"] = {
"primary": first_val,
"conflicts": mismatches,
}
# Hours cross-reference
if gbp_data and apple_data:
gbp_hours = gbp_data.get("hours") or {}
apple_hours = apple_data.get("hours") or {}
if gbp_hours and apple_hours:
hour_mismatches = {}
for day in gbp_hours:
if day in apple_hours and gbp_hours[day] != apple_hours[day]:
hour_mismatches[day] = {
"google": gbp_hours[day],
"apple": apple_hours[day],
}
if hour_mismatches:
contract["audit"]["verification"]["hours_mismatch"] = hour_mismatches
# Use GBP as primary source — but only if it has real data
gbp_usable = gbp_data and gbp_data.get("name") and gbp_data.get("name") not in ("Hours", "Open", "Closed", "") and gbp_data.get("address")
if gbp_usable:
contract["name"] = gbp_data.get("name")
contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("address") or "").strip()
contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("phone") or "").strip()
contract["website"] = gbp_data.get("website")
contract["rating"] = gbp_data.get("rating")
contract["reviews"] = gbp_data.get("reviews")
contract["reviews_sample"] = apple_data.get("reviews_sample") if apple_data else None
contract["hours"] = gbp_data.get("hours")
contract["category"] = gbp_data.get("category")
contract["price_level"] = gbp_data.get("price_level")
contract["photos_count"] = gbp_data.get("photos_count")
contract["description"] = gbp_data.get("description")
contract["coordinates"] = {
"lat": gbp_data.get("latitude"),
"lon": gbp_data.get("longitude"),
}
contract["closed_status"] = {
"permanently_closed": gbp_data.get("permanently_closed"),
"temporarily_closed": gbp_data.get("temporarily_closed"),
}
elif apple_data:
# ponytail: fallback to Apple if GBP returned None fields
contract["name"] = apple_data.get("name") or "Unknown"
contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("address") or "").strip()
contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("phone") or "").strip()
contract["website"] = apple_data.get("website") or ""
contract["rating"] = apple_data.get("rating")
contract["reviews"] = apple_data.get("reviews")
contract["hours"] = apple_data.get("hours")
contract["primary_source"] = "apple_maps"
return contract
async def main():
if len(sys.argv) < 3:
print("Usage: multi_scraper.py <business_name> <city, state>")
sys.exit(1)
business = sys.argv[1]
location = sys.argv[2]
query = f"{business} {location}"
print(f"Scraping: {query}")
print("=" * 60)
# Step 1: Geocode
print("Geocoding location...")
coords = await geocode(query)
if coords:
print(f" Coords: {coords[0]}, {coords[1]}")
# Step 2: Scrape GBP (primary)
print("\nScraping Google Business Profile...")
gbp_data = await scrape_google(query, coords, business=business)
if gbp_data:
print(f" [OK] Google: {gbp_data['name']} ({gbp_data['rating']} ★, {gbp_data['reviews']} reviews)")
print(f" Hours: {gbp_data.get('hours', {})}")
# Step 3: Scrape Apple Maps
print("\nScraping Apple Maps...")
apple_data = await scrape_apple(query)
if apple_data:
print(f" [OK] Apple: {apple_data.get('name')} ({apple_data.get('rating')} ★, {apple_data.get('reviews')} reviews)")
if apple_data.get('hours'):
print(f" Hours: {apple_data.get('hours')}")
else:
print(" [FAIL] Apple Maps returned no data")
# Step 3b: Scrape Bing Places (via web search)
print("\nScraping Bing Places...")
await asyncio.sleep(5) # ponytail: rate limit, separate from Apple Maps browser launch
bing_data = await scrape_bing(query)
if bing_data:
print(f" [OK] Bing: {bing_data.get('name')} ({bing_data.get('rating')} ★, {bing_data.get('reviews')} reviews)")
if bing_data.get('hours'):
print(f" Hours: {bing_data.get('hours')}")
else:
print(" [FAIL] Bing returned no data")
# Step 4: Fetch website
print("\nFetching website...")
website_url = gbp_data.get("website") if gbp_data else None
if not website_url and apple_data:
website_url = apple_data.get("website")
website_data = fetch_website_data(website_url)
if website_data:
print(f" [OK] Website: {website_data.get('source_url')}")
else:
print(" [SKIP] No website URL found")
# Step 5: Merge and validate
print("\nMerging sources...")
contract = merge_and_validate(gbp_data=gbp_data, apple_data=apple_data, website_data=website_data, bing_data=bing_data)
# Step 6: Save
safe_name = re.sub(r"[^\w\s-]", "", business).strip().replace(" ", "_").lower()
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
filename = f"{safe_name}_multi_surface_{date_str}.json"
# Accept optional output path as 3rd argument; default to script dir
output_dir = sys.argv[3] if len(sys.argv) > 3 else BASE_DIR
os.makedirs(output_dir, exist_ok=True)
filepath = os.path.join(output_dir, filename)
with open(filepath, "w") as f:
json.dump(contract, f, indent=2, ensure_ascii=False)
print(f"\nSaved: {filepath}")
print(f"Primary source: {contract['primary_source']}")
# Print verification summary
if contract["audit"]["verification"]:
print("\n⚠️ Mismatches detected:")
for key, val in contract["audit"]["verification"].items():
print(f" {key}: {json.dumps(val)}")
if __name__ == "__main__":
asyncio.run(main())