Files
veripath/implementation/auditing/multi_scraper.py
T
Leonard 8bb1d7b791 fix(auditing): reject call-tracking placeholders in website phone extraction
_valid_phone() rejects all-same-digit numbers (999-999-9999, 000-000-0000)
that call-tracking widgets emit before they are wired; _phone_from_text now
scans past an invalid first match instead of returning it. Selftest covers
placeholder rejection and visible-text-wins-over-tel: semantics.
2026-08-15 17:54:03 +00:00

422 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
Unified Multi-Surface Scraper for VeriPath Audits.
Extracts GBP + Apple Maps + Website, merges into verified JSON contract.
Working surfaces:
- Google Business Profile (via noworneverev/google-maps-scraper)
- Apple Maps (headless Firefox → /data/search JSON)
- Website (direct HTTP fetch)
Bot-walled (headless detection, no free bypass):
- Bing Places, Yelp
"""
import asyncio
import json
import sys
import re
import os
from datetime import datetime, timezone
# Ensure scraper is importable
sys.path.insert(0, os.path.expanduser("~/deps/google-maps-scraper/src"))
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from gmaps_scraper.scraper import GoogleMapsScraper
from apple_scraper import scrape_apple
from bing_provider import scrape_bing
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
async def geocode(query):
"""Get coordinates from Google Maps search."""
import urllib.request
import urllib.parse
search_url = f"https://www.google.com/maps/search/{urllib.parse.quote(query)}/"
req = urllib.request.Request(search_url, headers={"User-Agent": "Mozilla/5.0"})
try:
with urllib.request.urlopen(req, timeout=10) as resp:
html = resp.read().decode("utf-8", errors="replace")
except Exception:
return None
m = re.search(r'"center":\{"lat":(-?\d+\.\d+),"lng":(-?\d+\.\d+)}', html)
if m:
return float(m.group(1)), float(m.group(2))
m = re.search(r'@(-?\d+\.\d+),(-?\d+\.\d+),\d+z', html)
if m:
return float(m.group(1)), float(m.group(2))
return None
def build_gmaps_url(query, coords):
"""Build Google Maps search URL."""
import urllib.parse
encoded = urllib.parse.quote(query)
if coords:
lat, lon = coords
return f"https://www.google.com/maps/search/{encoded}/@{lat},{lon},14z/data=!3m1!4b1"
return f"https://www.google.com/maps/search/{encoded}/data=!3m1!4b1"
# Known search-page titles (not real places) — same set the lib rejects
_SEARCH_PAGE_TITLES = {"results", "Results", "検索結果", "搜尋結果", "搜索结果"}
def _looks_like_match(place_name, business):
"""Reject search-page titles and names that share no significant token with the query."""
if not place_name:
return False
name = place_name.strip()
if name.lower() in _SEARCH_PAGE_TITLES:
return False
# Significant tokens = words >=4 chars from the business query
biz_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', business)}
name_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', name)}
if not biz_tokens:
return True
# Require at least one shared significant token (e.g. "Crystal", "Plumbing")
return bool(biz_tokens & name_tokens)
async def scrape_google(query, coords, business=""):
"""Scrape Google Business Profile data, validating the matched result."""
url = build_gmaps_url(query, coords)
try:
async with GoogleMapsScraper() as scraper:
result = await scraper.scrape(url)
if result and result.place:
p = result.place
# Validation gate: reject wrong-entity matches before trusting
if not _looks_like_match(p.name, business or query):
print(f" [REJECT] GBP matched wrong entity: '{p.name}' (expected ~'{business}')")
return None
# Hours: ["Thursday9 AM6 PM", "SundayClosed", ...]
hours_dict = {}
if p.hours:
for h in p.hours:
m = re.match(r'(Monday|Tuesday|Wednesday|Thursday|Friday|Saturday|Sunday)(.+)', h, re.IGNORECASE)
if m:
day = m.group(1).lower()
time_str = m.group(2).strip()
time_str = time_str.replace('\u202f', ' ').replace('\u2013', '-').replace('\ue14d', '').strip()
hours_dict[day] = time_str
return {
# Strip PUA glyphs (map-pin \ue0b0 etc.) from address + phone — BMP PUA (U+E000U+F8FF) breaks downstream normalize
"name": p.name,
"address": re.sub(r'[\ue000-\uf8ff]', '', p.address or '').strip(),
"phone": re.sub(r'[\ue000-\uf8ff]', '', p.phone or '').strip() if p.phone else p.phone,
"website": p.website,
"rating": p.rating,
"reviews": p.review_count,
"hours": hours_dict,
"category": p.category,
"price_level": p.price_level,
"description": p.description,
"photos_count": p.photos_count,
"latitude": p.latitude,
"longitude": p.longitude,
"url": p.google_maps_url,
"permanently_closed": p.permanently_closed,
"temporarily_closed": p.temporarily_closed,
}
except Exception as e:
print(f" [FAIL] Google: {e}")
return None
def _valid_phone(num):
"""10 US digits, not an all-same-digit placeholder (999-999-9999, the
value call-tracking widgets emit before they're wired)."""
d = re.sub(r"\D", "", num)
return len(d) == 10 and len(set(d)) > 1
def _phone_from_text(text):
"""First standalone, non-placeholder US phone number in rendered text.
Lookarounds reject a phone-like substring glued to other digits (the
raw-HTML path that previously returned JS artifact numbers);
_valid_phone rejects all-same-digit call-tracking placeholders.
"""
for pat in (r"(?<!\d)(\(\d{3}\)\s*\d{3}[-.]\d{4})(?!\d)",
r"(?<!\d)(\d{3}[-.]\d{3}[-.]\d{4})(?!\d)"):
for mm in re.finditer(pat, text):
if _valid_phone(mm.group(1)):
return mm.group(1)
return None
def _jina_markdown(website_url):
"""Render a URL to clean text via Jina Reader (no key, free tier)."""
import urllib.request
req = urllib.request.Request(
f"https://r.jina.ai/{website_url}",
headers={"User-Agent": "Mozilla/5.0", "Accept": "text/plain"},
)
with urllib.request.urlopen(req, timeout=30) as resp:
return resp.read().decode("utf-8", errors="replace")
def fetch_website_data(website_url):
"""Extract basic info from business website.
Direct fetch for title/description (in <head>, present on JS sites too).
Phone is the NAP-critical field: if the direct HTML has no visible phone,
re-render via Jina Reader, because business sites usually expose the
phone client-side and a false 'no phone' would corrupt the cross-check.
"""
if not website_url:
return None
import urllib.request
html = None
try:
req = urllib.request.Request(website_url, headers={"User-Agent": "Mozilla/5.0"})
with urllib.request.urlopen(req, timeout=10) as resp:
html = resp.read().decode("utf-8", errors="replace")
except Exception as e:
print(f" [WARN] Website direct fetch failed ({e}); using Jina Reader")
data = {"source_url": website_url}
if html:
m = re.search(r"<title>([^<]+)</title>", html, re.IGNORECASE)
if m:
data["title"] = m.group(1).strip()
m = re.search(r'<meta[^>]*name=["\']description["\'][^>]*content=["\']([^"\']+)["\']', html, re.IGNORECASE)
if not m:
m = re.search(r'<meta[^>]*content=["\']([^"\']+)["\'][^>]*name=["\']description["\']', html, re.IGNORECASE)
if m:
data["description"] = m.group(1).strip()
data["phone_on_page"] = _phone_from_text(html)
if data.get("phone_on_page"):
return data
# No phone in direct HTML (or fetch failed) -> render via Jina Reader
try:
md = _jina_markdown(website_url)
except Exception as e:
print(f" [FAIL] Website (direct + Jina): {e}")
return data if html else None
if not data.get("title"):
m = re.search(r"^Title:\s*(.+)$", md, re.MULTILINE)
if m:
data["title"] = m.group(1).strip()
data["phone_on_page"] = _phone_from_text(md)
data["renderer"] = "jina_reader"
return data
def merge_and_validate(gbp_data=None, apple_data=None, website_data=None, bing_data=None):
"""Merge data from all surfaces and flag mismatches."""
contract = {
"audit": {
"timestamp": datetime.now(timezone.utc).isoformat(),
"tool": "multi_scraper_v1",
"surfaces_checked": {
"google_business_profile": bool(gbp_data),
"apple_maps": bool(apple_data),
"bing_places": bool(bing_data),
"website": bool(website_data),
},
"verification": {},
},
"primary_source": "google_business_profile" if gbp_data else None,
"name": None,
"address": None,
"phone": None,
"website": None,
"rating": None,
"reviews": None,
"reviews_sample": None, # Yelp-sourced via Apple Maps proxy
"hours": None,
"category": None,
"price_level": None,
"photos_count": None,
"description": None,
"coordinates": None,
"closed_status": None,
"sources": {
"google_business_profile": gbp_data,
"apple_maps": apple_data,
"bing_places": bing_data,
"website": website_data,
},
}
# Cross-source NAP verification
sources = [("google", gbp_data), ("apple", apple_data)]
for field in ["name", "phone", "website"]:
values = {}
for src_name, src_data in sources:
if src_data and src_data.get(field):
values[src_name] = src_data[field]
if len(values) > 1:
first_val = list(values.values())[0]
mismatches = {k: v for k, v in values.items() if v != first_val}
if mismatches:
contract["audit"]["verification"][f"{field}_mismatch"] = {
"primary": first_val,
"conflicts": mismatches,
}
# Hours cross-reference
if gbp_data and apple_data:
gbp_hours = gbp_data.get("hours") or {}
apple_hours = apple_data.get("hours") or {}
if gbp_hours and apple_hours:
hour_mismatches = {}
for day in gbp_hours:
if day in apple_hours and gbp_hours[day] != apple_hours[day]:
hour_mismatches[day] = {
"google": gbp_hours[day],
"apple": apple_hours[day],
}
if hour_mismatches:
contract["audit"]["verification"]["hours_mismatch"] = hour_mismatches
# Use GBP as primary source — but only if it has real data
gbp_usable = gbp_data and gbp_data.get("name") and gbp_data.get("name") not in ("Hours", "Open", "Closed", "") and gbp_data.get("address")
if gbp_usable:
contract["name"] = gbp_data.get("name")
contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("address") or "").strip()
contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("phone") or "").strip()
contract["website"] = gbp_data.get("website")
contract["rating"] = gbp_data.get("rating")
contract["reviews"] = gbp_data.get("reviews")
contract["reviews_sample"] = apple_data.get("reviews_sample") if apple_data else None
contract["hours"] = gbp_data.get("hours")
contract["category"] = gbp_data.get("category")
contract["price_level"] = gbp_data.get("price_level")
contract["photos_count"] = gbp_data.get("photos_count")
contract["description"] = gbp_data.get("description")
contract["coordinates"] = {
"lat": gbp_data.get("latitude"),
"lon": gbp_data.get("longitude"),
}
contract["closed_status"] = {
"permanently_closed": gbp_data.get("permanently_closed"),
"temporarily_closed": gbp_data.get("temporarily_closed"),
}
elif apple_data:
# ponytail: fallback to Apple if GBP returned None fields
contract["name"] = apple_data.get("name") or "Unknown"
contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("address") or "").strip()
contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("phone") or "").strip()
contract["website"] = apple_data.get("website") or ""
contract["rating"] = apple_data.get("rating")
contract["reviews"] = apple_data.get("reviews")
contract["hours"] = apple_data.get("hours")
contract["primary_source"] = "apple_maps"
return contract
async def main():
if len(sys.argv) < 3:
print("Usage: multi_scraper.py <business_name> <city, state>")
sys.exit(1)
business = sys.argv[1]
location = sys.argv[2]
query = f"{business} {location}"
print(f"Scraping: {query}")
print("=" * 60)
# Step 1: Geocode
print("Geocoding location...")
coords = await geocode(query)
if coords:
print(f" Coords: {coords[0]}, {coords[1]}")
# Step 2: Scrape GBP (primary)
print("\nScraping Google Business Profile...")
gbp_data = await scrape_google(query, coords, business=business)
if gbp_data:
print(f" [OK] Google: {gbp_data['name']} ({gbp_data['rating']} ★, {gbp_data['reviews']} reviews)")
print(f" Hours: {gbp_data.get('hours', {})}")
# Step 3: Scrape Apple Maps
print("\nScraping Apple Maps...")
apple_data = await scrape_apple(query)
if apple_data:
print(f" [OK] Apple: {apple_data.get('name')} ({apple_data.get('rating')} ★, {apple_data.get('reviews')} reviews)")
if apple_data.get('hours'):
print(f" Hours: {apple_data.get('hours')}")
else:
print(" [FAIL] Apple Maps returned no data")
# Step 3b: Scrape Bing Places (via web search)
print("\nScraping Bing Places...")
await asyncio.sleep(5) # ponytail: rate limit, separate from Apple Maps browser launch
bing_data = await scrape_bing(query)
if bing_data:
print(f" [OK] Bing: {bing_data.get('name')} ({bing_data.get('rating')} ★, {bing_data.get('reviews')} reviews)")
if bing_data.get('hours'):
print(f" Hours: {bing_data.get('hours')}")
else:
print(" [FAIL] Bing returned no data")
# Step 4: Fetch website
print("\nFetching website...")
website_url = gbp_data.get("website") if gbp_data else None
if not website_url and apple_data:
website_url = apple_data.get("website")
website_data = fetch_website_data(website_url)
if website_data:
print(f" [OK] Website: {website_data.get('source_url')}")
else:
print(" [SKIP] No website URL found")
# Step 5: Merge and validate
print("\nMerging sources...")
contract = merge_and_validate(gbp_data=gbp_data, apple_data=apple_data, website_data=website_data, bing_data=bing_data)
# Step 6: Save
safe_name = re.sub(r"[^\w\s-]", "", business).strip().replace(" ", "_").lower()
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
filename = f"{safe_name}_multi_surface_{date_str}.json"
# Accept optional output path as 3rd argument; default to script dir
output_dir = sys.argv[3] if len(sys.argv) > 3 else BASE_DIR
os.makedirs(output_dir, exist_ok=True)
filepath = os.path.join(output_dir, filename)
with open(filepath, "w") as f:
json.dump(contract, f, indent=2, ensure_ascii=False)
print(f"\nSaved: {filepath}")
print(f"Primary source: {contract['primary_source']}")
# Print verification summary
if contract["audit"]["verification"]:
print("\n⚠️ Mismatches detected:")
for key, val in contract["audit"]["verification"].items():
print(f" {key}: {json.dumps(val)}")
def _selftest():
"""Runnable check for phone extraction (no network)."""
assert _phone_from_text("Call us: (530) 350-9197 today") == "(530) 350-9197"
assert _phone_from_text("tel: (530) 350-9197") == "(530) 350-9197"
assert _phone_from_text("530-350-9197") == "530-350-9197"
# JS artifact: phone-like substring glued to other digits -> reject
assert _phone_from_text("var id=1234567890123;") is None
assert _phone_from_text("no numbers here") is None
# call-tracking placeholder (all-same digit) -> reject
assert _phone_from_text("Call (999) 999-9999 now") is None
# visible text wins over tel: href (audit semantics: report what's displayed;
# tel: vs visible discrepancies are audit FINDINGS, not extraction noise)
txt = 'Call: (916) 238-3838 <a href="tel:9165716448">call</a>'
assert _phone_from_text(txt) == "(916) 238-3838"
print("[SELFTEST] phone extraction: PASS")
if __name__ == "__main__":
if "--selftest" in sys.argv:
_selftest()
else:
asyncio.run(main())