diff --git a/implementation/auditing/multi_scraper.py b/implementation/auditing/multi_scraper.py
index e99711b..597f813 100644
--- a/implementation/auditing/multi_scraper.py
+++ b/implementation/auditing/multi_scraper.py
@@ -5,7 +5,7 @@ Extracts GBP + Apple Maps + Website, merges into verified JSON contract.
Working surfaces:
- Google Business Profile (via noworneverev/google-maps-scraper)
- - Apple Maps (headless Firefox + /data/search JSON)
+ - Apple Maps (headless Firefox → /data/search JSON)
- Website (direct HTTP fetch)
Bot-walled (headless detection, no free bypass):
@@ -42,7 +42,7 @@ async def geocode(query):
except Exception:
return None
- m = re.search(r'"center":\{"lat":(-?\d+\.\d+),"lng":(-?\d+\.\d+)\}', html)
+ m = re.search(r'"center":\{"lat":(-?\d+\.\d+),"lng":(-?\d+\.\d+)}', html)
if m:
return float(m.group(1)), float(m.group(2))
@@ -63,15 +63,39 @@ def build_gmaps_url(query, coords):
return f"https://www.google.com/maps/search/{encoded}/data=!3m1!4b1"
-async def scrape_google(query, coords):
- """Scrape Google Business Profile data."""
+# Known search-page titles (not real places) — same set the lib rejects
+_SEARCH_PAGE_TITLES = {"results", "Results", "検索結果", "搜尋結果", "搜索结果"}
+
+
+def _looks_like_match(place_name, business):
+ """Reject search-page titles and names that share no significant token with the query."""
+ if not place_name:
+ return False
+ name = place_name.strip()
+ if name.lower() in _SEARCH_PAGE_TITLES:
+ return False
+ # Significant tokens = words >=4 chars from the business query
+ biz_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', business)}
+ name_tokens = {t.lower() for t in re.findall(r'[A-Za-z]{4,}', name)}
+ if not biz_tokens:
+ return True
+ # Require at least one shared significant token (e.g. "Crystal", "Plumbing")
+ return bool(biz_tokens & name_tokens)
+
+
+async def scrape_google(query, coords, business=""):
+ """Scrape Google Business Profile data, validating the matched result."""
url = build_gmaps_url(query, coords)
try:
async with GoogleMapsScraper() as scraper:
result = await scraper.scrape(url)
if result and result.place:
p = result.place
- # Hours: ["Thursday9 AM٥ PM", "SundayClosed", ...]
+ # Validation gate: reject wrong-entity matches before trusting
+ if not _looks_like_match(p.name, business or query):
+ print(f" [REJECT] GBP matched wrong entity: '{p.name}' (expected ~'{business}')")
+ return None
+ # Hours: ["Thursday9 AM–6 PM", "SundayClosed", ...]
hours_dict = {}
if p.hours:
for h in p.hours:
@@ -83,9 +107,10 @@ async def scrape_google(query, coords):
hours_dict[day] = time_str
return {
+ # Strip PUA glyphs (map-pin \ue0b0 etc.) from address + phone — BMP PUA (U+E000–U+F8FF) breaks downstream normalize
"name": p.name,
- "address": p.address,
- "phone": p.phone,
+ "address": re.sub(r'[\ue000-\uf8ff]', '', p.address or '').strip(),
+ "phone": re.sub(r'[\ue000-\uf8ff]', '', p.phone or '').strip() if p.phone else p.phone,
"website": p.website,
"rating": p.rating,
"reviews": p.review_count,
@@ -123,14 +148,14 @@ def fetch_website_data(website_url):
data["title"] = m.group(1).strip()
# Extract meta description
- m = re.search(r']*content=["\']([^"\']+)["\']', html, re.IGNORECASE)
+ m = re.search(r']*name=["\']description["\'][^>]*content=["\']([^"\']+)["\']', html, re.IGNORECASE)
if not m:
- m = re.search(r']*name=["\']description["\']', html, re.IGNORECASE)
+ m = re.search(r']*content=["\']([^"\']+)["\'][^>]*name=["\']description["\']', html, re.IGNORECASE)
if m:
data["description"] = m.group(1).strip()
# Extract phone from page
- phone_match = re.search(r'(d{3}d{3}d{4})', html)
+ phone_match = re.search(r'(\(?\d{3}\)?[\s-]?\d{3}[\s-]?\d{4})', html)
if phone_match:
data["phone_on_page"] = phone_match.group(1)
@@ -209,11 +234,12 @@ def merge_and_validate(gbp_data=None, apple_data=None, website_data=None, bing_d
if hour_mismatches:
contract["audit"]["verification"]["hours_mismatch"] = hour_mismatches
- # Use GBP as primary source
- if gbp_data:
+ # Use GBP as primary source — but only if it has real data
+ gbp_usable = gbp_data and gbp_data.get("name") and gbp_data.get("name") not in ("Hours", "Open", "Closed", "") and gbp_data.get("address")
+ if gbp_usable:
contract["name"] = gbp_data.get("name")
- contract["address"] = re.sub(r'[\U000E0000-\U000EFFFF]', '', gbp_data.get("address", "")).strip()
- contract["phone"] = re.sub(r'[\U000E0000-\U000EFFFF]', '', gbp_data.get("phone", "")).strip()
+ contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("address") or "").strip()
+ contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', gbp_data.get("phone") or "").strip()
contract["website"] = gbp_data.get("website")
contract["rating"] = gbp_data.get("rating")
contract["reviews"] = gbp_data.get("reviews")
@@ -231,6 +257,16 @@ def merge_and_validate(gbp_data=None, apple_data=None, website_data=None, bing_d
"permanently_closed": gbp_data.get("permanently_closed"),
"temporarily_closed": gbp_data.get("temporarily_closed"),
}
+ elif apple_data:
+ # ponytail: fallback to Apple if GBP returned None fields
+ contract["name"] = apple_data.get("name") or "Unknown"
+ contract["address"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("address") or "").strip()
+ contract["phone"] = re.sub(r'[\ue000-\uf8ff\U000E0000-\U000EFFFF]', '', apple_data.get("phone") or "").strip()
+ contract["website"] = apple_data.get("website") or ""
+ contract["rating"] = apple_data.get("rating")
+ contract["reviews"] = apple_data.get("reviews")
+ contract["hours"] = apple_data.get("hours")
+ contract["primary_source"] = "apple_maps"
return contract
@@ -255,7 +291,7 @@ async def main():
# Step 2: Scrape GBP (primary)
print("\nScraping Google Business Profile...")
- gbp_data = await scrape_google(query, coords)
+ gbp_data = await scrape_google(query, coords, business=business)
if gbp_data:
print(f" [OK] Google: {gbp_data['name']} ({gbp_data['rating']} ★, {gbp_data['reviews']} reviews)")
print(f" Hours: {gbp_data.get('hours', {})}")
@@ -300,7 +336,10 @@ async def main():
safe_name = re.sub(r"[^\w\s-]", "", business).strip().replace(" ", "_").lower()
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
filename = f"{safe_name}_multi_surface_{date_str}.json"
- filepath = os.path.join(BASE_DIR, filename)
+ # Accept optional output path as 3rd argument; default to script dir
+ output_dir = sys.argv[3] if len(sys.argv) > 3 else BASE_DIR
+ os.makedirs(output_dir, exist_ok=True)
+ filepath = os.path.join(output_dir, filename)
with open(filepath, "w") as f:
json.dump(contract, f, indent=2, ensure_ascii=False)