audit pipeline v1.2: canonical business record as diff baseline (owner-verified truth), website JSON-LD hours, single-command e2e with gate
This commit is contained in:
@@ -63,14 +63,12 @@ def _normalize_phone(phone):
|
||||
|
||||
|
||||
def _normalize_address(addr):
|
||||
"""Lowercase, strip extra whitespace, remove suite abbrev variances, drop commas."""
|
||||
"""Lowercase, strip extra whitespace, remove suite abbrev variances."""
|
||||
if not addr:
|
||||
return ""
|
||||
s = re.sub(r'[\U000E0000-\U000EFFFF]', '', addr or '')
|
||||
s = re.sub(r'\s+', ' ', s.strip().lower())
|
||||
s = re.sub(r'\bst[e]*\.?\b', 'ste', s)
|
||||
s = re.sub(r'\bsuite\b', 'ste', s)
|
||||
s = s.replace(',', '') # ponytail: comma is pure formatting, not a material diff
|
||||
return s
|
||||
|
||||
|
||||
@@ -215,37 +213,57 @@ def check_nap_website(contract):
|
||||
|
||||
|
||||
def check_hours_mismatch(contract):
|
||||
"""Hours consistency. GBP is authoritative."""
|
||||
findings = []
|
||||
gbp_hours = _safe_get(contract, "sources", "google_business_profile", "hours", default={})
|
||||
apple_hours = _safe_get(contract, "sources", "apple_maps", "hours", default={})
|
||||
bing_hours = _safe_get(contract, "sources", "bing_places", "hours", default={})
|
||||
"""Hours consistency vs reference: canonical record (owner-verified truth)
|
||||
if the client has one, else GBP (legacy surface-vs-surface).
|
||||
|
||||
if not gbp_hours:
|
||||
Evidence shape: {reference, reference_hours, provenance, deviations[]}
|
||||
where each deviation day carries the RAW value of every captured surface.
|
||||
"""
|
||||
findings = []
|
||||
from canonical_baseline import resolve_reference, normalize_hours_dict, parse_hours_value
|
||||
|
||||
label, ref_hours, prov = resolve_reference(contract, contract.get("_canonical_record"))
|
||||
ref_norm = normalize_hours_dict(ref_hours or {})
|
||||
if not ref_norm:
|
||||
return findings # nothing to compare against
|
||||
|
||||
mismatched_days = []
|
||||
for other_name, other_hours in [("apple_maps", apple_hours), ("bing_places", bing_hours)]:
|
||||
if not other_hours:
|
||||
sources = contract.get("sources", {})
|
||||
SURF_KEYS = ("google_business_profile", "apple_maps", "bing_places", "website")
|
||||
deviations = []
|
||||
for day, ref_val in sorted(ref_norm.items()):
|
||||
raw_vals = {}
|
||||
for sname in SURF_KEYS:
|
||||
v = (sources.get(sname) or {}).get("hours") or {}
|
||||
if v.get(day) is not None:
|
||||
raw_vals[sname] = v[day]
|
||||
if not raw_vals:
|
||||
continue
|
||||
for day, gbp_val in gbp_hours.items():
|
||||
if day in other_hours:
|
||||
other_val = other_hours[day]
|
||||
if not _hours_overlap(gbp_val, other_val):
|
||||
mismatched_days.append({
|
||||
"day": day,
|
||||
"google": gbp_val,
|
||||
other_name: other_val
|
||||
})
|
||||
# ponytail: skip the reference surface itself in legacy mode (it can't deviate from itself anyway)
|
||||
bad = any(
|
||||
parse_hours_value(v) is not None and parse_hours_value(v) != ref_val
|
||||
for v in raw_vals.values()
|
||||
)
|
||||
if bad:
|
||||
entry = {"day": day, "reference": ref_val}
|
||||
entry.update(raw_vals)
|
||||
deviations.append(entry)
|
||||
|
||||
if mismatched_days:
|
||||
if deviations:
|
||||
suffix = "owner-verified hours" if prov else "GBP is authoritative"
|
||||
findings.append({
|
||||
"id": "hours_mismatch",
|
||||
"severity": SEVERITY_IMMEDIATE,
|
||||
"title": f"Hours mismatch on {len(mismatched_days)} day(s) — GBP is authoritative",
|
||||
"title": f"Hours mismatch on {len(deviations)} day(s) — {suffix}",
|
||||
"sub_score": GOOGLE_BUSINESS,
|
||||
"evidence": mismatched_days,
|
||||
"recommendation": "Correct hours on non-GBP surfaces to match GBP. Mismatched hours cause customer complaints and trust loss."
|
||||
"evidence": {
|
||||
"reference": label,
|
||||
"reference_hours": ref_norm,
|
||||
"provenance": prov,
|
||||
"deviations": deviations,
|
||||
},
|
||||
"recommendation": ("Correct hours on the deviating surfaces to match the owner-verified schedule (canonical record)."
|
||||
if prov else
|
||||
"Correct hours on non-GBP surfaces to match GBP. Mismatched hours cause customer complaints and trust loss."),
|
||||
})
|
||||
|
||||
return findings
|
||||
@@ -321,6 +339,76 @@ def check_review_count_delta(contract):
|
||||
return findings
|
||||
|
||||
|
||||
def check_onpage_title_meta(contract):
|
||||
"""On-page title/meta audit (technical SEO): presence, length, name, phone.
|
||||
|
||||
Reads the website fields captured at scrape time (title, description,
|
||||
phone_on_page) so the audit is reproducible from the capture. Skips
|
||||
cleanly when no website surface was captured."""
|
||||
findings = []
|
||||
ws = _safe_get(contract, "sources", "website", default={}) or {}
|
||||
title = ws.get("title") or ""
|
||||
desc = ws.get("description") or ""
|
||||
if not title and not desc:
|
||||
return findings # no website surface captured
|
||||
|
||||
name = (contract.get("name") or "").lower()
|
||||
phone_norm = _normalize_phone(contract.get("phone") or "")
|
||||
page_phone_norm = _normalize_phone(ws.get("phone_on_page") or "")
|
||||
|
||||
# <title> presence & length (Google truncates ~60 chars)
|
||||
if not title:
|
||||
findings.append({
|
||||
"id": "missing_title", "severity": SEVERITY_HIGH, "sub_score": TECHNICAL_SEO,
|
||||
"title": "Website has no <title> tag — search engines fall back to the URL.",
|
||||
"evidence": {"title": ""},
|
||||
"recommendation": "Add a descriptive <title> tag including the business name and main service.",
|
||||
})
|
||||
else:
|
||||
if len(title) > 65:
|
||||
findings.append({
|
||||
"id": "title_too_long", "severity": SEVERITY_ENHANCEMENT, "sub_score": TECHNICAL_SEO,
|
||||
"title": f"Title tag is {len(title)} chars (>65) — truncated in Google results.",
|
||||
"evidence": {"title": title, "length": len(title)},
|
||||
"recommendation": "Shorten the <title> to ~50-60 characters.",
|
||||
})
|
||||
if name and name not in title.lower():
|
||||
findings.append({
|
||||
"id": "title_missing_name", "severity": SEVERITY_MEDIUM, "sub_score": TECHNICAL_SEO,
|
||||
"title": "Title tag does not contain the business name.",
|
||||
"evidence": {"title": title, "business_name": name},
|
||||
"recommendation": "Include the business name in the <title> for brand recognition.",
|
||||
})
|
||||
|
||||
# Meta description presence & length (Google truncates ~160 chars)
|
||||
if not desc:
|
||||
findings.append({
|
||||
"id": "missing_meta_description", "severity": SEVERITY_MEDIUM, "sub_score": TECHNICAL_SEO,
|
||||
"title": "No meta description — Google may use arbitrary page text as the snippet.",
|
||||
"evidence": {"description": ""},
|
||||
"recommendation": "Write a 150-160 character meta description naming the business and service.",
|
||||
})
|
||||
elif len(desc) > 170:
|
||||
findings.append({
|
||||
"id": "meta_description_too_long", "severity": SEVERITY_ENHANCEMENT, "sub_score": TECHNICAL_SEO,
|
||||
"title": f"Meta description is {len(desc)} chars (>170) — truncated in results.",
|
||||
"evidence": {"length": len(desc)},
|
||||
"recommendation": "Shorten the meta description to ~150-160 characters.",
|
||||
})
|
||||
|
||||
# On-page NAP: flag only when a phone WAS captured and it differs from the
|
||||
# business phone (evidence-based; no capture = can't conclude, don't flag).
|
||||
if phone_norm and page_phone_norm and phone_norm != page_phone_norm:
|
||||
findings.append({
|
||||
"id": "phone_not_on_page", "severity": SEVERITY_MEDIUM, "sub_score": DIGITAL_IDENTITY,
|
||||
"title": "Phone number found on the website differs from the business phone (on-page NAP gap).",
|
||||
"evidence": {"business_phone": phone_norm, "phone_on_page": page_phone_norm},
|
||||
"recommendation": "Make the website phone match the business phone in the header/footer.",
|
||||
})
|
||||
|
||||
return findings
|
||||
|
||||
|
||||
def check_review_velocity(contract):
|
||||
"""Review recency — are recent reviews flowing?"""
|
||||
findings = []
|
||||
@@ -333,7 +421,7 @@ def check_review_velocity(contract):
|
||||
if not dates:
|
||||
findings.append({
|
||||
"id": "review_dates_unavailable",
|
||||
"severity": SEVERITY_MEDIUM,
|
||||
"severity": SEVERITY_ENHANCEMENT, # tooling limitation, not a business finding (matches classify() FP in report_generate.py)
|
||||
"title": "Cannot determine review recency — no dates available in samples",
|
||||
"sub_score": REVIEWS,
|
||||
"evidence": {"sources_with_dates": []},
|
||||
@@ -364,62 +452,78 @@ def check_review_velocity(contract):
|
||||
return findings
|
||||
|
||||
|
||||
def _service_keywords(contract):
|
||||
"""Derive the business's own service/category vocabulary (vertical-agnostic).
|
||||
|
||||
Grounded in what the business actually sells — its category, description,
|
||||
and name — so the specificity signal adapts to any vertical instead of
|
||||
assuming one (the old version hardcoded salon terms).
|
||||
"""
|
||||
words = set()
|
||||
# Top-level capture fields are the most reliable vocabulary source
|
||||
for field in ("name", "category", "description"):
|
||||
val = contract.get(field) or ""
|
||||
for w in re.findall(r"[a-z]{3,}", val.lower()):
|
||||
words.add(w)
|
||||
for src in ["google_business_profile", "apple_maps", "bing_places", "website"]:
|
||||
data = _safe_get(contract, "sources", src, default={}) or {}
|
||||
for field in ("category", "description", "name", "title"):
|
||||
val = data.get(field) or ""
|
||||
for w in re.findall(r"[a-z]{3,}", val.lower()):
|
||||
words.add(w)
|
||||
# Drop generic words that don't indicate a specific service/product
|
||||
stop = {"the", "and", "for", "with", "this", "that", "your", "you", "our",
|
||||
"are", "was", "have", "has", "been", "will", "would", "like", "get",
|
||||
"got", "good", "great", "best", "very", "really", "service",
|
||||
"services", "business", "place", "store", "shop", "location", "area",
|
||||
"city", "staff", "customer", "customers", "experience", "quality",
|
||||
"professional", "professionals", "helpful", "friendly", "recommend",
|
||||
"recommended", "price", "prices", "paid", "cost", "costs", "time",
|
||||
"times", "day", "days", "week", "weeks", "month", "months", "year",
|
||||
"years", "online", "presence", "audit", "veripath"}
|
||||
return words - stop
|
||||
|
||||
|
||||
def check_review_text_quality(contract):
|
||||
"""Check if reviews mention specific services/practitioners (E-E-A-T signal)."""
|
||||
"""Check if reviews mention the business's own services/products (E-E-A-T signal).
|
||||
|
||||
Vertical-agnostic: the keyword set is derived from the business's own
|
||||
category/description/name, not a hardcoded vertical.
|
||||
"""
|
||||
findings = []
|
||||
all_samples = []
|
||||
for src in ["apple_maps", "bing_places"]:
|
||||
samples = _safe_get(contract, "sources", src, "reviews_sample", default=[])
|
||||
all_samples.extend(samples or [])
|
||||
|
||||
if not all_samples:
|
||||
if len(all_samples) < 3:
|
||||
return findings
|
||||
|
||||
keywords = _service_keywords(contract)
|
||||
if not keywords:
|
||||
return findings # no vocabulary to match against; can't judge specificity
|
||||
|
||||
specific_signals = 0
|
||||
for r in all_samples:
|
||||
text = (r.get('text') or '').lower()
|
||||
if any(word in text for word in ['ashley', 'grace', 'mallory', 'facial', 'hair', 'waxing', 'lashes', 'spray tan', 'massage']):
|
||||
if any(w in text for w in keywords):
|
||||
specific_signals += 1
|
||||
|
||||
ratio = specific_signals / len(all_samples) if all_samples else 0
|
||||
if ratio < 0.5 and len(all_samples) >= 3:
|
||||
ratio = specific_signals / len(all_samples)
|
||||
if ratio < 0.5:
|
||||
findings.append({
|
||||
"id": "reviews_low_specificity",
|
||||
"severity": SEVERITY_ENHANCEMENT,
|
||||
"title": f"Only {ratio:.0%} of sample reviews mention specific services or practitioners",
|
||||
"title": f"Only {ratio:.0%} of sample reviews mention specific services or staff",
|
||||
"sub_score": REVIEWS,
|
||||
"evidence": {"sample_size": len(all_samples), "specific_count": specific_signals},
|
||||
"recommendation": "Encourage detailed reviews mentioning services and staff. Specific reviews rank higher and convert better."
|
||||
"evidence": {"sample_size": len(all_samples), "specific_count": specific_signals,
|
||||
"keyword_basis": "category+description+name"},
|
||||
"recommendation": "Encourage reviews that name specific services, products, or staff. Specific reviews rank higher and convert better."
|
||||
})
|
||||
|
||||
return findings
|
||||
|
||||
|
||||
# ponytail: small explicit synonym map; extend when new variants appear in the wild
|
||||
_CATEGORY_SYNONYMS = {
|
||||
"beauty salon": "beauty salon",
|
||||
"beauty and spa": "beauty salon",
|
||||
"beauty & spa": "beauty salon",
|
||||
"beauty & spa": "beauty salon",
|
||||
"salon and spa": "beauty salon",
|
||||
"hair salon": "hair salon",
|
||||
"hair & beauty salon": "hair salon",
|
||||
"hair and beauty salon": "hair salon",
|
||||
"day spa": "day spa",
|
||||
"spa": "spa",
|
||||
"medical spa": "medical spa",
|
||||
"barber shop": "barber shop",
|
||||
"barbershop": "barber shop",
|
||||
}
|
||||
|
||||
|
||||
def _normalize_category(cat):
|
||||
if not cat:
|
||||
return ""
|
||||
c = re.sub(r'\s+', ' ', cat.strip().lower()).replace('&', '&')
|
||||
return _CATEGORY_SYNONYMS.get(c, c)
|
||||
|
||||
|
||||
def check_category_alignment(contract):
|
||||
"""Category consistency and fragmentation across surfaces."""
|
||||
findings = []
|
||||
@@ -430,7 +534,7 @@ def check_category_alignment(contract):
|
||||
data = _safe_get(contract, "sources", src_data, default={})
|
||||
cat = data.get('category')
|
||||
if cat:
|
||||
categories[src_name] = _normalize_category(cat)
|
||||
categories[src_name] = cat.lower().strip()
|
||||
|
||||
if len(categories) < 2:
|
||||
return findings
|
||||
@@ -567,13 +671,32 @@ def check_website_schema(contract):
|
||||
})
|
||||
return findings
|
||||
|
||||
# Fetch and check for JSON-LD
|
||||
try:
|
||||
import urllib.request
|
||||
req = urllib.request.Request(website_url, headers={"User-Agent": "Mozilla/5.0"})
|
||||
with urllib.request.urlopen(req, timeout=10) as resp:
|
||||
html = resp.read().decode("utf-8", errors="replace")
|
||||
|
||||
# Prefer capture-time evidence (reproducible audit). Live refetch only for
|
||||
# pre-fix captures that lack the fields.
|
||||
ws = _safe_get(contract, "sources", "website", default={}) or {}
|
||||
if "has_jsonld" in ws:
|
||||
has_jsonld = bool(ws.get("has_jsonld"))
|
||||
types = [t.lower() for t in ws.get("schema_types") or []]
|
||||
has_local_business = any(t in ("localbusiness", "beautysalon", "dayspa", "healthandbeautybusiness") for t in types)
|
||||
has_canonical = bool(ws.get("canonical"))
|
||||
has_og = bool(ws.get("has_og"))
|
||||
else:
|
||||
# Fetch and check for JSON-LD (legacy captures)
|
||||
try:
|
||||
import urllib.request
|
||||
req = urllib.request.Request(website_url, headers={"User-Agent": "Mozilla/5.0"})
|
||||
with urllib.request.urlopen(req, timeout=10) as resp:
|
||||
html = resp.read().decode("utf-8", errors="replace")
|
||||
except Exception as e:
|
||||
findings.append({
|
||||
"id": "website_unreachable",
|
||||
"severity": SEVERITY_HIGH,
|
||||
"title": f"Website fetch failed: {e}",
|
||||
"sub_score": TECHNICAL_SEO,
|
||||
"evidence": {"url": website_url, "error": str(e)},
|
||||
"recommendation": "Verify website is live and accessible. A broken site kills all local signals."
|
||||
})
|
||||
return findings
|
||||
# Check for JSON-LD
|
||||
has_jsonld = bool(re.search(r'<script[^>]*type=["\']application/ld\+json["\']', html, re.IGNORECASE))
|
||||
has_local_business = bool(re.search(r'"@type"\s*:\s*"(?:LocalBusiness|BeautySalon|DaySpa|HealthAndBeautyBusiness)', html, re.IGNORECASE))
|
||||
@@ -584,53 +707,43 @@ def check_website_schema(contract):
|
||||
# Check for OG tags
|
||||
has_og = bool(re.search(r'<meta[^>]*property=["\']og:', html, re.IGNORECASE))
|
||||
|
||||
if not has_jsonld:
|
||||
findings.append({
|
||||
"id": "missing_jsonld_schema",
|
||||
"severity": SEVERITY_HIGH,
|
||||
"title": "No JSON-LD schema found on website",
|
||||
"sub_score": TECHNICAL_SEO,
|
||||
"evidence": {"url": website_url, "has_jsonld": False},
|
||||
"recommendation": "Add LocalBusiness/BeautySalon JSON-LD schema. Critical for rich results and local ranking."
|
||||
})
|
||||
elif not has_local_business:
|
||||
findings.append({
|
||||
"id": "schema_wrong_type",
|
||||
"severity": SEVERITY_MEDIUM,
|
||||
"title": "JSON-LD present but no LocalBusiness/BeautySalon type",
|
||||
"sub_score": TECHNICAL_SEO,
|
||||
"evidence": {"url": website_url, "has_jsonld": True, "has_local_business": False},
|
||||
"recommendation": "Ensure schema includes LocalBusiness or BeautySalon @type for local search relevance."
|
||||
})
|
||||
|
||||
if not has_canonical:
|
||||
findings.append({
|
||||
"id": "missing_canonical",
|
||||
"severity": SEVERITY_MEDIUM,
|
||||
"title": "No canonical URL tag on website",
|
||||
"sub_score": TECHNICAL_SEO,
|
||||
"evidence": {"url": website_url},
|
||||
"recommendation": "Add <link rel='canonical' href='...'> to prevent duplicate content issues."
|
||||
})
|
||||
|
||||
if not has_og:
|
||||
findings.append({
|
||||
"id": "missing_og_tags",
|
||||
"severity": SEVERITY_ENHANCEMENT,
|
||||
"title": "No Open Graph tags on website",
|
||||
"sub_score": TECHNICAL_SEO,
|
||||
"evidence": {"url": website_url},
|
||||
"recommendation": "Add OG tags for proper link preview on social media and messaging apps."
|
||||
})
|
||||
|
||||
except Exception as e:
|
||||
if not has_jsonld:
|
||||
findings.append({
|
||||
"id": "website_unreachable",
|
||||
"id": "missing_jsonld_schema",
|
||||
"severity": SEVERITY_HIGH,
|
||||
"title": f"Website fetch failed: {e}",
|
||||
"title": "No JSON-LD schema found on website",
|
||||
"sub_score": TECHNICAL_SEO,
|
||||
"evidence": {"url": website_url, "error": str(e)},
|
||||
"recommendation": "Verify website is live and accessible. A broken site kills all local signals."
|
||||
"evidence": {"url": website_url, "has_jsonld": False},
|
||||
"recommendation": "Add LocalBusiness/BeautySalon JSON-LD schema. Critical for rich results and local ranking."
|
||||
})
|
||||
elif not has_local_business:
|
||||
findings.append({
|
||||
"id": "schema_wrong_type",
|
||||
"severity": SEVERITY_MEDIUM,
|
||||
"title": "JSON-LD present but no LocalBusiness/BeautySalon type",
|
||||
"sub_score": TECHNICAL_SEO,
|
||||
"evidence": {"url": website_url, "has_jsonld": True, "has_local_business": False},
|
||||
"recommendation": "Ensure schema includes LocalBusiness or BeautySalon @type for local search relevance."
|
||||
})
|
||||
|
||||
if not has_canonical:
|
||||
findings.append({
|
||||
"id": "missing_canonical",
|
||||
"severity": SEVERITY_MEDIUM,
|
||||
"title": "No canonical URL tag on website",
|
||||
"sub_score": TECHNICAL_SEO,
|
||||
"evidence": {"url": website_url},
|
||||
"recommendation": "Add <link rel='canonical' href='...'> to prevent duplicate content issues."
|
||||
})
|
||||
|
||||
if not has_og:
|
||||
findings.append({
|
||||
"id": "missing_og_tags",
|
||||
"severity": SEVERITY_ENHANCEMENT,
|
||||
"title": "No Open Graph tags on website",
|
||||
"sub_score": TECHNICAL_SEO,
|
||||
"evidence": {"url": website_url},
|
||||
"recommendation": "Add OG tags for proper link preview on social media and messaging apps."
|
||||
})
|
||||
|
||||
return findings
|
||||
@@ -693,6 +806,7 @@ ALL_CHECKS = [
|
||||
check_price_level,
|
||||
check_photos,
|
||||
check_website_schema,
|
||||
check_onpage_title_meta,
|
||||
check_google_maps_url,
|
||||
check_coordinates_precision,
|
||||
]
|
||||
@@ -830,6 +944,18 @@ def main():
|
||||
with open(filepath) as f:
|
||||
contract = json.load(f)
|
||||
|
||||
# Canonical baseline: owner-verified record from Gitea (None -> legacy GBP reference)
|
||||
try:
|
||||
from canonical_baseline import fetch_canonical_record, canonical_hours
|
||||
record = fetch_canonical_record(contract.get("name") or "")
|
||||
if record and canonical_hours(record):
|
||||
contract["_canonical_record"] = record
|
||||
print(f" [OK] Canonical baseline: {record['_path']} @ {record['_commit']}", file=sys.stderr)
|
||||
else:
|
||||
print(" [INFO] No verified canonical record — legacy surface-vs-surface reference", file=sys.stderr)
|
||||
except Exception as e:
|
||||
print(f" [WARN] canonical fetch failed ({e}) — legacy reference", file=sys.stderr)
|
||||
|
||||
findings = analyze(contract)
|
||||
summary = summarize(findings, contract)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user