diff --git a/implementation/auditing/audit_engine.py b/implementation/auditing/audit_engine.py
index 351317e..3629e89 100644
--- a/implementation/auditing/audit_engine.py
+++ b/implementation/auditing/audit_engine.py
@@ -63,14 +63,12 @@ def _normalize_phone(phone):
def _normalize_address(addr):
- """Lowercase, strip extra whitespace, remove suite abbrev variances, drop commas."""
+ """Lowercase, strip extra whitespace, remove suite abbrev variances."""
if not addr:
return ""
s = re.sub(r'[\U000E0000-\U000EFFFF]', '', addr or '')
s = re.sub(r'\s+', ' ', s.strip().lower())
s = re.sub(r'\bst[e]*\.?\b', 'ste', s)
- s = re.sub(r'\bsuite\b', 'ste', s)
- s = s.replace(',', '') # ponytail: comma is pure formatting, not a material diff
return s
@@ -215,37 +213,57 @@ def check_nap_website(contract):
def check_hours_mismatch(contract):
- """Hours consistency. GBP is authoritative."""
- findings = []
- gbp_hours = _safe_get(contract, "sources", "google_business_profile", "hours", default={})
- apple_hours = _safe_get(contract, "sources", "apple_maps", "hours", default={})
- bing_hours = _safe_get(contract, "sources", "bing_places", "hours", default={})
+ """Hours consistency vs reference: canonical record (owner-verified truth)
+ if the client has one, else GBP (legacy surface-vs-surface).
- if not gbp_hours:
+ Evidence shape: {reference, reference_hours, provenance, deviations[]}
+ where each deviation day carries the RAW value of every captured surface.
+ """
+ findings = []
+ from canonical_baseline import resolve_reference, normalize_hours_dict, parse_hours_value
+
+ label, ref_hours, prov = resolve_reference(contract, contract.get("_canonical_record"))
+ ref_norm = normalize_hours_dict(ref_hours or {})
+ if not ref_norm:
return findings # nothing to compare against
- mismatched_days = []
- for other_name, other_hours in [("apple_maps", apple_hours), ("bing_places", bing_hours)]:
- if not other_hours:
+ sources = contract.get("sources", {})
+ SURF_KEYS = ("google_business_profile", "apple_maps", "bing_places", "website")
+ deviations = []
+ for day, ref_val in sorted(ref_norm.items()):
+ raw_vals = {}
+ for sname in SURF_KEYS:
+ v = (sources.get(sname) or {}).get("hours") or {}
+ if v.get(day) is not None:
+ raw_vals[sname] = v[day]
+ if not raw_vals:
continue
- for day, gbp_val in gbp_hours.items():
- if day in other_hours:
- other_val = other_hours[day]
- if not _hours_overlap(gbp_val, other_val):
- mismatched_days.append({
- "day": day,
- "google": gbp_val,
- other_name: other_val
- })
+ # ponytail: skip the reference surface itself in legacy mode (it can't deviate from itself anyway)
+ bad = any(
+ parse_hours_value(v) is not None and parse_hours_value(v) != ref_val
+ for v in raw_vals.values()
+ )
+ if bad:
+ entry = {"day": day, "reference": ref_val}
+ entry.update(raw_vals)
+ deviations.append(entry)
- if mismatched_days:
+ if deviations:
+ suffix = "owner-verified hours" if prov else "GBP is authoritative"
findings.append({
"id": "hours_mismatch",
"severity": SEVERITY_IMMEDIATE,
- "title": f"Hours mismatch on {len(mismatched_days)} day(s) — GBP is authoritative",
+ "title": f"Hours mismatch on {len(deviations)} day(s) — {suffix}",
"sub_score": GOOGLE_BUSINESS,
- "evidence": mismatched_days,
- "recommendation": "Correct hours on non-GBP surfaces to match GBP. Mismatched hours cause customer complaints and trust loss."
+ "evidence": {
+ "reference": label,
+ "reference_hours": ref_norm,
+ "provenance": prov,
+ "deviations": deviations,
+ },
+ "recommendation": ("Correct hours on the deviating surfaces to match the owner-verified schedule (canonical record)."
+ if prov else
+ "Correct hours on non-GBP surfaces to match GBP. Mismatched hours cause customer complaints and trust loss."),
})
return findings
@@ -321,6 +339,76 @@ def check_review_count_delta(contract):
return findings
+def check_onpage_title_meta(contract):
+ """On-page title/meta audit (technical SEO): presence, length, name, phone.
+
+ Reads the website fields captured at scrape time (title, description,
+ phone_on_page) so the audit is reproducible from the capture. Skips
+ cleanly when no website surface was captured."""
+ findings = []
+ ws = _safe_get(contract, "sources", "website", default={}) or {}
+ title = ws.get("title") or ""
+ desc = ws.get("description") or ""
+ if not title and not desc:
+ return findings # no website surface captured
+
+ name = (contract.get("name") or "").lower()
+ phone_norm = _normalize_phone(contract.get("phone") or "")
+ page_phone_norm = _normalize_phone(ws.get("phone_on_page") or "")
+
+ #
presence & length (Google truncates ~60 chars)
+ if not title:
+ findings.append({
+ "id": "missing_title", "severity": SEVERITY_HIGH, "sub_score": TECHNICAL_SEO,
+ "title": "Website has no tag — search engines fall back to the URL.",
+ "evidence": {"title": ""},
+ "recommendation": "Add a descriptive tag including the business name and main service.",
+ })
+ else:
+ if len(title) > 65:
+ findings.append({
+ "id": "title_too_long", "severity": SEVERITY_ENHANCEMENT, "sub_score": TECHNICAL_SEO,
+ "title": f"Title tag is {len(title)} chars (>65) — truncated in Google results.",
+ "evidence": {"title": title, "length": len(title)},
+ "recommendation": "Shorten the to ~50-60 characters.",
+ })
+ if name and name not in title.lower():
+ findings.append({
+ "id": "title_missing_name", "severity": SEVERITY_MEDIUM, "sub_score": TECHNICAL_SEO,
+ "title": "Title tag does not contain the business name.",
+ "evidence": {"title": title, "business_name": name},
+ "recommendation": "Include the business name in the for brand recognition.",
+ })
+
+ # Meta description presence & length (Google truncates ~160 chars)
+ if not desc:
+ findings.append({
+ "id": "missing_meta_description", "severity": SEVERITY_MEDIUM, "sub_score": TECHNICAL_SEO,
+ "title": "No meta description — Google may use arbitrary page text as the snippet.",
+ "evidence": {"description": ""},
+ "recommendation": "Write a 150-160 character meta description naming the business and service.",
+ })
+ elif len(desc) > 170:
+ findings.append({
+ "id": "meta_description_too_long", "severity": SEVERITY_ENHANCEMENT, "sub_score": TECHNICAL_SEO,
+ "title": f"Meta description is {len(desc)} chars (>170) — truncated in results.",
+ "evidence": {"length": len(desc)},
+ "recommendation": "Shorten the meta description to ~150-160 characters.",
+ })
+
+ # On-page NAP: flag only when a phone WAS captured and it differs from the
+ # business phone (evidence-based; no capture = can't conclude, don't flag).
+ if phone_norm and page_phone_norm and phone_norm != page_phone_norm:
+ findings.append({
+ "id": "phone_not_on_page", "severity": SEVERITY_MEDIUM, "sub_score": DIGITAL_IDENTITY,
+ "title": "Phone number found on the website differs from the business phone (on-page NAP gap).",
+ "evidence": {"business_phone": phone_norm, "phone_on_page": page_phone_norm},
+ "recommendation": "Make the website phone match the business phone in the header/footer.",
+ })
+
+ return findings
+
+
def check_review_velocity(contract):
"""Review recency — are recent reviews flowing?"""
findings = []
@@ -333,7 +421,7 @@ def check_review_velocity(contract):
if not dates:
findings.append({
"id": "review_dates_unavailable",
- "severity": SEVERITY_MEDIUM,
+ "severity": SEVERITY_ENHANCEMENT, # tooling limitation, not a business finding (matches classify() FP in report_generate.py)
"title": "Cannot determine review recency — no dates available in samples",
"sub_score": REVIEWS,
"evidence": {"sources_with_dates": []},
@@ -364,62 +452,78 @@ def check_review_velocity(contract):
return findings
+def _service_keywords(contract):
+ """Derive the business's own service/category vocabulary (vertical-agnostic).
+
+ Grounded in what the business actually sells — its category, description,
+ and name — so the specificity signal adapts to any vertical instead of
+ assuming one (the old version hardcoded salon terms).
+ """
+ words = set()
+ # Top-level capture fields are the most reliable vocabulary source
+ for field in ("name", "category", "description"):
+ val = contract.get(field) or ""
+ for w in re.findall(r"[a-z]{3,}", val.lower()):
+ words.add(w)
+ for src in ["google_business_profile", "apple_maps", "bing_places", "website"]:
+ data = _safe_get(contract, "sources", src, default={}) or {}
+ for field in ("category", "description", "name", "title"):
+ val = data.get(field) or ""
+ for w in re.findall(r"[a-z]{3,}", val.lower()):
+ words.add(w)
+ # Drop generic words that don't indicate a specific service/product
+ stop = {"the", "and", "for", "with", "this", "that", "your", "you", "our",
+ "are", "was", "have", "has", "been", "will", "would", "like", "get",
+ "got", "good", "great", "best", "very", "really", "service",
+ "services", "business", "place", "store", "shop", "location", "area",
+ "city", "staff", "customer", "customers", "experience", "quality",
+ "professional", "professionals", "helpful", "friendly", "recommend",
+ "recommended", "price", "prices", "paid", "cost", "costs", "time",
+ "times", "day", "days", "week", "weeks", "month", "months", "year",
+ "years", "online", "presence", "audit", "veripath"}
+ return words - stop
+
+
def check_review_text_quality(contract):
- """Check if reviews mention specific services/practitioners (E-E-A-T signal)."""
+ """Check if reviews mention the business's own services/products (E-E-A-T signal).
+
+ Vertical-agnostic: the keyword set is derived from the business's own
+ category/description/name, not a hardcoded vertical.
+ """
findings = []
all_samples = []
for src in ["apple_maps", "bing_places"]:
samples = _safe_get(contract, "sources", src, "reviews_sample", default=[])
all_samples.extend(samples or [])
- if not all_samples:
+ if len(all_samples) < 3:
return findings
+ keywords = _service_keywords(contract)
+ if not keywords:
+ return findings # no vocabulary to match against; can't judge specificity
+
specific_signals = 0
for r in all_samples:
text = (r.get('text') or '').lower()
- if any(word in text for word in ['ashley', 'grace', 'mallory', 'facial', 'hair', 'waxing', 'lashes', 'spray tan', 'massage']):
+ if any(w in text for w in keywords):
specific_signals += 1
- ratio = specific_signals / len(all_samples) if all_samples else 0
- if ratio < 0.5 and len(all_samples) >= 3:
+ ratio = specific_signals / len(all_samples)
+ if ratio < 0.5:
findings.append({
"id": "reviews_low_specificity",
"severity": SEVERITY_ENHANCEMENT,
- "title": f"Only {ratio:.0%} of sample reviews mention specific services or practitioners",
+ "title": f"Only {ratio:.0%} of sample reviews mention specific services or staff",
"sub_score": REVIEWS,
- "evidence": {"sample_size": len(all_samples), "specific_count": specific_signals},
- "recommendation": "Encourage detailed reviews mentioning services and staff. Specific reviews rank higher and convert better."
+ "evidence": {"sample_size": len(all_samples), "specific_count": specific_signals,
+ "keyword_basis": "category+description+name"},
+ "recommendation": "Encourage reviews that name specific services, products, or staff. Specific reviews rank higher and convert better."
})
return findings
-# ponytail: small explicit synonym map; extend when new variants appear in the wild
-_CATEGORY_SYNONYMS = {
- "beauty salon": "beauty salon",
- "beauty and spa": "beauty salon",
- "beauty & spa": "beauty salon",
- "beauty & spa": "beauty salon",
- "salon and spa": "beauty salon",
- "hair salon": "hair salon",
- "hair & beauty salon": "hair salon",
- "hair and beauty salon": "hair salon",
- "day spa": "day spa",
- "spa": "spa",
- "medical spa": "medical spa",
- "barber shop": "barber shop",
- "barbershop": "barber shop",
-}
-
-
-def _normalize_category(cat):
- if not cat:
- return ""
- c = re.sub(r'\s+', ' ', cat.strip().lower()).replace('&', '&')
- return _CATEGORY_SYNONYMS.get(c, c)
-
-
def check_category_alignment(contract):
"""Category consistency and fragmentation across surfaces."""
findings = []
@@ -430,7 +534,7 @@ def check_category_alignment(contract):
data = _safe_get(contract, "sources", src_data, default={})
cat = data.get('category')
if cat:
- categories[src_name] = _normalize_category(cat)
+ categories[src_name] = cat.lower().strip()
if len(categories) < 2:
return findings
@@ -567,13 +671,32 @@ def check_website_schema(contract):
})
return findings
- # Fetch and check for JSON-LD
- try:
- import urllib.request
- req = urllib.request.Request(website_url, headers={"User-Agent": "Mozilla/5.0"})
- with urllib.request.urlopen(req, timeout=10) as resp:
- html = resp.read().decode("utf-8", errors="replace")
-
+ # Prefer capture-time evidence (reproducible audit). Live refetch only for
+ # pre-fix captures that lack the fields.
+ ws = _safe_get(contract, "sources", "website", default={}) or {}
+ if "has_jsonld" in ws:
+ has_jsonld = bool(ws.get("has_jsonld"))
+ types = [t.lower() for t in ws.get("schema_types") or []]
+ has_local_business = any(t in ("localbusiness", "beautysalon", "dayspa", "healthandbeautybusiness") for t in types)
+ has_canonical = bool(ws.get("canonical"))
+ has_og = bool(ws.get("has_og"))
+ else:
+ # Fetch and check for JSON-LD (legacy captures)
+ try:
+ import urllib.request
+ req = urllib.request.Request(website_url, headers={"User-Agent": "Mozilla/5.0"})
+ with urllib.request.urlopen(req, timeout=10) as resp:
+ html = resp.read().decode("utf-8", errors="replace")
+ except Exception as e:
+ findings.append({
+ "id": "website_unreachable",
+ "severity": SEVERITY_HIGH,
+ "title": f"Website fetch failed: {e}",
+ "sub_score": TECHNICAL_SEO,
+ "evidence": {"url": website_url, "error": str(e)},
+ "recommendation": "Verify website is live and accessible. A broken site kills all local signals."
+ })
+ return findings
# Check for JSON-LD
has_jsonld = bool(re.search(r'