#!/usr/bin/env python3 """ Canonical baseline for VeriPath audits. Owner-verified ground truth lives on Gitea (Tony_tech/veripath, docs/clients//canonical-business-record-vX.Y.Z.json). Only fields with status == "verified" qualify as diff baseline. When no record exists (or the fetch fails), callers fall back to surface-vs-surface comparison. Shared by audit_engine.py, report_generate.py, report_gate.py so all three derive the same reference and the same per-day consistency verdict. """ import json import os import re import urllib.request GITEA_HOST = os.environ.get("GITEA_HOST", "http://localhost:3000") GITEA_REPO = os.environ.get("GITEA_REPO", "Tony_tech/veripath") TOKEN_FILE = os.path.expanduser("~/.hermes/profiles/leonard/.env") _CLOSED = {"closed", "none", "n/a", "", "-", "—"} _DAY_MAP = { "monday": "monday", "tuesday": "tuesday", "wednesday": "wednesday", "thursday": "thursday", "friday": "friday", "saturday": "saturday", "sunday": "sunday", "mon": "monday", "tue": "tuesday", "tues": "tuesday", "wed": "wednesday", "thu": "thursday", "thur": "thursday", "thurs": "thursday", "fri": "friday", "sat": "saturday", "sun": "sunday", } def _gitea_token(): try: for line in open(TOKEN_FILE): if line.startswith("GITEA_ACCESS_TOKEN="): return line.strip().split("=", 1)[1].strip() except OSError: pass return None def _gitea_get(path, token): url = f"{GITEA_HOST}/api/v1/repos/{GITEA_REPO}/contents/{path}" req = urllib.request.Request(url, headers={"Authorization": f"token {token}"}) with urllib.request.urlopen(req, timeout=15) as resp: return json.loads(resp.read().decode("utf-8")) def slugify(name): s = re.sub(r"[^a-zA-Z0-9]+", "-", (name or "").strip()) return s.strip("-").lower() def fetch_canonical_record(business_name): """Fetch the latest canonical record for a client. None if absent. ponytail: prefix-match longest-slug-first against actual client dirs ("phoenix-salon-spa" -> dir "phoenix-salons"); upgrade to explicit record_id lookup when client naming gets fuzzy. """ token = _gitea_token() if not token: return None try: entries = _gitea_get("docs/clients", token) except Exception: return None client_dirs = sorted(e["name"] for e in entries if e.get("type") == "dir") words = slugify(business_name).split("-") for k in range(len(words), 0, -1): cand = "-".join(words[:k]) for d in client_dirs: if not (d == cand or d.startswith(cand)): continue try: files = _gitea_get(f"docs/clients/{d}", token) except Exception: continue names = sorted( (e["name"] for e in files if e["type"] == "file" and re.match(r"canonical-business-record-v[\d.]+\.json$", e["name"])), key=lambda n: [int(x) for x in re.search(r"v([\d.]+)\.json$", n).group(1).split(".")]) if not names: continue body = _gitea_get(f"docs/clients/{d}/{names[-1]}", token) data = json.loads(base64_decode(body["content"])) data["_path"] = f"docs/clients/{d}/{names[-1]}" data["_commit"] = (body.get("last_commit_sha") or "")[:7] return data return None def base64_decode(s): import base64 return base64.b64decode(s).decode("utf-8") def canonical_hours(record): """Return verified hours.regular_hours as {day: "HH:MM-HH:MM"|"closed"}, else None.""" if not record: return None block = (record.get("domains", {}).get("hours", {}) or {}).get("regular_hours") or {} if block.get("status") != "verified": return None value = block.get("value") or {} days = {} for day, val in value.items(): d = _DAY_MAP.get(day.lower()) if not d: continue if isinstance(val, dict): if val.get("open") in (None, "", "null") or str(val.get("open", "")).lower() in _CLOSED: days[d] = "closed" else: days[d] = f"{val['open']}-{val.get('close')}" elif str(val).strip().lower() in _CLOSED: days[d] = "closed" else: norm = parse_hours_value(val) if norm: days[d] = norm return days or None def canonical_provenance(record): """Compact provenance for reports: where the truth came from.""" if not record: return None block = (record.get("domains", {}).get("hours", {}) or {}).get("regular_hours") or {} return { "record": record.get("_path"), "record_id": record.get("record_id"), "verified_at": block.get("verified_at"), "fresh_until": block.get("fresh_until"), "commit": record.get("_commit"), "sources": block.get("sources") or [], } def parse_hours_value(val): """Normalize one day's hours to a canonical string, or 'closed'. Handles: "Closed"/"closed", "9 AM-6 PM", "9 AM - 6 PM", "10:00 AM-6:00 PM", ISO "09:00-18:00". Returns None when unparseable (comparison skips the day). """ if val is None: return None s = str(val).strip() if s.lower() in _CLOSED: return "closed" times = re.split(r"\s*[-–—]\s*", s) if len(times) != 2: return None parts = [] for t in times: m = re.match(r"^(\d{1,2}):?(\d{2})?\s*(am|pm)?$", t.strip().lower().replace(".", "")) if not m: return None hh = int(m.group(1)) mm = m.group(2) or "00" ap = m.group(3) if ap == "pm" and hh != 12: hh += 12 elif ap == "am" and hh == 12: hh = 0 parts.append(f"{hh:02d}:{mm}") return f"{parts[0]}-{parts[1]}" def normalize_hours_dict(hours): """{any-day: any-format} -> {canonical-day: canonical-value|'closed'}. Unparseable days dropped.""" out = {} for day, val in (hours or {}).items(): d = _DAY_MAP.get(str(day).strip().lower()) if not d: continue norm = parse_hours_value(val) if norm is not None: out[d] = norm return out def hours_consistent(a_val, b_val): """True if two already-normalized day values agree (or either is missing).""" if a_val is None or b_val is None: return True # can't compare return a_val == b_val def hours_deviations(reference_hours, surface_hours): """Days where the surface disagrees with the reference. Both dicts normalized.""" ref = normalize_hours_dict(reference_hours) surf = normalize_hours_dict(surface_hours) dev = [] for day in sorted(set(ref) | set(surf)): rv, sv = ref.get(day), surf.get(day) if rv is not None and sv is not None and rv != sv: dev.append({"day": day, "reference": rv, "surface": sv}) return dev def resolve_reference(contract, record=None): """Pick the hours reference for an audit. Returns (label, reference_hours, provenance|None): - canonical record with verified hours -> owner truth - otherwise -> primary source (legacy surface-vs-surface) """ ch = canonical_hours(record) if ch: return ("canonical record (owner-verified ground truth)", ch, canonical_provenance(record)) primary = contract.get("primary_source") or "google_business_profile" ph = (contract.get("sources", {}).get(primary) or {}).get("hours") if ph: label = "Google Business Profile" if primary == "google_business_profile" else primary return (label, ph, None) return (None, None, None) def jsonld_hours(html): """Extract per-day hours from JSON-LD (LocalBusiness.openingHours(Specification)).""" days = {} for m in re.finditer(r"]*type=[\"']application/ld\+json[\"'][^>]*>(.*?)", html, re.DOTALL | re.IGNORECASE): try: data = json.loads(m.group(1).strip()) except (json.JSONDecodeError, ValueError): continue nodes = data if isinstance(data, list) else [data] for node in nodes: if not isinstance(node, dict): continue oh = node.get("openingHours") if isinstance(oh, list): for entry in oh: m = re.match(r"^(Mon|Tue|Wed|Thu|Fri|Sat|Sun)\s*(\d{2}:\d{2})-(\d{2}:\d{2})$", str(entry).strip()) if m: days[_DAY_MAP[m.group(1).lower()]] = f"{m.group(2)}-{m.group(3)}" ohs = node.get("openingHoursSpecification") if isinstance(ohs, list): for spec in ohs: if not isinstance(spec, dict): continue dts = spec.get("dayOfWeek") dts = dts if isinstance(dts, list) else [dts] for dt in dts: d = _DAY_MAP.get(str(dt).strip().lower()) if d and spec.get("opens") and spec.get("closes"): days[d] = f"{spec['opens'][:5]}-{spec['closes'][:5]}" return days if __name__ == "__main__": # Self-check: fails if any of the normalizations above regress assert parse_hours_value("9 AM-6 PM") == "09:00-18:00" assert parse_hours_value("9 AM - 6 PM") == "09:00-18:00" assert parse_hours_value("10:00 AM-6:00 PM") == "10:00-18:00" assert parse_hours_value("Closed") == "closed" assert parse_hours_value("09:00-18:00") == "09:00-18:00" assert parse_hours_value("9am–5pm") is not None assert normalize_hours_dict({"Monday": "9 AM-6 PM", "Sunday": "Closed"}) == {"monday": "09:00-18:00", "sunday": "closed"} ref = {"monday": "09:00-18:00", "sunday": "closed"} dev = hours_deviations(ref, {"monday": "10:00-18:00", "sunday": "10:00-18:00"}) assert [d["day"] for d in dev] == ["monday", "sunday"], dev assert hours_deviations(ref, {"monday": "9 AM - 6 PM"}) == [] # format variance is not a deviation html = '' assert jsonld_hours(html).get("saturday") == "09:00-17:00" print("canonical_baseline: self-check OK")