"""Clickability Index for Athena entries. Read-only against the DB (SELECT only). Computes virality ranking with exponential time-decay so items sink as they age. """ import json import math import os import re import time from datetime import datetime, timezone from typing import Optional from oracle.config import DB_PATH # Virality weights (clickability = how viral/spreadable an item is right now) VEL_W = 0.50 ENG_W = 0.50 SIG_W = 0.0 NOW = None # set in fetch_items for age math # Category-specific half-lives (hours) CATEGORY_HALF_LIVES = { "breaking": 36.0, "update": 24.0, "OTHER": 18.0, } def get_connection(): return __import__("sqlite3").connect(str(DB_PATH)) def _classify(src: str, title: str, summary: str) -> str: t = (title + " " + (summary or "")).lower() if re.search(r"\bshow\s+hn\b", t) or (src == "hackernews" and re.search(r"\b(show|built|made|launched|shipped)\b", t)): return "SHOW_HN" if re.search(r"\b(gpt-|gpt5|gpt-5|deepseek|glm-|llama|qwen|claude|gemini|mistral|flux|stable-diffusion|sora|kimi|grok)\b", t) \ and re.search(r"\b(releases?|released|v\d|launch|unveil|model|new\s+model|update|version)\b", t): return "MODEL_RELEASE" if re.search(r"\b(releases?|released|launches?|unveils?|announces?|debut|new\s+model|gpt-5|deepseek-v|glm-5)\b", t) \ and re.search(r"\b(openai|anthropic|google|meta|microsoft|nvidia|ai)\b", t): return "MODEL_RELEASE" if src == "huggingface": return "MODEL_CARD" if src == "arxiv" or re.search(r"\b(paper|study|benchmark|arxiv|proposes|learns?|novel|framework\s+for|towards)\b", t): return "RESEARCH" if re.search(r"\b(sues|lawsuit|funding|raises|acqui|ipo|valued|stealing|trade secret|layoff|hire[ds]?|exec|ceo)\b", t) \ and not re.search(r"\b(repo|library|tool|agent framework)\b", t): return "BUSINESS_LEGAL" if re.search(r"\b(burnout|opinion|think|feel|why|essay|culture|linkedin|social media|future of|we made|i think|hot take|i believe|my view|in defense)\b", t): return "CULTURE_OPINION" if re.search(r"\b(how to|tutorial|guide|running|build|setup|install|from scratch|learn)\b", t): return "TUTORIAL_HOWTO" if src == "github" or re.search(r"\b(repo|library|framework|tool|agent|sdk|cli|extension|plugin|app|engine)\b", t): return "DEV_TOOL_DRAMA" return "OTHER" def _extract(src: str, md: dict) -> tuple: """Return (velocity_raw, engagement_raw, age_hours).""" if src == "hackernews": pts = md.get("score", 0) or 0 cmts = md.get("descendants", 0) or 0 age_h = None if md.get("time"): try: age_h = max((NOW - md["time"]) / 3600.0, 0.1) except Exception: age_h = None vel = (pts / age_h) if age_h else pts return vel, (pts + 2 * cmts), age_h if src == "reddit": ups = md.get("ups", 0) or 0 cmts = md.get("num_comments", 0) or 0 return ups, (ups + 2 * cmts), None if src == "huggingface": likes = md.get("likes", 0) or 0 return likes, likes, None if src == "github": spd = md.get("stars_per_day", 0) or 0 stars = md.get("stars", 0) or 0 return spd, stars, None if src == "arxiv": return 0.0, 0.0, None return 0.0, 0.0, None def fetch_items(conn) -> list[dict]: """Fetch all entries and compute raw engagement signals.""" global NOW NOW = time.time() cur = conn.cursor() cur.execute(""" SELECT id, title, url, source, summary, signal_score, raw_metadata, first_seen, curated_by, manual_section, manual_tier FROM entries """) cols = [d[0] for d in cur.description] out = [] for row in cur.fetchall(): d = dict(zip(cols, row)) try: md = json.loads(d.get("raw_metadata") or "{}") except Exception: md = {} vel, eng, age = _extract(d["source"], md) ct = _classify(d["source"], d.get("title") or "", d.get("summary") or "") created_at = md.get("createdAt") if d["source"] == "huggingface" else None out.append({ "id": d["id"], "title": d.get("title") or "", "url": d.get("url") or "", "source": d["source"], "summary": d.get("summary") or "", "signal_score": d.get("signal_score") or 0, "velocity_raw": vel, "engagement_raw": eng, "content_type": ct, "first_seen": d.get("first_seen") or "", "created_at": created_at or "", "age_hours": 0.0, "curated_by": d.get("curated_by") or "", "manual_section": d.get("manual_section") or "", "manual_tier": d.get("manual_tier") or "", }) return out def log1p_norm(values: list[float]) -> list[float]: """Log1p + min-max normalization.""" log_vals = [math.log1p(max(v, 0)) for v in values] if not log_vals: return [] min_v, max_v = min(log_vals), max(log_vals) if max_v == min_v: return [0.0] * len(values) return [(v - min_v) / (max_v - min_v) for v in log_vals] def compute_index(items: list[dict]) -> list[dict]: """Compute clickability index for all items.""" velocities = [it.get("velocity_raw", 0) or 0 for it in items] engagements = [it.get("engagement_raw", 0) or 0 for it in items] signals = [it.get("signal_score", 0) or 0 for it in items] vel_norm = log1p_norm(velocities) eng_norm = log1p_norm(engagements) sig_norm = log1p_norm(signals) for i, item in enumerate(items): raw = vel_norm[i] * VEL_W + eng_norm[i] * ENG_W + sig_norm[i] * SIG_W if raw == 0 and sig_norm[i] > 0: raw = 0.05 * sig_norm[i] item["clickability"] = round(raw, 4) item["section"] = "" return items def _age_hours(item: dict) -> float: """Effective news-age in hours.""" if item.get("source") == "huggingface" and item.get("created_at"): s = item["created_at"] else: s = item.get("first_seen") or "" if not s: return 0.0 try: ts = datetime.strptime(s[:19], "%Y-%m-%dT%H:%M:%S").replace( tzinfo=timezone.utc ).timestamp() return max((time.time() - ts) / 3600.0, 0.0) except Exception: return 0.0 def _get_half_life(item: dict) -> Optional[float]: """Return section/tier-specific half-life in hours, or None for default.""" ms = (item.get("manual_section") or "").upper() if ms in ("HARDWARE", "TIPS"): return 336.0 tier = item.get("tier", "normal") if tier == "breaking": return CATEGORY_HALF_LIVES["breaking"] if tier == "update": return CATEGORY_HALF_LIVES["update"] return None def decay_index(items: list[dict], half_life_h: float = 18.0) -> list[dict]: """Apply exponential time-decay to clickability.""" cutoff = datetime.now(timezone.utc).timestamp() - 24 * 3600 for it in items: age = _age_hours(it) it["age_hours"] = round(age, 1) base = it.get("clickability", 0) or 0 hl = _get_half_life(it) if hl is None: hl = half_life_h k = math.log(2) / hl it["clickability_decayed"] = round(base * math.exp(-k * age), 4) it["effective_half_life"] = hl fs = it.get("first_seen") or "" try: ts = datetime.fromisoformat(fs.replace("Z", "+00:00")).timestamp() except ValueError: ts = 0 it["fresh"] = ts >= cutoff return items