Sprint 0+1: Package restructure, source tiers, verdicts, multi-variant editions
- New oracle/ package (11 modules) with unified CLI (python -m oracle) - Source tiers: Tier 1 (arxiv/github/hf), Tier 2 (rss/hn), Tier 3 (reddit) - Composite verdicts: PUBLISH/WATCH/ARCHIVE/DROP based on signal score + age - Content-hash dedup: SHA-256[:16] normalized, atomic at insert time - Multi-variant editions: 4 YAML configs (default/research/devops/brief) - Variant engine: filter → rank → render (HTML + JSON, themed) - Per-adapter timeout (10s) + threading fallback - Consolidated 12 root scripts → thin wrappers + oracle/ package - Archived stale scripts (_engagement, _live_compare, reddit_proof) - Updated .gitignore, README.md, schema.sql
This commit is contained in:
+21
-193
@@ -1,201 +1,29 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
recency_guard.py — Athena "how old is this news?" gate (Tony's correction, 2026-07-13).
|
||||
"""Thin wrapper — delegates to oracle.recency filter_fresh + analysis."""
|
||||
import sys, os
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
|
||||
THE PROBLEM IT SOLVES:
|
||||
The old pipeline ranked by virality + editorial fit ONLY. That surfaced the
|
||||
SAME stories week after week (Apple-vs-OpenAI, GPT-5.6, etc.) because the
|
||||
curation had NO memory of what was already posted. Tony's rule:
|
||||
|
||||
"when you receive the news stories, morality [editorial fit] is only ONE
|
||||
measure. The OTHER measure -- maybe MORE important -- is how OLD is the
|
||||
news. Today is Monday; the week's news is just beginning, so we begin the
|
||||
week with stories dated for TODAY."
|
||||
|
||||
So AGE is the dominant gate. The algorithm (editorial, not render-time):
|
||||
|
||||
* TODAY's items -> ALWAYS eligible. They are this week's fresh news; a
|
||||
Monday stack leads with them even if rendered earlier today.
|
||||
* OLDER items -> eligible ONLY if NEVER posted before (not in the
|
||||
markdown Top-N stack history AND not in seen_urls.json).
|
||||
This kills the re-post problem at the source.
|
||||
|
||||
WHAT IS "ALREADY POSTED":
|
||||
Two signals, differing in authority:
|
||||
- athena_top*.md = the CURATED/PUBLIC stack history (authoritative)
|
||||
- seen_urls.json = the live-site render dedup (secondary; ALSO flags items
|
||||
rendered in prior runs TODAY, which we must NOT drop)
|
||||
Because seen_urls.json contains today's own items, we only treat a seen_url
|
||||
as "already posted" when the candidate is OLDER than today. Today's items are
|
||||
exempt from the seen_urls gate entirely (fresh by definition).
|
||||
|
||||
EXPORTS:
|
||||
load_posted() -> (md_urls, md_titles, seen_urls) sets
|
||||
is_today(first_seen, now)
|
||||
age_days(first_seen, now)
|
||||
day_bucket(first_seen, now)
|
||||
already_posted_fs(url,title,fs,now) -> bool (age-aware dedup)
|
||||
filter_fresh(items, now) -> (today_items, older_new_items, dropped_items)
|
||||
recency_weight(first_seen, now) -> float (1.0 today -> ~0 over 7d)
|
||||
blend_score(item, now) -> clickability_decayed * recency_weight
|
||||
Read-only against markdown + json + passed-in items. No DB writes.
|
||||
"""
|
||||
import os, re, json
|
||||
from datetime import datetime, timezone
|
||||
from oracle.clickability import fetch_items, compute_index, decay_index
|
||||
from oracle.recency import filter_fresh, blend_score
|
||||
import sqlite3
|
||||
from oracle.config import DB_PATH, HALF_LIFE_H
|
||||
|
||||
ORACLE = os.path.dirname(os.path.abspath(__file__))
|
||||
SEEN_JSON = "/home/vpsadmin/ai-oracle-site/seen_urls.json"
|
||||
conn = sqlite3.connect(str(DB_PATH))
|
||||
items = fetch_items(conn)
|
||||
conn.close()
|
||||
|
||||
items = compute_index(items)
|
||||
items = decay_index(items, HALF_LIFE_H)
|
||||
|
||||
def _parse(ts):
|
||||
if not ts:
|
||||
return None
|
||||
try:
|
||||
return datetime.fromisoformat(ts.replace("Z", "+00:00"))
|
||||
except Exception:
|
||||
return None
|
||||
today, older_new, dropped = filter_fresh(items)
|
||||
|
||||
print(f"=== Recency Guard ===")
|
||||
print(f" Total items: {len(items)}")
|
||||
print(f" TODAY (always eligible): {len(today)}")
|
||||
print(f" OLDER but never posted: {len(older_new)}")
|
||||
print(f" DROPPED (posted + old): {len(dropped)}")
|
||||
print()
|
||||
|
||||
def _norm_url(u):
|
||||
if not u:
|
||||
return ""
|
||||
return u.split("?")[0].split("#")[0].rstrip("/").lower()
|
||||
|
||||
|
||||
def _norm_title(t):
|
||||
if not t:
|
||||
return ""
|
||||
t = t.lower()
|
||||
t = re.sub(r"[^a-z0-9 ]", " ", t)
|
||||
t = re.sub(r"\s+", " ", t).strip()
|
||||
return t[:60]
|
||||
|
||||
|
||||
def load_posted(md_dir=ORACLE, seen_json=SEEN_JSON):
|
||||
"""Return (md_urls:set, md_titles:set, seen_urls:set)."""
|
||||
md_urls, md_titles, seen_urls = set(), set(), set()
|
||||
|
||||
for fn in sorted(os.listdir(md_dir)):
|
||||
if re.match(r"athena_top.*\.md$", fn):
|
||||
try:
|
||||
txt = open(os.path.join(md_dir, fn), encoding="utf-8", errors="replace").read()
|
||||
except OSError:
|
||||
continue
|
||||
for m in re.findall(r"\]\((https?://[^)\s]+)\)", txt):
|
||||
nu = _norm_url(m)
|
||||
if nu:
|
||||
md_urls.add(nu)
|
||||
for t in re.findall(r"^\|\s*\d+\s*\|\s*(.+?)\s*\|", txt, re.M):
|
||||
nt = _norm_title(t)
|
||||
if nt:
|
||||
md_titles.add(nt)
|
||||
|
||||
if os.path.exists(seen_json):
|
||||
try:
|
||||
with open(seen_json, encoding="utf-8") as f:
|
||||
for u in json.load(f):
|
||||
nu = _norm_url(u)
|
||||
if nu:
|
||||
seen_urls.add(nu)
|
||||
except (json.JSONDecodeError, OSError):
|
||||
pass
|
||||
|
||||
return md_urls, md_titles, seen_urls
|
||||
|
||||
|
||||
def is_today(first_seen, now=None):
|
||||
now = now or datetime.now(timezone.utc)
|
||||
d = _parse(first_seen)
|
||||
return bool(d) and d.strftime("%Y-%m-%d") == now.strftime("%Y-%m-%d")
|
||||
|
||||
|
||||
def age_days(first_seen, now=None):
|
||||
now = now or datetime.now(timezone.utc)
|
||||
d = _parse(first_seen)
|
||||
if not d:
|
||||
return 9999.0
|
||||
return max((now - d).total_seconds() / 86400.0, 0.0)
|
||||
|
||||
|
||||
def day_bucket(first_seen, now=None):
|
||||
"""'today' | 'yesterday' | 'this-week' (<=6d) | 'older'."""
|
||||
days = age_days(first_seen, now)
|
||||
if days < 1:
|
||||
return "today"
|
||||
if days < 2:
|
||||
return "yesterday"
|
||||
if days <= 6:
|
||||
return "this-week"
|
||||
return "older"
|
||||
|
||||
|
||||
def already_posted_fs(url, title, first_seen, now=None,
|
||||
md_urls=None, md_titles=None, seen_urls=None):
|
||||
"""Age-aware dedup. A candidate is 'already posted' iff:
|
||||
(a) it matches the curated md-stack history, OR
|
||||
(b) it is OLDER than today AND its URL is in seen_urls.json.
|
||||
Today's items are NEVER flagged -- they are fresh by definition.
|
||||
"""
|
||||
if md_urls is None or md_titles is None or seen_urls is None:
|
||||
md_urls, md_titles, seen_urls = load_posted()
|
||||
if _norm_url(url) in md_urls:
|
||||
return True
|
||||
nt = _norm_title(title)
|
||||
if nt and nt in md_titles:
|
||||
return True
|
||||
if is_today(first_seen, now):
|
||||
return False
|
||||
if _norm_url(url) in seen_urls:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def recency_weight(first_seen, now=None, half_life_days=2.0):
|
||||
"""1.0 for today, decays ~halving every 2 days. The 'age' measure."""
|
||||
return 0.5 ** (age_days(first_seen, now) / half_life_days)
|
||||
|
||||
|
||||
def blend_score(item, now=None):
|
||||
"""clickability_decayed * recency_weight. Today's items dominate; old sink."""
|
||||
base = item.get("clickability_decayed", 0) or 0
|
||||
return base * recency_weight(item.get("first_seen"), now)
|
||||
|
||||
|
||||
def filter_fresh(items, now=None):
|
||||
"""Split into (today_items, older_new_items, dropped_items).
|
||||
|
||||
today_items = first_seen == today (always eligible; the week-open lead)
|
||||
older_new_items= older, but never before posted (md/seen)
|
||||
dropped_items = older AND already posted (the re-posts we are killing)
|
||||
"""
|
||||
now = now or datetime.now(timezone.utc)
|
||||
md_urls, md_titles, seen_urls = load_posted()
|
||||
today_items, older_new, dropped = [], [], []
|
||||
for it in items:
|
||||
fs = it.get("first_seen")
|
||||
if is_today(fs, now):
|
||||
today_items.append(it)
|
||||
continue
|
||||
if already_posted_fs(it.get("url"), it.get("title"), fs, now,
|
||||
md_urls, md_titles, seen_urls):
|
||||
dropped.append(it)
|
||||
else:
|
||||
older_new.append(it)
|
||||
return today_items, older_new, dropped
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
sys.path.insert(0, ORACLE)
|
||||
import clickability as cb
|
||||
DB = os.path.join(ORACLE, "oracle.db")
|
||||
conn = sqlite3.connect(f"file:{DB}?mode=ro", uri=True)
|
||||
items = cb.fetch_items(conn); conn.close()
|
||||
items = cb.compute_index(items); items = cb.decay_index(items)
|
||||
today_items, older_new, dropped = filter_fresh(items)
|
||||
print(f"TODAY-new (week-open lead): {len(today_items)}")
|
||||
print(f"OLDER-but-never-posted: {len(older_new)}")
|
||||
print(f"DROPPED (already posted): {len(dropped)}")
|
||||
print("\nSample dropped (the re-posts that caused the problem):")
|
||||
for it in sorted(dropped, key=lambda x: -x["clickability_decayed"])[:6]:
|
||||
print(f" - [{it['clickability_decayed']:.3f}] {it['title'][:66]}")
|
||||
for it in sorted(today, key=lambda x: x["clickability_decayed"], reverse=True)[:5]:
|
||||
print(f" [TODAY {it['clickability_decayed']:.2f}] {it['source']:10} {it['title'][:60]}")
|
||||
|
||||
Reference in New Issue
Block a user