23cce4d609
- adapters/__init__.py: add http_get() unified retry (429/5xx only, max 2 attempts, capped exp backoff) + AdapterHTTPError carrying failure_class; SourceAdapter.last_failure_class set on failure for pipeline capture. - arxiv/github/huggingface/hackernews/reddit: route HTTP through http_get. Preserves GitHub 403 rate-limit retry and Reddit 403/429 fast-bail. - schema.sql + pipeline.py: add run_log.failure_class column; rollup most- severe class across sources (5xx>4xx>429>error>zero_fetch>ok). - pipeline.py: ENABLE RSS in ENABLED_SOURCES (was registered, disabled). - RSS smoke test surfaced 3 broken feeds (anthropic 404, googleai 404, metaai 301) — left as-is, captured in feed_failures; URL fix is separate discovery task, not guessed. Verified: full dry-run fetches all 6 sources; github live fetch OK; Reddit 429 fast-bail preserved; no import/syntax errors.
50 lines
2.2 KiB
SQL
50 lines
2.2 KiB
SQL
CREATE TABLE IF NOT EXISTS entries (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
source TEXT NOT NULL,
|
|
source_id TEXT NOT NULL,
|
|
url TEXT,
|
|
title TEXT,
|
|
extracted_text TEXT,
|
|
summary TEXT,
|
|
category_tags TEXT,
|
|
signal_score REAL,
|
|
raw_metadata TEXT,
|
|
first_seen TEXT DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
|
|
last_updated TEXT DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
|
|
UNIQUE(source, source_id)
|
|
);
|
|
|
|
CREATE INDEX IF NOT EXISTS idx_entries_source ON entries(source);
|
|
CREATE INDEX IF NOT EXISTS idx_entries_signal ON entries(signal_score DESC);
|
|
CREATE INDEX IF NOT EXISTS idx_entries_category ON entries(category_tags);
|
|
|
|
-- Run log: records each pipeline invocation for failure visibility + growth control.
|
|
-- Partial failures (e.g. Reddit rate-limited) are detectable here, not hidden
|
|
-- as a "complete" run. Also enables future pruning decisions (entries older
|
|
-- than N days with no re-fetch can be archived).
|
|
-- failure_class (issue #2): one of 4xx / 5xx / 429 / zero_fetch / ok, derived
|
|
-- from the real HTTP response via adapters.http_get, not guessed after the fact.
|
|
CREATE TABLE IF NOT EXISTS run_log (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
run_time TEXT DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
|
|
total_fetched INTEGER DEFAULT 0,
|
|
total_stored INTEGER DEFAULT 0,
|
|
sources_ok TEXT, -- JSON list of sources that succeeded
|
|
sources_failed TEXT, -- JSON list of sources that errored/skipped
|
|
failure_class TEXT, -- 4xx / 5xx / 429 / zero_fetch / ok
|
|
notes TEXT
|
|
);
|
|
|
|
-- Theme tags: Phase 6 trend-tracking. Tags entries by the 4 practitioner
|
|
-- resource-discipline themes so we can measure RECURRING theme frequency
|
|
-- across FRESH entries (not persistence of specific rows). Counts new
|
|
-- arrivals per cron cycle -> the falsification check for the "one-day cluster
|
|
-- vs real trend" question. Separate table, never mutates the core entries schema.
|
|
CREATE TABLE IF NOT EXISTS theme_tags (
|
|
entry_id INTEGER NOT NULL,
|
|
theme TEXT NOT NULL,
|
|
first_seen_cycle TEXT DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
|
|
PRIMARY KEY (entry_id, theme),
|
|
FOREIGN KEY (entry_id) REFERENCES entries(id)
|
|
);
|