729760fb27
- theme_scan.py: tags entries by 4 practitioner themes (tool-call/context/ compute/trust), counts NEW arrivals per cron cycle (falsification check for one-day-cluster vs trend). Idempotent: re-run = 0 new. - schema.sql: theme_tags table (separate from core entries schema) - oracle-pipeline.sh: wire theme_scan after summarize - A result (footnote): unified discipline layer unoccupied; adjacent OSS entrants exist (agentgateway, lelu) but no portable unified layer. Verified: theme_scan classifies 5 seed/extra items, 2nd run = 0 new.
47 lines
1.9 KiB
SQL
47 lines
1.9 KiB
SQL
CREATE TABLE IF NOT EXISTS entries (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
source TEXT NOT NULL,
|
|
source_id TEXT NOT NULL,
|
|
url TEXT,
|
|
title TEXT,
|
|
extracted_text TEXT,
|
|
summary TEXT,
|
|
category_tags TEXT,
|
|
signal_score REAL,
|
|
raw_metadata TEXT,
|
|
first_seen TEXT DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
|
|
last_updated TEXT DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
|
|
UNIQUE(source, source_id)
|
|
);
|
|
|
|
CREATE INDEX IF NOT EXISTS idx_entries_source ON entries(source);
|
|
CREATE INDEX IF NOT EXISTS idx_entries_signal ON entries(signal_score DESC);
|
|
CREATE INDEX IF NOT EXISTS idx_entries_category ON entries(category_tags);
|
|
|
|
-- Run log: records each pipeline invocation for failure visibility + growth control.
|
|
-- Partial failures (e.g. Reddit rate-limited) are detectable here, not hidden
|
|
-- as a "complete" run. Also enables future pruning decisions (entries older
|
|
-- than N days with no re-fetch can be archived).
|
|
CREATE TABLE IF NOT EXISTS run_log (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
run_time TEXT DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
|
|
total_fetched INTEGER DEFAULT 0,
|
|
total_stored INTEGER DEFAULT 0,
|
|
sources_ok TEXT, -- JSON list of sources that succeeded
|
|
sources_failed TEXT, -- JSON list of sources that errored/skipped
|
|
notes TEXT
|
|
);
|
|
|
|
-- Theme tags: Phase 6 trend-tracking. Tags entries by the 4 practitioner
|
|
-- resource-discipline themes so we can measure RECURRING theme frequency
|
|
-- across FRESH entries (not persistence of specific rows). Counts new
|
|
-- arrivals per cron cycle -> the falsification check for the "one-day cluster
|
|
-- vs real trend" question. Separate table, never mutates the core entries schema.
|
|
CREATE TABLE IF NOT EXISTS theme_tags (
|
|
entry_id INTEGER NOT NULL,
|
|
theme TEXT NOT NULL,
|
|
first_seen_cycle TEXT DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
|
|
PRIMARY KEY (entry_id, theme),
|
|
FOREIGN KEY (entry_id) REFERENCES entries(id)
|
|
);
|