Discover feeds behind aggregator-only articles

A best-effort pipeline stage after social enrichment asks Miniflux to
discover the feeds behind each aggregator-only article, validates every
result by sniffing the body, and records the survivors as feed candidates
with a per-host memo. Ranking is a shrunk mean over the linked articles'
existing telemetry and ratings. `daily-epub feeds discover` runs the same
pass over recent articles for seeding.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01QVPagF6jfDv78CC5Jv2wp4
This commit is contained in:
2026-09-07 04:31:13 +00:00
co-authored by Claude Fable 5.1
parent 4a2fc6d1c4
commit 2bbb4809b7
7 changed files with 1939 additions and 36 deletions
+27
View File
@@ -0,0 +1,27 @@
-- Feed discovery: proposed Miniflux subscriptions found behind aggregator-only
-- articles (feed discovery plan §4 step 1).
CREATE TABLE feed_candidates (
id INTEGER PRIMARY KEY AUTOINCREMENT,
feed_url TEXT NOT NULL UNIQUE,
host TEXT NOT NULL,
title TEXT,
status TEXT NOT NULL CHECK (status IN ('candidate', 'added', 'dismissed')),
first_seen TEXT NOT NULL,
last_seen TEXT NOT NULL,
miniflux_feed_id INTEGER,
decided_at TEXT
);
CREATE INDEX idx_feed_candidates_status_host ON feed_candidates(status, host);
CREATE TABLE feed_candidate_articles (
candidate_id INTEGER NOT NULL REFERENCES feed_candidates(id) ON DELETE CASCADE,
article_id INTEGER NOT NULL REFERENCES articles(id) ON DELETE CASCADE,
PRIMARY KEY (candidate_id, article_id)
);
CREATE TABLE feed_discovery_hosts (
host TEXT PRIMARY KEY,
checked_at TEXT NOT NULL,
candidates INTEGER NOT NULL
);