DeepSeek triage over every eligible article (plan §10) with cached assessments in article_assessments, the union admission with quotas and exploration slots (§11), hygiene moved to admit.rs with the churn rule reading assessments, prefilter.rs reduced to hygiene and text heuristic, prefilter_keep removed in favour of curation.ranking.deep_keep, the scores table dropped (migration 0003), and --rescore on generate. Implemented by Codex (gpt-5.4, high effort) from docs/plans/curation-v2-briefs/step4.md; reviewed against plan §10–§11. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01A1rCLQeKBgnBo3oTgHuTMe
187 lines
6.5 KiB
TOML
187 lines
6.5 KiB
TOML
# The Daily EPUB — example configuration (spec §3.14).
|
|
#
|
|
# Load order (later wins): built-in defaults ← this file ← `DAILY_EPUB_*` env vars.
|
|
# Nested keys use a double underscore in env vars, e.g.
|
|
# DAILY_EPUB_MINIFLUX__API_KEY=...
|
|
# DAILY_EPUB_DEEPSEEK__API_KEY=...
|
|
# DAILY_EPUB_ANTHROPIC__API_KEY=...
|
|
# DAILY_EPUB_VOYAGE__API_KEY=...
|
|
# DAILY_EPUB_SERVER__HMAC_SECRET=...
|
|
# DAILY_EPUB_LOOKBACK_HOURS=30
|
|
|
|
timezone = "America/New_York"
|
|
lookback_hours = 26
|
|
target_article_count = 20
|
|
retention_days = 21 # EPUBs, by age
|
|
xtc_retention_count = 5 # XTC issues, by count (~80-100 MB each)
|
|
max_daily_usd = 2.0 # DeepSeek ceiling per UTC day; [anthropic] and [voyage] have their own
|
|
world_briefing = true
|
|
|
|
# SQLite database file. Parent directories are created on demand.
|
|
database_path = "/var/lib/daily-epub/daily-epub.db"
|
|
|
|
# Default output directory for generated artifacts (overridden by `--out`).
|
|
out_dir = "/var/lib/daily-epub/out"
|
|
|
|
# Hand-maintained reader profile and Scour interests merged into the system prompt.
|
|
profile_path = "data/profile.md"
|
|
interests_opml = "data/scour-interests.opml"
|
|
|
|
[miniflux]
|
|
base_url = "http://127.0.0.1:8082"
|
|
# api_key via DAILY_EPUB_MINIFLUX__API_KEY env
|
|
page_limit = 250
|
|
|
|
[deepseek]
|
|
base_url = "https://api.deepseek.com/v1"
|
|
model = "deepseek-v4-flash" # DeepSeek-V4-Flash-0731 (confirmed 2026-08-15)
|
|
# api_key via DAILY_EPUB_DEEPSEEK__API_KEY env
|
|
score_batch_size = 12
|
|
triage_batch_size = 25 # articles per first-pass triage request
|
|
max_concurrent_requests = 4 # triage and stage-A batches in flight at once
|
|
score_temperature = 0.3
|
|
editorial_temperature = 0.8 # used only when DeepSeek is the fallback editor
|
|
# USD per 1M tokens, used for the cost guardrail.
|
|
price_input_per_mtok = 0.14
|
|
price_cached_input_per_mtok = 0.0028
|
|
price_output_per_mtok = 0.28
|
|
|
|
# Claude is the editor: selection, summaries, The Brief and the weekly profile
|
|
# rebuild. Every call degrades to DeepSeek when the key is missing, the daily
|
|
# ceiling is hit, or the API refuses/fails. Server-side refusal fallback
|
|
# (`fallbacks = "default"`) is always on. Set a spend limit in the Anthropic
|
|
# dashboard too: `max_daily_usd` is a runaway guard, not accounting.
|
|
[anthropic]
|
|
enabled = true
|
|
base_url = "https://api.anthropic.com"
|
|
model = "claude-opus-5"
|
|
# api_key via DAILY_EPUB_ANTHROPIC__API_KEY env
|
|
effort = "high" # low | medium | high | xhigh | max
|
|
price_input_per_mtok = 5.0
|
|
price_cache_write_per_mtok = 6.25
|
|
price_cache_read_per_mtok = 0.5
|
|
price_output_per_mtok = 25.0
|
|
max_daily_usd = 3.0
|
|
max_concurrent_requests = 4
|
|
|
|
# Voyage AI embeddings behind the interest and rated-neighbour signals. Set
|
|
# `enabled = false` (or leave the key unset) and the paper still builds: the
|
|
# learned signals are simply absent, never a penalty.
|
|
[voyage]
|
|
enabled = true
|
|
base_url = "https://api.voyageai.com/v1"
|
|
model = "voyage-4-lite"
|
|
# api_key via DAILY_EPUB_VOYAGE__API_KEY env
|
|
output_dimension = 512 # 256 | 512 | 1024 | 2048
|
|
batch_size = 32
|
|
max_concurrent_requests = 4
|
|
max_input_chars = 60000 # per article, cut on a char boundary
|
|
max_daily_usd = 0.50 # runaway guard ($0.02 / M tokens)
|
|
|
|
[curation]
|
|
max_article_count = 28 # hard ceiling; there is no minimum (§13)
|
|
recent_rejection_days = 7
|
|
recent_rejection_floor = 3.0
|
|
always_include_feeds = [] # miniflux feed ids or site urls
|
|
blocked_domains = []
|
|
# Extra paywalled hosts, merged with the built-in list (nytimes, wsj, ft, …).
|
|
# A short body from one of these is marked "excerpt only" and penalized (§3.3).
|
|
paywall_domains = []
|
|
sections = [
|
|
"Top Stories",
|
|
"Tech & Engineering",
|
|
"Science & Space",
|
|
"AI & Machine Learning",
|
|
"Culture & Essays",
|
|
"Boston & Local",
|
|
"Niche Corner",
|
|
"From the Blogroll",
|
|
]
|
|
|
|
[curation.feedback]
|
|
loved_value = 1.0
|
|
good_value = 0.35
|
|
not_for_me_value = -1.0
|
|
verdicts_in_prompt = 60
|
|
|
|
# Every weight, quota, gate and threshold of the personalized ranker. The
|
|
# learned signals (`knn`, `feed`) contribute nothing until their gates open:
|
|
# the weight ramps linearly from `*_floor` to `*_full` rated articles.
|
|
[curation.ranking]
|
|
triage_max = 800 # eligible articles the triage LLM reads
|
|
deep_keep = 120 # deep-assessment set
|
|
shortlist_keep = 60 # what the editor sees
|
|
assessment_reuse_days = 3
|
|
rating_lookback_days = 180
|
|
rating_half_life_days = 60
|
|
neighbour_k = 5
|
|
negative_coefficient = 0.75
|
|
knn_floor = 8
|
|
knn_full = 25
|
|
feed_floor = 15
|
|
feed_full = 40
|
|
semantic_min_words = 300
|
|
exploration_slots = 5
|
|
embedding_retention_days = 120 # `features prune`: unrated, unpublished vectors
|
|
telemetry_retention_days = 180 # `features prune`: candidate_runs rows
|
|
|
|
[curation.ranking.quotas]
|
|
triage = 60
|
|
interest = 20
|
|
knn = 20
|
|
|
|
# Weights need not sum to 1; they are renormalized over the present signals.
|
|
[curation.ranking.weights.preliminary]
|
|
interest = 0.35
|
|
knn = 0.25
|
|
heuristic = 0.20
|
|
feed = 0.10
|
|
social = 0.10
|
|
|
|
[curation.ranking.weights.utility]
|
|
quality = 0.40
|
|
fit = 0.20
|
|
knn = 0.15
|
|
interest = 0.10
|
|
feed = 0.05
|
|
triage = 0.05
|
|
social = 0.03
|
|
heuristic = 0.02
|
|
|
|
[curation.ranking.diversity]
|
|
cluster_threshold = 0.85
|
|
per_cluster_cap = 2
|
|
utility_protected = 10
|
|
|
|
[editorial]
|
|
summary_model = "editor" # editor (Claude) | bulk (DeepSeek)
|
|
summary_input_tokens = 3000 # article text offered per summary
|
|
|
|
[publish]
|
|
# Where both EPUB editions land, and what the OPDS feed lists. BookOrbit is
|
|
# optional — it just watches this folder if you run it.
|
|
# (Renamed from `bookorbit_dir`; the old key is now a hard config error.)
|
|
epub_dir = "/srv/bookorbit/libraries/daily-epub"
|
|
# XTC artifacts. Not listed in OPDS (CrossPoint cannot acquire them); reachable
|
|
# at /files/xtc/<name> for sideloading.
|
|
xtc_dir = "/var/lib/daily-epub/xtc"
|
|
|
|
[xtc]
|
|
enabled = true
|
|
# epub-to-xtc-converter has no global npm bin; it is invoked through node.
|
|
# The code appends: <input.epub> -o <output.xtch> -f <format> -c <settings>
|
|
command = "node"
|
|
args = ["/opt/epub-to-xtc-converter/cli/index.js", "convert"]
|
|
format = "xtch" # xtc (1-bit) | xtch (grayscale)
|
|
# REQUIRED in practice: the converter refuses to start without `font.path`,
|
|
# which can only be given through this file. Start from xtc-settings.example.json.
|
|
settings = "/etc/daily-epub/xtc-settings.json"
|
|
|
|
[server]
|
|
bind = "127.0.0.1:3499"
|
|
public_url = "https://daily.hallada.net"
|
|
# hmac_secret via DAILY_EPUB_SERVER__HMAC_SECRET env (32+ random bytes)
|
|
# Optional Basic auth for the XTC OPDS feed and file downloads:
|
|
# basic_auth_user = "daily"
|
|
# basic_auth_pass = "..."
|