Files
the-daily-epub/config.example.toml
T
thalladaandClaude Fable 5.1 05a74a0dcf Curation v2 step 5: deep assessment, utility, diversified shortlist
assess.rs replaces score.rs (DEEP_INSTRUCTIONS, representative sample
with [BEGINNING]/[MIDDLE]/[END], facets, cached deep rows with --rescore
bypass), rank.rs adds the utility blend over present signals with gate
ramps and the leader-clustered shortlist (cap 2 → 3 → uncapped, protected
top-N, exploration reserve), editor.rs replaces select.rs with the §13
rendering and utility-ordered fallbacks. ScoredArticle is gone; Candidate
is the only flow type. deep_batch_size replaces score_batch_size.

Started by Codex (cut off by its usage limit mid-verification) and
finished by a Claude agent from docs/plans/curation-v2-briefs/step5.md;
reviewed against plan §12–§13.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01A1rCLQeKBgnBo3oTgHuTMe
2026-09-02 16:04:43 +00:00

187 lines
6.6 KiB
TOML

# The Daily EPUB — example configuration (spec §3.14).
#
# Load order (later wins): built-in defaults ← this file ← `DAILY_EPUB_*` env vars.
# Nested keys use a double underscore in env vars, e.g.
# DAILY_EPUB_MINIFLUX__API_KEY=...
# DAILY_EPUB_DEEPSEEK__API_KEY=...
# DAILY_EPUB_ANTHROPIC__API_KEY=...
# DAILY_EPUB_VOYAGE__API_KEY=...
# DAILY_EPUB_SERVER__HMAC_SECRET=...
# DAILY_EPUB_LOOKBACK_HOURS=30
timezone = "America/New_York"
lookback_hours = 26
target_article_count = 20
retention_days = 21 # EPUBs, by age
xtc_retention_count = 5 # XTC issues, by count (~80-100 MB each)
max_daily_usd = 2.0 # DeepSeek ceiling per UTC day; [anthropic] and [voyage] have their own
world_briefing = true
# SQLite database file. Parent directories are created on demand.
database_path = "/var/lib/daily-epub/daily-epub.db"
# Default output directory for generated artifacts (overridden by `--out`).
out_dir = "/var/lib/daily-epub/out"
# Hand-maintained reader profile and Scour interests merged into the system prompt.
profile_path = "data/profile.md"
interests_opml = "data/scour-interests.opml"
[miniflux]
base_url = "http://127.0.0.1:8082"
# api_key via DAILY_EPUB_MINIFLUX__API_KEY env
page_limit = 250
[deepseek]
base_url = "https://api.deepseek.com/v1"
model = "deepseek-v4-flash" # DeepSeek-V4-Flash-0731 (confirmed 2026-08-15)
# api_key via DAILY_EPUB_DEEPSEEK__API_KEY env
deep_batch_size = 8 # articles per close-reading assessment request
triage_batch_size = 25 # articles per first-pass triage request
max_concurrent_requests = 4 # triage and deep-assessment batches in flight
score_temperature = 0.3
editorial_temperature = 0.8 # used only when DeepSeek is the fallback editor
# USD per 1M tokens, used for the cost guardrail.
price_input_per_mtok = 0.14
price_cached_input_per_mtok = 0.0028
price_output_per_mtok = 0.28
# Claude is the editor: selection, summaries, The Brief and the weekly profile
# rebuild. Every call degrades to DeepSeek when the key is missing, the daily
# ceiling is hit, or the API refuses/fails. Server-side refusal fallback
# (`fallbacks = "default"`) is always on. Set a spend limit in the Anthropic
# dashboard too: `max_daily_usd` is a runaway guard, not accounting.
[anthropic]
enabled = true
base_url = "https://api.anthropic.com"
model = "claude-opus-5"
# api_key via DAILY_EPUB_ANTHROPIC__API_KEY env
effort = "high" # low | medium | high | xhigh | max
price_input_per_mtok = 5.0
price_cache_write_per_mtok = 6.25
price_cache_read_per_mtok = 0.5
price_output_per_mtok = 25.0
max_daily_usd = 3.0
max_concurrent_requests = 4
# Voyage AI embeddings behind the interest and rated-neighbour signals. Set
# `enabled = false` (or leave the key unset) and the paper still builds: the
# learned signals are simply absent, never a penalty.
[voyage]
enabled = true
base_url = "https://api.voyageai.com/v1"
model = "voyage-4-lite"
# api_key via DAILY_EPUB_VOYAGE__API_KEY env
output_dimension = 512 # 256 | 512 | 1024 | 2048
batch_size = 32
max_concurrent_requests = 4
max_input_chars = 60000 # per article, cut on a char boundary
max_daily_usd = 0.50 # runaway guard ($0.02 / M tokens)
[curation]
max_article_count = 28 # hard ceiling; there is no minimum (§13)
recent_rejection_days = 7
recent_rejection_floor = 3.0
always_include_feeds = [] # miniflux feed ids or site urls
blocked_domains = []
# Extra paywalled hosts, merged with the built-in list (nytimes, wsj, ft, …).
# A short body from one of these is marked "excerpt only" and penalized (§3.3).
paywall_domains = []
sections = [
"Top Stories",
"Tech & Engineering",
"Science & Space",
"AI & Machine Learning",
"Culture & Essays",
"Boston & Local",
"Niche Corner",
"From the Blogroll",
]
[curation.feedback]
loved_value = 1.0
good_value = 0.35
not_for_me_value = -1.0
verdicts_in_prompt = 60
# Every weight, quota, gate and threshold of the personalized ranker. The
# learned signals (`knn`, `feed`) contribute nothing until their gates open:
# the weight ramps linearly from `*_floor` to `*_full` rated articles.
[curation.ranking]
triage_max = 800 # eligible articles the triage LLM reads
deep_keep = 120 # deep-assessment set
shortlist_keep = 60 # what the editor sees
assessment_reuse_days = 3
rating_lookback_days = 180
rating_half_life_days = 60
neighbour_k = 5
negative_coefficient = 0.75
knn_floor = 8
knn_full = 25
feed_floor = 15
feed_full = 40
semantic_min_words = 300
exploration_slots = 5
embedding_retention_days = 120 # `features prune`: unrated, unpublished vectors
telemetry_retention_days = 180 # `features prune`: candidate_runs rows
[curation.ranking.quotas]
triage = 60
interest = 20
knn = 20
# Weights need not sum to 1; they are renormalized over the present signals.
[curation.ranking.weights.preliminary]
interest = 0.35
knn = 0.25
heuristic = 0.20
feed = 0.10
social = 0.10
[curation.ranking.weights.utility]
quality = 0.40
fit = 0.20
knn = 0.15
interest = 0.10
feed = 0.05
triage = 0.05
social = 0.03
heuristic = 0.02
[curation.ranking.diversity]
cluster_threshold = 0.85
per_cluster_cap = 2
utility_protected = 10
[editorial]
summary_model = "editor" # editor (Claude) | bulk (DeepSeek)
summary_input_tokens = 3000 # article text offered per summary
[publish]
# Where both EPUB editions land, and what the OPDS feed lists. BookOrbit is
# optional — it just watches this folder if you run it.
# (Renamed from `bookorbit_dir`; the old key is now a hard config error.)
epub_dir = "/srv/bookorbit/libraries/daily-epub"
# XTC artifacts. Not listed in OPDS (CrossPoint cannot acquire them); reachable
# at /files/xtc/<name> for sideloading.
xtc_dir = "/var/lib/daily-epub/xtc"
[xtc]
enabled = true
# epub-to-xtc-converter has no global npm bin; it is invoked through node.
# The code appends: <input.epub> -o <output.xtch> -f <format> -c <settings>
command = "node"
args = ["/opt/epub-to-xtc-converter/cli/index.js", "convert"]
format = "xtch" # xtc (1-bit) | xtch (grayscale)
# REQUIRED in practice: the converter refuses to start without `font.path`,
# which can only be given through this file. Start from xtc-settings.example.json.
settings = "/etc/daily-epub/xtc-settings.json"
[server]
bind = "127.0.0.1:3499"
public_url = "https://daily.hallada.net"
# hmac_secret via DAILY_EPUB_SERVER__HMAC_SECRET env (32+ random bytes)
# Optional Basic auth for the XTC OPDS feed and file downloads:
# basic_auth_user = "daily"
# basic_auth_pass = "..."