Files
the-daily-epub/config.example.toml
T
thalladaandClaude Fable 5.1 69cd736455 Web dashboard step 1: migration, sessions, users CLI, public site
Migration 0004 (users, sessions, config_changes, profile_versions, jobs,
rating_events.user_id, runs.report_json, issues.issue_json), the issue
snapshot writer and loader, the web module skeleton with layout and static
assets, axum-login/tower-sessions over a sqlx session store, password-auth
users with a CLI, the login throttle, the origin check, security headers,
and the public issue pages, archive, Atom feed and robots.txt.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01NHyYupFdBiR4VfoUM7NjSM
2026-09-03 04:38:22 +00:00

222 lines
8.6 KiB
TOML

# The Daily EPUB — example configuration (spec §3.14).
#
# Load order (later wins): built-in defaults ← this file ← `DAILY_EPUB_*` env vars.
# Nested keys use a double underscore in env vars, e.g.
# DAILY_EPUB_MINIFLUX__API_KEY=...
# DAILY_EPUB_PROVIDERS__DEEPSEEK__API_KEY=...
# DAILY_EPUB_PROVIDERS__ANTHROPIC__API_KEY=...
# DAILY_EPUB_PROVIDERS__GEMINI__API_KEY=...
# DAILY_EPUB_VOYAGE__API_KEY=...
# DAILY_EPUB_SERVER__HMAC_SECRET=...
# DAILY_EPUB_LOOKBACK_HOURS=30
# DAILY_EPUB_LLM__EDITOR=gemini # one-off role override, no file edit
timezone = "America/New_York"
lookback_hours = 26
target_article_count = 20
retention_days = 21 # EPUBs, by age
xtc_retention_count = 5 # XTC issues, by count (~80-100 MB each)
world_briefing = true
# SQLite database file. Parent directories are created on demand.
database_path = "/var/lib/daily-epub/daily-epub.db"
# Default output directory for generated artifacts (overridden by `--out`).
out_dir = "/var/lib/daily-epub/out"
# Hand-maintained reader profile and Scour interests merged into the system prompt.
profile_path = "data/profile.md"
interests_opml = "data/scour-interests.opml"
[miniflux]
base_url = "http://127.0.0.1:8082"
# api_key via DAILY_EPUB_MINIFLUX__API_KEY env
page_limit = 250
# The two LLM roles, each assigned to a provider declared in [providers.*]
# below. Switching the editor to Gemini is `editor = "gemini"` plus its key in
# the env file; nothing else changes. An empty editor ("") runs everything on
# the bulk provider. Every editor call degrades to bulk when the editor's key
# is missing, its daily ceiling is hit, or the API refuses/fails.
[llm]
bulk = "deepseek" # triage, deep assessment, and every fallback
editor = "anthropic" # lineup, summaries, The Brief, the weekly profile rebuild
triage_batch_size = 25 # articles per first-pass triage request
deep_batch_size = 8 # articles per close-reading assessment request
score_temperature = 0.3 # sent only by providers that take a temperature (kind = "openai")
editorial_temperature = 0.8 # summaries and The Brief on an openai-kind provider
# The provider registry. Any number of entries; a role above names one by its
# table name. Keys never live here: DAILY_EPUB_PROVIDERS__<NAME>__API_KEY.
# `max_daily_usd` is a per-provider runaway guard per UTC day (0 = none), not
# accounting — set hard spend limits in each provider's dashboard as well.
# Prices are USD per 1M tokens and only feed the guard's arithmetic.
[providers.deepseek]
kind = "openai" # openai | anthropic
base_url = "https://api.deepseek.com/v1"
model = "deepseek-v4-flash" # DeepSeek-V4-Flash-0731 (confirmed 2026-08-15)
# api_key via DAILY_EPUB_PROVIDERS__DEEPSEEK__API_KEY env
max_daily_usd = 2.0
max_concurrent_requests = 4 # triage and deep-assessment batches in flight
price_input_per_mtok = 0.14
price_cache_read_per_mtok = 0.0028 # prefix-cache hits
price_cache_write_per_mtok = 0.0 # DeepSeek caches implicitly, no write charge
price_output_per_mtok = 0.28
# Claude Opus 5 over the Messages API. Requests carry `output_config.effort`,
# a cached system block, and `fallbacks = "default"` so a classifier refusal is
# re-routed server-side; a refusal that still comes back degrades to bulk.
[providers.anthropic]
kind = "anthropic"
base_url = "https://api.anthropic.com"
model = "claude-opus-5"
# api_key via DAILY_EPUB_PROVIDERS__ANTHROPIC__API_KEY env
effort = "high" # low | medium | high | xhigh | max → output_config.effort
max_daily_usd = 3.0
max_concurrent_requests = 4 # summaries in flight when this is the summary provider
price_input_per_mtok = 5.0
price_cache_read_per_mtok = 0.5
price_cache_write_per_mtok = 6.25
price_output_per_mtok = 25.0
# Gemini 3.8 Flash through Google's OpenAI-compatible endpoint (beta, verified
# 2026-09-02). Declared but unreferenced until a role names it.
[providers.gemini]
kind = "openai"
base_url = "https://generativelanguage.googleapis.com/v1beta/openai"
model = "gemini-3.8-flash"
# api_key via DAILY_EPUB_PROVIDERS__GEMINI__API_KEY env
effort = "high" # minimal | low | medium | high → reasoning_effort
max_daily_usd = 3.0
max_concurrent_requests = 4
price_input_per_mtok = 0.75 # promotional through 2026-12-31; $1.50 from 2027-01-01
price_cache_read_per_mtok = 0.075 # implicit cache hits; $0.15 from 2027-01-01
price_cache_write_per_mtok = 0.0
price_output_per_mtok = 3.75 # includes thinking tokens; $7.50 from 2027-01-01
# Voyage AI embeddings behind the interest and rated-neighbour signals. Set
# `enabled = false` (or leave the key unset) and the paper still builds: the
# learned signals are simply absent, never a penalty.
[voyage]
enabled = true
base_url = "https://api.voyageai.com/v1"
model = "voyage-4-lite"
# api_key via DAILY_EPUB_VOYAGE__API_KEY env
output_dimension = 512 # 256 | 512 | 1024 | 2048
batch_size = 32
max_concurrent_requests = 4
max_input_chars = 60000 # per article, cut on a char boundary
max_daily_usd = 0.50 # runaway guard ($0.02 / M tokens)
[curation]
max_article_count = 28 # hard ceiling; there is no minimum (§13)
recent_rejection_days = 7
recent_rejection_floor = 3.0
always_include_feeds = [] # miniflux feed ids or site urls
blocked_domains = []
# Extra paywalled hosts, merged with the built-in list (nytimes, wsj, ft, …).
# A short body from one of these is marked "excerpt only" and penalized (§3.3).
paywall_domains = []
sections = [
"Top Stories",
"Tech & Engineering",
"Science & Space",
"AI & Machine Learning",
"Culture & Essays",
"Boston & Local",
"Niche Corner",
"From the Blogroll",
]
[curation.feedback]
loved_value = 1.0
good_value = 0.35
not_for_me_value = -1.0
verdicts_in_prompt = 60
# Every weight, quota, gate and threshold of the personalized ranker. The
# learned signals (`knn`, `feed`) contribute nothing until their gates open:
# the weight ramps linearly from `*_floor` to `*_full` rated articles.
[curation.ranking]
triage_max = 800 # eligible articles the triage LLM reads
deep_keep = 120 # deep-assessment set
shortlist_keep = 60 # what the editor sees
assessment_reuse_days = 3
rating_lookback_days = 180
rating_half_life_days = 60
neighbour_k = 5
negative_coefficient = 0.75
knn_floor = 8
knn_full = 25
feed_floor = 15
feed_full = 40
semantic_min_words = 300
exploration_slots = 5
embedding_retention_days = 120 # `features prune`: unrated, unpublished vectors
telemetry_retention_days = 180 # `features prune`: candidate_runs and article_assessments
[curation.ranking.quotas]
triage = 60
interest = 20
knn = 20
# Weights need not sum to 1; they are renormalized over the present signals.
[curation.ranking.weights.preliminary]
interest = 0.35
knn = 0.25
heuristic = 0.20
feed = 0.10
social = 0.10
[curation.ranking.weights.utility]
quality = 0.40
fit = 0.20
knn = 0.15
interest = 0.10
feed = 0.05
triage = 0.05
social = 0.03
heuristic = 0.02
[curation.ranking.diversity]
cluster_threshold = 0.85
per_cluster_cap = 2
utility_protected = 10
[editorial]
summary_model = "editor" # editor | bulk — which [llm] role writes the summaries
summary_input_tokens = 3000 # article text offered per summary
[publish]
# Where both EPUB editions land, and what the OPDS feed lists. BookOrbit is
# optional — it just watches this folder if you run it.
# (Renamed from `bookorbit_dir`; the old key is now a hard config error.)
epub_dir = "/srv/bookorbit/libraries/daily-epub"
# XTC artifacts. Not listed in OPDS (CrossPoint cannot acquire them); reachable
# at /files/xtc/<name> for sideloading.
xtc_dir = "/var/lib/daily-epub/xtc"
[xtc]
enabled = true
# epub-to-xtc-converter has no global npm bin; it is invoked through node.
# The code appends: <input.epub> -o <output.xtch> -f <format> -c <settings>
command = "node"
args = ["/opt/epub-to-xtc-converter/cli/index.js", "convert"]
format = "xtch" # xtc (1-bit) | xtch (grayscale)
# REQUIRED in practice: the converter refuses to start without `font.path`,
# which can only be given through this file. Start from xtc-settings.example.json.
settings = "/etc/daily-epub/xtc-settings.json"
[server]
bind = "127.0.0.1:3499"
public_url = "https://daily.hallada.net"
session_days = 30
login_attempts = 10
login_window_minutes = 15
jobs_enabled = true
journal_lines = 300
# hmac_secret via DAILY_EPUB_SERVER__HMAC_SECRET env (32+ random bytes)
# Optional Basic auth for the XTC OPDS feed and file downloads:
# basic_auth_user = "daily"
# basic_auth_pass = "..."