Curation v2 step 3: Voyage embeddings, cheap signals, candidate telemetry

- embedding.rs: EmbeddingBackend + VoyageBackend, batched bounded-concurrency
  client with its own UsageMeter, f32 BLOB codec, article/interest embedding
  cache keyed by model, dimension and sha256 of the embedded text.
- signals.rs: z-scored interest match, decayed rated-neighbour preference
  with the knn gate, feed affinity with the feed gate, social, text heuristic
  without social terms, mid-rank percentile normalizer, preliminary blend.
- telemetry.rs: candidate_runs writer with §7.5 signals_json, explain and
  near-misses renderers, prune.
- [voyage] and the full [curation.ranking] config with validation.
- CLI: explain, features backfill|prune, generate --skip-embeddings.
- Pipeline: hygiene rows, embed and signals stages before the old prefilter;
  same-date regeneration no longer excludes its own picks.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01A1rCLQeKBgnBo3oTgHuTMe
This commit is contained in:
2026-09-02 04:08:45 +00:00
co-authored by Claude Fable 5.1
parent 3a9f4b99e0
commit ea3b141373
13 changed files with 4888 additions and 17 deletions
+63
View File
@@ -43,6 +43,20 @@ price_input_per_mtok = 0.14
price_cached_input_per_mtok = 0.0028
price_output_per_mtok = 0.28
# Voyage AI embeddings behind the interest and rated-neighbour signals. Set
# `enabled = false` (or leave the key unset) and the paper still builds: the
# learned signals are simply absent, never a penalty.
[voyage]
enabled = true
base_url = "https://api.voyageai.com/v1"
model = "voyage-4-lite"
# api_key via DAILY_EPUB_VOYAGE__API_KEY env
output_dimension = 512 # 256 | 512 | 1024 | 2048
batch_size = 32
max_concurrent_requests = 4
max_input_chars = 60000 # per article, cut on a char boundary
max_daily_usd = 0.50 # runaway guard ($0.02 / M tokens)
[curation]
always_include_feeds = [] # miniflux feed ids or site urls
blocked_domains = []
@@ -66,6 +80,55 @@ good_value = 0.35
not_for_me_value = -1.0
verdicts_in_prompt = 60
# Every weight, quota, gate and threshold of the personalized ranker. The
# learned signals (`knn`, `feed`) contribute nothing until their gates open:
# the weight ramps linearly from `*_floor` to `*_full` rated articles.
[curation.ranking]
triage_max = 800 # eligible articles the triage LLM reads
deep_keep = 120 # deep-assessment set
shortlist_keep = 60 # what the editor sees
assessment_reuse_days = 3
rating_lookback_days = 180
rating_half_life_days = 60
neighbour_k = 5
negative_coefficient = 0.75
knn_floor = 8
knn_full = 25
feed_floor = 15
feed_full = 40
semantic_min_words = 300
exploration_slots = 5
embedding_retention_days = 120 # `features prune`: unrated, unpublished vectors
telemetry_retention_days = 180 # `features prune`: candidate_runs rows
[curation.ranking.quotas]
triage = 60
interest = 20
knn = 20
# Weights need not sum to 1; they are renormalized over the present signals.
[curation.ranking.weights.preliminary]
interest = 0.35
knn = 0.25
heuristic = 0.20
feed = 0.10
social = 0.10
[curation.ranking.weights.utility]
quality = 0.40
fit = 0.20
knn = 0.15
interest = 0.10
feed = 0.05
triage = 0.05
social = 0.03
heuristic = 0.02
[curation.ranking.diversity]
cluster_threshold = 0.85
per_cluster_cap = 2
utility_protected = 10
[publish]
# Where both EPUB editions land, and what the OPDS feed lists. BookOrbit is
# optional — it just watches this folder if you run it.