Write match rows per run, categorize interests, and add the interests CLI (steps 4 and 8)
The signals stage upserts each eligible article's matched interests; an LLM batch files uncategorized interests before the prompt is built (and on demand as the interests-categorize job); `interests import` and `interests backfill` load the OPML and seed match rows once. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01K9PrjtUS16PAQve8D4bHgc
This commit is contained in:
@@ -135,6 +135,8 @@ daily-epub features backfill [--days 30] [--rated-only] [--all] [--yes]
|
|||||||
daily-epub features prune # stale embeddings, old telemetry and assessments
|
daily-epub features prune # stale embeddings, old telemetry and assessments
|
||||||
daily-epub backfill-social [--days 7] # re-poll social scores for recent articles
|
daily-epub backfill-social [--days 7] # re-poll social scores for recent articles
|
||||||
daily-epub feeds discover [--days 14] [--limit 50] # seed feed candidates from recent aggregator articles
|
daily-epub feeds discover [--days 14] [--limit 50] # seed feed candidates from recent aggregator articles
|
||||||
|
daily-epub interests import [--opml PATH] [--profile PATH] # one-time standing-interest import
|
||||||
|
daily-epub interests backfill # match compatible cached article embeddings to interests
|
||||||
daily-epub db migrate # run migrations (also automatic on every start)
|
daily-epub db migrate # run migrations (also automatic on every start)
|
||||||
daily-epub config check # validate the config, print the resolved roles, keys and paths
|
daily-epub config check # validate the config, print the resolved roles, keys and paths
|
||||||
daily-epub users add USER [--admin] [--password-stdin]
|
daily-epub users add USER [--admin] [--password-stdin]
|
||||||
@@ -929,6 +931,7 @@ images/ article images comments.rs discussion chapters
|
|||||||
encode re-encode publish.rs BookOrbit + XTC
|
encode re-encode publish.rs BookOrbit + XTC
|
||||||
embed into the page server.rs ratings, OPDS
|
embed into the page server.rs ratings, OPDS
|
||||||
html.rs markup helpers db.rs SQLite
|
html.rs markup helpers db.rs SQLite
|
||||||
|
interests.rs standing interests and stored article matches
|
||||||
lock.rs one writer at a time (flock on <database_path>.lock)
|
lock.rs one writer at a time (flock on <database_path>.lock)
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
+221
-9
@@ -4,11 +4,16 @@
|
|||||||
//! the pipeline's shared records.
|
//! the pipeline's shared records.
|
||||||
|
|
||||||
use std::collections::{BTreeSet, HashMap};
|
use std::collections::{BTreeSet, HashMap};
|
||||||
|
use std::fmt::Write as _;
|
||||||
|
|
||||||
use anyhow::{Result, bail};
|
use anyhow::{Context as _, Result, bail};
|
||||||
use jiff::Timestamp;
|
use jiff::Timestamp;
|
||||||
|
use serde::Deserialize;
|
||||||
use sqlx::Row as _;
|
use sqlx::Row as _;
|
||||||
|
|
||||||
|
use crate::config::Config;
|
||||||
|
use crate::curate::llm::{LlmClient, Llms, provider_meters};
|
||||||
|
use crate::curate::profile;
|
||||||
use crate::curate::signals::TopInterest;
|
use crate::curate::signals::TopInterest;
|
||||||
use crate::db::{Db, fmt_ts};
|
use crate::db::{Db, fmt_ts};
|
||||||
use crate::types::ArticleId;
|
use crate::types::ArticleId;
|
||||||
@@ -234,29 +239,155 @@ pub async fn replace_matches(
|
|||||||
matches: &[(ArticleId, Vec<TopInterest>)],
|
matches: &[(ArticleId, Vec<TopInterest>)],
|
||||||
ids: &HashMap<String, i64>,
|
ids: &HashMap<String, i64>,
|
||||||
) -> Result<()> {
|
) -> Result<()> {
|
||||||
|
write_matches(db, run_id, matches, ids, true).await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Insert backfilled matches without disturbing rows a real run wrote.
|
||||||
|
pub async fn insert_matches_if_absent(
|
||||||
|
db: &Db,
|
||||||
|
matches: &[(ArticleId, Vec<TopInterest>)],
|
||||||
|
ids: &HashMap<String, i64>,
|
||||||
|
) -> Result<u64> {
|
||||||
|
write_matches(db, None, matches, ids, false).await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One transaction over the top interests of every article, either upserting
|
||||||
|
/// (a run's own rows) or leaving whatever is already stored alone (a backfill).
|
||||||
|
async fn write_matches(
|
||||||
|
db: &Db,
|
||||||
|
run_id: Option<i64>,
|
||||||
|
matches: &[(ArticleId, Vec<TopInterest>)],
|
||||||
|
ids: &HashMap<String, i64>,
|
||||||
|
replace: bool,
|
||||||
|
) -> Result<u64> {
|
||||||
|
let sql = if replace {
|
||||||
|
"INSERT INTO article_interests (article_id, interest_id, cos, z, run_id)
|
||||||
|
VALUES (?, ?, ?, ?, ?)
|
||||||
|
ON CONFLICT(article_id, interest_id) DO UPDATE SET
|
||||||
|
cos = excluded.cos, z = excluded.z, run_id = excluded.run_id"
|
||||||
|
} else {
|
||||||
|
"INSERT OR IGNORE INTO article_interests (article_id, interest_id, cos, z, run_id)
|
||||||
|
VALUES (?, ?, ?, ?, ?)"
|
||||||
|
};
|
||||||
let mut tx = db.pool().begin().await?;
|
let mut tx = db.pool().begin().await?;
|
||||||
|
let mut written = 0;
|
||||||
for (article_id, top_interests) in matches {
|
for (article_id, top_interests) in matches {
|
||||||
for top in top_interests {
|
for top in top_interests {
|
||||||
let Some(interest_id) = ids.get(&top.name) else {
|
let Some(interest_id) = ids.get(&top.name) else {
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
sqlx::query(
|
written += sqlx::query(sql)
|
||||||
"INSERT INTO article_interests (article_id, interest_id, cos, z, run_id)
|
|
||||||
VALUES (?, ?, ?, ?, ?)
|
|
||||||
ON CONFLICT(article_id, interest_id) DO UPDATE SET
|
|
||||||
cos = excluded.cos, z = excluded.z, run_id = excluded.run_id",
|
|
||||||
)
|
|
||||||
.bind(article_id)
|
.bind(article_id)
|
||||||
.bind(interest_id)
|
.bind(interest_id)
|
||||||
.bind(top.cos)
|
.bind(top.cos)
|
||||||
.bind(top.z)
|
.bind(top.z)
|
||||||
.bind(run_id)
|
.bind(run_id)
|
||||||
.execute(&mut *tx)
|
.execute(&mut *tx)
|
||||||
.await?;
|
.await?
|
||||||
|
.rows_affected();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
tx.commit().await?;
|
tx.commit().await?;
|
||||||
Ok(())
|
Ok(written)
|
||||||
|
}
|
||||||
|
|
||||||
|
const CATEGORIZE_PROMPT: &str = r#"TASK: file each new standing interest under one of the reader's interest categories.
|
||||||
|
Existing categories (reuse these names verbatim): {categories}
|
||||||
|
Create a new category only when none of the existing ones fits; a new category must be broad enough to hold several interests and named like the existing ones (two to five words, sentence case). Every interest gets exactly one category.
|
||||||
|
New interests: {interests}
|
||||||
|
Return JSON exactly: {"assignments":[{"interest":"…","category":"…"}]}"#;
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
struct CategorizeResponse {
|
||||||
|
#[serde(default)]
|
||||||
|
assignments: Vec<CategoryAssignment>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
struct CategoryAssignment {
|
||||||
|
interest: String,
|
||||||
|
category: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// File every currently uncategorized interest in one bulk-model call.
|
||||||
|
pub async fn categorize(config: &Config, db: &Db) -> Result<String> {
|
||||||
|
let pending = uncategorized(db).await?;
|
||||||
|
if pending.is_empty() {
|
||||||
|
return Ok("nothing to categorize".into());
|
||||||
|
}
|
||||||
|
let taste = profile::load_or_build(
|
||||||
|
db,
|
||||||
|
&config.profile_path,
|
||||||
|
config.curation.feedback.verdicts_in_prompt,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
let llms = Llms::from_config(config, taste.text, &provider_meters(config));
|
||||||
|
let llm = llms
|
||||||
|
.bulk
|
||||||
|
.as_ref()
|
||||||
|
.or_else(|| llms.editor_or_bulk())
|
||||||
|
.context("no LLM provider is available for interest categorization")?;
|
||||||
|
categorize_with_llm(db, &pending, llm).await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn categorize_with_llm(db: &Db, pending: &[Interest], llm: &LlmClient) -> Result<String> {
|
||||||
|
let all = list(db).await?;
|
||||||
|
let categories = all
|
||||||
|
.iter()
|
||||||
|
.filter_map(|interest| interest.category.as_deref())
|
||||||
|
.collect::<BTreeSet<_>>()
|
||||||
|
.into_iter()
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
let names = pending
|
||||||
|
.iter()
|
||||||
|
.map(|interest| interest.name.as_str())
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
let prompt = CATEGORIZE_PROMPT
|
||||||
|
.replace("{categories}", &categories.join(", "))
|
||||||
|
.replace("{interests}", &format!("\n{}", names.join("\n")));
|
||||||
|
let response: CategorizeResponse = llm.complete_json(&prompt, 0.2).await?;
|
||||||
|
|
||||||
|
let pending_by_name = pending
|
||||||
|
.iter()
|
||||||
|
.map(|interest| (interest.name.to_lowercase(), interest))
|
||||||
|
.collect::<HashMap<_, _>>();
|
||||||
|
let existing_categories = categories
|
||||||
|
.iter()
|
||||||
|
.map(|category| category.to_lowercase())
|
||||||
|
.collect::<BTreeSet<_>>();
|
||||||
|
let mut assigned = BTreeSet::new();
|
||||||
|
let mut new_categories = BTreeSet::new();
|
||||||
|
let now = Timestamp::now();
|
||||||
|
for assignment in response.assignments {
|
||||||
|
let Some(interest) = pending_by_name.get(&assignment.interest.trim().to_lowercase()) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let category = assignment.category.trim();
|
||||||
|
if !(1..=60).contains(&category.chars().count()) || !assigned.insert(interest.id) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
set_category(db, interest.id, Some(category), now).await?;
|
||||||
|
if !existing_categories.contains(&category.to_lowercase()) {
|
||||||
|
new_categories.insert(category.to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut message = format!(
|
||||||
|
"categorized {} ({} new categories:",
|
||||||
|
assigned.len(),
|
||||||
|
new_categories.len()
|
||||||
|
);
|
||||||
|
if !new_categories.is_empty() {
|
||||||
|
let _ = write!(
|
||||||
|
message,
|
||||||
|
" {}",
|
||||||
|
new_categories.into_iter().collect::<Vec<_>>().join(", ")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
message.push(')');
|
||||||
|
tracing::info!(%message);
|
||||||
|
Ok(message)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Stored matches for the requested articles.
|
/// Stored matches for the requested articles.
|
||||||
@@ -583,4 +714,85 @@ mod tests {
|
|||||||
let (_dir, db) = test_db().await;
|
let (_dir, db) = test_db().await;
|
||||||
assert!(matches_for_articles(&db, &[]).await.unwrap().is_empty());
|
assert!(matches_for_articles(&db, &[]).await.unwrap().is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn categorizer_short_circuits_without_uncategorized_interests() {
|
||||||
|
let (_dir, db) = test_db().await;
|
||||||
|
add(
|
||||||
|
&db,
|
||||||
|
"Databases",
|
||||||
|
Some("Software"),
|
||||||
|
ts("2026-09-12T12:00:00Z"),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
categorize(&Config::default(), &db).await.unwrap(),
|
||||||
|
"nothing to categorize"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn categorizer_assigns_known_names_and_tracks_new_categories() {
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use crate::config::ProviderConfig;
|
||||||
|
use crate::curate::llm::{MockBackend, UsageMeter};
|
||||||
|
use crate::types::TokenUsage;
|
||||||
|
|
||||||
|
let (_dir, db) = test_db().await;
|
||||||
|
let now = ts("2026-09-12T12:00:00Z");
|
||||||
|
add(&db, "Databases", Some("Software"), now).await.unwrap();
|
||||||
|
add(&db, "Rust macros", None, now).await.unwrap();
|
||||||
|
add(&db, "Wheel-thrown pottery", None, now).await.unwrap();
|
||||||
|
let pending = uncategorized(&db).await.unwrap();
|
||||||
|
|
||||||
|
let backend = Arc::new(MockBackend::new());
|
||||||
|
backend.push(
|
||||||
|
r#"{"assignments":[
|
||||||
|
{"interest":"RUST MACROS","category":" Software "},
|
||||||
|
{"interest":"Wheel-thrown pottery","category":"Creative crafts"},
|
||||||
|
{"interest":"Not in the batch","category":"Made up"}
|
||||||
|
]}"#,
|
||||||
|
TokenUsage::default(),
|
||||||
|
);
|
||||||
|
let llm = LlmClient::with_backend(
|
||||||
|
"mock",
|
||||||
|
"taste prompt".into(),
|
||||||
|
UsageMeter::for_provider(&ProviderConfig::deepseek()),
|
||||||
|
backend.clone(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let message = categorize_with_llm(&db, &pending, &llm).await.unwrap();
|
||||||
|
assert_eq!(message, "categorized 2 (1 new categories: Creative crafts)");
|
||||||
|
let stored = list(&db).await.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
stored
|
||||||
|
.iter()
|
||||||
|
.find(|interest| interest.name == "Rust macros")
|
||||||
|
.and_then(|interest| interest.category.as_deref()),
|
||||||
|
Some("Software")
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
stored
|
||||||
|
.iter()
|
||||||
|
.find(|interest| interest.name == "Wheel-thrown pottery")
|
||||||
|
.and_then(|interest| interest.category.as_deref()),
|
||||||
|
Some("Creative crafts")
|
||||||
|
);
|
||||||
|
assert_eq!(backend.calls(), 1);
|
||||||
|
let request = &backend.prompts()[0];
|
||||||
|
assert_eq!(request.temperature, 0.2);
|
||||||
|
assert!(request.json);
|
||||||
|
assert!(
|
||||||
|
request
|
||||||
|
.user
|
||||||
|
.contains("Existing categories (reuse these names verbatim): Software")
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
request
|
||||||
|
.user
|
||||||
|
.contains("New interests: \nRust macros\nWheel-thrown pottery")
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+20
-2
@@ -44,11 +44,13 @@ pub enum Job {
|
|||||||
FeaturesPrune,
|
FeaturesPrune,
|
||||||
/// `import-ratings` → process ratings-dashboard URL imports.
|
/// `import-ratings` → process ratings-dashboard URL imports.
|
||||||
ImportRatings,
|
ImportRatings,
|
||||||
|
/// `interests-categorize` → file uncategorized standing interests.
|
||||||
|
InterestsCategorize,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Job {
|
impl Job {
|
||||||
/// The catalogue in the order the Jobs page lists it.
|
/// The catalogue in the order the Jobs page lists it.
|
||||||
pub const CATALOGUE: [Job; 7] = [
|
pub const CATALOGUE: [Job; 8] = [
|
||||||
Job::Generate { date: None },
|
Job::Generate { date: None },
|
||||||
Job::DryRun,
|
Job::DryRun,
|
||||||
Job::ProfileRebuild,
|
Job::ProfileRebuild,
|
||||||
@@ -56,6 +58,7 @@ impl Job {
|
|||||||
Job::BackfillSocial,
|
Job::BackfillSocial,
|
||||||
Job::FeaturesPrune,
|
Job::FeaturesPrune,
|
||||||
Job::ImportRatings,
|
Job::ImportRatings,
|
||||||
|
Job::InterestsCategorize,
|
||||||
];
|
];
|
||||||
|
|
||||||
/// `^[a-z0-9-]+$`: the only characters a job (and so a unit instance) name
|
/// `^[a-z0-9-]+$`: the only characters a job (and so a unit instance) name
|
||||||
@@ -81,6 +84,7 @@ impl Job {
|
|||||||
"backfill-social" => Some(Job::BackfillSocial),
|
"backfill-social" => Some(Job::BackfillSocial),
|
||||||
"features-prune" => Some(Job::FeaturesPrune),
|
"features-prune" => Some(Job::FeaturesPrune),
|
||||||
"import-ratings" => Some(Job::ImportRatings),
|
"import-ratings" => Some(Job::ImportRatings),
|
||||||
|
"interests-categorize" => Some(Job::InterestsCategorize),
|
||||||
_ => {
|
_ => {
|
||||||
let date = name.strip_prefix("generate-")?;
|
let date = name.strip_prefix("generate-")?;
|
||||||
// Exactly `YYYY-MM-DD`; the round trip rejects `2026-9-3`.
|
// Exactly `YYYY-MM-DD`; the round trip rejects `2026-9-3`.
|
||||||
@@ -100,6 +104,7 @@ impl Job {
|
|||||||
Job::BackfillSocial => "backfill-social".into(),
|
Job::BackfillSocial => "backfill-social".into(),
|
||||||
Job::FeaturesPrune => "features-prune".into(),
|
Job::FeaturesPrune => "features-prune".into(),
|
||||||
Job::ImportRatings => "import-ratings".into(),
|
Job::ImportRatings => "import-ratings".into(),
|
||||||
|
Job::InterestsCategorize => "interests-categorize".into(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -128,6 +133,9 @@ impl Job {
|
|||||||
"Drop stale embeddings, old candidate telemetry and old assessments per the retention config."
|
"Drop stale embeddings, old candidate telemetry and old assessments per the retention config."
|
||||||
}
|
}
|
||||||
Job::ImportRatings => "Fetch, embed and rate the URLs queued from the Ratings page.",
|
Job::ImportRatings => "Fetch, embed and rate the URLs queued from the Ratings page.",
|
||||||
|
Job::InterestsCategorize => {
|
||||||
|
"File uncategorized interests under categories with the bulk model, creating new ones only when needed."
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -139,7 +147,7 @@ impl Job {
|
|||||||
Job::ProfileRebuild => Some("profile rebuild"),
|
Job::ProfileRebuild => Some("profile rebuild"),
|
||||||
Job::FeaturesBackfill => Some("features backfill"),
|
Job::FeaturesBackfill => Some("features backfill"),
|
||||||
Job::BackfillSocial => Some("backfill-social"),
|
Job::BackfillSocial => Some("backfill-social"),
|
||||||
Job::FeaturesPrune | Job::ImportRatings => None,
|
Job::FeaturesPrune | Job::ImportRatings | Job::InterestsCategorize => None,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -459,6 +467,16 @@ mod tests {
|
|||||||
Job::ImportRatings.description(),
|
Job::ImportRatings.description(),
|
||||||
"Fetch, embed and rate the URLs queued from the Ratings page."
|
"Fetch, embed and rate the URLs queued from the Ratings page."
|
||||||
);
|
);
|
||||||
|
assert_eq!(
|
||||||
|
Job::parse("interests-categorize"),
|
||||||
|
Some(Job::InterestsCategorize)
|
||||||
|
);
|
||||||
|
assert_eq!(Job::InterestsCategorize.takes_lock(), None);
|
||||||
|
assert_eq!(
|
||||||
|
Job::InterestsCategorize.description(),
|
||||||
|
"File uncategorized interests under categories with the bulk model, creating new ones only when needed."
|
||||||
|
);
|
||||||
|
assert!(!Job::InterestsCategorize.dangerous());
|
||||||
assert!(Job::parse("generate-2026-09-03").unwrap().dangerous());
|
assert!(Job::parse("generate-2026-09-03").unwrap().dangerous());
|
||||||
assert!(!Job::parse("dry-run").unwrap().dangerous());
|
assert!(!Job::parse("dry-run").unwrap().dangerous());
|
||||||
}
|
}
|
||||||
|
|||||||
+353
-5
@@ -3,6 +3,7 @@
|
|||||||
//! Everything of substance lives in the library (`src/lib.rs`); this binary only
|
//! Everything of substance lives in the library (`src/lib.rs`); this binary only
|
||||||
//! parses flags, loads config, opens the database and dispatches.
|
//! parses flags, loads config, opens the database and dispatches.
|
||||||
|
|
||||||
|
use std::collections::{HashMap, HashSet};
|
||||||
use std::io::Write as _;
|
use std::io::Write as _;
|
||||||
use std::path::PathBuf;
|
use std::path::PathBuf;
|
||||||
|
|
||||||
@@ -17,7 +18,7 @@ use daily_epub::db::Db;
|
|||||||
use daily_epub::pipeline::{self, GenerateOptions, GenerateOutcome};
|
use daily_epub::pipeline::{self, GenerateOptions, GenerateOutcome};
|
||||||
use daily_epub::report::{RunReport, VOYAGE_PROVIDER};
|
use daily_epub::report::{RunReport, VOYAGE_PROVIDER};
|
||||||
use daily_epub::types::{ArticleId, Vote};
|
use daily_epub::types::{ArticleId, Vote};
|
||||||
use daily_epub::{curate, discovery, http, imports, jobs, lock, rate, server, social};
|
use daily_epub::{curate, discovery, http, imports, interests, jobs, lock, rate, server, social};
|
||||||
|
|
||||||
/// A personalized daily newspaper, delivered as an EPUB.
|
/// A personalized daily newspaper, delivered as an EPUB.
|
||||||
#[derive(Debug, Parser)]
|
#[derive(Debug, Parser)]
|
||||||
@@ -64,6 +65,9 @@ enum Command {
|
|||||||
/// Feed subscription candidates (feed discovery plan §4 step 6).
|
/// Feed subscription candidates (feed discovery plan §4 step 6).
|
||||||
#[command(subcommand)]
|
#[command(subcommand)]
|
||||||
Feeds(FeedsCommand),
|
Feeds(FeedsCommand),
|
||||||
|
/// Import and backfill standing interests.
|
||||||
|
#[command(subcommand)]
|
||||||
|
Interests(InterestsCommand),
|
||||||
/// Operator jobs (what `daily-epub-job@<name>.service` runs).
|
/// Operator jobs (what `daily-epub-job@<name>.service` runs).
|
||||||
#[command(subcommand)]
|
#[command(subcommand)]
|
||||||
Job(JobCommand),
|
Job(JobCommand),
|
||||||
@@ -74,7 +78,8 @@ enum JobCommand {
|
|||||||
/// Run one catalogue job in-process and record it in the `jobs` table.
|
/// Run one catalogue job in-process and record it in the `jobs` table.
|
||||||
Run {
|
Run {
|
||||||
/// `generate`, `generate-YYYY-MM-DD`, `dry-run`, `profile-rebuild`,
|
/// `generate`, `generate-YYYY-MM-DD`, `dry-run`, `profile-rebuild`,
|
||||||
/// `features-backfill`, `backfill-social`, `features-prune` or `import-ratings`.
|
/// `features-backfill`, `backfill-social`, `features-prune`, `import-ratings`
|
||||||
|
/// or `interests-categorize`.
|
||||||
name: String,
|
name: String,
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
@@ -86,6 +91,24 @@ enum FeedsCommand {
|
|||||||
Discover(FeedsDiscoverArgs),
|
Discover(FeedsDiscoverArgs),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Subcommand)]
|
||||||
|
enum InterestsCommand {
|
||||||
|
/// Import standing interests from OPML and the profile's Interests section.
|
||||||
|
Import(InterestsImportArgs),
|
||||||
|
/// Match every compatible cached article embedding to standing interests.
|
||||||
|
Backfill,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, clap::Args)]
|
||||||
|
struct InterestsImportArgs {
|
||||||
|
/// OPML input (defaults to data/scour-interests.opml).
|
||||||
|
#[arg(long, value_name = "PATH")]
|
||||||
|
opml: Option<PathBuf>,
|
||||||
|
/// Profile input (defaults to profile_path from the configuration).
|
||||||
|
#[arg(long, value_name = "PATH")]
|
||||||
|
profile: Option<PathBuf>,
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Debug, clap::Args)]
|
#[derive(Debug, clap::Args)]
|
||||||
struct FeedsDiscoverArgs {
|
struct FeedsDiscoverArgs {
|
||||||
/// How far back to look for articles.
|
/// How far back to look for articles.
|
||||||
@@ -410,6 +433,10 @@ async fn main() -> Result<()> {
|
|||||||
let db = Db::open_and_migrate(&config.database_path).await?;
|
let db = Db::open_and_migrate(&config.database_path).await?;
|
||||||
println!("{}", cmd_feeds_discover(&config, &db, args).await?);
|
println!("{}", cmd_feeds_discover(&config, &db, args).await?);
|
||||||
}
|
}
|
||||||
|
Command::Interests(command) => {
|
||||||
|
let db = Db::open_and_migrate(&config.database_path).await?;
|
||||||
|
println!("{}", cmd_interests(&config, &db, command).await?);
|
||||||
|
}
|
||||||
Command::Job(JobCommand::Run { name }) => {
|
Command::Job(JobCommand::Run { name }) => {
|
||||||
let Some(job) = jobs::Job::parse(&name) else {
|
let Some(job) = jobs::Job::parse(&name) else {
|
||||||
eprintln!("unknown job {name:?}; the catalogue is:");
|
eprintln!("unknown job {name:?}; the catalogue is:");
|
||||||
@@ -435,8 +462,8 @@ async fn main() -> Result<()> {
|
|||||||
|
|
||||||
/// The commands that write the database and provider budgets and so hold the
|
/// The commands that write the database and provider budgets and so hold the
|
||||||
/// run lock (§5): `generate`, `profile rebuild`, `features backfill`,
|
/// run lock (§5): `generate`, `profile rebuild`, `features backfill`,
|
||||||
/// `backfill-social`, and a `job run` of any of them. Everything else is
|
/// `backfill-social`, `interests backfill`, and a `job run` of any of them.
|
||||||
/// read-only or its own writer.
|
/// Everything else is read-only or its own writer.
|
||||||
fn lock_holder(command: &Command) -> Option<&'static str> {
|
fn lock_holder(command: &Command) -> Option<&'static str> {
|
||||||
match command {
|
match command {
|
||||||
Command::Generate(_) => Some("generate"),
|
Command::Generate(_) => Some("generate"),
|
||||||
@@ -445,6 +472,7 @@ fn lock_holder(command: &Command) -> Option<&'static str> {
|
|||||||
Command::Profile(ProfileCommand::Rebuild) => Some("profile rebuild"),
|
Command::Profile(ProfileCommand::Rebuild) => Some("profile rebuild"),
|
||||||
Command::Features(FeaturesCommand::Backfill(_)) => Some("features backfill"),
|
Command::Features(FeaturesCommand::Backfill(_)) => Some("features backfill"),
|
||||||
Command::BackfillSocial(_) => Some("backfill-social"),
|
Command::BackfillSocial(_) => Some("backfill-social"),
|
||||||
|
Command::Interests(InterestsCommand::Backfill) => Some("interests backfill"),
|
||||||
// An unknown name takes no lock; the dispatch exits 2 before opening
|
// An unknown name takes no lock; the dispatch exits 2 before opening
|
||||||
// the database.
|
// the database.
|
||||||
Command::Job(JobCommand::Run { name }) => {
|
Command::Job(JobCommand::Run { name }) => {
|
||||||
@@ -457,7 +485,8 @@ fn lock_holder(command: &Command) -> Option<&'static str> {
|
|||||||
| Command::Features(FeaturesCommand::Prune)
|
| Command::Features(FeaturesCommand::Prune)
|
||||||
| Command::Db(_)
|
| Command::Db(_)
|
||||||
| Command::Config(_)
|
| Command::Config(_)
|
||||||
| Command::Users(_) => None,
|
| Command::Users(_)
|
||||||
|
| Command::Interests(InterestsCommand::Import(_)) => None,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -494,6 +523,145 @@ async fn cmd_feeds_discover(config: &Config, db: &Db, args: FeedsDiscoverArgs) -
|
|||||||
))
|
))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn cmd_interests(config: &Config, db: &Db, command: InterestsCommand) -> Result<String> {
|
||||||
|
match command {
|
||||||
|
InterestsCommand::Import(args) => cmd_interests_import(config, db, args).await,
|
||||||
|
InterestsCommand::Backfill => cmd_interests_backfill(config, db).await,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn cmd_interests_import(
|
||||||
|
config: &Config,
|
||||||
|
db: &Db,
|
||||||
|
args: InterestsImportArgs,
|
||||||
|
) -> Result<String> {
|
||||||
|
let opml_path = args
|
||||||
|
.opml
|
||||||
|
.unwrap_or_else(|| PathBuf::from("data/scour-interests.opml"));
|
||||||
|
let profile_path = args.profile.unwrap_or_else(|| config.profile_path.clone());
|
||||||
|
let opml = std::fs::read_to_string(&opml_path)
|
||||||
|
.with_context(|| format!("reading interests OPML at {}", opml_path.display()))?;
|
||||||
|
let profile = curate::profile::load_profile(&profile_path)?;
|
||||||
|
let mut seen = HashSet::new();
|
||||||
|
let names = interests::parse_opml(&opml)
|
||||||
|
.into_iter()
|
||||||
|
.chain(profile.interests)
|
||||||
|
.filter(|name| seen.insert(name.to_lowercase()))
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
|
||||||
|
let now = jiff::Timestamp::now();
|
||||||
|
let mut added = Vec::new();
|
||||||
|
let mut skipped = 0;
|
||||||
|
for name in names {
|
||||||
|
match interests::add(db, &name, None, now).await? {
|
||||||
|
interests::AddOutcome::Added(id) => added.push((id, name)),
|
||||||
|
interests::AddOutcome::Duplicate => skipped += 1,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let new_names = added
|
||||||
|
.iter()
|
||||||
|
.map(|(_, name)| name.clone())
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
let categories = curate::profile::themes::group_into_themes(&new_names)
|
||||||
|
.into_iter()
|
||||||
|
.flat_map(|(category, members)| {
|
||||||
|
members.into_iter().map(move |name| {
|
||||||
|
let category = (category != "Other standing interests").then(|| category.clone());
|
||||||
|
(name.to_lowercase(), category)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect::<HashMap<_, _>>();
|
||||||
|
let mut uncategorized = 0;
|
||||||
|
for (id, name) in &added {
|
||||||
|
match categories
|
||||||
|
.get(&name.to_lowercase())
|
||||||
|
.and_then(Option::as_deref)
|
||||||
|
{
|
||||||
|
Some(category) => interests::set_category(db, *id, Some(category), now).await?,
|
||||||
|
None => uncategorized += 1,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(format!(
|
||||||
|
"imported {}, skipped {} existing, {} left for the categorizer",
|
||||||
|
added.len(),
|
||||||
|
skipped,
|
||||||
|
uncategorized
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Backfill uses z-scores over the whole compatible cache as a stand-in for a
|
||||||
|
/// run's per-day article cohort.
|
||||||
|
async fn cmd_interests_backfill(config: &Config, db: &Db) -> Result<String> {
|
||||||
|
use sqlx::Row as _;
|
||||||
|
|
||||||
|
let mut article_embeddings = HashMap::new();
|
||||||
|
let mut after = i64::MIN;
|
||||||
|
loop {
|
||||||
|
let rows = sqlx::query(
|
||||||
|
"SELECT article_id, embedding FROM article_embeddings
|
||||||
|
WHERE model = ? AND dimension = ? AND article_id > ?
|
||||||
|
ORDER BY article_id LIMIT 500",
|
||||||
|
)
|
||||||
|
.bind(&config.voyage.model)
|
||||||
|
.bind(config.voyage.output_dimension as i64)
|
||||||
|
.bind(after)
|
||||||
|
.fetch_all(db.pool())
|
||||||
|
.await?;
|
||||||
|
if rows.is_empty() {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
for row in &rows {
|
||||||
|
let article_id: ArticleId = row.get("article_id");
|
||||||
|
after = article_id;
|
||||||
|
match embedding::decode_blob(
|
||||||
|
&row.get::<Vec<u8>, _>("embedding"),
|
||||||
|
config.voyage.output_dimension,
|
||||||
|
) {
|
||||||
|
Ok(vector) => {
|
||||||
|
article_embeddings.insert(article_id, vector);
|
||||||
|
}
|
||||||
|
Err(error) => {
|
||||||
|
tracing::warn!(article_id, %error, "ignoring a malformed cached embedding")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let rows = interests::list(db).await?;
|
||||||
|
let names = rows
|
||||||
|
.iter()
|
||||||
|
.map(|interest| interest.name.clone())
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
let ids = rows
|
||||||
|
.into_iter()
|
||||||
|
.map(|interest| (interest.name, interest.id))
|
||||||
|
.collect::<HashMap<_, _>>();
|
||||||
|
let service = if config.voyage.enabled {
|
||||||
|
embedding::EmbeddingService::real(db.clone(), config.voyage.clone())
|
||||||
|
.context("building the Voyage client")?
|
||||||
|
} else {
|
||||||
|
embedding::EmbeddingService::cached_only(db.clone(), config.voyage.clone())
|
||||||
|
};
|
||||||
|
let interest_embeddings = service.interests(&names).await?;
|
||||||
|
let mut matches = curate::signals::interest_matches(&article_embeddings, &interest_embeddings)
|
||||||
|
.into_iter()
|
||||||
|
.filter_map(|(article_id, matched)| {
|
||||||
|
(!matched.top_interests.is_empty()).then_some((article_id, matched.top_interests))
|
||||||
|
})
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
matches.sort_by_key(|(article_id, _)| *article_id);
|
||||||
|
let inserted = interests::insert_matches_if_absent(db, &matches, &ids).await?;
|
||||||
|
if config.voyage.enabled {
|
||||||
|
Ok(format!("wrote {inserted} interest match rows"))
|
||||||
|
} else {
|
||||||
|
Ok(format!(
|
||||||
|
"voyage disabled; used cached interest vectors and wrote {inserted} interest match rows"
|
||||||
|
))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
async fn cmd_users(db: &Db, command: UsersCommand) -> Result<()> {
|
async fn cmd_users(db: &Db, command: UsersCommand) -> Result<()> {
|
||||||
use daily_epub::web::users;
|
use daily_epub::web::users;
|
||||||
match command {
|
match command {
|
||||||
@@ -1006,6 +1174,7 @@ async fn run_job(config: &Config, db: &Db, job: &jobs::Job) -> Result<(String, O
|
|||||||
None,
|
None,
|
||||||
)),
|
)),
|
||||||
jobs::Job::ImportRatings => Ok((imports::run(config, db).await?, None)),
|
jobs::Job::ImportRatings => Ok((imports::run(config, db).await?, None)),
|
||||||
|
jobs::Job::InterestsCategorize => Ok((interests::categorize(config, db).await?, None)),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1149,6 +1318,10 @@ mod tests {
|
|||||||
lock_holder(&parse(&["backfill-social"])),
|
lock_holder(&parse(&["backfill-social"])),
|
||||||
Some("backfill-social")
|
Some("backfill-social")
|
||||||
);
|
);
|
||||||
|
assert_eq!(
|
||||||
|
lock_holder(&parse(&["interests", "backfill"])),
|
||||||
|
Some("interests backfill")
|
||||||
|
);
|
||||||
for args in [
|
for args in [
|
||||||
vec!["serve"],
|
vec!["serve"],
|
||||||
vec!["explain", "--date", "2026-09-02", "--near-misses"],
|
vec!["explain", "--date", "2026-09-02", "--near-misses"],
|
||||||
@@ -1157,7 +1330,9 @@ mod tests {
|
|||||||
vec!["db", "migrate"],
|
vec!["db", "migrate"],
|
||||||
vec!["features", "prune"],
|
vec!["features", "prune"],
|
||||||
vec!["config", "check"],
|
vec!["config", "check"],
|
||||||
|
vec!["interests", "import"],
|
||||||
vec!["job", "run", "features-prune"],
|
vec!["job", "run", "features-prune"],
|
||||||
|
vec!["job", "run", "interests-categorize"],
|
||||||
vec!["job", "run", "not-a-job"],
|
vec!["job", "run", "not-a-job"],
|
||||||
] {
|
] {
|
||||||
assert_eq!(lock_holder(&parse(&args)), None, "{args:?}");
|
assert_eq!(lock_holder(&parse(&args)), None, "{args:?}");
|
||||||
@@ -1209,6 +1384,34 @@ mod tests {
|
|||||||
assert!(Cli::try_parse_from(["daily-epub", "job"]).is_err());
|
assert!(Cli::try_parse_from(["daily-epub", "job"]).is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parses_interests_commands() {
|
||||||
|
match Cli::try_parse_from([
|
||||||
|
"daily-epub",
|
||||||
|
"interests",
|
||||||
|
"import",
|
||||||
|
"--opml",
|
||||||
|
"/tmp/interests.opml",
|
||||||
|
"--profile",
|
||||||
|
"/tmp/profile.md",
|
||||||
|
])
|
||||||
|
.unwrap()
|
||||||
|
.command
|
||||||
|
{
|
||||||
|
Command::Interests(InterestsCommand::Import(args)) => {
|
||||||
|
assert_eq!(args.opml, Some(PathBuf::from("/tmp/interests.opml")));
|
||||||
|
assert_eq!(args.profile, Some(PathBuf::from("/tmp/profile.md")));
|
||||||
|
}
|
||||||
|
other => panic!("expected interests import, got {other:?}"),
|
||||||
|
}
|
||||||
|
assert!(matches!(
|
||||||
|
Cli::try_parse_from(["daily-epub", "interests", "backfill"])
|
||||||
|
.unwrap()
|
||||||
|
.command,
|
||||||
|
Command::Interests(InterestsCommand::Backfill)
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
/// Dashboard plan §17 "Jobs": `job run` flips the dashboard's `requested`
|
/// Dashboard plan §17 "Jobs": `job run` flips the dashboard's `requested`
|
||||||
/// row to `running` and then `ok` with the command's message, in-process
|
/// row to `running` and then `ok` with the command's message, in-process
|
||||||
/// and without systemd.
|
/// and without systemd.
|
||||||
@@ -1495,4 +1698,149 @@ mod tests {
|
|||||||
assert_eq!(rows[1].get::<f64, _>("value"), 0.0);
|
assert_eq!(rows[1].get::<f64, _>("value"), 0.0);
|
||||||
assert!(db.current_ratings(36500).await.unwrap().is_empty());
|
assert!(db.current_ratings(36500).await.unwrap().is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn interests_import_is_idempotent_and_preserves_theme_groups() {
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let db = Db::open_and_migrate(&dir.path().join("interests.db"))
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let opml_path = dir.path().join("interests.opml");
|
||||||
|
let profile_path = dir.path().join("profile.md");
|
||||||
|
std::fs::write(
|
||||||
|
&opml_path,
|
||||||
|
r#"<opml><body><outline text="Rust"/><outline text="Flibbertigibbet"/></body></opml>"#,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
std::fs::write(
|
||||||
|
&profile_path,
|
||||||
|
"# Reader\n\n## Interests\n- rust\n- Postgres query plans\n",
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let config = Config {
|
||||||
|
profile_path: profile_path.clone(),
|
||||||
|
..Config::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let args = || InterestsImportArgs {
|
||||||
|
opml: Some(opml_path.clone()),
|
||||||
|
profile: None,
|
||||||
|
};
|
||||||
|
assert_eq!(
|
||||||
|
cmd_interests_import(&config, &db, args()).await.unwrap(),
|
||||||
|
"imported 3, skipped 0 existing, 1 left for the categorizer"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
cmd_interests_import(&config, &db, args()).await.unwrap(),
|
||||||
|
"imported 0, skipped 3 existing, 0 left for the categorizer"
|
||||||
|
);
|
||||||
|
|
||||||
|
let rows = interests::list(&db).await.unwrap();
|
||||||
|
assert_eq!(rows.len(), 3);
|
||||||
|
assert_eq!(
|
||||||
|
rows.iter()
|
||||||
|
.find(|interest| interest.name == "Rust")
|
||||||
|
.and_then(|interest| interest.category.as_deref()),
|
||||||
|
Some("Systems & languages")
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
rows.iter()
|
||||||
|
.find(|interest| interest.name == "Flibbertigibbet")
|
||||||
|
.and_then(|interest| interest.category.as_deref()),
|
||||||
|
None
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn interests_backfill_inserts_matches_without_replacing_run_rows() {
|
||||||
|
use daily_epub::curate::embedding::encode_blob;
|
||||||
|
use daily_epub::curate::signals::TopInterest;
|
||||||
|
use sqlx::Row as _;
|
||||||
|
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let db = Db::open_and_migrate(&dir.path().join("interests.db"))
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
sqlx::query(
|
||||||
|
"INSERT INTO articles (id, canonical_url, title, first_seen) VALUES
|
||||||
|
(1, 'https://example.com/1', 'One', '2026-09-12T00:00:00Z'),
|
||||||
|
(2, 'https://example.com/2', 'Two', '2026-09-12T00:00:00Z')",
|
||||||
|
)
|
||||||
|
.execute(db.pool())
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let now = jiff::Timestamp::now();
|
||||||
|
let interests::AddOutcome::Added(one_id) =
|
||||||
|
interests::add(&db, "First axis", None, now).await.unwrap()
|
||||||
|
else {
|
||||||
|
unreachable!();
|
||||||
|
};
|
||||||
|
let interests::AddOutcome::Added(two_id) =
|
||||||
|
interests::add(&db, "Second axis", None, now).await.unwrap()
|
||||||
|
else {
|
||||||
|
unreachable!();
|
||||||
|
};
|
||||||
|
for (article_id, vector) in [(1, [1.0_f32, 0.0]), (2, [0.0, 1.0])] {
|
||||||
|
sqlx::query(
|
||||||
|
"INSERT INTO article_embeddings
|
||||||
|
(article_id, model, dimension, input_hash, embedding, created_at)
|
||||||
|
VALUES (?, 'test-model', 2, 'hash', ?, '2026-09-12T00:00:00Z')",
|
||||||
|
)
|
||||||
|
.bind(article_id)
|
||||||
|
.bind(encode_blob(&vector).unwrap())
|
||||||
|
.execute(db.pool())
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
for (name, vector) in [("First axis", [1.0_f32, 0.0]), ("Second axis", [0.0, 1.0])] {
|
||||||
|
sqlx::query(
|
||||||
|
"INSERT INTO interest_embeddings
|
||||||
|
(interest, model, dimension, embedding, created_at)
|
||||||
|
VALUES (?, 'test-model', 2, ?, '2026-09-12T00:00:00Z')",
|
||||||
|
)
|
||||||
|
.bind(name)
|
||||||
|
.bind(encode_blob(&vector).unwrap())
|
||||||
|
.execute(db.pool())
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
interests::replace_matches(
|
||||||
|
&db,
|
||||||
|
Some(77),
|
||||||
|
&[(
|
||||||
|
1,
|
||||||
|
vec![TopInterest {
|
||||||
|
name: "First axis".into(),
|
||||||
|
cos: 0.75,
|
||||||
|
z: 1.5,
|
||||||
|
}],
|
||||||
|
)],
|
||||||
|
&HashMap::from([("First axis".to_string(), one_id)]),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
let mut config = Config::default();
|
||||||
|
config.voyage.enabled = false;
|
||||||
|
config.voyage.model = "test-model".into();
|
||||||
|
config.voyage.output_dimension = 2;
|
||||||
|
|
||||||
|
let message = cmd_interests_backfill(&config, &db).await.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
message,
|
||||||
|
"voyage disabled; used cached interest vectors and wrote 1 interest match rows"
|
||||||
|
);
|
||||||
|
let rows = sqlx::query(
|
||||||
|
"SELECT article_id, interest_id, run_id, cos, z
|
||||||
|
FROM article_interests ORDER BY article_id",
|
||||||
|
)
|
||||||
|
.fetch_all(db.pool())
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(rows.len(), 2);
|
||||||
|
assert_eq!(rows[0].get::<i64, _>("interest_id"), one_id);
|
||||||
|
assert_eq!(rows[0].get::<Option<i64>, _>("run_id"), Some(77));
|
||||||
|
assert_eq!(rows[0].get::<f64, _>("cos"), 0.75);
|
||||||
|
assert_eq!(rows[1].get::<i64, _>("interest_id"), two_id);
|
||||||
|
assert_eq!(rows[1].get::<Option<i64>, _>("run_id"), None);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+70
-1
@@ -943,13 +943,21 @@ async fn prepare_features(
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
report.counts.embedded = article_embeddings.len() as i64;
|
report.counts.embedded = article_embeddings.len() as i64;
|
||||||
let interest_names = match interests::names(db).await {
|
let interest_rows = match interests::list(db).await {
|
||||||
Ok(interests) => interests,
|
Ok(interests) => interests,
|
||||||
Err(error) => {
|
Err(error) => {
|
||||||
tracing::warn!(%error, "could not load standing interests for embeddings");
|
tracing::warn!(%error, "could not load standing interests for embeddings");
|
||||||
Vec::new()
|
Vec::new()
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
let interest_names = interest_rows
|
||||||
|
.iter()
|
||||||
|
.map(|interest| interest.name.clone())
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
let interest_ids = interest_rows
|
||||||
|
.into_iter()
|
||||||
|
.map(|interest| (interest.name, interest.id))
|
||||||
|
.collect::<HashMap<_, _>>();
|
||||||
let interest_embeddings = match service.interests(&interest_names).await {
|
let interest_embeddings = match service.interests(&interest_names).await {
|
||||||
Ok(embeddings) => embeddings,
|
Ok(embeddings) => embeddings,
|
||||||
Err(error) => {
|
Err(error) => {
|
||||||
@@ -1008,6 +1016,21 @@ async fn prepare_features(
|
|||||||
.remove(&candidate.article.id)
|
.remove(&candidate.article.id)
|
||||||
.unwrap_or_else(|| signals::Signals::baseline(&candidate.article));
|
.unwrap_or_else(|| signals::Signals::baseline(&candidate.article));
|
||||||
}
|
}
|
||||||
|
let matches = candidates
|
||||||
|
.iter()
|
||||||
|
.filter(|candidate| !candidate.signals.top_interests.is_empty())
|
||||||
|
.map(|candidate| {
|
||||||
|
(
|
||||||
|
candidate.article.id,
|
||||||
|
candidate.signals.top_interests.clone(),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
if let Err(error) =
|
||||||
|
interests::replace_matches(db, Some(ctx.run_id), &matches, &interest_ids).await
|
||||||
|
{
|
||||||
|
report.warn(format!("could not record interest matches: {error}"));
|
||||||
|
}
|
||||||
if let Err(error) = record_candidates(ctx, candidates).await {
|
if let Err(error) = record_candidates(ctx, candidates).await {
|
||||||
report.warn(format!("could not record eligible candidates: {error}"));
|
report.warn(format!("could not record eligible candidates: {error}"));
|
||||||
}
|
}
|
||||||
@@ -1159,6 +1182,39 @@ async fn build_llms(
|
|||||||
let make_clients = |prompt: String| Llms::from_config(ctx.config, prompt, meters);
|
let make_clients = |prompt: String| Llms::from_config(ctx.config, prompt, meters);
|
||||||
|
|
||||||
let mut llms = make_clients(profile.text);
|
let mut llms = make_clients(profile.text);
|
||||||
|
// Categorizing regroups the prompt's standing interests, so the prompt is
|
||||||
|
// reloaded and the clients remade before anything uses them.
|
||||||
|
if !llms.is_empty() {
|
||||||
|
match interests::uncategorized(ctx.db).await {
|
||||||
|
Ok(pending) if !pending.is_empty() => {
|
||||||
|
match interests::categorize(ctx.config, ctx.db).await {
|
||||||
|
Ok(message) => {
|
||||||
|
tracing::info!(%message);
|
||||||
|
match profile::load_or_build(
|
||||||
|
ctx.db,
|
||||||
|
&ctx.config.profile_path,
|
||||||
|
ctx.config.curation.feedback.verdicts_in_prompt,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(regrouped) => {
|
||||||
|
report.counts.verdicts_in_prompt = regrouped.verdicts as i64;
|
||||||
|
llms = make_clients(regrouped.text);
|
||||||
|
}
|
||||||
|
Err(error) => report.warn(format!(
|
||||||
|
"could not rebuild the taste prompt after categorizing: {error:#}"
|
||||||
|
)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(error) => report.warn(format!("interest categorization failed: {error:#}")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(error) => report.warn(format!(
|
||||||
|
"could not check for uncategorized interests: {error:#}"
|
||||||
|
)),
|
||||||
|
}
|
||||||
|
}
|
||||||
let Some(rebuild_client) = llms.editor_or_bulk() else {
|
let Some(rebuild_client) = llms.editor_or_bulk() else {
|
||||||
report.warn("no LLM provider is available; curating heuristically");
|
report.warn("no LLM provider is available; curating heuristically");
|
||||||
return llms;
|
return llms;
|
||||||
@@ -1719,6 +1775,19 @@ mod tests {
|
|||||||
);
|
);
|
||||||
assert!(signals.preliminary.is_some());
|
assert!(signals.preliminary.is_some());
|
||||||
|
|
||||||
|
let expected_matches = features
|
||||||
|
.iter()
|
||||||
|
.map(|candidate| candidate.signals.top_interests.len() as i64)
|
||||||
|
.sum::<i64>();
|
||||||
|
let stored_matches: i64 =
|
||||||
|
sqlx::query_scalar("SELECT COUNT(*) FROM article_interests WHERE run_id = ?")
|
||||||
|
.bind(h.run_id)
|
||||||
|
.fetch_one(h.db.pool())
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
assert!(expected_matches > 0);
|
||||||
|
assert_eq!(stored_matches, expected_matches);
|
||||||
|
|
||||||
let rows = stage_rows(&h.db, h.run_id).await;
|
let rows = stage_rows(&h.db, h.run_id).await;
|
||||||
assert_eq!(rows.len(), 4, "one row per considered article");
|
assert_eq!(rows.len(), 4, "one row per considered article");
|
||||||
assert_eq!(rows[&blocked].0, "excluded");
|
assert_eq!(rows[&blocked].0, "excluded");
|
||||||
|
|||||||
Reference in New Issue
Block a user