diff --git a/CHANGELOG.md b/CHANGELOG.md index ab03ab4..6447e00 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -48,6 +48,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed +- Checking a feed with a long history is much quicker: items already stored are no longer written again on every check. - A scan no longer asks a feed's website for its icon every time; it asks again when the artwork changes or you refresh the feed. - The feed list loads several times faster: it was asking the database six questions per feed. - Artwork a feed offers only over plain http shows on the https site too. diff --git a/src/db.rs b/src/db.rs index d1a5dc9..cea91f8 100644 --- a/src/db.rs +++ b/src/db.rs @@ -1779,6 +1779,27 @@ impl Db { /// A feed's enclosures skipped by one of its filters, by URL, with the reason: the verdicts a /// change of settings can overturn. A torrent held back while torrents are off is not a /// filter's call. + /// The guids of a feed's stored items and the URLs of its files: a scan inserts only what is + /// not among them. Inserting every item it read, stored or not, cost a round trip each, about + /// 10 ms: 13 s a scan for Clarkesworld's 1200 items, and 117 s of a full 315 s scan (#96). + #[tracing::instrument(skip_all)] + pub async fn stored_items( + &self, + feed_id: &str, + ) -> Result<(std::collections::HashSet, std::collections::HashSet)> { + let col = |sql: &'static str, name: &'static str| async move { + self.rows(sql, vec![feed_id.into()]) + .await? + .iter() + .map(|r| Ok(r.try_get::("", name)?)) + .collect::>>() + }; + Ok(( + col("SELECT guid FROM entries WHERE feed_id = $1", "guid").await?, + col("SELECT url FROM enclosures WHERE feed_id = $1", "url").await?, + )) + } + #[tracing::instrument(skip_all)] pub async fn skipped_by_filter(&self, feed_id: &str) -> Result> { self.rows( @@ -1913,6 +1934,19 @@ mod tests { assert_eq!((list["f"].unread, list["g"].unread), (1, 1)); // a read, c hidden; sam's read is sam's } + #[tokio::test] + async fn a_scan_knows_a_feeds_stored_items_and_files() { + let db = Db::memory().await.unwrap(); + db.exec_for_test( + "INSERT INTO entries (feed_id, guid, first_seen) VALUES ('f','a',0),('f','b',0),('g','c',0); + INSERT INTO enclosures (id, feed_id, guid, url, state) VALUES (1,'f','a','u1','pending'),(2,'g','c','u2','pending');", + ).await + .unwrap(); + let (items, files) = db.stored_items("f").await.unwrap(); + assert_eq!(items, ["a", "b"].map(String::from).into()); + assert_eq!(files, ["u1"].map(String::from).into()); // u2 is g's: left to the insert to find + } + #[tokio::test] async fn only_artwork_a_feed_names_is_fetched_for_the_page() { let db = Db::memory().await.unwrap(); diff --git a/src/main.rs b/src/main.rs index f4b33a3..53dab98 100644 --- a/src/main.rs +++ b/src/main.rs @@ -1316,16 +1316,19 @@ async fn scan_one( // discovery, it outlived the setting behind it, and allowing explicit items afterwards // changed nothing however often the feed was scanned. let skipped = ctx.db.skipped_by_filter(id).await?; + let (known_items, known_files) = ctx.db.stored_items(id).await?; let mut scan = Scan::default(); // Its own span: the time a feed spends after its fetch was untraced (#96). let store = tracing::info_span!("store", items = parsed.entries.len()); tracing::Instrument::instrument(async { for entry in &parsed.entries { - if ctx.db.record_entry(id, entry).await? { + // Only what is not stored yet is inserted; the insert would find the rest and do nothing. + if !known_items.contains(&entry.guid) && ctx.db.record_entry(id, entry).await? { scan.new_entries += 1; } for enc in &entry.enclosures { - let was = if ctx.db.record_enclosure(id, &entry.guid, enc).await? { + // A URL not among this feed's files may still be another feed's: the insert says. + let was = if !known_files.contains(&enc.url) && ctx.db.record_enclosure(id, &entry.guid, enc).await? { None } else if let Some(reason) = skipped.get(&enc.url) { Some(reason.as_str())