A download limit means a show's newest episodes, not a pace (#97)
pending() took the newest files still pending, up to the limit, so once a show's latest three
were down, each full read of its feed took the three before them, working back through its
whole history. In production 4420 files (about 310 GB) were queued this way across 12 shows,
all on the default limit of 3, which is meant as "the latest three". It now takes only from the
feed's newest `limit` items with a file. 0, unlimited, still takes the whole back catalogue:
that is how the shows kept as an archive are set, along with limits of 100 and 10000.
The settings' wording followed the old behaviour ("The rest wait for the next scan"); the field
is now "Newest episodes to download", and says what 0 does.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
41
src/db.rs
41
src/db.rs
@@ -623,16 +623,26 @@ pub struct Pending {
|
||||
}
|
||||
|
||||
impl Db {
|
||||
/// The download queue is the table, not the parse result: an enclosure held back by
|
||||
/// `max_new_per_check` is simply picked up by the next scan, in feed order.
|
||||
/// The download queue is the table, not the parse result: what a scan could not finish is
|
||||
/// picked up by the next. Only from the feed's `limit` newest items with a file, though: a
|
||||
/// limit of 3 means the three latest episodes. Taking the newest three still pending, each
|
||||
/// full read of a feed took the three before the ones already downloaded, working back
|
||||
/// through every show's history: 4420 files queued across 12 shows on the default of 3
|
||||
/// (#97). Unlimited, 0 in the settings, is the whole back catalogue, for an archive. An item
|
||||
/// whose files were all skipped by a filter does not hold one of the places.
|
||||
#[tracing::instrument(skip_all)]
|
||||
pub async fn pending(&self, feed_id: &str, limit: usize) -> Result<Vec<Pending>> {
|
||||
// Newest first: a cap of 3 should mean the three latest episodes, not the three that
|
||||
// happen to have been recorded first.
|
||||
self.rows(
|
||||
"SELECT x.id, x.url, x.mime FROM enclosures x
|
||||
JOIN entries e ON e.feed_id = x.feed_id AND e.guid = x.guid
|
||||
WHERE x.feed_id = $1 AND x.state = 'pending'
|
||||
AND e.guid IN (SELECT n.guid FROM entries n
|
||||
WHERE n.feed_id = $1
|
||||
AND EXISTS (SELECT 1 FROM enclosures y
|
||||
WHERE y.feed_id = n.feed_id AND y.guid = n.guid
|
||||
AND y.state <> 'skipped')
|
||||
ORDER BY coalesce(n.published, n.first_seen) DESC, n.guid DESC
|
||||
LIMIT $2)
|
||||
ORDER BY coalesce(e.published, e.first_seen) DESC, x.id DESC
|
||||
LIMIT $2",
|
||||
// Unlimited is usize::MAX, which `as i64` makes -1: SQLite reads LIMIT -1 as no
|
||||
@@ -1936,6 +1946,29 @@ mod tests {
|
||||
assert_eq!((list["f"].unread, list["g"].unread), (1, 1)); // a read, c hidden; sam's read is sam's
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn a_limit_takes_the_newest_episodes_not_the_back_catalogue() {
|
||||
let db = Db::memory().await.unwrap();
|
||||
db.exec_for_test(
|
||||
"INSERT INTO entries (feed_id, guid, first_seen, published) VALUES
|
||||
('f','e1',0,100),('f','e2',0,200),('f','e3',0,300),('f','e4',0,400),('f','e5',0,500);
|
||||
INSERT INTO enclosures (id, feed_id, guid, url, state, path) VALUES
|
||||
(1,'f','e1','u1','pending',null),(2,'f','e2','u2','pending',null),
|
||||
(3,'f','e3','u3','done','/tmp/3'),(4,'f','e4','u4','done','/tmp/4'),
|
||||
(5,'f','e5','u5','skipped',null);",
|
||||
).await
|
||||
.unwrap();
|
||||
let ids = |p: Vec<Pending>| p.into_iter().map(|p| p.id).collect::<Vec<_>>();
|
||||
// The newest three with a file worth having are e4, e3 and e2 (e5's was skipped): only
|
||||
// e2 is waiting. e1 is back catalogue.
|
||||
assert_eq!(ids(db.pending("f", 3).await.unwrap()), vec![2]);
|
||||
// Once e2 is down, nothing: before #97 the next scan took e1.
|
||||
db.exec_for_test("UPDATE enclosures SET state = 'done', path = '/tmp/2' WHERE id = 2").await.unwrap();
|
||||
assert!(db.pending("f", 3).await.unwrap().is_empty());
|
||||
// Unlimited is the archive: everything still waiting.
|
||||
assert_eq!(ids(db.pending("f", usize::MAX).await.unwrap()), vec![1]);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn an_unlimited_queue_takes_everything_waiting() {
|
||||
let db = Db::memory().await.unwrap();
|
||||
|
||||
Reference in New Issue
Block a user