diff --git a/CHANGELOG.md b/CHANGELOG.md index c99747c..c2a184f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- Artwork is kept on disk once shown and loads from iPX, not from each publisher's server: fast after the first time, and still there when the publisher's server is not. The admin page sets how much is kept (500 MB by default). - `ipx add --list --category News ` puts a feed in the Directory for anyone to subscribe to, and it stays there when its last subscriber leaves. Run for a feed already in the catalogue, it lists it or sets its category. diff --git a/docs/configuration.md b/docs/configuration.md index 6797d24..3a8a07f 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -4,7 +4,7 @@ Two places. **config.toml** holds what ipx needs before it reaches its database, who gets in: where things are (`download_dir`, `socket`, `organize`), `[torrent]` and `[web]`. **The database** holds the catalogue of feeds (`[feeds.]` below) and the server settings the admin page edits (`schedule`, `max_total_gb`, `max_age_days`, `max_new_per_check`, -`media_types`). Change those in the web UI, or with `ipx add`, `ipx rm` and `ipx import`; they +`media_types`, `art_cache_mb`). Change those in the web UI, or with `ipx add`, `ipx rm` and `ipx import`; they take effect without a restart. The first time ipx meets a database that holds no catalogue, it takes the feeds and those @@ -39,9 +39,10 @@ max_total_gb = 50 # 0 = unlimited max_age_days = 30 # 0 = keep forever max_new_per_check = 3 # per feed, per scan. 0 = unlimited media_types = ["audio", "video"] +art_cache_mb = 500 # artwork kept on disk. 0 = none ``` -`schedule`, `max_total_gb`, `max_age_days`, `max_new_per_check` and `media_types` move into the +`schedule`, `max_total_gb`, `max_age_days`, `max_new_per_check`, `media_types` and `art_cache_mb` move into the database as described above; `download_dir`, `socket` and `organize` stay in config.toml. * **`schedule`** — how often feeds are re-checked. A feed's own `` still wins when it asks to @@ -54,6 +55,10 @@ database as described above; `download_dir`, `socket` and `organize` stay in con read it. `0` disables it entirely. * **`max_age_days`** — items older than this with no file on disk are pruned from the database. Kept ones stay. `0` disables it. +* **`art_cache_mb`** — how much show and episode artwork iPX keeps on disk, in `art/` beside the + database, so the page loads it from iPX rather than from every publisher's server. Over this, + what has gone longest unshown goes first. `0` keeps none: artwork still comes through iPX, fetched + each time. 500 by default. * **`max_new_per_check`** — how many of a feed's newest episodes are downloaded; older ones stay listed to download by hand. It stops a new subscription pulling a whole back catalogue. `0` means every episode, for an archive; set it on the feeds you want archived, since on the diff --git a/src/art.rs b/src/art.rs new file mode 100644 index 0000000..34d3394 --- /dev/null +++ b/src/art.rs @@ -0,0 +1,92 @@ +//! Artwork kept on disk, so the page loads it from iPX: fast after the first time, nothing asked +//! of each publisher's server for every visit, and still there when a publisher is not. + +use std::path::{Path, PathBuf}; + +pub fn dir() -> PathBuf { + crate::config::data_dir().join("art") +} + +/// Where an image's bytes are kept; its content type is beside it, in `.type`. +// ponytail: std's hasher, 64 bits, keyed by the URL. Its algorithm may change with Rust, which +// only makes the cache miss once and refill; a collision among a few thousand images is about +// one in 10^12. Move to a SHA if either ever matters. +pub fn path(url: &str) -> PathBuf { + use std::hash::{Hash, Hasher}; + let mut h = std::collections::hash_map::DefaultHasher::new(); + url.hash(&mut h); + dir().join(format!("{:016x}", h.finish())) +} + +/// A kept image and its type, marked as just used, which is what keeps it from being trimmed. +pub fn read(file: &Path) -> Option<(String, Vec)> { + let kind = std::fs::read_to_string(file.with_extension("type")).ok()?; + let body = std::fs::read(file).ok()?; + if let Ok(f) = std::fs::File::options().write(true).open(file) { + let _ = f.set_modified(std::time::SystemTime::now()); + } + Some((kind, body)) +} + +/// Keeps an image, written aside and renamed into place, so a page asking for it meanwhile +/// never reads half of one. +pub fn write(file: &Path, kind: &str, body: &[u8]) -> std::io::Result<()> { + std::fs::create_dir_all(file.parent().unwrap_or(Path::new(".")))?; + let tmp = file.with_extension("part"); + std::fs::write(&tmp, body)?; + std::fs::write(file.with_extension("type"), kind)?; + std::fs::rename(&tmp, file) +} + +/// Removes the least recently used images until what is kept fits in `limit` bytes; 0 empties it. +/// Returns how many went. +pub fn trim(dir: &Path, limit: u64) -> usize { + let Ok(entries) = std::fs::read_dir(dir) else { return 0 }; + let mut kept: Vec<(std::time::SystemTime, u64, PathBuf)> = entries + .flatten() + .map(|e| e.path()) + .filter(|p| p.extension().is_none()) + .filter_map(|p| { + let m = p.metadata().ok()?; + Some((m.modified().ok()?, m.len(), p)) + }) + .collect(); + let mut total: u64 = kept.iter().map(|(_, n, _)| n).sum(); + kept.sort(); + let mut gone = 0; + for (_, n, p) in kept { + if total <= limit { + break; + } + let _ = std::fs::remove_file(p.with_extension("type")); + if std::fs::remove_file(&p).is_ok() { + total -= n; + gone += 1; + } + } + gone +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_least_recently_used_go_first_until_it_fits() { + let d = std::env::temp_dir().join(format!("ipx-art-test-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&d); + for (name, age) in [("old", 300), ("mid", 200), ("new", 100)] { + let f = d.join(name); + write(&f, "image/png", &[0; 1000]).unwrap(); + let t = std::time::SystemTime::now() - std::time::Duration::from_secs(age); + std::fs::File::options().write(true).open(&f).unwrap().set_modified(t).unwrap(); + } + // Shown just now: no longer the oldest. + assert_eq!(read(&d.join("old")).unwrap().0, "image/png"); + assert_eq!(trim(&d, 2000), 1); + assert!(!d.join("mid").exists() && !d.join("mid.type").exists()); + assert!(d.join("old").exists() && d.join("new").exists()); + assert_eq!(trim(&d, 0), 2); + let _ = std::fs::remove_dir_all(&d); + } +} diff --git a/src/config.rs b/src/config.rs index 2bf6040..7d027a2 100644 --- a/src/config.rs +++ b/src/config.rs @@ -39,6 +39,9 @@ pub struct General { /// image in an , so taking everything filled the disk with artwork and /// counted it as episodes. Empty means take anything. pub media_types: Vec, + /// How much artwork iPX keeps on disk, in MB, so the page loads it from iPX rather than from + /// every publisher; the least recently shown goes first. 0 keeps none. + pub art_cache_mb: u64, } #[derive(Debug, Clone, Copy, PartialEq, Eq, Deserialize, Serialize)] @@ -184,6 +187,7 @@ impl Default for General { max_age_days: 0, max_new_per_check: 3, media_types: vec!["audio".into(), "video".into()], + art_cache_mb: 500, } } } @@ -294,10 +298,18 @@ pub struct Stored { pub max_age_days: u64, pub max_new_per_check: usize, pub media_types: Vec, + /// Settings saved before it existed have none: they get the default. + #[serde(default = "art_cache_mb")] + pub art_cache_mb: u64, +} + +fn art_cache_mb() -> u64 { + General::default().art_cache_mb } /// `[general]` keys that live in the database once it holds the configuration. -const STORED_KEYS: [&str; 5] = ["schedule", "max_total_gb", "max_age_days", "max_new_per_check", "media_types"]; +const STORED_KEYS: [&str; 6] = + ["schedule", "max_total_gb", "max_age_days", "max_new_per_check", "media_types", "art_cache_mb"]; impl Stored { pub fn of(cfg: &Config) -> Self { @@ -308,6 +320,7 @@ impl Stored { max_age_days: g.max_age_days, max_new_per_check: g.max_new_per_check, media_types: g.media_types.clone(), + art_cache_mb: g.art_cache_mb, } } @@ -318,6 +331,7 @@ impl Stored { g.max_age_days = self.max_age_days; g.max_new_per_check = self.max_new_per_check; g.media_types = self.media_types; + g.art_cache_mb = self.art_cache_mb; } } @@ -575,6 +589,12 @@ mod tests { assert!(wanted_media(Some("image/jpeg"), &["image/jpeg".to_string()])); } + #[test] + fn settings_saved_before_the_art_cache_keep_the_default() { + let old = r#"{"schedule":"every 1h","max_total_gb":0.0,"max_age_days":0,"max_new_per_check":3,"media_types":["audio"]}"#; + assert_eq!(serde_json::from_str::(old).unwrap().art_cache_mb, 500); + } + #[test] fn slugs_are_readable_and_unique() { assert_eq!(slug("Accidental Tech Podcast"), "accidental-tech-podcast"); diff --git a/src/main.rs b/src/main.rs index 7c71730..d34268b 100644 --- a/src/main.rs +++ b/src/main.rs @@ -1,4 +1,5 @@ mod access; +mod art; mod auth; mod config; mod db; @@ -1007,6 +1008,10 @@ async fn reap(ctx: &Ctx, dry_run: bool, standalone: bool) -> Result<()> { let r = retention::run(&ctx.cfg(), &ctx.db, dry_run).await?; if !dry_run { clean_directory(ctx).await?; + let gone = art::trim(&art::dir(), ctx.cfg().general.art_cache_mb * 1_048_576); + if gone > 0 { + tracing::debug!(gone, "trimmed the artwork kept on disk"); + } } for c in r.aged_out.iter().chain(r.over_quota.iter()) { ctx.out.emit(Event::Reaped { diff --git a/src/web.rs b/src/web.rs index df1d817..9945f1d 100644 --- a/src/web.rs +++ b/src/web.rs @@ -1730,16 +1730,25 @@ struct ArtQuery { u: String, } -/// Artwork a feed names on plain http, fetched here for the https page. The browser upgrades an -/// http image on an https page to https, and a host with no https, such as The Secret Cabal's -/// CDN, then answers nothing (issue #90). +/// Artwork, from iPX: the page asks here for every image a feed or item names, and the first +/// time it is fetched from the publisher and kept on disk (`art`, up to `art_cache_mb`), so +/// later it is fast, the publisher is not asked on every visit, and it outlives the publisher's +/// server. Only an address some feed or item names as its artwork, and only an image up to +/// 5 MB, so the route cannot be pointed at anything else. It began as a way round http-only +/// artwork on the https page (#90). async fn art(State(state): State, Query(q): Query) -> Response { const MAX: usize = 5 << 20; - if !q.u.starts_with("http://") || !state.ctx.db.names_image(&q.u).await.unwrap_or(false) { + let web = q.u.starts_with("http://") || q.u.starts_with("https://"); + if !web || !state.ctx.db.names_image(&q.u).await.unwrap_or(false) { return (StatusCode::NOT_FOUND, "no feed names that artwork").into_response(); } + let keep = state.ctx.cfg().general.art_cache_mb > 0; + let file = crate::art::path(&q.u); + if keep && let Some((kind, body)) = crate::art::read(&file) { + return art_response(kind, body); + } let got = async { - let mut r = state.ctx.client.get(&q.u).send().await?.error_for_status()?; + let mut r = state.ctx.client.get(&q.u).timeout(crate::feed::FEED_TIMEOUT).send().await?.error_for_status()?; let kind = r.headers().get(header::CONTENT_TYPE).and_then(|v| v.to_str().ok()).unwrap_or("").to_owned(); anyhow::ensure!(kind.starts_with("image/"), "not an image: {kind}"); let mut body = Vec::new(); @@ -1751,12 +1760,30 @@ async fn art(State(state): State, Query(q): Query) -> Respon }; match got.await { Ok((kind, body)) => { - ([(header::CONTENT_TYPE, kind), (header::CACHE_CONTROL, "private, max-age=86400".into())], body).into_response() + if keep && let Err(e) = crate::art::write(&file, &kind, &body) { + tracing::debug!(error = %e, "could not keep the artwork"); + } + art_response(kind, body) } Err(e) => (StatusCode::BAD_GATEWAY, format!("{e:#}")).into_response(), } } +/// An image from someone else's server, served from iPX's own address: it must stay an image. +/// An SVG opened on its own would otherwise run its script as iPX's page, with its cookies. +fn art_response(kind: String, body: Vec) -> Response { + ( + [ + (header::CONTENT_TYPE, kind), + (header::CACHE_CONTROL, "private, max-age=2592000".into()), + (header::X_CONTENT_TYPE_OPTIONS, "nosniff".into()), + (header::CONTENT_SECURITY_POLICY, "default-src 'none'; style-src 'unsafe-inline'; sandbox".into()), + ], + body, + ) + .into_response() +} + #[derive(Deserialize)] struct Position { secs: i64, @@ -1922,6 +1949,7 @@ struct Settings { download_dir: String, max_total_gb: f64, max_age_days: u64, + art_cache_mb: u64, } async fn get_settings(State(state): State) -> Json { @@ -1934,6 +1962,7 @@ async fn get_settings(State(state): State) -> Json { download_dir: cfg.general.download_dir.display().to_string(), max_total_gb: cfg.general.max_total_gb, max_age_days: cfg.general.max_age_days, + art_cache_mb: cfg.general.art_cache_mb, }) } @@ -1944,6 +1973,7 @@ struct SettingsPatch { media_types: Option>, max_total_gb: Option, max_age_days: Option, + art_cache_mb: Option, } async fn patch_settings( @@ -1980,6 +2010,9 @@ async fn patch_settings( if let Some(v) = body.max_age_days { cfg.general.max_age_days = v; } + if let Some(v) = body.art_cache_mb { + cfg.general.art_cache_mb = v; + } state.ctx.store_cfg(cfg).await?; Ok(StatusCode::NO_CONTENT) } diff --git a/tests/ui/app.spec.js b/tests/ui/app.spec.js index ad3adda..8816ae9 100644 --- a/tests/ui/app.spec.js +++ b/tests/ui/app.spec.js @@ -956,6 +956,22 @@ test('reading an item updates its feed\'s count without reloading the list', asy expect(lists, 'the row came back with the read, not by reloading the list').toEqual([]); }); +test('artwork comes from iPX, kept, and only as an image', async ({ page }) => { + // Every image a feed names is drawn from iPX's address. + expect(await page.evaluate(() => artHTML('https://example.com/a.jpg', 'A'))).toContain('src="/api/art?u=https%3A%2F%2Fexample.com%2Fa.jpg"'); + // Multi Show's items name art.jpg. + await expect(page.locator('.feed', { hasText: 'Multi Show' })).toBeVisible({ timeout: 20_000 }); + const src = '/api/art?u=' + encodeURIComponent('http://127.0.0.1:8792/art.jpg'); + for (const _ of [1, 2]) { // fetched, then kept + const r = await page.request.get(src); + expect(r.status()).toBe(200); + expect(r.headers()['content-type']).toMatch(/^image\//); + expect(r.headers()['content-security-policy']).toContain('sandbox'); + } + // An address no feed names is not fetched. + expect((await page.request.get('/api/art?u=' + encodeURIComponent('http://127.0.0.1:8792/show.xml'))).status()).toBe(404); +}); + test('a deleted file looks as if it was never downloaded', async ({ page }) => { // Other people subscribe to Picture Blog by now, so both prompts come; take them. page.on('dialog', d => d.accept()); diff --git a/tests/ui/fixtures/serve.js b/tests/ui/fixtures/serve.js index a435b2a..eac54b6 100644 --- a/tests/ui/fixtures/serve.js +++ b/tests/ui/fixtures/serve.js @@ -12,7 +12,10 @@ http.createServer((req, res) => { fs.readFile(file, (err, body) => { if (err) { res.writeHead(404).end('no'); return; } const type = file.endsWith('.mp3') ? 'audio/mpeg' - : file.endsWith('.opml') ? 'text/x-opml' : 'application/xml'; + : file.endsWith('.opml') ? 'text/x-opml' + // Artwork as an image, or /api/art refuses it as not one. + : file.endsWith('.jpg') ? 'image/jpeg' + : file.endsWith('.html') ? 'text/html' : 'application/xml'; res.writeHead(200, { 'content-type': type, 'content-length': body.length }); res.end(body); }); diff --git a/web/src/admin.ts b/web/src/admin.ts index 61f351d..8a770e2 100644 --- a/web/src/admin.ts +++ b/web/src/admin.ts @@ -49,6 +49,10 @@ async function drawServer(){ items are never touched.
+
+ + Show and episode artwork is kept here once shown, so it loads from iPX + instead of each publisher. Over this, what has gone longest unshown is dropped first.
${esc(g.download_dir)}
`; @@ -59,7 +63,8 @@ async function drawServer(){ max_new_per_check: Math.max(0, Number($('#gmax').value) || 0), media_types: $('#gtypes').value.split(',').map(t => t.trim()).filter(Boolean), max_total_gb: Number($('#gquota').value) || 0, - max_age_days: Number($('#gage').value) || 0})}); + max_age_days: Number($('#gage').value) || 0, + art_cache_mb: Math.max(0, Number($('#gart').value) || 0)})}); toast('Settings saved'); }catch(e){ toast(e.message, true); } }; diff --git a/web/src/player.ts b/web/src/player.ts index 20a9c58..1e62d6e 100644 --- a/web/src/player.ts +++ b/web/src/player.ts @@ -46,7 +46,7 @@ function mediaSession(e,f){ if(!('mediaSession' in navigator)) return; navigator.mediaSession.metadata=new MediaMetadata({ title:e.title||'', artist:f?(f.title||f.id):'', album:f?(f.title||''):'', - artwork:(e.image||(f&&f.image))?[{src:e.image||f.image,sizes:'512x512'}]:[], + artwork:(e.image||(f&&f.image))?[{src:artSrc(e.image||f.image),sizes:'512x512'}]:[], }); const h={play:()=>audio.play(),pause:()=>audio.pause(), seekbackward:()=>audio.currentTime-=15,seekforward:()=>audio.currentTime+=30}; diff --git a/web/src/util.ts b/web/src/util.ts index 916b61c..b281098 100644 --- a/web/src/util.ts +++ b/web/src/util.ts @@ -145,10 +145,11 @@ const tint=name=>{ let h=0; for(const c of name||'?') h=(h*31+c.charCodeAt(0))>> function tileHTML(name,cls){ return `
${esc(initials(name))}
`; } -/// On the https page, the browser upgrades an http:// image to https, and a host that has no -/// https shows nothing (issue #90); ipx fetches those itself. +/// Every image a feed names comes from iPX, which keeps a copy: fast after the first time, and +/// no publisher's server asked on every visit. It began with http-only artwork, which the https +/// page could not load (#90). function artSrc(url: string){ - return location.protocol==='https:'&&/^http:\/\//i.test(url) ? '/api/art?u='+encodeURIComponent(url) : url; + return /^https?:\/\//i.test(url) ? '/api/art?u='+encodeURIComponent(url) : url; } function artHTML(url: string | null, name: string, cls?: string){ return url