Substack puts a post's subtitle in <description> and leaves it out of content:encoded, and body() took content:encoded alone, so the subtitle was lost (a known gap in CLAUDE.md). A description is now shown above the body, as <p><em>, when it is short plain text the body does not already contain. Podcast feeds that repeat their notes in both, whole or cut short with an ellipsis, are unchanged; the comparison is by words, since a tag taken out of the body leaves stray spaces around punctuation. Checked against Experimental History's feed (subtitles appear) and The Daily's (notes in both fields, shown once). Entries are inserted with ON CONFLICT DO NOTHING, so only posts first seen from now on get it. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
1213 lines
52 KiB
Rust
1213 lines
52 KiB
Rust
//! Feed fetching and parsing. Replaces FeedData.__getFeed / __getEntries.
|
|
|
|
use anyhow::{Context, Result, anyhow};
|
|
use reqwest::StatusCode;
|
|
use reqwest::header::{ETAG, IF_MODIFIED_SINCE, IF_NONE_MATCH, LAST_MODIFIED};
|
|
|
|
use crate::config::Feed as FeedCfg;
|
|
|
|
#[derive(Debug, Default)]
|
|
pub struct ParsedFeed {
|
|
pub title: Option<String>,
|
|
pub ttl_mins: Option<u64>,
|
|
pub image: Option<String>,
|
|
/// The channel's first `<itunes:category>`, for the Directory's chips.
|
|
pub category: Option<String>,
|
|
pub entries: Vec<Entry>,
|
|
}
|
|
|
|
#[derive(Debug, Default)]
|
|
pub struct Entry {
|
|
pub guid: String,
|
|
pub title: Option<String>,
|
|
pub link: Option<String>,
|
|
pub published: Option<i64>,
|
|
pub description: Option<String>,
|
|
pub categories: Vec<String>,
|
|
pub explicit: bool,
|
|
/// Episode artwork; falls back to the feed's in the UI.
|
|
pub image: Option<String>,
|
|
/// Seconds.
|
|
pub duration: Option<i64>,
|
|
pub episode: Option<i64>,
|
|
pub season: Option<i64>,
|
|
pub enclosures: Vec<Enclosure>,
|
|
}
|
|
|
|
#[derive(Debug, Default, Clone, PartialEq)]
|
|
pub struct Enclosure {
|
|
pub url: String,
|
|
pub mime: Option<String>,
|
|
pub length: Option<i64>,
|
|
}
|
|
|
|
pub enum Fetched {
|
|
/// Server said 304, or returned a body we already have.
|
|
NotModified,
|
|
Body {
|
|
bytes: Vec<u8>,
|
|
etag: Option<String>,
|
|
last_modified: Option<String>,
|
|
},
|
|
}
|
|
|
|
/// Conditional GET. reqwest handles gzip and redirects; the original's hand-rolled
|
|
/// CONNECT/socket.ssl proxy path is gone -- `system-proxy` reads http_proxy/https_proxy.
|
|
pub async fn fetch(
|
|
client: &reqwest::Client,
|
|
cfg: &FeedCfg,
|
|
etag: Option<&str>,
|
|
last_modified: Option<&str>,
|
|
) -> Result<Fetched> {
|
|
let mut req = client.get(&cfg.url);
|
|
if let Some(tag) = etag {
|
|
req = req.header(IF_NONE_MATCH, tag);
|
|
}
|
|
if let Some(lm) = last_modified {
|
|
req = req.header(IF_MODIFIED_SINCE, lm);
|
|
}
|
|
if let Some(user) = &cfg.username {
|
|
req = req.basic_auth(user, cfg.password());
|
|
}
|
|
|
|
let resp = req.send().await.context("connecting")?;
|
|
if resp.status() == StatusCode::NOT_MODIFIED {
|
|
return Ok(Fetched::NotModified);
|
|
}
|
|
let status = resp.status();
|
|
if !status.is_success() {
|
|
// The original surfaced 401/407 specially; the code is enough for a UI to switch on.
|
|
return Err(anyhow!("HTTP {status}"));
|
|
}
|
|
|
|
let header = |h: reqwest::header::HeaderName| {
|
|
resp.headers().get(&h).and_then(|v| v.to_str().ok()).map(str::to_owned)
|
|
};
|
|
let etag = header(ETAG);
|
|
let last_modified = header(LAST_MODIFIED);
|
|
let bytes = resp.bytes().await.context("reading body")?.to_vec();
|
|
Ok(Fetched::Body { bytes, etag, last_modified })
|
|
}
|
|
|
|
/// A stored `last_error`, translated into plain words for whoever subscribes: whose problem
|
|
/// it is, and whether there is a new address to switch to.
|
|
pub struct Failure {
|
|
pub reason: &'static str,
|
|
pub new_url: Option<String>,
|
|
}
|
|
|
|
/// Reads a `last_error` the same way `set_feed_error` received it (`format!("{e:#}")` on the
|
|
/// anyhow chain from `fetch` or `parse`) and says what it means, for the errors worth telling
|
|
/// someone about. Everything else -- a timeout, a 5xx, a 429, a feed that is simply garbled --
|
|
/// comes back `None`: transient by nature, or with nothing more useful to say than the raw
|
|
/// text already shown once a feed is open.
|
|
///
|
|
/// ponytail: matches on the fixed strings this crate itself produces (`anyhow!("HTTP
|
|
/// {status}")`, and `parse`'s "got a web page" and "the site sent") plus the substrings a DNS failure
|
|
/// reliably contains. Fragile if reqwest's own wording changes; the fallback is just showing
|
|
/// nothing extra, so a miss costs a clearer message, not a wrong one.
|
|
pub fn explain_failure(msg: &str) -> Option<Failure> {
|
|
if let Some(rest) = msg.strip_prefix("got a web page, not a feed") {
|
|
let new_url = rest
|
|
.strip_prefix("; it links ")
|
|
.and_then(|r| r.strip_suffix(" as its feed"))
|
|
.map(str::to_owned);
|
|
return Some(Failure { reason: "The feed moved; this address now shows a web page.", new_url });
|
|
}
|
|
if msg.contains("the site sent ") {
|
|
return Some(Failure { reason: "The site sent a message instead of the feed; the publisher has to fix it.", new_url: None });
|
|
}
|
|
let low = msg.to_ascii_lowercase();
|
|
if low.contains("http 404") {
|
|
return Some(Failure { reason: "The publisher took this feed down, or moved it.", new_url: None });
|
|
}
|
|
if low.contains("http 401") || low.contains("http 403") {
|
|
return Some(Failure { reason: "The site refuses ipx's requests.", new_url: None });
|
|
}
|
|
if low.contains("http 402") {
|
|
return Some(Failure { reason: "The feed now needs a paid plan.", new_url: None });
|
|
}
|
|
if low.contains("dns error")
|
|
|| low.contains("failed to lookup address")
|
|
|| low.contains("no address associated")
|
|
{
|
|
return Some(Failure { reason: "This address no longer resolves; the site is gone.", new_url: None });
|
|
}
|
|
None
|
|
}
|
|
|
|
/// True when a body is an OPML document rather than a feed.
|
|
///
|
|
/// The original matched on the URL ending in ".opml" (iPXClass.py:34), which misses an
|
|
/// OPML served from a URL without that extension. Sniffing the body catches both.
|
|
pub fn is_opml(bytes: &[u8]) -> bool {
|
|
let head = &bytes[..bytes.len().min(1024)];
|
|
let text = String::from_utf8_lossy(head).to_lowercase();
|
|
text.contains("<opml")
|
|
}
|
|
|
|
/// The feeds listed in an OPML document, as (title, xml_url), walking nested folders.
|
|
pub fn parse_opml(bytes: &[u8]) -> Result<Vec<(String, String)>> {
|
|
let text = String::from_utf8_lossy(bytes);
|
|
let doc = opml::OPML::from_str(&text)
|
|
.map_err(|e| anyhow!("that does not parse as OPML: {e}"))?;
|
|
let mut out = vec![];
|
|
crate::collect_outlines(&doc.body.outlines, &mut out);
|
|
Ok(out)
|
|
}
|
|
|
|
/// The <head><title> of an OPML document.
|
|
pub fn opml_title(bytes: &[u8]) -> Option<String> {
|
|
let text = String::from_utf8_lossy(bytes);
|
|
let doc = opml::OPML::from_str(&text).ok()?;
|
|
doc.head
|
|
.and_then(|h| h.title)
|
|
.map(|t| t.trim().to_owned())
|
|
.filter(|t| !t.is_empty())
|
|
}
|
|
|
|
/// The token and show of a Patreon feed link, or None for any other URL.
|
|
///
|
|
/// Patreon gives each patron one token per creator. With no show it stands for the creator,
|
|
/// whose feed carries every show at once.
|
|
fn patreon_parts(url: &str) -> Option<(String, Option<String>)> {
|
|
let u = url::Url::parse(url).ok()?;
|
|
if !matches!(u.host_str()?, "patreon.com" | "www.patreon.com") || !u.path().starts_with("/rss") {
|
|
return None;
|
|
}
|
|
let param = |name: &str| u.query_pairs().find(|(k, _)| k == name).map(|(_, v)| v.into_owned());
|
|
Some((param("auth")?, param("show")))
|
|
}
|
|
|
|
/// A Patreon link naming a creator but no show.
|
|
pub fn is_patreon_creator(url: &str) -> bool {
|
|
matches!(patreon_parts(url), Some((_, None)))
|
|
}
|
|
|
|
/// What was typed into Add feed, as a URL. A bare Patreon token is taken as its creator's
|
|
/// feed, since the token alone says whose it is.
|
|
pub fn expand_input(input: &str) -> String {
|
|
let s = input.trim();
|
|
let token = s.len() >= 20 && s.chars().all(|c| c.is_ascii_alphanumeric() || c == '-' || c == '_');
|
|
if token { format!("https://www.patreon.com/rss?auth={s}") } else { s.to_owned() }
|
|
}
|
|
|
|
/// Whether two URLs are the same feed. One Patreon show has several spellings -- by the
|
|
/// creator's name, by number, or with no creator at all -- and the token and show are what
|
|
/// identify it.
|
|
pub fn same_feed(a: &str, b: &str) -> bool {
|
|
a == b || patreon_parts(a).is_some_and(|p| Some(p) == patreon_parts(b))
|
|
}
|
|
|
|
/// A Patreon creator's name and shows, each show as (title, feed URL).
|
|
///
|
|
/// ponytail: Patreon's own web API, undocumented, asked without signing in. If it changes,
|
|
/// finding shows stops and the show feeds already found keep working. The documented API
|
|
/// needs an OAuth client per install and does not list shows.
|
|
pub async fn patreon_shows(
|
|
client: &reqwest::Client,
|
|
url: &str,
|
|
) -> Result<(Option<String>, Vec<(String, String)>)> {
|
|
// The creator feed names its campaign by number in its self link, a few hundred bytes in.
|
|
// The whole feed runs to megabytes and Patreon ignores Range, so read until it turns up.
|
|
let mut resp = client.get(url).send().await.context("connecting")?;
|
|
if !resp.status().is_success() {
|
|
return Err(anyhow!("Patreon refused the feed: HTTP {}", resp.status()));
|
|
}
|
|
let mut head = Vec::new();
|
|
while patreon_campaign(&head).is_none() && head.len() < 64 * 1024 {
|
|
let Some(chunk) = resp.chunk().await.context("reading the feed")? else { break };
|
|
head.extend_from_slice(&chunk);
|
|
}
|
|
let campaign = patreon_campaign(&head)
|
|
.ok_or_else(|| anyhow!("the Patreon feed does not say whose it is"))?;
|
|
|
|
let api = format!(
|
|
"https://www.patreon.com/api/campaigns/{campaign}\
|
|
?include=shows&fields%5Bcampaign%5D=name&fields%5Bcollection%5D=title"
|
|
);
|
|
let resp = client.get(api).send().await.context("asking Patreon for the shows")?;
|
|
if !resp.status().is_success() {
|
|
return Err(anyhow!("Patreon would not list the shows: HTTP {}", resp.status()));
|
|
}
|
|
let (name, shows) = parse_patreon_shows(&resp.bytes().await.context("reading the shows")?)?;
|
|
Ok((name, shows.into_iter().map(|(id, title)| (title, format!("{url}&show={id}"))).collect()))
|
|
}
|
|
|
|
/// The campaign number in the start of a Patreon feed.
|
|
fn patreon_campaign(head: &[u8]) -> Option<String> {
|
|
let text = String::from_utf8_lossy(head);
|
|
text.match_indices("patreon.com/rss/").find_map(|(i, m)| {
|
|
let id: String = text[i + m.len()..].chars().take_while(char::is_ascii_digit).collect();
|
|
(!id.is_empty()).then_some(id)
|
|
})
|
|
}
|
|
|
|
/// A campaign's name and its shows as (id, title), from Patreon's JSON:API answer.
|
|
fn parse_patreon_shows(json: &[u8]) -> Result<(Option<String>, Vec<(String, String)>)> {
|
|
let v: serde_json::Value = serde_json::from_slice(json).context("Patreon's answer is not JSON")?;
|
|
// Missing is not the same as none. Read as no shows, the creator feed would be scanned as
|
|
// a plain feed, claim every show's files, and leave the shows empty once the list returned.
|
|
let ids = v["data"]["relationships"]["shows"]["data"]
|
|
.as_array()
|
|
.ok_or_else(|| anyhow!("Patreon's answer does not list the shows"))?;
|
|
let title = |id: &str| -> Option<String> {
|
|
let show = v["included"].as_array()?.iter().find(|x| x["type"] == "collection" && x["id"] == id)?;
|
|
show["attributes"]["title"].as_str().map(|t| t.trim().to_owned())
|
|
};
|
|
let shows = ids
|
|
.iter()
|
|
.filter_map(|s| s["id"].as_str())
|
|
.map(|id| (id.to_owned(), title(id).unwrap_or_else(|| format!("Show {id}"))))
|
|
.collect();
|
|
Ok((v["data"]["attributes"]["name"].as_str().map(str::to_owned), shows))
|
|
}
|
|
|
|
/// RSS first, then Atom -- the same split the original made on `parsedFeed.version`.
|
|
pub fn parse(bytes: &[u8]) -> Result<ParsedFeed> {
|
|
match rss::Channel::read_from(bytes) {
|
|
Ok(ch) => Ok(from_rss(ch, bytes)),
|
|
Err(rss_err) => match atom_syndication::Feed::read_from(bytes) {
|
|
Ok(feed) => Ok(from_atom(feed)),
|
|
Err(atom_err) => {
|
|
if let Some(said) = plain_text(bytes) {
|
|
return Err(anyhow!("the site sent {said} instead of a feed"));
|
|
}
|
|
// Some publishers (kcpw, feedland) write a bare "&" in a URL instead of
|
|
// "&". Strict XML parsers refuse it; browsers don't. Retry once with
|
|
// every offending "&" escaped rather than fail outright.
|
|
let escaped = escape_bare_ampersands(bytes);
|
|
if escaped != bytes {
|
|
if let Ok(ch) = rss::Channel::read_from(escaped.as_slice()) {
|
|
return Ok(from_rss(ch, &escaped));
|
|
}
|
|
if let Ok(feed) = atom_syndication::Feed::read_from(escaped.as_slice()) {
|
|
return Ok(from_atom(feed));
|
|
}
|
|
}
|
|
Err(match alternate_feed_link(bytes) {
|
|
Some(href) if looks_like_html(bytes) => {
|
|
anyhow!("got a web page, not a feed; it links {href} as its feed")
|
|
}
|
|
None if looks_like_html(bytes) => anyhow!("got a web page, not a feed"),
|
|
_ => anyhow!("not RSS ({rss_err}) and not Atom ({atom_err})"),
|
|
})
|
|
}
|
|
},
|
|
}
|
|
}
|
|
|
|
/// Whether a body is a web page rather than a feed: most of the errors traced back to a feed
|
|
/// that moved or a domain that lapsed, with the old URL now serving the site instead (or a
|
|
/// redirect to it). `is_opml` already sniffs the other "not actually a feed" case.
|
|
fn looks_like_html(bytes: &[u8]) -> bool {
|
|
let head = String::from_utf8_lossy(&bytes[..bytes.len().min(2048)]).to_lowercase();
|
|
head.contains("<!doctype html") || head.contains("<html")
|
|
}
|
|
|
|
/// What a site sent when it sent a sentence instead of markup. doghouse's feed answered 200 with
|
|
/// "Unable to establish a DB connection", and the two parsers' errors about end of input buried
|
|
/// it. Anything starting with `<` is markup, however broken, and keeps the parsers' errors.
|
|
fn plain_text(bytes: &[u8]) -> Option<String> {
|
|
let head = String::from_utf8_lossy(&bytes[..bytes.len().min(512)]);
|
|
let text = head.trim_start_matches(|c: char| c.is_whitespace() || c == '\u{feff}');
|
|
if text.starts_with('<') {
|
|
return None;
|
|
}
|
|
let line = text.lines().next().unwrap_or("").trim_end();
|
|
if line.is_empty() {
|
|
return Some("an empty reply".into());
|
|
}
|
|
let mut said: String = line.chars().take(80).collect();
|
|
if said.len() < line.len() {
|
|
said.push('…');
|
|
}
|
|
Some(format!("\"{said}\""))
|
|
}
|
|
|
|
/// The feed a web page names as its own via `<link rel="alternate" type="application/rss+xml"
|
|
/// href="...">` (or the Atom equivalent) -- how the new address was found for om.co, ms.now,
|
|
/// Letters of Note, the Daily Dot, Hell Gate, The Frame Lab and Daily Kos.
|
|
fn alternate_feed_link(bytes: &[u8]) -> Option<String> {
|
|
let text = String::from_utf8_lossy(bytes);
|
|
let lower = text.to_lowercase();
|
|
let mut pos = 0;
|
|
while let Some(rel) = lower[pos..].find("<link") {
|
|
let start = pos + rel;
|
|
let Some(end) = lower[start..].find('>').map(|e| start + e) else { break };
|
|
pos = end + 1;
|
|
let tag = &text[start..end];
|
|
let tag_lower = &lower[start..end];
|
|
let is_alternate = tag_lower.contains("rel=\"alternate\"") || tag_lower.contains("rel='alternate'");
|
|
let is_feed_type = tag_lower.contains("rss+xml") || tag_lower.contains("atom+xml");
|
|
if is_alternate && is_feed_type
|
|
&& let Some(href) = tag_attr(tag, "href")
|
|
{
|
|
return Some(href);
|
|
}
|
|
}
|
|
None
|
|
}
|
|
|
|
/// The value of one attribute in an HTML/XML start tag, however it is quoted.
|
|
fn tag_attr(tag: &str, name: &str) -> Option<String> {
|
|
let key = format!("{name}=");
|
|
let idx = tag.to_lowercase().find(&key)?;
|
|
let after = &tag[idx + key.len()..];
|
|
let quote = after.chars().next()?;
|
|
if quote != '"' && quote != '\'' {
|
|
return None;
|
|
}
|
|
let rest = &after[1..];
|
|
let close = rest.find(quote)?;
|
|
Some(rest[..close].trim().to_owned())
|
|
}
|
|
|
|
/// Escapes every `&` that does not already start a recognized XML entity
|
|
/// (`&`, `<`, `>`, `"`, `'`, or a numeric reference like `'`).
|
|
fn escape_bare_ampersands(bytes: &[u8]) -> Vec<u8> {
|
|
fn is_entity_start(rest: &[u8]) -> bool {
|
|
for named in [&b"amp;"[..], b"lt;", b"gt;", b"quot;", b"apos;"] {
|
|
if rest.starts_with(named) {
|
|
return true;
|
|
}
|
|
}
|
|
let digits = if rest.starts_with(b"#x") || rest.starts_with(b"#X") {
|
|
&rest[2..]
|
|
} else if rest.starts_with(b"#") {
|
|
&rest[1..]
|
|
} else {
|
|
return false;
|
|
};
|
|
let len = digits.iter().take_while(|b| b.is_ascii_alphanumeric()).count();
|
|
len > 0 && digits.get(len) == Some(&b';')
|
|
}
|
|
|
|
let mut out = Vec::with_capacity(bytes.len());
|
|
let mut i = 0;
|
|
while i < bytes.len() {
|
|
if bytes[i] == b'&' && !is_entity_start(&bytes[i + 1..]) {
|
|
out.extend_from_slice(b"&");
|
|
} else {
|
|
out.push(bytes[i]);
|
|
}
|
|
i += 1;
|
|
}
|
|
out
|
|
}
|
|
|
|
/// Every `<enclosure>` of every `<item>`, in document order.
|
|
///
|
|
/// The `rss` crate models an item as having at most one enclosure -- which is what RSS 2.0
|
|
/// says -- and when a feed carries several it keeps only the *last*, silently losing the
|
|
/// rest. Feeds do ship several, so read them from the XML directly.
|
|
fn enclosures_by_item(bytes: &[u8]) -> Vec<Vec<Enclosure>> {
|
|
use quick_xml::events::Event;
|
|
|
|
let mut reader = quick_xml::Reader::from_reader(bytes);
|
|
reader.config_mut().trim_text(true);
|
|
let mut buf = Vec::new();
|
|
let mut out: Vec<Vec<Enclosure>> = Vec::new();
|
|
let mut current: Option<Vec<Enclosure>> = None;
|
|
|
|
let read_enclosure = |e: &quick_xml::events::BytesStart| -> Option<Enclosure> {
|
|
let (mut url, mut mime, mut length) = (String::new(), None, None);
|
|
for attr in e.attributes().flatten() {
|
|
// Values arrive escaped: a feed URL's "&" is "&" in the document.
|
|
let val = quick_xml::escape::unescape(&attr.value)
|
|
.map(|v| v.trim().to_string())
|
|
.unwrap_or_default();
|
|
match attr.key.local_name().as_ref() {
|
|
"url" => url = val,
|
|
"type" => mime = Some(val).filter(|v| !v.is_empty()),
|
|
"length" => length = val.parse().ok(),
|
|
_ => {}
|
|
}
|
|
}
|
|
(!url.is_empty()).then_some(Enclosure { url, mime, length })
|
|
};
|
|
|
|
loop {
|
|
match reader.read_event_into(&mut buf) {
|
|
Ok(Event::Start(e)) => match e.name().local_name().as_ref() {
|
|
"item" => current = Some(Vec::new()),
|
|
"enclosure" => {
|
|
if let (Some(list), Some(enc)) = (current.as_mut(), read_enclosure(&e)) {
|
|
list.push(enc);
|
|
}
|
|
}
|
|
_ => {}
|
|
},
|
|
Ok(Event::Empty(e)) => match e.name().local_name().as_ref() {
|
|
// <item/> with no children still counts, so the indexes stay aligned.
|
|
"item" => out.push(Vec::new()),
|
|
"enclosure" => {
|
|
if let (Some(list), Some(enc)) = (current.as_mut(), read_enclosure(&e)) {
|
|
list.push(enc);
|
|
}
|
|
}
|
|
_ => {}
|
|
},
|
|
Ok(Event::End(e)) => {
|
|
if e.name().local_name().as_ref() == "item"
|
|
&& let Some(list) = current.take()
|
|
{
|
|
out.push(list);
|
|
}
|
|
}
|
|
Ok(Event::Eof) | Err(_) => break,
|
|
_ => {}
|
|
}
|
|
buf.clear();
|
|
}
|
|
if let Some(list) = current.take() {
|
|
out.push(list);
|
|
}
|
|
out
|
|
}
|
|
|
|
fn from_rss(ch: rss::Channel, bytes: &[u8]) -> ParsedFeed {
|
|
let per_item = enclosures_by_item(bytes);
|
|
let explicit = ch
|
|
.itunes_ext()
|
|
.and_then(|it| it.explicit())
|
|
.is_some_and(is_yes);
|
|
|
|
let entries = ch
|
|
.items()
|
|
.iter()
|
|
.enumerate()
|
|
.filter_map(|(idx, item)| {
|
|
// Straight from the XML, so an item with several keeps all of them. Falls
|
|
// back to the parsed one if the scan and the parser disagree on item count.
|
|
let mut enclosures: Vec<Enclosure> = per_item.get(idx).cloned().unwrap_or_else(|| {
|
|
item.enclosure()
|
|
.into_iter()
|
|
.map(|e| Enclosure {
|
|
url: e.url().trim().to_owned(),
|
|
mime: non_empty(Some(e.mime_type())),
|
|
length: e.length().parse().ok(),
|
|
})
|
|
.filter(|e| !e.url.is_empty())
|
|
.collect()
|
|
});
|
|
drop_player_repeats(&mut enclosures);
|
|
|
|
let guid = pick_guid(
|
|
item.guid().map(|g| g.value()),
|
|
item.link(),
|
|
enclosures.first().map(|e| e.url.as_str()),
|
|
item.title(),
|
|
)?;
|
|
|
|
let entry_explicit = item
|
|
.itunes_ext()
|
|
.and_then(|it| it.explicit())
|
|
.is_some_and(is_yes);
|
|
let it = item.itunes_ext();
|
|
|
|
Some(Entry {
|
|
guid,
|
|
title: title_text(item.title()),
|
|
link: non_empty(item.link()),
|
|
published: item.pub_date().and_then(parse_date),
|
|
// Content wins over description, as __getEntries preferred entry.content.
|
|
description: body(item.content(), item.description()),
|
|
categories: item
|
|
.categories()
|
|
.iter()
|
|
.map(|c| c.name().to_owned())
|
|
.filter(|c| !c.is_empty() && !c.starts_with("http"))
|
|
.collect(),
|
|
explicit: explicit || entry_explicit,
|
|
image: item_image(item, &enclosures),
|
|
duration: it.and_then(|i| i.duration()).and_then(parse_duration),
|
|
episode: it.and_then(|i| i.episode()).and_then(|e| e.trim().parse().ok()),
|
|
season: it.and_then(|i| i.season()).and_then(|e| e.trim().parse().ok()),
|
|
enclosures,
|
|
})
|
|
})
|
|
.collect();
|
|
|
|
ParsedFeed {
|
|
title: title_text(Some(ch.title())),
|
|
ttl_mins: ch.ttl().and_then(|t| t.trim().parse().ok()),
|
|
// itunes:image is the square artwork; <image><url> is the older, often smaller one.
|
|
image: ch
|
|
.itunes_ext()
|
|
.and_then(|i| i.image())
|
|
.map(str::to_owned)
|
|
.or_else(|| ch.image().map(|i| i.url().to_owned())),
|
|
// Only the iTunes one: Apple's list is fixed, while a plain <category> is freeform and
|
|
// would fill the Directory with one-off tags. The subcategory where there is one: Apple
|
|
// files every tabletop and gaming show under Leisure, which says little; Games says it.
|
|
category: ch
|
|
.itunes_ext()
|
|
.and_then(|i| i.categories().first())
|
|
.map(|c| c.subcategory().filter(|s| !s.text().trim().is_empty()).unwrap_or(c))
|
|
.and_then(|c| non_empty(Some(c.text().trim()))),
|
|
entries,
|
|
}
|
|
}
|
|
|
|
fn from_atom(feed: atom_syndication::Feed) -> ParsedFeed {
|
|
let entries = feed
|
|
.entries()
|
|
.iter()
|
|
.filter_map(|e| {
|
|
// Atom carries enclosures as <link rel="enclosure">.
|
|
let mut enclosures: Vec<Enclosure> = e
|
|
.links()
|
|
.iter()
|
|
.filter(|l| l.rel() == "enclosure")
|
|
.map(|l| Enclosure {
|
|
url: l.href().trim().to_owned(),
|
|
mime: non_empty(l.mime_type()),
|
|
length: l.length().and_then(|s| s.parse().ok()),
|
|
})
|
|
.filter(|e| !e.url.is_empty())
|
|
.collect();
|
|
drop_player_repeats(&mut enclosures);
|
|
|
|
let alt = e
|
|
.links()
|
|
.iter()
|
|
.find(|l| l.rel() == "alternate" || l.rel().is_empty())
|
|
.map(|l| l.href());
|
|
|
|
let guid = pick_guid(
|
|
Some(e.id()),
|
|
alt,
|
|
enclosures.first().map(|x| x.url.as_str()),
|
|
Some(e.title().as_str()),
|
|
)?;
|
|
|
|
Some(Entry {
|
|
guid,
|
|
title: title_text(Some(e.title().as_str())),
|
|
link: alt.map(str::to_owned),
|
|
published: e.published().or(Some(e.updated())).map(|d| d.timestamp()),
|
|
description: body(e.content().and_then(|c| c.value()), e.summary().map(|s| s.as_str())),
|
|
categories: e.categories().iter().map(|c| c.term().to_owned()).collect(),
|
|
explicit: false,
|
|
image: None,
|
|
duration: None,
|
|
episode: None,
|
|
season: None,
|
|
enclosures,
|
|
})
|
|
})
|
|
.collect();
|
|
|
|
ParsedFeed {
|
|
title: title_text(Some(feed.title().as_str())),
|
|
ttl_mins: None,
|
|
image: feed.logo().or_else(|| feed.icon()).map(str::to_owned),
|
|
category: None,
|
|
entries,
|
|
}
|
|
}
|
|
|
|
/// The original fell back to hashing the title or description. A guid, permalink or
|
|
/// enclosure URL is a stable identifier already, so no hashing is needed; an entry with
|
|
/// none of them has nothing to download and is dropped.
|
|
fn pick_guid(
|
|
guid: Option<&str>,
|
|
link: Option<&str>,
|
|
enclosure: Option<&str>,
|
|
title: Option<&str>,
|
|
) -> Option<String> {
|
|
[guid, link, enclosure, title]
|
|
.into_iter()
|
|
.flatten()
|
|
.map(str::trim)
|
|
.find(|s| !s.is_empty())
|
|
.map(str::to_owned)
|
|
}
|
|
|
|
fn is_yes(s: &str) -> bool {
|
|
matches!(s.trim().to_ascii_lowercase().as_str(), "yes" | "true" | "explicit")
|
|
}
|
|
|
|
fn non_empty(s: Option<&str>) -> Option<String> {
|
|
s.map(str::trim).filter(|s| !s.is_empty()).map(str::to_owned)
|
|
}
|
|
|
|
/// WordPress numbers each audio player on a page by adding `?_=N` to its file's URL, so a post
|
|
/// that embeds the file it encloses lists the same file twice: Rands in Repose's "The Promotion
|
|
/// Paradox" was downloaded twice and offered two play buttons for one mp3. The first stays.
|
|
fn drop_player_repeats(encs: &mut Vec<Enclosure>) {
|
|
let mut seen = std::collections::HashSet::new();
|
|
encs.retain(|e| seen.insert(same_file_key(&e.url)));
|
|
}
|
|
|
|
/// An enclosure URL without WordPress's player number, for telling repeats of one file apart
|
|
/// from different files.
|
|
pub fn same_file_key(url: &str) -> String {
|
|
let Ok(mut u) = url::Url::parse(url) else { return url.to_owned() };
|
|
let kept: Vec<(String, String)> = u
|
|
.query_pairs()
|
|
.filter(|(k, v)| !(k == "_" && !v.is_empty() && v.bytes().all(|b| b.is_ascii_digit())))
|
|
.map(|(k, v)| (k.into_owned(), v.into_owned()))
|
|
.collect();
|
|
if kept.is_empty() {
|
|
u.set_query(None);
|
|
} else {
|
|
u.query_pairs_mut().clear().extend_pairs(kept);
|
|
}
|
|
u.to_string()
|
|
}
|
|
|
|
/// A title as plain text. An Atom title of `type="html"`, or an RSS one in CDATA, comes through
|
|
/// the XML parser with its HTML entities intact: The Verge's "Meta’s" reached the page as
|
|
/// typed. Decoded one entity at a time, so an `&` that starts none, as in "Q&A", stays as it is
|
|
/// instead of failing the whole title.
|
|
fn title_text(s: Option<&str>) -> Option<String> {
|
|
let s = non_empty(s)?;
|
|
let mut out = String::with_capacity(s.len());
|
|
let mut rest = s.as_str();
|
|
while let Some(at) = rest.find('&') {
|
|
out.push_str(&rest[..at]);
|
|
rest = &rest[at..];
|
|
let len = 1 + rest[1..]
|
|
.find(|c: char| !(c.is_ascii_alphanumeric() || c == '#'))
|
|
.unwrap_or(rest.len() - 1);
|
|
let decoded = rest[len..]
|
|
.starts_with(';')
|
|
.then(|| quick_xml::escape::unescape_with(&rest[..=len], quick_xml::escape::resolve_html5_entity).ok())
|
|
.flatten();
|
|
match decoded {
|
|
Some(v) => {
|
|
out.push_str(&v);
|
|
rest = &rest[len + 1..];
|
|
}
|
|
None => {
|
|
out.push('&');
|
|
rest = &rest[1..];
|
|
}
|
|
}
|
|
}
|
|
out.push_str(rest);
|
|
non_empty(Some(&out))
|
|
}
|
|
|
|
/// An item's show notes: its full body when that is whole, else its description.
|
|
///
|
|
/// libsyn served Daily Meditation Podcast's `content:encoded` cut at the `>` inside a class name
|
|
/// pasted from a web app (`[&:has([data-writing-block])>*]:pointer-events-auto`), so the body
|
|
/// began halfway through a tag and the page showed the rest of the tag as text. The same item's
|
|
/// `description` was whole. With no description to fall back on, a damaged body beats none.
|
|
fn body(content: Option<&str>, description: Option<&str>) -> Option<String> {
|
|
match non_empty(content).filter(|c| !starts_mid_tag(c)) {
|
|
Some(c) => Some(match subtitle(&c, description) {
|
|
Some(s) => format!("<p><em>{}</em></p>{c}", quick_xml::escape::escape(s.as_str())),
|
|
None => c,
|
|
}),
|
|
None => non_empty(description).or_else(|| non_empty(content)),
|
|
}
|
|
}
|
|
|
|
/// A description that is a subtitle rather than a second copy of the notes: Substack puts the
|
|
/// post's subtitle there and leaves it out of `content:encoded`, so taking the body alone lost it.
|
|
/// Podcast feeds mostly repeat their notes in both, whole or cut short with an ellipsis, and a
|
|
/// description found in the body is not shown twice.
|
|
///
|
|
/// ponytail: short plain text not found in the body. A summary a podcast writes apart from its
|
|
/// notes passes too and shows above them, which reads fine; a real subtitle field would need an
|
|
/// `entries` column.
|
|
fn subtitle(body: &str, description: Option<&str>) -> Option<String> {
|
|
let d = title_text(description)?;
|
|
if d.contains('<') || d.chars().count() > 300 {
|
|
return None;
|
|
}
|
|
// Words alone: a tag taken out leaves "tape ," where the description has "tape,", and a cut
|
|
// description ends in "…" or "[...]".
|
|
let words = |s: &str| {
|
|
s.split(|c: char| !c.is_alphanumeric()).filter(|w| !w.is_empty()).collect::<Vec<_>>().join(" ").to_lowercase()
|
|
};
|
|
let want = words(&d);
|
|
let text = title_text(Some(&text_of(body))).unwrap_or_default();
|
|
(!want.is_empty() && !words(&text).contains(&want)).then_some(d)
|
|
}
|
|
|
|
/// HTML with its tags taken out, each replaced by a space so words either side stay apart.
|
|
fn text_of(html: &str) -> String {
|
|
let mut out = String::with_capacity(html.len());
|
|
let mut in_tag = false;
|
|
for c in html.chars() {
|
|
match c {
|
|
'<' => in_tag = true,
|
|
'>' if in_tag => { in_tag = false; out.push(' '); }
|
|
_ if !in_tag => out.push(c),
|
|
_ => {}
|
|
}
|
|
}
|
|
out
|
|
}
|
|
|
|
/// Text that closes an attribute list (`">`) before any tag has opened is the tail of a tag whose
|
|
/// start was cut off.
|
|
fn starts_mid_tag(html: &str) -> bool {
|
|
html[..html.find('<').unwrap_or(html.len())].contains("\">")
|
|
}
|
|
|
|
/// The picture to show beside an item, in order of how deliberate it is:
|
|
/// `itunes:image`, then Media RSS `media:thumbnail`, then a `media:content` that is an
|
|
/// image, and finally an image enclosure -- which is how a blog's article picture arrives
|
|
/// (Substack puts it there), so those entries get artwork rather than a blank square.
|
|
fn item_image(item: &rss::Item, enclosures: &[Enclosure]) -> Option<String> {
|
|
if let Some(url) = item.itunes_ext().and_then(|i| i.image()) {
|
|
return non_empty(Some(url));
|
|
}
|
|
|
|
let media = item.extensions().get("media");
|
|
let attr = |name: &str, want_image: bool| -> Option<String> {
|
|
media?.get(name)?.iter().find_map(|e| {
|
|
if want_image {
|
|
// media:content carries anything; only take it when it says it is a picture.
|
|
let is_image = e.attrs.get("type").is_some_and(|t| t.starts_with("image/"))
|
|
|| e.attrs.get("medium").is_some_and(|m| m == "image");
|
|
if !is_image {
|
|
return None;
|
|
}
|
|
}
|
|
non_empty(e.attrs.get("url").map(String::as_str))
|
|
})
|
|
};
|
|
attr("thumbnail", false)
|
|
.or_else(|| attr("content", true))
|
|
.or_else(|| {
|
|
enclosures
|
|
.iter()
|
|
.find(|e| e.mime.as_deref().is_some_and(|m| m.starts_with("image/")))
|
|
.map(|e| e.url.clone())
|
|
})
|
|
}
|
|
|
|
/// itunes:duration is either plain seconds ("5649") or a clock ("1:34:09", "23:45").
|
|
fn parse_duration(s: &str) -> Option<i64> {
|
|
let s = s.trim();
|
|
if s.is_empty() {
|
|
return None;
|
|
}
|
|
if !s.contains(':') {
|
|
return s.parse().ok().filter(|n| *n > 0);
|
|
}
|
|
let mut total: i64 = 0;
|
|
for part in s.split(':') {
|
|
total = total * 60 + part.trim().parse::<i64>().ok()?;
|
|
}
|
|
Some(total).filter(|n| *n > 0)
|
|
}
|
|
|
|
/// RSS pubDate is RFC 2822; some feeds ship RFC 3339 instead.
|
|
fn parse_date(s: &str) -> Option<i64> {
|
|
let s = s.trim();
|
|
chrono::DateTime::parse_from_rfc2822(s)
|
|
.or_else(|_| chrono::DateTime::parse_from_rfc3339(s))
|
|
.ok()
|
|
.map(|d| d.timestamp())
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
|
|
#[test]
|
|
fn parses_rss_with_itunes_extensions() {
|
|
let bytes = include_bytes!("../tests/data/rss2.xml");
|
|
let feed = parse(bytes).unwrap();
|
|
|
|
assert_eq!(feed.title.as_deref(), Some("Test Cast"));
|
|
assert_eq!(feed.ttl_mins, Some(45));
|
|
assert_eq!(feed.category.as_deref(), Some("Podcasting"), "the first, by its subcategory");
|
|
assert_eq!(feed.entries.len(), 3);
|
|
|
|
let ep = &feed.entries[0];
|
|
assert_eq!(ep.guid, "https://example.com/ep/1");
|
|
assert_eq!(ep.title.as_deref(), Some("Episode One"));
|
|
assert_eq!(ep.published, Some(1_078_016_400));
|
|
assert_eq!(ep.categories, vec!["Tech"]);
|
|
assert!(!ep.explicit);
|
|
assert_eq!(
|
|
ep.enclosures,
|
|
vec![Enclosure {
|
|
url: "https://example.com/ep1.mp3".into(),
|
|
mime: Some("audio/mpeg".into()),
|
|
length: Some(12_345_678),
|
|
}]
|
|
);
|
|
|
|
assert!(feed.entries[1].explicit, "entry-level itunes:explicit=yes");
|
|
assert_eq!(
|
|
feed.entries[2].enclosures[0].mime.as_deref(),
|
|
Some("application/x-bittorrent")
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn a_file_wordpress_lists_twice_is_one_enclosure() {
|
|
let xml = br#"<?xml version="1.0"?><rss version="2.0"><channel><title>R</title>
|
|
<item><title>The Promotion Paradox</title><guid>p</guid>
|
|
<enclosure url="https://x/ep.mp3" length="1" type="audio/mpeg"/>
|
|
<enclosure url="https://x/ep.mp3?_=2" length="1" type="audio/mpeg"/>
|
|
<enclosure url="https://x/other.mp3?_=3&key=k" length="1" type="audio/mpeg"/>
|
|
</item></channel></rss>"#;
|
|
let urls: Vec<String> =
|
|
parse(xml).unwrap().entries[0].enclosures.iter().map(|e| e.url.clone()).collect();
|
|
assert_eq!(urls, ["https://x/ep.mp3", "https://x/other.mp3?_=3&key=k"], "the repeat goes, a different file stays");
|
|
assert_eq!(same_file_key("https://x/a.mp3?key=k&_=2"), same_file_key("https://x/a.mp3?key=k"));
|
|
assert_ne!(same_file_key("https://x/a.mp3?_=x"), same_file_key("https://x/a.mp3"), "only a number");
|
|
}
|
|
|
|
#[test]
|
|
fn titles_are_read_as_text_not_html() {
|
|
// The Verge: an Atom title of type="html", its entity inside CDATA.
|
|
let xml = br#"<?xml version="1.0"?>
|
|
<feed xmlns="http://www.w3.org/2005/Atom"><title type="text">V</title><id>v</id>
|
|
<updated>2026-09-15T00:00:00Z</updated>
|
|
<entry><title type="html"><![CDATA[Meta’s new One]]></title><id>e1</id>
|
|
<updated>2026-09-15T00:00:00Z</updated></entry></feed>"#;
|
|
assert_eq!(parse(xml).unwrap().entries[0].title.as_deref(), Some("Meta\u{2019}s new One"));
|
|
// HTML names as well as numbers; a bare `&` and an unknown name are left as they are.
|
|
assert_eq!(
|
|
title_text(Some("Peña & “Q&A” &bogus; AT&T;")).as_deref(),
|
|
Some("Pe\u{f1}a & \u{201c}Q&A\u{201d} &bogus; AT&T;")
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn a_body_cut_off_mid_tag_gives_way_to_the_description() {
|
|
// How libsyn served Daily Meditation Podcast #3477: content:encoded began inside a tag.
|
|
let cut = r#"*]:pointer-events-auto R6Vx5W_threadScrollVars" dir="auto" data-turn="assistant"> <p>What if</p>"#;
|
|
let whole = r#"<div class="[&:has([data-writing-block])>*]:pointer-events-auto"><p>What if</p></div>"#;
|
|
assert_eq!(body(Some(cut), Some(whole)).as_deref(), Some(whole));
|
|
assert_eq!(body(Some("<p>Notes</p>"), Some("<p>Notes</p>")).as_deref(), Some("<p>Notes</p>"), "a whole body wins");
|
|
assert_eq!(body(Some("Plain notes, no tags."), Some("Plain notes, no tags.")).as_deref(), Some("Plain notes, no tags."));
|
|
assert_eq!(body(Some(cut), None).as_deref(), Some(cut), "a damaged body beats none");
|
|
assert_eq!(body(None, Some("Summary")).as_deref(), Some("Summary"));
|
|
}
|
|
|
|
#[test]
|
|
fn a_subtitle_missing_from_the_body_is_kept_above_it() {
|
|
// Substack: the subtitle is the description, and content:encoded does not repeat it.
|
|
assert_eq!(
|
|
body(Some("<p>The post.</p>"), Some("Why the <b> tag & I fell out")).as_deref(),
|
|
Some("<p>The post.</p>"),
|
|
"a description with markup in it is notes, not a subtitle",
|
|
);
|
|
assert_eq!(
|
|
body(Some("<p>The post.</p>"), Some("Why Q&A threads go wrong")).as_deref(),
|
|
Some("<p><em>Why Q&A threads go wrong</em></p><p>The post.</p>"),
|
|
);
|
|
// A podcast repeating its notes, whole, cut short, or differently spaced: shown once.
|
|
let notes = "<p>This week we talk about <a href=\"x\">tape</a>, drums and a very long list.</p>";
|
|
for d in ["This week we talk about tape, drums and a very long list.",
|
|
"This week we talk about tape, drums…",
|
|
"This week we talk about\ntape [...]"] {
|
|
assert_eq!(body(Some(notes), Some(d)).as_deref(), Some(notes), "{d:?} is already in the body");
|
|
}
|
|
let long = "word ".repeat(80);
|
|
assert_eq!(body(Some("<p>The post.</p>"), Some(&long)).as_deref(), Some("<p>The post.</p>"),
|
|
"a long description is notes, not a subtitle");
|
|
}
|
|
|
|
#[test]
|
|
fn feed_level_explicit_overrides_entries() {
|
|
let xml = br#"<?xml version="1.0"?>
|
|
<rss version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd">
|
|
<channel><title>X</title><link>https://x</link><description>d</description>
|
|
<itunes:explicit>yes</itunes:explicit>
|
|
<item><title>a</title><guid>g1</guid>
|
|
<enclosure url="https://x/a.mp3" length="1" type="audio/mpeg"/></item>
|
|
</channel></rss>"#;
|
|
let feed = parse(xml).unwrap();
|
|
assert!(
|
|
feed.entries[0].explicit,
|
|
"the entry says nothing; the feed-level flag must still mark it explicit"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn explain_failure_translates_the_errors_the_ui_should_flag() {
|
|
assert_eq!(
|
|
explain_failure("HTTP 404 Not Found").unwrap().reason,
|
|
"The publisher took this feed down, or moved it."
|
|
);
|
|
assert_eq!(explain_failure("HTTP 401 Unauthorized").unwrap().reason, "The site refuses ipx's requests.");
|
|
assert_eq!(explain_failure("HTTP 403 Forbidden").unwrap().reason, "The site refuses ipx's requests.");
|
|
assert_eq!(explain_failure("HTTP 402 Payment Required").unwrap().reason, "The feed now needs a paid plan.");
|
|
let dns = explain_failure("connecting: dns error: failed to lookup address information").unwrap();
|
|
assert_eq!(dns.reason, "This address no longer resolves; the site is gone.");
|
|
let moved = explain_failure("got a web page, not a feed; it links https://x/feed as its feed").unwrap();
|
|
assert_eq!(moved.new_url.as_deref(), Some("https://x/feed"));
|
|
assert!(explain_failure("got a web page, not a feed").unwrap().new_url.is_none());
|
|
let down = explain_failure("the site sent \"Unable to establish a DB connection\" instead of a feed").unwrap();
|
|
assert_eq!(down.reason, "The site sent a message instead of the feed; the publisher has to fix it.");
|
|
for transient in [
|
|
"HTTP 500 Internal Server Error",
|
|
"HTTP 429 Too Many Requests",
|
|
"operation timed out",
|
|
"not RSS (reached end of input without finding a complete channel) and not Atom (unexpected end of input)",
|
|
] {
|
|
assert!(explain_failure(transient).is_none(), "{transient} must not be flagged");
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn a_web_page_says_so_and_names_the_feed_it_links() {
|
|
let html = br#"<!doctype html><html><head>
|
|
<link rel="alternate" type="application/rss+xml" href="https://x.example/feed">
|
|
</head><body>not a feed</body></html>"#;
|
|
let err = parse(html).unwrap_err().to_string();
|
|
assert_eq!(err, "got a web page, not a feed; it links https://x.example/feed as its feed");
|
|
}
|
|
|
|
#[test]
|
|
fn a_web_page_with_no_feed_link_still_says_so() {
|
|
let html = b"<!doctype html><html><body>moved</body></html>";
|
|
assert_eq!(parse(html).unwrap_err().to_string(), "got a web page, not a feed");
|
|
}
|
|
|
|
#[test]
|
|
fn malformed_xml_gets_the_original_parser_errors() {
|
|
let err = parse(b"<rss><channel><title>cut off").unwrap_err().to_string();
|
|
assert!(err.starts_with("not RSS ("), "{err}");
|
|
}
|
|
|
|
#[test]
|
|
fn a_body_with_no_markup_says_what_the_site_sent() {
|
|
let err = parse(b"\xef\xbb\xbf\r\n Unable to establish a DB connection\nmore").unwrap_err().to_string();
|
|
assert_eq!(err, "the site sent \"Unable to establish a DB connection\" instead of a feed");
|
|
let long = parse("x".repeat(200).as_bytes()).unwrap_err().to_string();
|
|
assert_eq!(long, format!("the site sent \"{}…\" instead of a feed", "x".repeat(80)));
|
|
assert_eq!(parse(b" \n").unwrap_err().to_string(), "the site sent an empty reply instead of a feed");
|
|
}
|
|
|
|
#[test]
|
|
fn parses_atom_enclosure_links() {
|
|
let bytes = include_bytes!("../tests/data/atom.xml");
|
|
let feed = parse(bytes).unwrap();
|
|
|
|
assert_eq!(feed.title.as_deref(), Some("Atom Cast"));
|
|
assert_eq!(feed.entries.len(), 1);
|
|
let ep = &feed.entries[0];
|
|
assert_eq!(ep.guid, "urn:uuid:1225c695-cfb8-4ebb-aaaa-80da344efa6a");
|
|
assert_eq!(ep.link.as_deref(), Some("https://example.org/ep/1"));
|
|
assert_eq!(ep.published, Some(1_078_016_400));
|
|
assert_eq!(
|
|
ep.enclosures,
|
|
vec![Enclosure {
|
|
url: "https://example.org/ep1.m4a".into(),
|
|
mime: Some("audio/mp4".into()),
|
|
length: Some(9_876_543),
|
|
}],
|
|
"rel=enclosure only; the alternate link must not become an enclosure"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn a_bare_ampersand_in_a_link_is_repaired_and_parsed() {
|
|
// kcpw.org: <link>https://kcpw.org/?post_type=post&p=125715</link> -- a bare "&"
|
|
// that strict XML rejects but browsers accept.
|
|
let xml = br#"<?xml version="1.0"?>
|
|
<rss version="2.0"><channel><title>X</title><link>https://x</link><description>d</description>
|
|
<item><title>a</title><guid>g1</guid>
|
|
<link>https://kcpw.org/?post_type=post&p=125715</link>
|
|
<enclosure url="https://x/a.mp3?a=1&b=2" length="1" type="audio/mpeg"/></item>
|
|
</channel></rss>"#;
|
|
let feed = parse(xml).unwrap();
|
|
assert_eq!(feed.entries[0].link.as_deref(), Some("https://kcpw.org/?post_type=post&p=125715"));
|
|
assert_eq!(feed.entries[0].enclosures[0].url, "https://x/a.mp3?a=1&b=2");
|
|
}
|
|
|
|
#[test]
|
|
fn escape_bare_ampersands_leaves_real_entities_alone() {
|
|
let out = escape_bare_ampersands(b"a&b <x> ' / c&d");
|
|
assert_eq!(out, b"a&b <x> ' / c&d");
|
|
}
|
|
|
|
#[test]
|
|
fn the_rss_title_always_wins_and_episode_numbers_stay_metadata() {
|
|
// Some feeds set a different itunes:title. The displayed title is always the RSS
|
|
// <title>, verbatim -- separators and all -- and season/episode are stored
|
|
// alongside it rather than folded into it.
|
|
let xml = br#"<?xml version="1.0"?>
|
|
<rss version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd">
|
|
<channel><title>Show</title><link>https://x</link><description>d</description>
|
|
<item>
|
|
<title>Music from a Darkened Room | Session Zero</title>
|
|
<itunes:title>Session Zero</itunes:title>
|
|
<guid>sz</guid>
|
|
<itunes:season>8</itunes:season>
|
|
<itunes:duration>6720</itunes:duration>
|
|
<enclosure url="https://x/sz.mp3" length="1" type="audio/mpeg"/>
|
|
</item>
|
|
<item>
|
|
<title>Music from a Darkened Room Part 1 | Murphy's Drawer</title>
|
|
<guid>p1</guid>
|
|
<itunes:season>8</itunes:season><itunes:episode>1</itunes:episode>
|
|
<enclosure url="https://x/p1.mp3" length="1" type="audio/mpeg"/>
|
|
</item>
|
|
</channel></rss>"#;
|
|
let feed = parse(xml).unwrap();
|
|
|
|
let sz = &feed.entries[0];
|
|
assert_eq!(
|
|
sz.title.as_deref(),
|
|
Some("Music from a Darkened Room | Session Zero"),
|
|
"itunes:title must not override the RSS title"
|
|
);
|
|
assert_eq!(sz.season, Some(8));
|
|
assert_eq!(sz.episode, None, "a missing episode number stays missing");
|
|
assert_eq!(sz.duration, Some(6720));
|
|
|
|
let p1 = &feed.entries[1];
|
|
assert_eq!(p1.title.as_deref(), Some("Music from a Darkened Room Part 1 | Murphy's Drawer"));
|
|
assert_eq!((p1.season, p1.episode), (Some(8), Some(1)));
|
|
}
|
|
|
|
#[test]
|
|
fn opml_is_recognised_and_its_feeds_listed() {
|
|
let xml = br#"<opml version="2.0"><head><title>My Subscriptions</title></head><body>
|
|
<outline text="Folder">
|
|
<outline type="rss" text="Alpha" xmlUrl="https://a.example/rss"/>
|
|
<outline type="rss" text="Beta" xmlUrl="https://b.example/rss"/>
|
|
</outline>
|
|
<outline text="Not a feed"/>
|
|
</body></opml>"#;
|
|
assert!(is_opml(xml));
|
|
assert_eq!(opml_title(xml).as_deref(), Some("My Subscriptions"));
|
|
|
|
let feeds = parse_opml(xml).unwrap();
|
|
assert_eq!(feeds.len(), 2, "nested folders are walked, non-feed outlines skipped");
|
|
assert_eq!(feeds[0], ("Alpha".into(), "https://a.example/rss".into()));
|
|
|
|
// A feed must never be mistaken for a subscription list.
|
|
assert!(!is_opml(include_bytes!("../tests/data/rss2.xml")));
|
|
assert!(!is_opml(include_bytes!("../tests/data/atom.xml")));
|
|
}
|
|
|
|
#[test]
|
|
fn a_patreon_creator_is_a_list_of_its_shows() {
|
|
let tok = "AbCdEfGhIjKlMnOpQrStUvWxYz012_-9";
|
|
assert_eq!(expand_input(&format!(" {tok} ")), format!("https://www.patreon.com/rss?auth={tok}"));
|
|
assert_eq!(expand_input("https://example.com/rss"), "https://example.com/rss");
|
|
|
|
assert!(is_patreon_creator(&format!("https://www.patreon.com/rss/glasscannon?auth={tok}")));
|
|
assert!(is_patreon_creator(&format!("https://www.patreon.com/rss?auth={tok}")));
|
|
assert!(!is_patreon_creator(&format!("https://www.patreon.com/rss/x?auth={tok}&show=1")), "one show is a feed");
|
|
assert!(!is_patreon_creator(&format!("https://example.com/rss?auth={tok}")));
|
|
|
|
// The show you already have by name is the one a bare token would add by number.
|
|
assert!(same_feed(
|
|
&format!("https://www.patreon.com/rss/glasscannon?auth={tok}&show=2073588"),
|
|
&format!("https://www.patreon.com/rss?auth={tok}&show=2073588"),
|
|
));
|
|
assert!(!same_feed(
|
|
&format!("https://www.patreon.com/rss?auth={tok}&show=1"),
|
|
&format!("https://www.patreon.com/rss?auth={tok}&show=2"),
|
|
));
|
|
|
|
// The self link carries the campaign by number, whichever spelling was asked for.
|
|
let head = br#"<rss><channel><link>https://www.patreon.com/glasscannon</link>
|
|
<atom:link href="https://www.patreon.com/rss/369921?auth=t" rel="self"/>"#;
|
|
assert_eq!(patreon_campaign(head).as_deref(), Some("369921"));
|
|
assert_eq!(patreon_campaign(b"<rss><channel><title>T"), None);
|
|
|
|
let json = br#"{"data":{"id":"369921","type":"campaign","attributes":{"name":"The Glass Cannon Network"},
|
|
"relationships":{"shows":{"data":[{"id":"2073588","type":"collection"},{"id":"2073636","type":"collection"}]}}},
|
|
"included":[{"id":"2073588","type":"collection","attributes":{"title":"Get in the Trunk "}},
|
|
{"id":"2073636","type":"collection","attributes":{"title":"Shadowdark"}}]}"#;
|
|
let (name, shows) = parse_patreon_shows(json).unwrap();
|
|
assert_eq!(name.as_deref(), Some("The Glass Cannon Network"));
|
|
assert_eq!(shows, [("2073588".into(), "Get in the Trunk".into()), ("2073636".into(), "Shadowdark".into())]);
|
|
|
|
// An answer that stops naming the shows is an error, never "this creator has none".
|
|
assert!(parse_patreon_shows(br#"{"data":{"attributes":{"name":"X"}}}"#).is_err());
|
|
}
|
|
|
|
#[test]
|
|
fn an_item_may_carry_several_enclosures() {
|
|
// The rss crate keeps only one per item -- the last -- so these come from the XML.
|
|
let xml = br#"<?xml version="1.0"?>
|
|
<rss version="2.0"><channel><title>M</title><link>https://x</link><description>d</description>
|
|
<item><title>Two files</title><guid>m1</guid>
|
|
<enclosure url="https://x/a.mp3?v=1&t=2" length="111" type="audio/mpeg"/>
|
|
<enclosure url="https://x/b.mp4" length="222" type="video/mp4"/>
|
|
</item>
|
|
<item><title>One file</title><guid>m2</guid>
|
|
<enclosure url="https://x/c.mp3" length="333" type="audio/mpeg"/></item>
|
|
<item><title>None</title><guid>m3</guid></item>
|
|
</channel></rss>"#;
|
|
let f = parse(xml).unwrap();
|
|
assert_eq!(f.entries.len(), 3);
|
|
|
|
let two = &f.entries[0].enclosures;
|
|
assert_eq!(two.len(), 2, "both enclosures survive");
|
|
assert_eq!(
|
|
two[0].url, "https://x/a.mp3?v=1&t=2",
|
|
"document order, and the escaped ampersand is decoded"
|
|
);
|
|
assert_eq!(two[0].length, Some(111));
|
|
assert_eq!(two[1].url, "https://x/b.mp4");
|
|
assert_eq!(two[1].mime.as_deref(), Some("video/mp4"));
|
|
|
|
assert_eq!(f.entries[1].enclosures.len(), 1);
|
|
assert_eq!(f.entries[2].enclosures.len(), 0, "an item may have none");
|
|
}
|
|
|
|
#[test]
|
|
fn an_items_picture_comes_from_the_most_deliberate_source() {
|
|
let xml = br#"<?xml version="1.0"?>
|
|
<rss version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd"
|
|
xmlns:media="http://search.yahoo.com/mrss/">
|
|
<channel><title>P</title><link>https://x</link><description>d</description>
|
|
<item><title>Has itunes</title><guid>a</guid>
|
|
<itunes:image href="https://x/itunes.jpg"/>
|
|
<media:thumbnail url="https://x/thumb.jpg"/>
|
|
<enclosure url="https://x/a.jpg" length="1" type="image/jpeg"/></item>
|
|
<item><title>Has thumbnail</title><guid>b</guid>
|
|
<media:thumbnail url="https://x/thumb.jpg"/>
|
|
<enclosure url="https://x/b.jpg" length="1" type="image/jpeg"/></item>
|
|
<item><title>Has media content</title><guid>c</guid>
|
|
<media:content url="https://x/pic.jpg" type="image/jpeg"/>
|
|
<media:content url="https://x/clip.mp4" type="video/mp4"/></item>
|
|
<item><title>Only an image enclosure</title><guid>d</guid>
|
|
<enclosure url="https://x/d.jpg" length="1" type="image/jpeg"/></item>
|
|
<item><title>Audio only</title><guid>e</guid>
|
|
<enclosure url="https://x/e.mp3" length="1" type="audio/mpeg"/></item>
|
|
</channel></rss>"#;
|
|
let f = parse(xml).unwrap();
|
|
let img = |i: usize| f.entries[i].image.as_deref();
|
|
assert_eq!(img(0), Some("https://x/itunes.jpg"), "itunes:image wins");
|
|
assert_eq!(img(1), Some("https://x/thumb.jpg"), "then media:thumbnail");
|
|
assert_eq!(img(2), Some("https://x/pic.jpg"), "media:content, and only the image one");
|
|
assert_eq!(img(3), Some("https://x/d.jpg"), "a blog's article picture arrives as an enclosure");
|
|
assert_eq!(img(4), None, "audio is not a picture");
|
|
}
|
|
|
|
#[test]
|
|
fn durations_parse_from_seconds_or_a_clock() {
|
|
assert_eq!(parse_duration("5649"), Some(5649));
|
|
assert_eq!(parse_duration("23:45"), Some(1425));
|
|
assert_eq!(parse_duration("1:34:09"), Some(5649));
|
|
assert_eq!(parse_duration("0"), None, "zero is not a duration");
|
|
assert_eq!(parse_duration(""), None);
|
|
assert_eq!(parse_duration("garbage"), None);
|
|
}
|
|
|
|
#[test]
|
|
fn rejects_html_masquerading_as_a_feed() {
|
|
assert!(parse(b"<html><body>nope</body></html>").is_err());
|
|
}
|
|
|
|
#[test]
|
|
fn guid_falls_back_through_link_then_enclosure() {
|
|
assert_eq!(pick_guid(Some(" "), Some("l"), Some("e"), None).as_deref(), Some("l"));
|
|
assert_eq!(pick_guid(None, None, Some("e"), Some("t")).as_deref(), Some("e"));
|
|
assert_eq!(pick_guid(None, None, None, None), None);
|
|
}
|
|
}
|
|
|