From 5e23916b40e29a1be5140711b67fe543f5a5a297 Mon Sep 17 00:00:00 2001 From: YoursFunny Date: Fri, 14 Aug 2026 18:19:39 +0800 Subject: [PATCH] refactor(site): move cache_key/is_retryable/media_headers into site modules --- crates/x-media/src/site/bsky/interface.rs | 19 ++ crates/x-media/src/site/bsky/mod.rs | 4 +- crates/x-media/src/site/mod.rs | 204 +++++++++---------- crates/x-media/src/site/pixiv/interface.rs | 76 ++++++- crates/x-media/src/site/pixiv/mod.rs | 4 +- crates/x-media/src/site/twitter/interface.rs | 43 ++++ crates/x-media/src/site/twitter/mod.rs | 4 +- 7 files changed, 237 insertions(+), 117 deletions(-) diff --git a/crates/x-media/src/site/bsky/interface.rs b/crates/x-media/src/site/bsky/interface.rs index 5a62c33..62a7bd5 100644 --- a/crates/x-media/src/site/bsky/interface.rs +++ b/crates/x-media/src/site/bsky/interface.rs @@ -59,6 +59,25 @@ pub async fn fetch_from_url(url: &str) -> Result { Ok(fetched) } +/// Cache key for a bsky URL: `"bsky:/"`. The prefix is the +/// site id used for caption-format lookup and link-cache keys. +pub fn cache_key(url: &str) -> Option { + PATTERN + .captures(url) + .map(|caps| format!("bsky:{}/{}", &caps[1], &caps[2])) +} + +/// Bluesky's fetch-retry policy: transient classes only. Not-found, blocked +/// and parse failures are permanent. +pub fn is_retryable(err: &FetchError) -> bool { + matches!(err, FetchError::Http(_) | FetchError::Transient(_)) +} + +/// bsky media (cdn.bsky.app) needs no extra headers. +pub fn media_headers(_url: &str) -> Option> { + None +} + /// Downloads an HLS playlist (master or media) and remuxes its segments to a /// single MP4 via ffmpeg. Returns the MP4 path plus the temp dir that must /// stay alive until the file is uploaded. `Ok(None)` when ffmpeg is missing. diff --git a/crates/x-media/src/site/bsky/mod.rs b/crates/x-media/src/site/bsky/mod.rs index 89d700f..6c9b1ba 100644 --- a/crates/x-media/src/site/bsky/mod.rs +++ b/crates/x-media/src/site/bsky/mod.rs @@ -1,4 +1,6 @@ mod interface; mod model; -pub use interface::{PATTERN, Post, enabled, fetch_from_url}; +pub use interface::{ + PATTERN, Post, cache_key, enabled, fetch_from_url, is_retryable, media_headers, +}; diff --git a/crates/x-media/src/site/mod.rs b/crates/x-media/src/site/mod.rs index 48544f1..d418a08 100644 --- a/crates/x-media/src/site/mod.rs +++ b/crates/x-media/src/site/mod.rs @@ -160,19 +160,17 @@ pub fn caption_from_fields( /// Stable per-post cache key derived from any supported URL, so variant /// domains (x.com / twitter.com / fxtwitter.com, mobile, `/photo/N` -/// suffixes) map to the same post. Returns `"twitter:"`, -/// `"pixiv:"` or `"bsky:/"`. +/// suffixes) map to the same post. Delegates to the per-site `cache_key` +/// implementations (dispatch order twitter → bsky → pixiv). pub fn cache_key(url: &str) -> Option { - if let Some(caps) = twitter::PATTERN.captures(url) { - return Some(format!("twitter:{}", &caps[1])); - } - if let Some(caps) = pixiv::PATTERN.captures(url) { - return Some(format!("pixiv:{}", &caps[1])); - } - if let Some(caps) = bsky::PATTERN.captures(url) { - return Some(format!("bsky:{}/{}", &caps[1], &caps[2])); - } - None + [ + twitter::cache_key(url), + bsky::cache_key(url), + pixiv::cache_key(url), + ] + .into_iter() + .flatten() + .next() } /// The site id carried by a cache key (`"twitter:123"` → `"twitter"`). @@ -317,35 +315,18 @@ pub(crate) fn log_once_ffmpeg_missing() { /// matches (unsupported links are silently ignored by the bot). /// /// Transient failures are retried: 3 total attempts with 1s then 2s delays. -/// Retried classes: bare HTTP errors, [`FetchError::Transient`] (429/5xx -/// from any site), pixiv network errors, and pixiv HTTP statuses that are -/// actually transient (429 / 5xx). Permanent classes are returned -/// immediately: Json, NotFound, Blocked, Sensitive, pixiv 4xx statuses -/// (bad/expired token, forbidden, not found) and pixiv API/auth errors. -/// Whether [`fetch`] should retry `err` (3 total attempts, 1s then 2s -/// backoff). Permanent classes — 4xx statuses, invalid tokens, unparseable -/// bodies, not-found/blocked/sensitive — are returned immediately; retrying -/// them only wastes attempts against the source site. -fn fetch_error_is_retryable(err: &FetchError) -> bool { - match err { - FetchError::Http(_) | FetchError::Transient(_) => true, - FetchError::Pixiv(e) => match e { - PixivError::Http(_) => true, - PixivError::Status(code) if *code == 429 || *code >= 500 => true, - // 4xx, invalid token, unparseable body: retrying cannot help. - PixivError::Status(_) - | PixivError::Api(_) - | PixivError::Json(_) - | PixivError::NoAuth => false, - }, - _ => false, - } -} - +/// What counts as transient is the matched site's own policy +/// (`SiteKind::is_retryable` — e.g. pixiv retries only network errors and +/// 429/5xx). Permanent classes (not-found, blocked, sensitive, parse +/// failures, pixiv 4xx/auth errors) are returned immediately; retrying them +/// only wastes attempts against the source site. pub async fn fetch(url: &str) -> Result, FetchError> { + let Some(site) = match_site(url) else { + return Ok(None); + }; for attempt in 0..3u32 { - match fetch_once(url).await { - Ok(Some(fetched)) => { + match fetch_once(url, site).await { + Ok(fetched) => { // Per-request detail: debug only, keyed by the post id. log::debug!( "fetched [key={}]: site {} returned {} media", @@ -355,9 +336,8 @@ pub async fn fetch(url: &str) -> Result, FetchError> { ); return Ok(Some(fetched)); } - Ok(None) => return Ok(None), Err(err) => { - if fetch_error_is_retryable(&err) && attempt < 2 { + if site.is_retryable(&err) && attempt < 2 { tokio::time::sleep(Duration::from_secs(1 << attempt)).await; } else { return Err(err); @@ -368,33 +348,80 @@ pub async fn fetch(url: &str) -> Result, FetchError> { unreachable!("retry loop always returns") } -async fn fetch_once(url: &str) -> Result, FetchError> { +/// Which site owns a URL (dispatch order twitter → bsky → pixiv), honoring +/// each site's `enabled()` gate. `None` for unsupported links. +fn match_site(url: &str) -> Option { if twitter::enabled() && twitter::PATTERN.is_match(url) { - return Ok(Some(twitter::fetch_from_url(url).await?)); + Some(SiteKind::Twitter) + } else if bsky::enabled() && bsky::PATTERN.is_match(url) { + Some(SiteKind::Bsky) + } else if pixiv::enabled() && pixiv::PATTERN.is_match(url) { + Some(SiteKind::Pixiv) + } else { + None } - if bsky::enabled() && bsky::PATTERN.is_match(url) { - return Ok(Some(bsky::fetch_from_url(url).await?)); +} + +/// Statically-dispatched site handle: keeps per-site fetch + retry policy +/// callable from the central dispatcher without a trait object (stage 2 of +/// the site-registry refactor; stage 3 replaces this with `dyn Site`). +#[derive(Clone, Copy)] +enum SiteKind { + Twitter, + Bsky, + Pixiv, +} + +impl SiteKind { + /// The matched site's own retry policy (see each site's `is_retryable`). + fn is_retryable(&self, err: &FetchError) -> bool { + match self { + SiteKind::Twitter => twitter::is_retryable(err), + SiteKind::Bsky => bsky::is_retryable(err), + SiteKind::Pixiv => pixiv::is_retryable(err), + } } - if pixiv::enabled() && pixiv::PATTERN.is_match(url) { - return Ok(Some(pixiv::fetch_from_url(url).await?)); +} + +async fn fetch_once(url: &str, site: SiteKind) -> Result { + match site { + SiteKind::Twitter => twitter::fetch_from_url(url).await, + SiteKind::Bsky => bsky::fetch_from_url(url).await, + SiteKind::Pixiv => pixiv::fetch_from_url(url).await, } - Ok(None) +} + +/// Applies every site's media-header rule to a download request (pixiv's +/// `Referer` for pximg.net hotlink protection). Sites contribute via their +/// `media_headers(url)` — the central download code carries no per-site logic. +fn apply_media_headers(mut request: reqwest::RequestBuilder, url: &str) -> reqwest::RequestBuilder { + for headers in [ + twitter::media_headers(url), + bsky::media_headers(url), + pixiv::media_headers(url), + ] + .into_iter() + .flatten() + { + for (name, value) in headers { + request = request.header(name, value); + } + } + request } /// Downloads media bytes for the bot's upload fallback: when Telegram's own /// fetch of a media URL is blocked (hotlink protection), the bot downloads -/// the file itself and uploads it via multipart. Site-appropriate headers: -/// pixiv image hosts need the `Referer` header. +/// the file itself and uploads it via multipart. Site-appropriate headers +/// come from each site's `media_headers` (pixiv image hosts need `Referer`). /// Returns the Content-Length of a media URL, or `None` when the server does /// not report one. Used to check whether a file fits Telegram's size limits /// before downloading/uploading it. pub async fn media_size(url: &str) -> Result, FetchError> { - let mut request = CLIENT.get(url); - let lower = url.to_ascii_lowercase(); - if lower.contains("pximg.net") { - request = request.header("Referer", "https://www.pixiv.net/"); - } - let response = request.send().await?.error_for_status()?; + let response = apply_media_headers(CLIENT.get(url), url) + .send() + .await? + .error_for_status()?; Ok(response.content_length()) } @@ -403,12 +430,10 @@ pub async fn media_size(url: &str) -> Result, FetchError> { /// crossed (or when a declared Content-Length already exceeds it). Keeps the /// bot from buffering arbitrarily large bodies into memory. pub async fn download_media_limited(url: &str, max_bytes: u64) -> Result { - let mut request = CLIENT.get(url); - let lower = url.to_ascii_lowercase(); - if lower.contains("pximg.net") { - request = request.header("Referer", "https://www.pixiv.net/"); - } - let response = request.send().await?.error_for_status()?; + let response = apply_media_headers(CLIENT.get(url), url) + .send() + .await? + .error_for_status()?; if let Some(len) = response.content_length() && len > max_bytes { @@ -441,12 +466,10 @@ pub async fn download_media_to_file( out: &mut std::fs::File, ) -> Result { use std::io::Write; - let mut request = CLIENT.get(url); - let lower = url.to_ascii_lowercase(); - if lower.contains("pximg.net") { - request = request.header("Referer", "https://www.pixiv.net/"); - } - let response = request.send().await?.error_for_status()?; + let response = apply_media_headers(CLIENT.get(url), url) + .send() + .await? + .error_for_status()?; if let Some(len) = response.content_length() && len > max_bytes { @@ -502,51 +525,6 @@ mod tests { assert_eq!(site_id_from_key("no-colon"), "unknown"); } - #[test] - fn fetch_error_retryability_classification() { - // Transient: network errors, explicit transient, pixiv 429/5xx. - assert!(fetch_error_is_retryable(&FetchError::Transient( - "429".into() - ))); - assert!(fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::Status(429) - ))); - assert!(fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::Status(500) - ))); - assert!(fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::Status(503) - ))); - // Permanent: pixiv 4xx (bad/expired token, forbidden, not found), - // api/auth errors, unparseable bodies, not-found/blocked/sensitive. - assert!(!fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::Status(400) - ))); - assert!(!fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::Status(401) - ))); - assert!(!fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::Status(403) - ))); - assert!(!fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::Status(404) - ))); - assert!(!fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::Api("invalid_grant".into()) - ))); - assert!(!fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::NoAuth - ))); - let json_err = serde_json::from_str::("x").unwrap_err(); - assert!(!fetch_error_is_retryable(&FetchError::Pixiv( - PixivError::Json(json_err) - ))); - assert!(!fetch_error_is_retryable(&FetchError::NotFound)); - assert!(!fetch_error_is_retryable(&FetchError::Blocked)); - assert!(!fetch_error_is_retryable(&FetchError::Sensitive)); - assert!(!fetch_error_is_retryable(&FetchError::TooLarge)); - } - #[test] fn caption_from_fields_substitutes_and_escapes() { // The format string is escaped, the field values are substituted diff --git a/crates/x-media/src/site/pixiv/interface.rs b/crates/x-media/src/site/pixiv/interface.rs index bc3eada..286fbd7 100644 --- a/crates/x-media/src/site/pixiv/interface.rs +++ b/crates/x-media/src/site/pixiv/interface.rs @@ -1,6 +1,6 @@ use super::model::{IllustrationModel, TypeModel}; use crate::media::Media; -use crate::site::{FetchError, Fetched}; +use crate::site::{FetchError, Fetched, PixivError}; use html_escape::{encode_double_quoted_attribute, encode_text}; use regex::Regex; use std::sync::LazyLock; @@ -23,6 +23,43 @@ pub async fn fetch_from_url(url: &str) -> Result { Ok(super::api::fetch(id).await?.into()) } +/// Cache key for a pixiv URL: `"pixiv:"`. The prefix is the site id used +/// for caption-format lookup and link-cache keys. +pub fn cache_key(url: &str) -> Option { + PATTERN + .captures(url) + .map(|caps| format!("pixiv:{}", &caps[1])) +} + +/// Pixiv's fetch-retry policy: transient classes only — network errors and +/// HTTP 429/5xx. Permanent 4xx (bad/expired token, forbidden, not found), +/// API/auth errors, unparseable bodies and missing auth are not retried. +pub fn is_retryable(err: &FetchError) -> bool { + match err { + FetchError::Http(_) | FetchError::Transient(_) => true, + FetchError::Pixiv(e) => match e { + PixivError::Http(_) => true, + PixivError::Status(code) if *code == 429 || *code >= 500 => true, + PixivError::Status(_) + | PixivError::Api(_) + | PixivError::Json(_) + | PixivError::NoAuth => false, + }, + _ => false, + } +} + +/// pximg.net is hotlink-protected: downloads must carry the pixiv Referer. +/// The match is on the media host, not the site PATTERN — pixiv's PATTERN +/// only matches `pixiv.net/artworks/...`, never `i.pximg.net`. +pub fn media_headers(url: &str) -> Option> { + if url.to_ascii_lowercase().contains("pximg.net") { + Some(vec![("Referer", "https://www.pixiv.net/".to_string())]) + } else { + None + } +} + #[derive(Debug)] pub struct Illustration { id: String, @@ -233,6 +270,43 @@ mod tests { } } + #[test] + fn is_retryable_classifies_transient_and_permanent() { + // Transient: network errors, explicit transient, pixiv 429/5xx. + assert!(is_retryable(&FetchError::Transient("429".into()))); + assert!(is_retryable(&FetchError::Pixiv(PixivError::Status(429)))); + assert!(is_retryable(&FetchError::Pixiv(PixivError::Status(500)))); + assert!(is_retryable(&FetchError::Pixiv(PixivError::Status(503)))); + // Permanent: pixiv 4xx (bad/expired token, forbidden, not found), + // api/auth errors, unparseable bodies, not-found/blocked/sensitive. + assert!(!is_retryable(&FetchError::Pixiv(PixivError::Status(400)))); + assert!(!is_retryable(&FetchError::Pixiv(PixivError::Status(401)))); + assert!(!is_retryable(&FetchError::Pixiv(PixivError::Status(403)))); + assert!(!is_retryable(&FetchError::Pixiv(PixivError::Status(404)))); + assert!(!is_retryable(&FetchError::Pixiv(PixivError::Api( + "invalid_grant".into() + )))); + assert!(!is_retryable(&FetchError::Pixiv(PixivError::NoAuth))); + let json_err = serde_json::from_str::("x").unwrap_err(); + assert!(!is_retryable(&FetchError::Pixiv(PixivError::Json( + json_err + )))); + assert!(!is_retryable(&FetchError::NotFound)); + assert!(!is_retryable(&FetchError::Blocked)); + assert!(!is_retryable(&FetchError::Sensitive)); + assert!(!is_retryable(&FetchError::TooLarge)); + } + + #[test] + fn media_headers_adds_referer_only_for_pximg() { + assert_eq!( + media_headers("https://i.pximg.net/img-original/img/1.png"), + Some(vec![("Referer", "https://www.pixiv.net/".to_string())]) + ); + assert_eq!(media_headers("https://www.pixiv.net/artworks/1"), None); + assert_eq!(media_headers("https://x.com/u/status/1"), None); + } + #[test] fn ugoira_yields_empty_media() { let v = illust_json( diff --git a/crates/x-media/src/site/pixiv/mod.rs b/crates/x-media/src/site/pixiv/mod.rs index f0c5f28..ee1189d 100644 --- a/crates/x-media/src/site/pixiv/mod.rs +++ b/crates/x-media/src/site/pixiv/mod.rs @@ -3,4 +3,6 @@ mod interface; mod model; pub use api::{PixivAPI, PixivError, disable, fetch, validate}; -pub use interface::{Illustration, PATTERN, enabled, fetch_from_url}; +pub use interface::{ + Illustration, PATTERN, cache_key, enabled, fetch_from_url, is_retryable, media_headers, +}; diff --git a/crates/x-media/src/site/twitter/interface.rs b/crates/x-media/src/site/twitter/interface.rs index 9c193d3..f7fc571 100644 --- a/crates/x-media/src/site/twitter/interface.rs +++ b/crates/x-media/src/site/twitter/interface.rs @@ -49,6 +49,26 @@ pub async fn fetch_from_url(url: &str) -> Result { } } +/// Cache key for a twitter URL: `"twitter:"`. The prefix is the site id +/// used for caption-format lookup and link-cache keys. +pub fn cache_key(url: &str) -> Option { + PATTERN + .captures(url) + .map(|caps| format!("twitter:{}", &caps[1])) +} + +/// Twitter's fetch-retry policy: transient classes only. Not-found, blocked, +/// sensitive (NSFW withholding) and parse failures are permanent — retrying +/// them only wastes attempts against the syndication endpoint. +pub fn is_retryable(err: &FetchError) -> bool { + matches!(err, FetchError::Http(_) | FetchError::Transient(_)) +} + +/// twimg URLs need no extra headers (no hotlink protection). +pub fn media_headers(_url: &str) -> Option> { + None +} + /// A Fetched with no media for withheld tweets: the bot replies /// "No media found" and moves on instead of erroring. fn empty_fetched(url: &str) -> Fetched { @@ -354,6 +374,29 @@ mod tests { } } + #[test] + fn cache_key_prefixes_tweet_id() { + assert_eq!( + cache_key("https://x.com/user/status/1234567890"), + Some("twitter:1234567890".into()) + ); + assert_eq!(cache_key("https://example.com/1"), None); + } + + #[test] + fn is_retryable_classifies_transient_and_permanent() { + // Transient: network errors and explicit transient statuses (the + // `Http` arm shares this match arm with `Transient`). + assert!(is_retryable(&FetchError::Transient("429".into()))); + // Permanent: gone, blocked, withheld, oversized, unparseable. + assert!(!is_retryable(&FetchError::NotFound)); + assert!(!is_retryable(&FetchError::Blocked)); + assert!(!is_retryable(&FetchError::Sensitive)); + assert!(!is_retryable(&FetchError::TooLarge)); + let json_err = serde_json::from_str::("x").unwrap_err(); + assert!(!is_retryable(&FetchError::Json(json_err))); + } + #[test] fn syndication_json_converts_to_fetched() { let raw = fixture(serde_json::json!([ diff --git a/crates/x-media/src/site/twitter/mod.rs b/crates/x-media/src/site/twitter/mod.rs index 77e1338..b5cb663 100644 --- a/crates/x-media/src/site/twitter/mod.rs +++ b/crates/x-media/src/site/twitter/mod.rs @@ -2,4 +2,6 @@ mod auth; mod interface; mod model; -pub use interface::{PATTERN, Tweet, enabled, fetch_from_url}; +pub use interface::{ + PATTERN, Tweet, cache_key, enabled, fetch_from_url, is_retryable, media_headers, +};